From 0804511a77bb949eb8b3c9b28b531ad941ecf481 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 10:20:43 +0800 Subject: [PATCH 01/64] refactor(raft): split commit_resolve into a directory module commit_resolve.rs held apply_tail, verdict, and vote handling in one 736-line file, over the per-file size limit. It becomes a commit_resolve/ directory with apply_tail.rs, verdict.rs, and vote.rs, each keeping its existing logic. The driver core tests also duplicated build_test_scheduler and make_sequenced_txn across catch_up.rs, process.rs, and scheduler.rs. They move into a shared test_support.rs so each test module imports the fixtures instead of redefining them. --- .../calvin/scheduler/driver/core/catch_up.rs | 104 +-- .../scheduler/driver/core/commit_resolve.rs | 736 ------------------ .../driver/core/commit_resolve/apply_tail.rs | 251 ++++++ .../driver/core/commit_resolve/mod.rs | 17 + .../driver/core/commit_resolve/verdict.rs | 275 +++++++ .../driver/core/commit_resolve/vote.rs | 109 +++ .../calvin/scheduler/driver/core/mod.rs | 2 + .../calvin/scheduler/driver/core/process.rs | 100 +-- .../calvin/scheduler/driver/core/scheduler.rs | 65 +- .../scheduler/driver/core/test_support.rs | 192 +++++ 10 files changed, 856 insertions(+), 995 deletions(-) delete mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs index bc5d3e4dc..623313670 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs @@ -157,111 +157,17 @@ impl Scheduler { #[cfg(test)] mod tests { use super::*; - use std::collections::{BTreeSet, HashMap}; use std::sync::atomic::Ordering; - use std::sync::{Arc, Mutex}; use std::time::{Duration, Instant}; - use nodedb_cluster::MultiRaft; - use nodedb_cluster::RoutingTable; - use nodedb_cluster::calvin::types::{ - EngineKeySet, EpochBatch, ReadWriteSet, SchedulerInput, SequencedTxn, SortedVec, TxClass, - VersionedReadSet, - }; - use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerEntry, SequencerStateMachine}; - use nodedb_types::TenantId; + use nodedb_cluster::calvin::SequencerEntry; + use nodedb_cluster::calvin::types::{EpochBatch, SchedulerInput, SequencedTxn}; use nodedb_types::id::{DatabaseId, VShardId}; - use super::super::scheduler::SchedulerParams; - use crate::bridge::dispatch::Dispatcher; - use crate::control::cluster::calvin::scheduler::lock_manager::{ - AcquireOutcome, LockManager, TxnId, + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler, make_sequenced_txn, }; - use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; - use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; - use crate::control::state::SharedState; - use crate::wal::WalManager; - - /// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data - /// Plane is NOT started — tests exercise Control-Plane routing, guards, and - /// request dispatch only, so no core loop is needed. The returned `TempDir` - /// must be kept alive for the scheduler's lifetime (backs the WAL and - /// Raft storage). - fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::TempDir) { - let registry = CalvinCompletionRegistry::new_detached(); - let dir = tempfile::tempdir().unwrap(); - let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); - let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); - let _data_side = data_sides - .pop() - .expect("one configured core has one data side"); - let shared = SharedState::new(dispatcher, wal).unwrap(); - - let rt = RoutingTable::uniform(1, &[1], 1); - let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); - - let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( - HashMap::new(), - Arc::clone(®istry), - ))); - - let (_tx, receiver) = tokio::sync::mpsc::channel(16); - let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); - let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); - let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); - registry.register_verdict_signal_sender(vshard_id, verdict_tx); - - let lock_manager = Arc::new(Mutex::new(LockManager::new())); - - let scheduler = Scheduler::new(SchedulerParams { - vshard_id, - receiver, - shared, - multi_raft, - sequencer_state_machine, - // A freshly-built scheduler has applied nothing, so its watermark is the - // not-yet-applied sentinel (matching `read_applied_recovery` for a clean - // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, - // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit - // every epoch-0 replay before it reaches the lock table — silently - // defeating the end-to-end drain tests below. - fully_applied_epoch: NOT_YET_APPLIED_EPOCH, - applied_tail: BTreeSet::new(), - rebuild_target_epoch: 0, - config: SchedulerConfig::default(), - metrics: SchedulerMetrics::new(), - read_result_rx, - lock_manager, - promotion_rx, - registry, - verdict_rx, - }); - (scheduler, dir) - } - - fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { - let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { - collection: "test_coll".to_string(), - surrogates: SortedVec::new(vec![1]), - }]); - let tx_class = TxClass::new_single_vshard( - ReadWriteSet::new(vec![]), - write_set, - vec![], - TenantId::new(1), - None, - VersionedReadSet::default(), - ) - .expect("valid TxClass"); - SequencedTxn { - epoch, - position, - tx_class, - epoch_system_ms: 1_700_000_000_000, - epoch_vshard_txn_count: 1, - lock_owner: None, - } - } + use crate::control::cluster::calvin::scheduler::lock_manager::{AcquireOutcome, TxnId}; #[tokio::test] async fn drain_catch_up_is_noop_when_no_drop_recorded() { diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve.rs deleted file mode 100644 index bdf2b6f5a..000000000 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve.rs +++ /dev/null @@ -1,736 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Verdict-driven commit resolution for staged static Calvin transactions. -//! -//! A static Calvin dispatch STAGES its transaction on the Data Plane (validate -//! the read-set + buffer the plans, no base mutation). Its executor response -//! carries the local commit vote on `read_set_valid`. This module drives the -//! final step: dispatch a flush (commit, after `commit_redo` has WAL-appended -//! the resolved `TransactionRedo`) or drop (abort) of the staged buffer, wait -//! for its response, then run the commit tail (deposit applied result, record -//! write versions — plus a `CalvinApplied` WAL fallback when no redo record -//! was appended — propose `CompletionAck`) for a flush, or ack-only for a -//! drop. - -use std::sync::atomic::Ordering; -use std::time::Instant; - -use nodedb_cluster::calvin::{SequencerEntry, VerdictSignal}; - -use super::super::types::CommitState; -use super::scheduler::Scheduler; -use super::staged_vote::{StagedVote, staged_commit_vote}; -use crate::bridge::envelope::{Response, Status}; -use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; -use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; - -impl Scheduler { - /// Cast this participant's local commit vote for a staged transaction, then - /// PARK it on the cross-shard commit barrier awaiting the durable GLOBAL - /// verdict — it does NOT self-decide flush-or-drop on its local vote. - /// - /// The staged executor response is validate-only: its `read_set_valid` is - /// this shard's local commit vote (`Some(true)` => commit, `Some(false)` => - /// abort; a `None` from the active/dependent path is treated as commit). The - /// leader proposes that vote via the sequencer Raft group; the sequencer - /// aggregates all participants' votes into a single authoritative - /// `SequencerEntry::Verdict`, applied on every replica. - /// - /// This method moves the txn to [`CommitState::AwaitingVerdict`] WITHOUT - /// dispatching a resolve or drop, then immediately probes - /// `registry.verdict(txn)`: if the verdict is already durable (replay, or a - /// push we raced) it resumes at once via [`Self::resume_on_verdict`]; - /// otherwise it stays parked, holding locks and its staged buffer, until the - /// verdict push, a later probe, or the stall re-probe sweep delivers the - /// verdict. Resuming (in `resume_on_verdict`) is where the flush/drop is - /// dispatched and the flushed/dropped counters bump — using the GLOBAL - /// verdict, never the local vote. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resolve_staged_commit( - &mut self, - txn_id: TxnId, - staged_response: &Response, - ) { - // A staged error is always an abort vote. Only successful staged - // responses may use `None` for the dependent-read path; accepting an - // error-plus-None as commit would let a failed participant flush after - // its peers received a global commit verdict. - let vote = staged_commit_vote(staged_response); - - // Durably propose this participant's commit vote via the sequencer - // Raft group, leader-guarded like `OllpMismatch`: only the data-group - // leader ran read-set validation, so only a leader's vote is - // authoritative. The sequencer aggregates every participant's vote into - // the single global verdict this txn parks on below. An abort travels as - // `AbortVote` so its cause survives to the coordinator. - if self.is_group_leader() { - let entry = match vote.abort_reason() { - Some(reason) => SequencerEntry::AbortVote { - epoch: txn_id.epoch, - position: txn_id.position, - vshard: self.vshard_id, - reason, - }, - None => SequencerEntry::Vote { - epoch: txn_id.epoch, - position: txn_id.position, - vshard: self.vshard_id, - commit: true, - }, - }; - self.propose_sequencer_entry(entry, txn_id, "commit vote"); - } - - if vote == StagedVote::SerializationConflict { - // The staged slice's read-set was no longer current: observe it, the - // same node-global signal the direct-apply path records. A - // participant error never validated a read-set, so it must not count - // here. - self.shared - .calvin_counters - .read_set_validation_failures - .fetch_add(1, Ordering::Relaxed); - } - - // PARK on the barrier: transition to `AwaitingVerdict` and arm the stall - // deadline. Do NOT dispatch resolve/drop here — the GLOBAL verdict, not - // this local vote, decides. If the txn already vanished (torn down - // elsewhere), there is nothing to park. - match self.pending.get_mut(&txn_id) { - Some(pending) => { - pending.commit_state = Some(CommitState::AwaitingVerdict); - // no-determinism: local stall-warning deadline only; the global replicated verdict, not this wall-clock, decides commit/abort. - pending.verdict_deadline = Some(Instant::now() + self.config.verdict_stall_warn()); - } - None => return, - } - - // PROBE on park (correctness backstop): the verdict may already be - // durable — on replay, or a push that raced ahead of this park. Resume - // immediately if so; the double-resume guard in `resume_on_verdict` - // makes a later duplicate push/probe a no-op. - if let Some(verdict) = self.registry.verdict(nodedb_cluster::calvin::TxnId::new( - txn_id.epoch, - txn_id.position, - )) { - self.resume_on_verdict(txn_id, verdict); - } - } - - /// Resume a txn parked in [`CommitState::AwaitingVerdict`] once the durable - /// GLOBAL verdict is known: dispatch its flush (commit) or drop (abort). - /// - /// `committed` is the authoritative cross-shard verdict — NOT this shard's - /// local vote. On commit, dispatches `MetaOp::CalvinResolve` and moves the - /// txn to [`CommitState::AwaitingRedoResolve`] (the resolved redo is - /// WAL-appended and the flush dispatched from [`Self::finish_redo_resolve`]). - /// On abort, dispatches the drop directly and moves the txn to - /// [`CommitState::AwaitingResolve`]. Bumps the flushed / dropped counter. The - /// commit tail runs later in [`Self::finish_resolved_commit`], once the - /// flush/drop response arrives. - /// - /// Double-resume guard: the verdict push and the probe-on-park (and the - /// stall re-probe sweep) can all fire for one txn, so this first confirms the - /// txn is still `Some(AwaitingVerdict)` — if it already transitioned out - /// (resolve/drop dispatched, or completed), this is a no-op. This guarantees - /// the flush/drop is dispatched exactly once. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resume_on_verdict( - &mut self, - txn_id: TxnId, - committed: bool, - ) { - // Guard: only a still-parked txn resumes. Mirrors `handle_completion`'s - // state-match so a duplicate push/probe/timeout is idempotent. - if !matches!( - self.pending.get(&txn_id).and_then(|p| p.commit_state), - Some(CommitState::AwaitingVerdict) - ) { - return; - } - - let dispatched = if committed { - // Resolve the staged post-images into a replayable `RedoRecord` - // first; the redo is WAL-appended (in `finish_redo_resolve`) before - // the flush is dispatched, restoring restart durability for this - // vShard's slice of a multi-shard Calvin commit. - self.dispatch_calvin_resolve(txn_id) - } else { - self.dispatch_commit_resolution(txn_id, false, None) - }; - if !dispatched { - // Resolve/drop dispatch failed: complete the txn as an infra error so - // its locks release and the epoch advances rather than stalling. The - // staged buffer is reclaimed by a later drop or on core teardown. - self.metrics.record_executor_error(); - self.metrics - .record_infra_abort(infra_abort_reason::IO_ERROR); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - return; - } - - if let Some(pending) = self.pending.get_mut(&txn_id) { - pending.commit_state = Some(if committed { - CommitState::AwaitingRedoResolve - } else { - CommitState::AwaitingResolve { - committed: false, - redo_lsn: None, - } - }); - // No longer parked: clear the stall deadline. - pending.verdict_deadline = None; - } - - if committed { - self.shared - .calvin_counters - .commits_flushed - .fetch_add(1, Ordering::Relaxed); - } else { - self.shared - .calvin_counters - .commits_dropped - .fetch_add(1, Ordering::Relaxed); - } - } - - /// Handle a pushed [`VerdictSignal`] from this node's completion registry. - /// - /// Matches the signal to the parked txn by `(epoch, position)` and resumes - /// it. A signal for a txn this scheduler does not host, or one that already - /// resumed, is a harmless no-op (the double-resume guard covers the latter). - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn handle_verdict_signal( - &mut self, - signal: VerdictSignal, - ) { - let txn_id = TxnId::new(signal.epoch, signal.position); - self.resume_on_verdict(txn_id, signal.verdict.is_commit()); - } - - /// Sweep parked `AwaitingVerdict` txns whose stall deadline has passed. - /// - /// For each stalled txn, RE-PROBE the durable verdict: if it is now known, - /// resume (a push we dropped on a full channel, or a verdict that landed - /// after the last probe). If it is STILL unknown, KEEP WAITING — hold locks, - /// emit a stall metric + warning, and re-arm the deadline so the warning is - /// rate-limited rather than per-iteration. It NEVER releases locks and NEVER - /// unilaterally aborts: a participant cannot know whether a peer already - /// flushed a COMMIT, so aborting one side while a peer committed would tear - /// the transaction. The verdict is guaranteed to arrive eventually — a - /// post-failover leader re-aggregates the replicated votes (seeded on every - /// replica) into the same verdict — so waiting is always the safe action. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn check_awaiting_verdict_stalls( - &mut self, - ) { - // no-determinism: stall-detection clock drives warnings/metrics only; this path holds locks and never aborts, so it cannot affect the replicated outcome. - let now = Instant::now(); - let stalled: Vec = self - .pending - .iter() - .filter(|(_, p)| matches!(p.commit_state, Some(CommitState::AwaitingVerdict))) - .filter(|(_, p)| p.verdict_deadline.is_some_and(|d| now >= d)) - .map(|(id, _)| *id) - .collect(); - - for txn_id in stalled { - if let Some(verdict) = self.registry.verdict(nodedb_cluster::calvin::TxnId::new( - txn_id.epoch, - txn_id.position, - )) { - self.resume_on_verdict(txn_id, verdict); - continue; - } - - // Verdict still unknown: keep waiting, hold locks, never abort. - self.metrics.record_verdict_stall(); - tracing::warn!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - "calvin: staged txn still awaiting the cross-shard verdict past its stall \ - deadline; HOLDING locks and waiting (never aborting — a peer may have already \ - flushed a commit). The verdict is guaranteed to arrive." - ); - if let Some(pending) = self.pending.get_mut(&txn_id) { - pending.verdict_deadline = Some(now + self.config.verdict_stall_warn()); - } - } - } - - /// Run the commit tail once a flush/drop response has returned. - /// - /// On a successful flush the full commit tail runs (deposit applied result, - /// `CalvinApplied` WAL + write-version recording, `CompletionAck`). On a - /// successful drop only the `CompletionAck` is proposed — the coordinator's - /// completion waiter still fires and the epoch advances, but nothing was - /// written so there is no result to deposit, no apply LSN, and no versions - /// to record. A non-`Ok` resolve response is treated as an executor error. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn finish_resolved_commit( - &mut self, - txn_id: TxnId, - response: Response, - committed: bool, - redo_lsn: Option, - ) { - let completed = if response.status == Status::Ok { - if committed { - self.commit_apply_tail(txn_id, response, redo_lsn) - } else { - self.propose_sequencer_entry( - SequencerEntry::CompletionAck { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }, - txn_id, - "completion ack (dropped)", - ); - true - } - } else { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - committed, - "calvin: flush/drop response was not Ok while applying an already-committed \ - verdict; forcing infra-abort completion so locks release and the epoch advances" - ); - false - }; - - if completed { - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } else { - // The cross-shard verdict is already globally durable, and a commit's - // resolved redo was WAL-appended before this flush — so recovery - // re-applies the write. A local flush/apply or WAL-marker failure is - // therefore an infrastructure event, NOT an outcome change. It must - // never leave the txn parked: holding its locks forever wedges every - // txn queued behind those keys and freezes this vShard's epoch - // watermark (which anchors cross-shard BEGIN snapshots), and nothing - // re-drives a non-`AwaitingVerdict` pending entry. Surface the infra - // abort and force completion — the same forward-progress contract the - // resolve/drop dispatch-failure path in `resume_on_verdict` follows. - self.metrics.record_executor_error(); - self.metrics - .record_infra_abort(infra_abort_reason::IO_ERROR); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } - } - - /// Deposit the applied result, durably mark the apply, record the apply's - /// write versions, and propose the `CompletionAck`. - /// - /// Shared by the flush-completion path and the direct-apply (dependent / - /// active) apply path. - /// - /// `redo_lsn` is `Some(lsn)` when a `TransactionRedo` record was already - /// WAL-appended for this commit's non-empty write set (`finish_redo_resolve`) - /// — that record already IS the durable applied marker, so only write - /// versions are recorded at it. `None` (a drop, an empty-ops staged commit, - /// or the direct-apply dependent/active path, which carries no redo record) - /// falls back to appending a `CalvinApplied` marker here, exactly as before - /// this record existed. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn commit_apply_tail( - &mut self, - txn_id: TxnId, - response: Response, - redo_lsn: Option, - ) -> bool { - // Deposit the FULL applied Response (affected-count + watermark + any - // RETURNING rows) into the local sidecar BEFORE proposing the replicated - // CompletionAck. The ack fires the coordinator's completion oneshot on - // every sequencer member, so depositing first guarantees the result is - // present by the time the coordinator drains it — no lost result, no - // race. - // - // Gated on the PRIMARY-WRITE participant: any participant whose slice - // carries the user's non-edge DML (Document/KV/Vector/etc.), as opposed - // to the implicit graph-edge cleanup that dual-homes alongside it. A - // multi-collection cross-shard COMMIT has MANY primary-write - // participants — each a plain affected-count write — and they coalesce: - // the first applied response stands for the coordinator (which discards - // it for a COMMIT tag anyway), and the plain-write siblings do not - // conflict. Only a genuine cross-shard RETURNING union — two - // participants each carrying RETURNING rows — records `Conflict`. - // Results travel via this in-process sidecar only — never the sequencer - // Raft log. - let (has_primary_write, has_returning) = self - .pending - .get(&txn_id) - .map(|p| (p.has_primary_write, p.has_returning)) - .unwrap_or((false, false)); - if has_primary_write { - use std::collections::hash_map::Entry; - - use crate::control::state::CalvinApplyResult; - - let key = nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position); - let mut results = self - .shared - .calvin_apply_results - .lock() - .unwrap_or_else(|p| p.into_inner()); - match results.entry(key) { - Entry::Vacant(slot) => { - slot.insert(CalvinApplyResult::Single { - response, - has_returning, - }); - } - Entry::Occupied(mut slot) => { - // Derive both facts from the existing entry BEFORE any - // insert, so the immutable borrow does not outlive the - // mutable one. - let existing_returning = matches!( - slot.get(), - CalvinApplyResult::Single { - has_returning: true, - .. - } - ); - let already_conflict = matches!(slot.get(), CalvinApplyResult::Conflict); - - if already_conflict { - // A RETURNING union was already recorded; stays Conflict. - } else if has_returning && existing_returning { - // Two RETURNING-bearing participants for one Calvin txn: - // a cross-shard RETURNING union, which is unsupported. - // Record Conflict so the coordinator fails the statement - // loudly rather than returning one shard's partial rows. - tracing::error!( - epoch = txn_id.epoch, - position = txn_id.position, - vshard = self.vshard_id, - "two RETURNING-bearing participants for one Calvin txn — cross-shard \ - RETURNING union unsupported" - ); - slot.insert(CalvinApplyResult::Conflict); - } else if has_returning { - // The incoming participant carries the rows; the existing - // entry was a plain affected-count sibling. Rows win. - slot.insert(CalvinApplyResult::Single { - response, - has_returning: true, - }); - } else { - // Incoming is a plain write; keep the existing entry — a - // multi-collection cross-shard COMMIT coalesces (the - // coordinator discards it for a COMMIT tag anyway). - } - } - } - } - let applied_lsn = match redo_lsn { - // The TransactionRedo record already durably marks this apply — the - // SAME shard-local WAL-LSN space fast-path writes and read - // watermarks use. Record the apply's per-key write versions at it; - // no second (CalvinApplied) marker is written. - Some(lsn) => { - self.record_calvin_write_versions(txn_id, lsn); - Some(lsn) - } - None => match self.shared.wal.append_calvin_applied( - crate::types::VShardId::new(self.vshard_id), - txn_id.epoch, - txn_id.position, - ) { - // The CalvinApplied WAL LSN is the committed write-LSN for this - // apply — the SAME shard-local WAL-LSN space fast-path writes and - // read watermarks use. Record the apply's per-key write versions - // at it once it exists; it does not exist yet at dispatch time. - Ok(applied_lsn) => { - self.record_calvin_write_versions(txn_id, applied_lsn); - Some(applied_lsn) - } - Err(e) => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: failed to write CalvinApplied WAL record" - ); - None - } - }, - }; - let Some(lsn) = applied_lsn else { - // The apply cannot be acknowledged without a durable participant - // LSN: CDC and write-version consumers would otherwise observe a - // successful commit with no authoritative ordering point. - return false; - }; - // Control change-stream events are distinct from Data-Plane - // WriteEvents. Publish the participant-local logical manifests once, - // from the data-group leader, at the authoritative committed LSN. - if self.is_group_leader() - && let Some(pending) = self.pending.get_mut(&txn_id) - { - let tenant_id = pending.txn.tx_class.tenant_id; - let database_id = pending.txn.tx_class.database_id; - for change_set in std::mem::take(&mut pending.change_sets) { - crate::control::server::dispatch_utils::publish_change_set_with_lsn( - &self.shared, - tenant_id, - database_id, - change_set, - lsn, - ); - } - } - self.propose_sequencer_entry( - SequencerEntry::CompletionAck { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }, - txn_id, - "completion ack", - ); - true - } -} - -#[cfg(test)] -mod tests { - use super::*; - use std::collections::HashMap; - use std::sync::{Arc, Mutex}; - - use nodedb_cluster::MultiRaft; - use nodedb_cluster::RoutingTable; - use nodedb_cluster::calvin::types::{ - EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet, - }; - use nodedb_cluster::calvin::{ - AbortReason, CalvinCompletionRegistry, ParticipantVote, SequencerStateMachine, - VerdictOutcome, - }; - use nodedb_physical::physical_plan::PhysicalPlan; - use nodedb_physical::physical_plan::meta::MetaOp; - use nodedb_types::TenantId; - - use super::super::scheduler::{Scheduler, SchedulerParams}; - use crate::bridge::dispatch::Dispatcher; - use crate::bridge::envelope::Payload; - use crate::control::cluster::calvin::scheduler::driver::types::PendingTxn; - use crate::control::cluster::calvin::scheduler::lock_manager::LockManager; - use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; - use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; - use crate::control::state::SharedState; - use crate::types::{Lsn, RequestId}; - use crate::wal::WalManager; - - /// Same minimal scheduler fixture as the process/catch_up driver tests, - /// retaining its Data-Plane request receiver for tests that must observe - /// scheduler dispatches. - fn build_test_scheduler_with_data_side( - vshard_id: u32, - registry: Arc, - ) -> ( - Scheduler, - tempfile::TempDir, - crate::bridge::dispatch::CoreChannelDataSide, - ) { - let dir = tempfile::tempdir().unwrap(); - let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); - let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); - let data_side = data_sides - .pop() - .expect("one configured core has one data side"); - let shared = SharedState::new(dispatcher, wal).unwrap(); - - let rt = RoutingTable::uniform(1, &[1], 1); - let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); - - let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( - HashMap::new(), - Arc::clone(®istry), - ))); - - let (_tx, receiver) = tokio::sync::mpsc::channel(16); - let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); - let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); - let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); - registry.register_verdict_signal_sender(vshard_id, verdict_tx); - - let lock_manager = Arc::new(Mutex::new(LockManager::new())); - - let scheduler = Scheduler::new(SchedulerParams { - vshard_id, - receiver, - shared, - multi_raft, - sequencer_state_machine, - fully_applied_epoch: NOT_YET_APPLIED_EPOCH, - applied_tail: std::collections::BTreeSet::new(), - rebuild_target_epoch: 0, - config: SchedulerConfig::default(), - metrics: SchedulerMetrics::new(), - read_result_rx, - lock_manager, - promotion_rx, - registry, - verdict_rx, - }); - (scheduler, dir, data_side) - } - - /// Build a static-write `SequencedTxn` at `(epoch, position)`. - fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { - PendingTxn { - txn, - lock_owner: txn_id, - dispatch_time: std::time::Instant::now(), - has_primary_write: true, - has_returning: false, - change_sets: Vec::new(), - commit_state: Some(CommitState::Staged), - verdict_deadline: None, - } - } - - fn staged_response(status: Status, read_set_valid: Option) -> Response { - Response { - request_id: RequestId::new(1), - status, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: None, - read_set_valid, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - } - } - - fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { - let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { - collection: "test_coll".to_string(), - surrogates: SortedVec::new(vec![1]), - }]); - let tx_class = TxClass::new_single_vshard( - ReadWriteSet::new(vec![]), - write_set, - vec![], - TenantId::new(1), - None, - VersionedReadSet::default(), - ) - .expect("valid TxClass"); - SequencedTxn { - epoch, - position, - tx_class, - epoch_system_ms: 1_700_000_000_000, - epoch_vshard_txn_count: 1, - lock_owner: None, - } - } - - /// A false vote from either participant makes the only global verdict abort; - /// applying that durable verdict broadcasts the abort to every parked local - /// participant. The scheduler's `resume_on_verdict(false)` then dispatches a - /// drop, never a resolve/flush, on each recipient. - #[tokio::test] - async fn two_participant_false_vote_broadcasts_global_abort_to_every_scheduler() { - let registry = CalvinCompletionRegistry::new_detached(); - let txn = nodedb_cluster::calvin::TxnId::new(14, 2); - let txn_id = TxnId::new(14, 2); - let (mut first_scheduler, _first_dir, mut first_data) = - build_test_scheduler_with_data_side(7, Arc::clone(®istry)); - let (mut second_scheduler, _second_dir, mut second_data) = - build_test_scheduler_with_data_side(9, Arc::clone(®istry)); - first_scheduler - .pending - .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); - second_scheduler - .pending - .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); - - // Local staging votes only park their own staged slices; neither the - // affirmative nor the failed participant may resolve or drop unilaterally. - first_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); - second_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Error, None)); - for (scheduler, data_side) in [ - (&first_scheduler, &mut first_data), - (&second_scheduler, &mut second_data), - ] { - assert!(matches!( - scheduler - .pending - .get(&txn_id) - .and_then(|pending| pending.commit_state), - Some(CommitState::AwaitingVerdict) - )); - assert!(data_side.request_rx.try_pop().is_err()); - } - - // Model the replicated vote entries and their resulting durable verdict. - // The shared registry sends each scheduler's actual registered channel. - registry.seed_expected(txn, 2); - registry.note_vote(txn, 7, ParticipantVote::Commit); - assert!(registry.drain_unproposed_verdicts().is_empty()); - registry.note_vote( - txn, - 9, - ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), - ); - assert_eq!( - registry.drain_unproposed_verdicts(), - vec![( - txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)) - )] - ); - registry.note_verdict( - txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)), - ); - assert_eq!(registry.verdict(txn), Some(false)); - - let first_signal = first_scheduler - .verdict_rx - .try_recv() - .expect("registry must signal the first registered scheduler"); - let second_signal = second_scheduler - .verdict_rx - .try_recv() - .expect("registry must signal the second registered scheduler"); - first_scheduler.handle_verdict_signal(first_signal); - second_scheduler.handle_verdict_signal(second_signal); - - for (scheduler, data_side) in [ - (&first_scheduler, &mut first_data), - (&second_scheduler, &mut second_data), - ] { - assert!(matches!( - scheduler - .pending - .get(&txn_id) - .and_then(|pending| pending.commit_state), - Some(CommitState::AwaitingResolve { - committed: false, - redo_lsn: None - }) - )); - let request = data_side - .request_rx - .try_pop() - .expect("global abort must dispatch a drop to every participant"); - assert!(matches!( - request.inner.plan, - PhysicalPlan::Meta(MetaOp::CalvinDrop { - epoch: 14, - position: 2 - }) - )); - assert!(data_side.request_rx.try_pop().is_err()); - } - } -} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs new file mode 100644 index 000000000..d9cd61074 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Commit tail: runs once a flush/drop response has returned, depositing the +//! applied result, marking the apply durable, recording write versions, and +//! proposing the `CompletionAck`. + +use nodedb_cluster::calvin::SequencerEntry; + +use crate::bridge::envelope::{Response, Status}; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; + +impl Scheduler { + /// Run the commit tail once a flush/drop response has returned. + /// + /// On a successful flush the full commit tail runs (deposit applied result, + /// `CalvinApplied` WAL + write-version recording, `CompletionAck`). On a + /// successful drop only the `CompletionAck` is proposed — the coordinator's + /// completion waiter still fires and the epoch advances, but nothing was + /// written so there is no result to deposit, no apply LSN, and no versions + /// to record. A non-`Ok` resolve response is treated as an executor error. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn finish_resolved_commit( + &mut self, + txn_id: TxnId, + response: Response, + committed: bool, + redo_lsn: Option, + ) { + let completed = if response.status == Status::Ok { + if committed { + self.commit_apply_tail(txn_id, response, redo_lsn) + } else { + self.propose_sequencer_entry( + SequencerEntry::CompletionAck { + epoch: txn_id.epoch, + position: txn_id.position, + vshard_id: self.vshard_id, + }, + txn_id, + "completion ack (dropped)", + ); + true + } + } else { + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + committed, + "calvin: flush/drop response was not Ok while applying an already-committed \ + verdict; forcing infra-abort completion so locks release and the epoch advances" + ); + false + }; + + if completed { + self.metrics.record_completed(); + self.on_txn_complete(txn_id); + } else { + // The cross-shard verdict is already globally durable, and a commit's + // resolved redo was WAL-appended before this flush — so recovery + // re-applies the write. A local flush/apply or WAL-marker failure is + // therefore an infrastructure event, NOT an outcome change. It must + // never leave the txn parked: holding its locks forever wedges every + // txn queued behind those keys and freezes this vShard's epoch + // watermark (which anchors cross-shard BEGIN snapshots), and nothing + // re-drives a non-`AwaitingVerdict` pending entry. Surface the infra + // abort and force completion — the same forward-progress contract the + // resolve/drop dispatch-failure path in `resume_on_verdict` follows. + self.metrics.record_executor_error(); + self.metrics + .record_infra_abort(infra_abort_reason::IO_ERROR); + self.metrics.record_completed(); + self.on_txn_complete(txn_id); + } + } + + /// Deposit the applied result, durably mark the apply, record the apply's + /// write versions, and propose the `CompletionAck`. + /// + /// Shared by the flush-completion path and the direct-apply (dependent / + /// active) apply path. + /// + /// `redo_lsn` is `Some(lsn)` when a `TransactionRedo` record was already + /// WAL-appended for this commit's non-empty write set (`finish_redo_resolve`) + /// — that record already IS the durable applied marker, so only write + /// versions are recorded at it. `None` (a drop, an empty-ops staged commit, + /// or the direct-apply dependent/active path, which carries no redo record) + /// falls back to appending a `CalvinApplied` marker here, exactly as before + /// this record existed. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn commit_apply_tail( + &mut self, + txn_id: TxnId, + response: Response, + redo_lsn: Option, + ) -> bool { + // Deposit the FULL applied Response (affected-count + watermark + any + // RETURNING rows) into the local sidecar BEFORE proposing the replicated + // CompletionAck. The ack fires the coordinator's completion oneshot on + // every sequencer member, so depositing first guarantees the result is + // present by the time the coordinator drains it — no lost result, no + // race. + // + // Gated on the PRIMARY-WRITE participant: any participant whose slice + // carries the user's non-edge DML (Document/KV/Vector/etc.), as opposed + // to the implicit graph-edge cleanup that dual-homes alongside it. A + // multi-collection cross-shard COMMIT has MANY primary-write + // participants — each a plain affected-count write — and they coalesce: + // the first applied response stands for the coordinator (which discards + // it for a COMMIT tag anyway), and the plain-write siblings do not + // conflict. Only a genuine cross-shard RETURNING union — two + // participants each carrying RETURNING rows — records `Conflict`. + // Results travel via this in-process sidecar only — never the sequencer + // Raft log. + let (has_primary_write, has_returning) = self + .pending + .get(&txn_id) + .map(|p| (p.has_primary_write, p.has_returning)) + .unwrap_or((false, false)); + if has_primary_write { + use std::collections::hash_map::Entry; + + use crate::control::state::CalvinApplyResult; + + let key = nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position); + let mut results = self + .shared + .calvin_apply_results + .lock() + .unwrap_or_else(|p| p.into_inner()); + match results.entry(key) { + Entry::Vacant(slot) => { + slot.insert(CalvinApplyResult::Single { + response, + has_returning, + }); + } + Entry::Occupied(mut slot) => { + // Derive both facts from the existing entry BEFORE any + // insert, so the immutable borrow does not outlive the + // mutable one. + let existing_returning = matches!( + slot.get(), + CalvinApplyResult::Single { + has_returning: true, + .. + } + ); + let already_conflict = matches!(slot.get(), CalvinApplyResult::Conflict); + + if already_conflict { + // A RETURNING union was already recorded; stays Conflict. + } else if has_returning && existing_returning { + // Two RETURNING-bearing participants for one Calvin txn: + // a cross-shard RETURNING union, which is unsupported. + // Record Conflict so the coordinator fails the statement + // loudly rather than returning one shard's partial rows. + tracing::error!( + epoch = txn_id.epoch, + position = txn_id.position, + vshard = self.vshard_id, + "two RETURNING-bearing participants for one Calvin txn — cross-shard \ + RETURNING union unsupported" + ); + slot.insert(CalvinApplyResult::Conflict); + } else if has_returning { + // The incoming participant carries the rows; the existing + // entry was a plain affected-count sibling. Rows win. + slot.insert(CalvinApplyResult::Single { + response, + has_returning: true, + }); + } else { + // Incoming is a plain write; keep the existing entry — a + // multi-collection cross-shard COMMIT coalesces (the + // coordinator discards it for a COMMIT tag anyway). + } + } + } + } + let applied_lsn = match redo_lsn { + // The TransactionRedo record already durably marks this apply — the + // SAME shard-local WAL-LSN space fast-path writes and read + // watermarks use. Record the apply's per-key write versions at it; + // no second (CalvinApplied) marker is written. + Some(lsn) => { + self.record_calvin_write_versions(txn_id, lsn); + Some(lsn) + } + None => match self.shared.wal.append_calvin_applied( + crate::types::VShardId::new(self.vshard_id), + txn_id.epoch, + txn_id.position, + ) { + // The CalvinApplied WAL LSN is the committed write-LSN for this + // apply — the SAME shard-local WAL-LSN space fast-path writes and + // read watermarks use. Record the apply's per-key write versions + // at it once it exists; it does not exist yet at dispatch time. + Ok(applied_lsn) => { + self.record_calvin_write_versions(txn_id, applied_lsn); + Some(applied_lsn) + } + Err(e) => { + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + error = %e, + "calvin: failed to write CalvinApplied WAL record" + ); + None + } + }, + }; + let Some(lsn) = applied_lsn else { + // The apply cannot be acknowledged without a durable participant + // LSN: CDC and write-version consumers would otherwise observe a + // successful commit with no authoritative ordering point. + return false; + }; + // Control change-stream events are distinct from Data-Plane + // WriteEvents. Publish the participant-local logical manifests once, + // from the data-group leader, at the authoritative committed LSN. + if self.is_group_leader() + && let Some(pending) = self.pending.get_mut(&txn_id) + { + let tenant_id = pending.txn.tx_class.tenant_id; + let database_id = pending.txn.tx_class.database_id; + for change_set in std::mem::take(&mut pending.change_sets) { + crate::control::server::dispatch_utils::publish_change_set_with_lsn( + &self.shared, + tenant_id, + database_id, + change_set, + lsn, + ); + } + } + self.propose_sequencer_entry( + SequencerEntry::CompletionAck { + epoch: txn_id.epoch, + position: txn_id.position, + vshard_id: self.vshard_id, + }, + txn_id, + "completion ack", + ); + true + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs new file mode 100644 index 000000000..be4a38649 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Verdict-driven commit resolution for staged static Calvin transactions. +//! +//! A static Calvin dispatch STAGES its transaction on the Data Plane (validate +//! the read-set + buffer the plans, no base mutation). Its executor response +//! carries the local commit vote on `read_set_valid`. This module drives the +//! final step: dispatch a flush (commit, after `commit_redo` has WAL-appended +//! the resolved `TransactionRedo`) or drop (abort) of the staged buffer, wait +//! for its response, then run the commit tail (deposit applied result, record +//! write versions — plus a `CalvinApplied` WAL fallback when no redo record +//! was appended — propose `CompletionAck`) for a flush, or ack-only for a +//! drop. + +mod apply_tail; +mod verdict; +mod vote; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs new file mode 100644 index 000000000..da27702da --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs @@ -0,0 +1,275 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Resume-on-verdict, verdict-signal handling, and the stall re-probe sweep +//! for a staged Calvin transaction parked on the cross-shard commit barrier. + +use std::sync::atomic::Ordering; +use std::time::Instant; + +use nodedb_cluster::calvin::VerdictSignal; + +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::driver::types::CommitState; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; + +impl Scheduler { + /// Resume a txn parked in [`CommitState::AwaitingVerdict`] once the durable + /// GLOBAL verdict is known: dispatch its flush (commit) or drop (abort). + /// + /// `committed` is the authoritative cross-shard verdict — NOT this shard's + /// local vote. On commit, dispatches `MetaOp::CalvinResolve` and moves the + /// txn to [`CommitState::AwaitingRedoResolve`] (the resolved redo is + /// WAL-appended and the flush dispatched from [`Self::finish_redo_resolve`]). + /// On abort, dispatches the drop directly and moves the txn to + /// [`CommitState::AwaitingResolve`]. Bumps the flushed / dropped counter. The + /// commit tail runs later in [`Self::finish_resolved_commit`], once the + /// flush/drop response arrives. + /// + /// Double-resume guard: the verdict push and the probe-on-park (and the + /// stall re-probe sweep) can all fire for one txn, so this first confirms the + /// txn is still `Some(AwaitingVerdict)` — if it already transitioned out + /// (resolve/drop dispatched, or completed), this is a no-op. This guarantees + /// the flush/drop is dispatched exactly once. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resume_on_verdict( + &mut self, + txn_id: TxnId, + committed: bool, + ) { + // Guard: only a still-parked txn resumes. Mirrors `handle_completion`'s + // state-match so a duplicate push/probe/timeout is idempotent. + if !matches!( + self.pending.get(&txn_id).and_then(|p| p.commit_state), + Some(CommitState::AwaitingVerdict) + ) { + return; + } + + let dispatched = if committed { + // Resolve the staged post-images into a replayable `RedoRecord` + // first; the redo is WAL-appended (in `finish_redo_resolve`) before + // the flush is dispatched, restoring restart durability for this + // vShard's slice of a multi-shard Calvin commit. + self.dispatch_calvin_resolve(txn_id) + } else { + self.dispatch_commit_resolution(txn_id, false, None) + }; + if !dispatched { + // Resolve/drop dispatch failed: complete the txn as an infra error so + // its locks release and the epoch advances rather than stalling. The + // staged buffer is reclaimed by a later drop or on core teardown. + self.metrics.record_executor_error(); + self.metrics + .record_infra_abort(infra_abort_reason::IO_ERROR); + self.metrics.record_completed(); + self.on_txn_complete(txn_id); + return; + } + + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.commit_state = Some(if committed { + CommitState::AwaitingRedoResolve + } else { + CommitState::AwaitingResolve { + committed: false, + redo_lsn: None, + } + }); + // No longer parked: clear the stall deadline. + pending.verdict_deadline = None; + } + + if committed { + self.shared + .calvin_counters + .commits_flushed + .fetch_add(1, Ordering::Relaxed); + } else { + self.shared + .calvin_counters + .commits_dropped + .fetch_add(1, Ordering::Relaxed); + } + } + + /// Handle a pushed [`VerdictSignal`] from this node's completion registry. + /// + /// Matches the signal to the parked txn by `(epoch, position)` and resumes + /// it. A signal for a txn this scheduler does not host, or one that already + /// resumed, is a harmless no-op (the double-resume guard covers the latter). + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn handle_verdict_signal( + &mut self, + signal: VerdictSignal, + ) { + let txn_id = TxnId::new(signal.epoch, signal.position); + self.resume_on_verdict(txn_id, signal.verdict.is_commit()); + } + + /// Sweep parked `AwaitingVerdict` txns whose stall deadline has passed. + /// + /// For each stalled txn, RE-PROBE the durable verdict: if it is now known, + /// resume (a push we dropped on a full channel, or a verdict that landed + /// after the last probe). If it is STILL unknown, KEEP WAITING — hold locks, + /// emit a stall metric + warning, and re-arm the deadline so the warning is + /// rate-limited rather than per-iteration. It NEVER releases locks and NEVER + /// unilaterally aborts: a participant cannot know whether a peer already + /// flushed a COMMIT, so aborting one side while a peer committed would tear + /// the transaction. The verdict is guaranteed to arrive eventually — a + /// post-failover leader re-aggregates the replicated votes (seeded on every + /// replica) into the same verdict — so waiting is always the safe action. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn check_awaiting_verdict_stalls( + &mut self, + ) { + // no-determinism: stall-detection clock drives warnings/metrics only; this path holds locks and never aborts, so it cannot affect the replicated outcome. + let now = Instant::now(); + let stalled: Vec = self + .pending + .iter() + .filter(|(_, p)| matches!(p.commit_state, Some(CommitState::AwaitingVerdict))) + .filter(|(_, p)| p.verdict_deadline.is_some_and(|d| now >= d)) + .map(|(id, _)| *id) + .collect(); + + for txn_id in stalled { + if let Some(verdict) = self.registry.verdict(nodedb_cluster::calvin::TxnId::new( + txn_id.epoch, + txn_id.position, + )) { + self.resume_on_verdict(txn_id, verdict); + continue; + } + + // Verdict still unknown: keep waiting, hold locks, never abort. + self.metrics.record_verdict_stall(); + tracing::warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + "calvin: staged txn still awaiting the cross-shard verdict past its stall \ + deadline; HOLDING locks and waiting (never aborting — a peer may have already \ + flushed a commit). The verdict is guaranteed to arrive." + ); + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.verdict_deadline = Some(now + self.config.verdict_stall_warn()); + } + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::calvin::{ + AbortReason, CalvinCompletionRegistry, ParticipantVote, VerdictOutcome, + }; + use nodedb_physical::physical_plan::PhysicalPlan; + use nodedb_physical::physical_plan::meta::MetaOp; + + use super::*; + use crate::bridge::envelope::Status; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler_with_data_side, make_sequenced_txn, staged_pending, staged_response, + }; + + /// A false vote from either participant makes the only global verdict abort; + /// applying that durable verdict broadcasts the abort to every parked local + /// participant. The scheduler's `resume_on_verdict(false)` then dispatches a + /// drop, never a resolve/flush, on each recipient. + #[tokio::test] + async fn two_participant_false_vote_broadcasts_global_abort_to_every_scheduler() { + let registry = CalvinCompletionRegistry::new_detached(); + let txn = nodedb_cluster::calvin::TxnId::new(14, 2); + let txn_id = TxnId::new(14, 2); + let (mut first_scheduler, _first_dir, mut first_data) = + build_test_scheduler_with_data_side(7, Arc::clone(®istry)); + let (mut second_scheduler, _second_dir, mut second_data) = + build_test_scheduler_with_data_side(9, Arc::clone(®istry)); + first_scheduler + .pending + .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); + second_scheduler + .pending + .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); + + // Local staging votes only park their own staged slices; neither the + // affirmative nor the failed participant may resolve or drop unilaterally. + first_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); + second_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Error, None)); + for (scheduler, data_side) in [ + (&first_scheduler, &mut first_data), + (&second_scheduler, &mut second_data), + ] { + assert!(matches!( + scheduler + .pending + .get(&txn_id) + .and_then(|pending| pending.commit_state), + Some(CommitState::AwaitingVerdict) + )); + assert!(data_side.request_rx.try_pop().is_err()); + } + + // Model the replicated vote entries and their resulting durable verdict. + // The shared registry sends each scheduler's actual registered channel. + registry.seed_expected(txn, 2); + registry.note_vote(txn, 7, ParticipantVote::Commit); + assert!(registry.drain_unproposed_verdicts().is_empty()); + registry.note_vote( + txn, + 9, + ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), + ); + assert_eq!( + registry.drain_unproposed_verdicts(), + vec![( + txn, + VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)) + )] + ); + registry.note_verdict( + txn, + VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)), + ); + assert_eq!(registry.verdict(txn), Some(false)); + + let first_signal = first_scheduler + .verdict_rx + .try_recv() + .expect("registry must signal the first registered scheduler"); + let second_signal = second_scheduler + .verdict_rx + .try_recv() + .expect("registry must signal the second registered scheduler"); + first_scheduler.handle_verdict_signal(first_signal); + second_scheduler.handle_verdict_signal(second_signal); + + for (scheduler, data_side) in [ + (&first_scheduler, &mut first_data), + (&second_scheduler, &mut second_data), + ] { + assert!(matches!( + scheduler + .pending + .get(&txn_id) + .and_then(|pending| pending.commit_state), + Some(CommitState::AwaitingResolve { + committed: false, + redo_lsn: None + }) + )); + let request = data_side + .request_rx + .try_pop() + .expect("global abort must dispatch a drop to every participant"); + assert!(matches!( + request.inner.plan, + PhysicalPlan::Meta(MetaOp::CalvinDrop { + epoch: 14, + position: 2 + }) + )); + assert!(data_side.request_rx.try_pop().is_err()); + } + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs new file mode 100644 index 000000000..62564932a --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs @@ -0,0 +1,109 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Local commit-vote casting for a staged static Calvin transaction. + +use std::sync::atomic::Ordering; +use std::time::Instant; + +use nodedb_cluster::calvin::SequencerEntry; + +use crate::bridge::envelope::Response; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::driver::core::staged_vote::{ + StagedVote, staged_commit_vote, +}; +use crate::control::cluster::calvin::scheduler::driver::types::CommitState; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Cast this participant's local commit vote for a staged transaction, then + /// PARK it on the cross-shard commit barrier awaiting the durable GLOBAL + /// verdict — it does NOT self-decide flush-or-drop on its local vote. + /// + /// The staged executor response is validate-only: its `read_set_valid` is + /// this shard's local commit vote (`Some(true)` => commit, `Some(false)` => + /// abort; a `None` from the active/dependent path is treated as commit). The + /// leader proposes that vote via the sequencer Raft group; the sequencer + /// aggregates all participants' votes into a single authoritative + /// `SequencerEntry::Verdict`, applied on every replica. + /// + /// This method moves the txn to [`CommitState::AwaitingVerdict`] WITHOUT + /// dispatching a resolve or drop, then immediately probes + /// `registry.verdict(txn)`: if the verdict is already durable (replay, or a + /// push we raced) it resumes at once via [`Self::resume_on_verdict`]; + /// otherwise it stays parked, holding locks and its staged buffer, until the + /// verdict push, a later probe, or the stall re-probe sweep delivers the + /// verdict. Resuming (in `resume_on_verdict`) is where the flush/drop is + /// dispatched and the flushed/dropped counters bump — using the GLOBAL + /// verdict, never the local vote. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resolve_staged_commit( + &mut self, + txn_id: TxnId, + staged_response: &Response, + ) { + // A staged error is always an abort vote. Only successful staged + // responses may use `None` for the dependent-read path; accepting an + // error-plus-None as commit would let a failed participant flush after + // its peers received a global commit verdict. + let vote = staged_commit_vote(staged_response); + + // Durably propose this participant's commit vote via the sequencer + // Raft group, leader-guarded like `OllpMismatch`: only the data-group + // leader ran read-set validation, so only a leader's vote is + // authoritative. The sequencer aggregates every participant's vote into + // the single global verdict this txn parks on below. An abort travels as + // `AbortVote` so its cause survives to the coordinator. + if self.is_group_leader() { + let entry = match vote.abort_reason() { + Some(reason) => SequencerEntry::AbortVote { + epoch: txn_id.epoch, + position: txn_id.position, + vshard: self.vshard_id, + reason, + }, + None => SequencerEntry::Vote { + epoch: txn_id.epoch, + position: txn_id.position, + vshard: self.vshard_id, + commit: true, + }, + }; + self.propose_sequencer_entry(entry, txn_id, "commit vote"); + } + + if vote == StagedVote::SerializationConflict { + // The staged slice's read-set was no longer current: observe it, the + // same node-global signal the direct-apply path records. A + // participant error never validated a read-set, so it must not count + // here. + self.shared + .calvin_counters + .read_set_validation_failures + .fetch_add(1, Ordering::Relaxed); + } + + // PARK on the barrier: transition to `AwaitingVerdict` and arm the stall + // deadline. Do NOT dispatch resolve/drop here — the GLOBAL verdict, not + // this local vote, decides. If the txn already vanished (torn down + // elsewhere), there is nothing to park. + match self.pending.get_mut(&txn_id) { + Some(pending) => { + pending.commit_state = Some(CommitState::AwaitingVerdict); + // no-determinism: local stall-warning deadline only; the global replicated verdict, not this wall-clock, decides commit/abort. + pending.verdict_deadline = Some(Instant::now() + self.config.verdict_stall_warn()); + } + None => return, + } + + // PROBE on park (correctness backstop): the verdict may already be + // durable — on replay, or a push that raced ahead of this park. Resume + // immediately if so; the double-resume guard in `resume_on_verdict` + // makes a later duplicate push/probe a no-op. + if let Some(verdict) = self.registry.verdict(nodedb_cluster::calvin::TxnId::new( + txn_id.epoch, + txn_id.position, + )) { + self.resume_on_verdict(txn_id, verdict); + } + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index e0f25ee2b..52cfb1c8e 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -58,6 +58,8 @@ pub mod request; pub mod routing; pub mod scheduler; pub mod staged_vote; +#[cfg(test)] +mod test_support; pub mod write_version_record; pub use propose::{CalvinReadResultProposal, propose_calvin_read_result}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs index 6eb7c58da..7f20cef25 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs @@ -361,106 +361,12 @@ impl Scheduler { #[cfg(test)] mod tests { use super::*; - use std::collections::{BTreeSet, HashMap}; - use std::sync::Mutex; + use std::collections::BTreeSet; use std::sync::atomic::Ordering; - use nodedb_cluster::MultiRaft; - use nodedb_cluster::RoutingTable; - use nodedb_cluster::calvin::types::{ - EngineKeySet, ReadWriteSet, SortedVec, TxClass, VersionedReadSet, + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler, make_sequenced_txn, }; - use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerStateMachine}; - use nodedb_types::TenantId; - - use super::super::scheduler::SchedulerParams; - use crate::bridge::dispatch::Dispatcher; - use crate::control::cluster::calvin::scheduler::lock_manager::LockManager; - use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; - use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; - use crate::control::state::SharedState; - use crate::wal::WalManager; - - /// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data - /// Plane is NOT started — tests exercise Control-Plane routing, guards, and - /// request dispatch only, so no core loop is needed. The returned `TempDir` - /// must be kept alive for the scheduler's lifetime (backs the WAL and - /// Raft storage). - fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::TempDir) { - let registry = CalvinCompletionRegistry::new_detached(); - let dir = tempfile::tempdir().unwrap(); - let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); - let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); - let _data_side = data_sides - .pop() - .expect("one configured core has one data side"); - let shared = SharedState::new(dispatcher, wal).unwrap(); - - let rt = RoutingTable::uniform(1, &[1], 1); - let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); - - let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( - HashMap::new(), - Arc::clone(®istry), - ))); - - let (_tx, receiver) = tokio::sync::mpsc::channel(16); - let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); - let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); - let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); - registry.register_verdict_signal_sender(vshard_id, verdict_tx); - - let lock_manager = Arc::new(Mutex::new(LockManager::new())); - - let scheduler = Scheduler::new(SchedulerParams { - vshard_id, - receiver, - shared, - multi_raft, - sequencer_state_machine, - // A freshly-built scheduler has applied nothing, so its watermark is the - // not-yet-applied sentinel (matching `read_applied_recovery` for a clean - // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, - // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit - // every epoch-0 replay before it reaches the lock table — silently - // defeating the end-to-end drain tests below. - fully_applied_epoch: NOT_YET_APPLIED_EPOCH, - applied_tail: BTreeSet::new(), - rebuild_target_epoch: 0, - config: SchedulerConfig::default(), - metrics: SchedulerMetrics::new(), - read_result_rx, - lock_manager, - promotion_rx, - registry, - verdict_rx, - }); - (scheduler, dir) - } - - fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { - let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { - collection: "test_coll".to_string(), - surrogates: SortedVec::new(vec![1]), - }]); - let tx_class = TxClass::new_single_vshard( - ReadWriteSet::new(vec![]), - write_set, - vec![], - TenantId::new(1), - None, - VersionedReadSet::default(), - ) - .expect("valid TxClass"); - SequencedTxn { - epoch, - position, - tx_class, - epoch_system_ms: 1_700_000_000_000, - epoch_vshard_txn_count: 1, - lock_owner: None, - } - } #[tokio::test] async fn in_flight_guard_skips_replayed_txn_already_in_flight() { diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index 40aecbc67..b9656b352 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -437,70 +437,9 @@ impl Scheduler { #[cfg(test)] mod tests { use super::*; - use std::collections::{BTreeSet, HashMap}; - - use nodedb_cluster::RoutingTable; - - use crate::bridge::dispatch::Dispatcher; - - /// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data - /// Plane is NOT started — tests exercise Control-Plane routing, guards, and - /// request dispatch only, so no core loop is needed. The returned `TempDir` - /// must be kept alive for the scheduler's lifetime (backs the WAL and - /// Raft storage). - fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::TempDir) { - let registry = CalvinCompletionRegistry::new_detached(); - let dir = tempfile::tempdir().unwrap(); - let wal = Arc::new( - crate::wal::WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap(), - ); - let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); - let _data_side = data_sides - .pop() - .expect("one configured core has one data side"); - let shared = SharedState::new(dispatcher, wal).unwrap(); - - let rt = RoutingTable::uniform(1, &[1], 1); - let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); - - let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( - HashMap::new(), - Arc::clone(®istry), - ))); - - let (_tx, receiver) = mpsc::channel(16); - let (_rr_tx, read_result_rx) = mpsc::channel(16); - let (_prom_tx, promotion_rx) = mpsc::unbounded_channel(); - let (verdict_tx, verdict_rx) = mpsc::channel(16); - registry.register_verdict_signal_sender(vshard_id, verdict_tx); - - let lock_manager = Arc::new(Mutex::new(LockManager::new())); + use std::collections::BTreeSet; - let scheduler = Scheduler::new(SchedulerParams { - vshard_id, - receiver, - shared, - multi_raft, - sequencer_state_machine, - // A freshly-built scheduler has applied nothing, so its watermark is the - // not-yet-applied sentinel (matching `read_applied_recovery` for a clean - // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, - // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit - // every epoch-0 replay before it reaches the lock table — silently - // defeating the end-to-end drain tests below. - fully_applied_epoch: NOT_YET_APPLIED_EPOCH, - applied_tail: BTreeSet::new(), - rebuild_target_epoch: 0, - config: SchedulerConfig::default(), - metrics: SchedulerMetrics::new(), - read_result_rx, - lock_manager, - promotion_rx, - registry, - verdict_rx, - }); - (scheduler, dir) - } + use crate::control::cluster::calvin::scheduler::driver::core::test_support::build_test_scheduler; /// A freshly-recovered scheduler (`fully_applied_epoch` still the /// `NOT_YET_APPLIED_EPOCH` sentinel) with a REAL, non-zero rebuild target must diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs new file mode 100644 index 000000000..00ebba4fa --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -0,0 +1,192 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared test fixtures for the Calvin scheduler driver's `core` unit tests. + +use std::collections::{BTreeSet, HashMap}; +use std::sync::{Arc, Mutex}; +use std::time::Instant; + +use nodedb_cluster::MultiRaft; +use nodedb_cluster::RoutingTable; +use nodedb_cluster::calvin::types::{ + EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet, +}; +use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerStateMachine}; +use nodedb_types::TenantId; + +use crate::bridge::dispatch::{CoreChannelDataSide, Dispatcher}; +use crate::bridge::envelope::{Payload, Response, Status}; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::{ + Scheduler, SchedulerParams, +}; +use crate::control::cluster::calvin::scheduler::driver::types::{CommitState, PendingTxn}; +use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; +use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; +use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; +use crate::control::state::SharedState; +use crate::types::{Lsn, RequestId}; +use crate::wal::WalManager; + +/// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data +/// Plane is NOT started — tests exercise Control-Plane routing, guards, and +/// request dispatch only, so no core loop is needed. The returned `TempDir` +/// must be kept alive for the scheduler's lifetime (backs the WAL and Raft +/// storage). +pub(super) fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::TempDir) { + let registry = CalvinCompletionRegistry::new_detached(); + let dir = tempfile::tempdir().unwrap(); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); + let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let _data_side = data_sides + .pop() + .expect("one configured core has one data side"); + let shared = SharedState::new(dispatcher, wal).unwrap(); + + let rt = RoutingTable::uniform(1, &[1], 1); + let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); + + let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( + HashMap::new(), + Arc::clone(®istry), + ))); + + let (_tx, receiver) = tokio::sync::mpsc::channel(16); + let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); + let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); + let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); + registry.register_verdict_signal_sender(vshard_id, verdict_tx); + + let lock_manager = Arc::new(Mutex::new(LockManager::new())); + + let scheduler = Scheduler::new(SchedulerParams { + vshard_id, + receiver, + shared, + multi_raft, + sequencer_state_machine, + // A freshly-built scheduler has applied nothing, so its watermark is the + // not-yet-applied sentinel (matching `read_applied_recovery` for a clean + // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, + // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit + // every epoch-0 replay before it reaches the lock table — silently + // defeating the end-to-end drain tests below. + fully_applied_epoch: NOT_YET_APPLIED_EPOCH, + applied_tail: BTreeSet::new(), + rebuild_target_epoch: 0, + config: SchedulerConfig::default(), + metrics: SchedulerMetrics::new(), + read_result_rx, + lock_manager, + promotion_rx, + registry, + verdict_rx, + }); + (scheduler, dir) +} + +/// Same minimal scheduler fixture as [`build_test_scheduler`], sharing a +/// caller-supplied completion `registry` (so several schedulers can register +/// against it) and retaining its Data-Plane request receiver for tests that +/// must observe scheduler dispatches. +pub(super) fn build_test_scheduler_with_data_side( + vshard_id: u32, + registry: Arc, +) -> (Scheduler, tempfile::TempDir, CoreChannelDataSide) { + let dir = tempfile::tempdir().unwrap(); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal")).unwrap()); + let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let data_side = data_sides + .pop() + .expect("one configured core has one data side"); + let shared = SharedState::new(dispatcher, wal).unwrap(); + + let rt = RoutingTable::uniform(1, &[1], 1); + let multi_raft = Arc::new(Mutex::new(MultiRaft::new(1, rt, dir.path().to_path_buf()))); + + let sequencer_state_machine = Arc::new(Mutex::new(SequencerStateMachine::new( + HashMap::new(), + Arc::clone(®istry), + ))); + + let (_tx, receiver) = tokio::sync::mpsc::channel(16); + let (_rr_tx, read_result_rx) = tokio::sync::mpsc::channel(16); + let (_prom_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); + let (verdict_tx, verdict_rx) = tokio::sync::mpsc::channel(16); + registry.register_verdict_signal_sender(vshard_id, verdict_tx); + + let lock_manager = Arc::new(Mutex::new(LockManager::new())); + + let scheduler = Scheduler::new(SchedulerParams { + vshard_id, + receiver, + shared, + multi_raft, + sequencer_state_machine, + fully_applied_epoch: NOT_YET_APPLIED_EPOCH, + applied_tail: BTreeSet::new(), + rebuild_target_epoch: 0, + config: SchedulerConfig::default(), + metrics: SchedulerMetrics::new(), + read_result_rx, + lock_manager, + promotion_rx, + registry, + verdict_rx, + }); + (scheduler, dir, data_side) +} + +/// Build a static-write `SequencedTxn` at `(epoch, position)`. +pub(super) fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { + let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { + collection: "test_coll".to_string(), + surrogates: SortedVec::new(vec![1]), + }]); + let tx_class = TxClass::new_single_vshard( + ReadWriteSet::new(vec![]), + write_set, + vec![], + TenantId::new(1), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + SequencedTxn { + epoch, + position, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + } +} + +/// A `PendingTxn` staged and parked awaiting the cross-shard commit verdict. +pub(super) fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { + PendingTxn { + txn, + lock_owner: txn_id, + dispatch_time: Instant::now(), + has_primary_write: true, + has_returning: false, + change_sets: Vec::new(), + commit_state: Some(CommitState::Staged), + verdict_deadline: None, + } +} + +/// A staged executor `Response` carrying the given status and read-set vote. +pub(super) fn staged_response(status: Status, read_set_valid: Option) -> Response { + Response { + request_id: RequestId::new(1), + status, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } +} From 6e6e9243a6a76a4d0224ef65999913d188841075 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 12:00:49 +0800 Subject: [PATCH 02/64] feat(bridge): defer sequenced dispatch on capacity refusal The bridge dispatcher had no way to signal "not enqueued, retry later" distinct from a terminal failure. It gains `DispatchCapacity`, a new error carrying which limit refused the request (tenant in-flight cap, a suspended per-database virtual queue, or a full per-core queue), and notifies waiters once an abandoned or completed request frees a slot. The Calvin scheduler is the first caller that cannot tolerate a terminal refusal for sequenced work: every replica must apply a sequenced txn, so a capacity refusal now parks the request in a FIFO (`deferred.rs`) instead of aborting it. The txn keeps its locks and its pending entry; the run loop re-sends parked requests once capacity frees, refreshing the deadline and group-leadership flag on each resend. Catch-up replay stops at the first deferred dispatch and re-arms itself at that Raft index instead of replaying the whole range again. `dispatcher.rs` splits into `enqueue.rs` (admission and slot accounting), `refusal.rs` (the refusal outcome type), and `response_poll.rs` (response draining), with shared test fixtures moved into `test_requests.rs`. The new error variant is wired through the data-plane wire format, the classify table (mapped to the retryable server-overload class), and every gateway error map (HTTP, pgwire, RESP, native). --- .../src/rpc_codec/data_plane_error.rs | 5 + nodedb/src/bridge/dispatch/dispatcher.rs | 676 +----------------- nodedb/src/bridge/dispatch/drain.rs | 59 +- nodedb/src/bridge/dispatch/enqueue.rs | 351 +++++++++ nodedb/src/bridge/dispatch/mod.rs | 6 + nodedb/src/bridge/dispatch/refusal.rs | 24 + nodedb/src/bridge/dispatch/response_poll.rs | 389 ++++++++++ nodedb/src/bridge/dispatch/test_requests.rs | 75 ++ nodedb/src/bridge/envelope/error_code.rs | 9 + .../calvin/scheduler/driver/core/catch_up.rs | 137 +++- .../scheduler/driver/core/commit_redo.rs | 66 +- .../driver/core/commit_resolution_dispatch.rs | 42 +- .../driver/core/commit_resolve/verdict.rs | 211 +++++- .../calvin/scheduler/driver/core/deferred.rs | 291 ++++++++ .../driver/core/dispatch/active_dispatch.rs | 127 +++- .../driver/core/dispatch/bind_identities.rs | 5 +- .../driver/core/dispatch/static_dispatch.rs | 62 +- .../calvin/scheduler/driver/core/mod.rs | 3 + .../calvin/scheduler/driver/core/process.rs | 148 +++- .../scheduler/driver/core/read_result.rs | 4 +- .../calvin/scheduler/driver/core/request.rs | 18 +- .../calvin/scheduler/driver/core/scheduler.rs | 46 +- .../scheduler/driver/core/test_support.rs | 259 ++++++- .../driver/core/write_version_record.rs | 95 ++- .../cluster/calvin/scheduler/driver/types.rs | 17 +- .../cluster/calvin/scheduler/metrics.rs | 42 ++ .../control/cluster/data_plane_error_wire.rs | 24 + nodedb/src/control/gateway/error_map/http.rs | 3 +- .../src/control/gateway/error_map/pgwire.rs | 6 + .../control/gateway/error_map/remote_code.rs | 3 +- nodedb/src/control/gateway/error_map/resp.rs | 1 + .../server/dispatch_utils/write_abort.rs | 9 + .../server/native/dispatch/raw_dispatch.rs | 11 +- .../control/server/pgwire/types/error_map.rs | 7 + .../control/server/resp/gateway_dispatch.rs | 4 +- .../ddl/neutral/graph_ops/edge_stage.rs | 6 +- .../shared/ddl/neutral/topic/publish.rs | 3 + .../src/control/server/shared/ddl/sqlstate.rs | 4 + nodedb/src/control/server/sync/refusal.rs | 24 +- nodedb/src/error/dispatch_capacity.rs | 66 ++ nodedb/src/error/mod.rs | 3 + nodedb/src/error/types.rs | 7 + nodedb/src/error_classify.rs | 2 + nodedb/src/error_from_data_plane.rs | 3 + nodedb/src/lib.rs | 2 +- 45 files changed, 2422 insertions(+), 933 deletions(-) create mode 100644 nodedb/src/bridge/dispatch/enqueue.rs create mode 100644 nodedb/src/bridge/dispatch/refusal.rs create mode 100644 nodedb/src/bridge/dispatch/response_poll.rs create mode 100644 nodedb/src/bridge/dispatch/test_requests.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs create mode 100644 nodedb/src/error/dispatch_capacity.rs diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index 45927be57..59e59a02d 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -121,4 +121,9 @@ pub enum DataPlaneErrorCode { status_column: String, row_identity: String, }, + /// The bridge dispatcher refused the request at a capacity limit; nothing + /// was enqueued. `reason` names the limit and its counts. + DispatchCapacity { + reason: String, + }, } diff --git a/nodedb/src/bridge/dispatch/dispatcher.rs b/nodedb/src/bridge/dispatch/dispatcher.rs index 7e1783f3e..728d0ca8b 100644 --- a/nodedb/src/bridge/dispatch/dispatcher.rs +++ b/nodedb/src/bridge/dispatch/dispatcher.rs @@ -1,21 +1,23 @@ // SPDX-License-Identifier: BUSL-1.1 -use std::collections::{HashMap, HashSet}; +//! The bridge [`Dispatcher`]: its per-core channels, construction, and +//! read-only accessors. +//! +//! Admission and enqueue live in `enqueue`. Response polling and dead-core +//! synthesis live in `response_poll`. The shutdown drain lives in `drain`. -use tracing::warn; +use std::collections::{HashMap, HashSet}; +use std::sync::Arc; use nodedb_bridge::backpressure::{BackpressureConfig, BackpressureController, PressureState}; use nodedb_bridge::buffer::RingBuffer; use nodedb_bridge::wfq::WeightedFairQueue; use nodedb_types::PriorityClass; +use tokio::sync::Notify; use crate::bridge::envelope; -use crate::bridge::envelope::{ErrorCode, Payload, Status}; use crate::control::router::vshard::VShardRouter; use crate::data::eventfd::EventFdNotifier; -use crate::types::{Lsn, RequestId}; - -use crate::bridge::admission_chokepoint::{assert_write_admitted, reject_uninjected_write}; use super::core_channel::{CoreChannel, CoreChannelDataSide}; @@ -73,7 +75,7 @@ pub struct Dispatcher { pub(super) cores: Vec, /// Routes vShards to core IDs. - router: VShardRouter, + pub(super) router: VShardRouter, /// Per-tenant in-flight request count across all cores. pub(super) tenant_inflight: HashMap, @@ -82,13 +84,13 @@ pub struct Dispatcher { pub(super) request_tenant: HashMap, /// Maximum in-flight requests per tenant (0 = unlimited). - max_per_tenant_inflight: u32, + pub(super) max_per_tenant_inflight: u32, /// Per-core queue capacity (used in tenant fairness recalculation). - per_core_capacity: u32, + pub(super) per_core_capacity: u32, /// Resolves priority class for a database_id (consulted on enqueue). - priority_resolver: Box, + pub(super) priority_resolver: Box, /// True once the shutdown bus has entered `DrainingDataPlane`. /// @@ -96,6 +98,10 @@ pub struct Dispatcher { /// enqueue takes, so a dispatch that observed `false` has already pushed by /// the time the drain starts. Nothing reaches a core after that. pub(super) data_plane_draining: bool, + + /// Capacity freed on the bridge dispatcher. Every path that releases an + /// in-flight slot wakes all waiters once per call. + pub(super) capacity_freed: Arc, } impl Dispatcher { @@ -149,171 +155,20 @@ impl Dispatcher { per_core_capacity: queue_capacity as u32, priority_resolver, data_plane_draining: false, + capacity_freed: Arc::new(Notify::new()), }, data_sides, ) } - /// Dispatch a request to the correct Data Plane core. - /// - /// Enqueues into the per-core weighted-fair queue keyed by `DatabaseId`, - /// then flushes WFQ → physical ring. Returns `Err` when the WFQ itself is - /// full (total capacity reached across all active databases on that core). - pub fn dispatch(&mut self, request: envelope::Request) -> crate::Result<()> { - reject_uninjected_write(&request)?; - assert_write_admitted(&request); - self.reject_if_draining()?; - let tenant_id = request.tenant_id.as_u64(); - let req_id = request.request_id.as_u64(); - let database_id = request.database_id.as_u64(); - - // Per-tenant fairness: reject if this tenant has too many in-flight requests. - if self.max_per_tenant_inflight > 0 { - let inflight = self.tenant_inflight.get(&tenant_id).copied().unwrap_or(0); - if inflight >= self.max_per_tenant_inflight { - return Err(crate::Error::Dispatch { - detail: format!( - "tenant {tenant_id}: queue full ({inflight}/{} in-flight)", - self.max_per_tenant_inflight - ), - }); - } - } - - let core_id = - self.router - .resolve(request.vshard_id) - .ok_or_else(|| crate::Error::Dispatch { - detail: format!("no core for vshard {}", request.vshard_id), - })?; - - let channel = &mut self.cores[core_id]; - - // Refresh priority for this DB in the WFQ. - let cls = self.priority_resolver.priority_for(database_id); - channel.wfq.set_priority(database_id, cls); - - // Check per-DB suspended state (≥95% of fair share). - if channel.wfq.is_suspended_for(database_id) { - return Err(crate::Error::Dispatch { - detail: format!( - "database {database_id}: virtual queue suspended (≥95% of fair share on core {core_id})" - ), - }); - } - - // Enqueue into the WFQ — returns Err if total capacity is full. - channel - .wfq - .try_enqueue(database_id, request) - .map_err(|_| crate::Error::Dispatch { - detail: format!("core {core_id}: total WFQ capacity exhausted"), - })?; - - // Update per-DB pressure. - channel.update_db_pressure(database_id); - - // Flush WFQ → physical ring. - channel.flush_wfq(); - - // Update global backpressure based on ring utilization. - let util = channel.request_tx.utilization(); - if let Some(new_state) = channel.backpressure.update(util) { - warn!( - core_id, - utilization = util, - state = ?new_state, - "backpressure transition" - ); - } - - // Track the request as outstanding on this core, so a later core death - // can fail it instead of stranding the caller's waiter. - channel.outstanding.insert(req_id); - - // Track per-tenant in-flight + request→tenant mapping for response routing. - *self.tenant_inflight.entry(tenant_id).or_insert(0) += 1; - self.request_tenant.insert(req_id, tenant_id); - - // Wake the Data Plane core via eventfd. - if let Some(ref notifier) = channel.wake_notifier { - notifier.notify(); - } - - Ok(()) - } - - /// Record a response received for a tenant (decrements in-flight count). - pub fn tenant_response_received(&mut self, tenant_id: u64) { - if let Some(count) = self.tenant_inflight.get_mut(&tenant_id) { - *count = count.saturating_sub(1); - } - } - - /// Recalculate the per-tenant in-flight limit based on active tenants. - pub fn recalculate_tenant_limits(&mut self) { - let active = self.tenant_inflight.len().max(1) as u32; - let total_capacity: u32 = self.cores.len() as u32 * self.per_core_capacity; - self.max_per_tenant_inflight = (total_capacity / active).max(2); - self.tenant_inflight.retain(|_, count| *count > 0); - } - - /// Dispatch a request directly to a specific core by index. + /// The signal fired whenever the dispatcher frees at least one in-flight + /// slot. /// - /// Bypasses vShard routing. Used by the checkpoint manager to send - /// checkpoint requests to every core regardless of vShard assignment. - pub fn dispatch_to_core( - &mut self, - core_id: usize, - request: envelope::Request, - ) -> crate::Result<()> { - reject_uninjected_write(&request)?; - assert_write_admitted(&request); - self.reject_if_draining()?; - if core_id >= self.cores.len() { - return Err(crate::Error::Dispatch { - detail: format!("core {core_id} out of range (have {})", self.cores.len()), - }); - } - - let tenant_id = request.tenant_id.as_u64(); - let req_id = request.request_id.as_u64(); - let database_id = request.database_id.as_u64(); - let channel = &mut self.cores[core_id]; - - let cls = self.priority_resolver.priority_for(database_id); - channel.wfq.set_priority(database_id, cls); - - channel - .wfq - .try_enqueue(database_id, request) - .map_err(|_| crate::Error::Dispatch { - detail: format!("core {core_id}: total WFQ capacity exhausted"), - })?; - - channel.update_db_pressure(database_id); - channel.flush_wfq(); - - let util = channel.request_tx.utilization(); - if let Some(new_state) = channel.backpressure.update(util) { - warn!( - core_id, - utilization = util, - state = ?new_state, - "backpressure transition" - ); - } - - channel.outstanding.insert(req_id); - - *self.tenant_inflight.entry(tenant_id).or_insert(0) += 1; - self.request_tenant.insert(req_id, tenant_id); - - if let Some(ref notifier) = channel.wake_notifier { - notifier.notify(); - } - - Ok(()) + /// A caller refused with [`crate::Error::DispatchCapacity`] waits on it + /// before it retries. It wakes only registered waiters, so a caller + /// enables its `notified()` future before it checks for refused work. + pub fn capacity_freed(&self) -> Arc { + Arc::clone(&self.capacity_freed) } /// Maximum SPSC request queue utilization across all cores (0-100). @@ -336,97 +191,6 @@ impl Dispatcher { .unwrap_or(PressureState::Normal) } - /// Poll responses from all Data Plane cores. - /// - /// A core whose channel has been observed dead contributes a synthesized - /// error `Response` for every request still outstanding on it: the one a - /// failed `try_push` consumed, everything still staged in its WFQ, and - /// everything dispatched earlier that it never answered. Those travel back - /// with the real responses so the single completion loop in the caller - /// finishes each waiter, and the loop below releases each request's - /// `tenant_inflight` slot exactly as a real response would — without which - /// one dead core ratchets the tenant's in-flight count until the tenant is - /// rejected on healthy cores too. - pub fn poll_responses(&mut self) -> Vec { - let mut responses = Vec::new(); - for (core_id, channel) in self.cores.iter_mut().enumerate() { - let mut batch = Vec::new(); - let (_drained, producer_gone) = channel.response_rx.drain_into(&mut batch, 64); - for br in batch { - let rid = br.inner.request_id.as_u64(); - // A streaming scan answers with many partials before its final - // response. The request is still executing on the core until - // that final one arrives, so releasing it here would let the - // shutdown drain call a live scan finished and would drop the - // tenant's in-flight slot mid-stream. - if !br.inner.partial { - channel.outstanding.remove(&rid); - if let Some(tid) = self.request_tenant.remove(&rid) - && let Some(count) = self.tenant_inflight.get_mut(&tid) - { - *count = count.saturating_sub(1); - } - } - responses.push(br.inner); - } - - if !(producer_gone || channel.request_tx.is_disconnected()) { - // Opportunistically flush WFQ after draining responses to fill headroom. - channel.flush_wfq(); - continue; - } - - // The core is gone. Collect every request it can no longer answer: - // items still staged in the WFQ first (dispatch order), then the - // rest of the outstanding set. A staged item is also in - // `outstanding`, so `seen` keeps each id to a single response. - let mut seen = HashSet::new(); - let mut lost = Vec::new(); - for staged in channel.wfq.drain() { - let rid = staged.request_id.as_u64(); - if seen.insert(rid) { - lost.push(rid); - } - } - for rid in channel.outstanding.drain() { - if seen.insert(rid) { - lost.push(rid); - } - } - - // Idempotence: both sources are emptied here — `wfq.drain` leaves - // the staging queue empty and `outstanding.drain` clears the set — - // and `flush_wfq` refuses to stage anything new onto a - // disconnected producer. A later poll therefore finds both empty - // and emits nothing, so a permanently dead core costs one pass - // over two empty containers rather than a repeating failure storm. - for rid in lost { - if let Some(tid) = self.request_tenant.remove(&rid) - && let Some(count) = self.tenant_inflight.get_mut(&tid) - { - *count = count.saturating_sub(1); - } - responses.push(envelope::Response { - request_id: RequestId::new(rid), - status: Status::Error, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: Some(Box::new(ErrorCode::Internal { - detail: format!( - "core-{core_id} is gone; the request can never be executed" - ), - })), - read_set_valid: None, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - }); - } - } - responses - } - /// Number of Data Plane cores. pub fn num_cores(&self) -> usize { self.cores.len() @@ -444,395 +208,3 @@ impl Dispatcher { &self.router } } - -#[cfg(test)] -mod tests { - use super::*; - use crate::bridge::envelope::*; - use crate::types::*; - use nodedb_physical::physical_plan::DocumentOp; - use std::time::{Duration, Instant}; - - fn make_request(vshard: u32) -> envelope::Request { - envelope::Request { - request_id: RequestId::new(1), - tenant_id: TenantId::new(1), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(vshard), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), - document_id: "u1".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: Admission::Exempt(ExemptReason::Read), - } - } - - fn make_request_for_db(vshard: u32, db: u64, req_id: u64) -> envelope::Request { - envelope::Request { - request_id: RequestId::new(req_id), - tenant_id: TenantId::new(1), - database_id: DatabaseId::new(db), - vshard_id: VShardId::new(vshard), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::new(db), "c"), - document_id: "d".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: Admission::Exempt(ExemptReason::Read), - } - } - - #[test] - fn dispatch_routes_to_correct_core() { - let (mut dispatcher, data_sides) = Dispatcher::new(4, 64); - - dispatcher.dispatch(make_request(0)).unwrap(); - dispatcher.dispatch(make_request(1)).unwrap(); - dispatcher.dispatch(make_request(4)).unwrap(); // Wraps to core 0. - - assert_eq!(data_sides[0].request_rx.len(), 2); - assert_eq!(data_sides[1].request_rx.len(), 1); - assert_eq!(data_sides[2].request_rx.len(), 0); - } - - #[test] - fn response_roundtrip() { - let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); - - dispatcher.dispatch(make_request(0)).unwrap(); - - let _req = data_sides[0].request_rx.try_pop().unwrap(); - data_sides[0] - .response_tx - .try_push(BridgeResponse { - inner: envelope::Response { - request_id: RequestId::new(1), - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::from_vec(b"result".to_vec()), - watermark_lsn: Lsn::new(42), - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }, - }) - .unwrap(); - - let responses = dispatcher.poll_responses(); - assert_eq!(responses.len(), 1); - assert_eq!(responses[0].status, Status::Ok); - assert_eq!(&*responses[0].payload, b"result"); - } - - #[test] - fn full_queue_returns_error() { - // With WFQ capacity == ring capacity, filling WFQ should eventually - // cause total-capacity exhaustion. - let (mut dispatcher, _data_sides) = Dispatcher::new(1, 4); - - for i in 0..4u64 { - dispatcher - .dispatch(make_request_for_db(0, i + 1, i + 1)) - .unwrap(); - } - - // Next dispatch should fail — WFQ total capacity exhausted. - let result = dispatcher.dispatch(make_request_for_db(0, 99, 99)); - assert!(result.is_err()); - } - - #[test] - fn dispatch_to_core_tracks_request_lifecycle() { - let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); - let request = make_request(0); - let tenant_id = request.tenant_id.as_u64(); - let request_id = request.request_id.as_u64(); - - dispatcher.dispatch_to_core(1, request).unwrap(); - - assert_eq!(dispatcher.tenant_inflight.get(&tenant_id), Some(&1)); - assert_eq!(dispatcher.request_tenant.get(&request_id), Some(&tenant_id)); - assert_eq!(data_sides[1].request_rx.len(), 1); - - let _req = data_sides[1].request_rx.try_pop().unwrap(); - data_sides[1] - .response_tx - .try_push(BridgeResponse { - inner: envelope::Response { - request_id: RequestId::new(request_id), - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }, - }) - .unwrap(); - - let responses = dispatcher.poll_responses(); - assert_eq!(responses.len(), 1); - assert_eq!(dispatcher.tenant_inflight.get(&tenant_id), Some(&0)); - assert!(!dispatcher.request_tenant.contains_key(&request_id)); - } - - #[test] - fn per_db_pressure_reported() { - let (mut dispatcher, _) = Dispatcher::new(1, 8); - // Fill fair share for DB 1 using 4 of 8 slots. - // With one DB initially, fair share = 8. With two DBs = 4 each. - // First enqueue DB1 + DB2, so fair_share = 4. - for i in 0..4u64 { - dispatcher - .dispatch(make_request_for_db(0, 1, i + 10)) - .unwrap(); - } - for i in 0..4u64 { - dispatcher - .dispatch(make_request_for_db(0, 2, i + 20)) - .unwrap(); - } - // After filling DB1's fair share, it should be suspended on core 0. - // (exact state depends on WFQ flush draining items to ring first) - // The test confirms per-DB pressure is being tracked without panic. - let _ = dispatcher.db_pressure_on_core(0, 1); - let _ = dispatcher.db_pressure_on_core(0, 2); - } - - // --- Dead-core request loss (GitHub #265) --- - // - // When a Data Plane core's consumer/producer is dropped (the core thread - // died), `Dispatcher` must synthesize an error `Response` for every - // request it knows is outstanding on that core, rather than dropping the - // request silently and leaking the caller's waiter + `tenant_inflight` - // slot forever. Dropping one element of the `data_sides` vector handed - // back by `Dispatcher::new`/`with_resolver` simulates that core thread - // dying, matching how `dispatch_routes_to_correct_core` and - // `response_roundtrip` above obtain the data-plane side of the channel. - - #[test] - fn dead_core_synthesizes_error_response_for_lost_request() { - let (mut dispatcher, mut data_sides) = Dispatcher::new(3, 64); - - // Core 2's thread has died: both halves of its data-plane side are gone. - let dead_core = 2; - drop(data_sides.remove(dead_core)); - - let request = make_request_for_db(0, 1, 7); - let request_id = request.request_id.as_u64(); - - // `dispatch_to_core` still reports success: the request was already - // moved into the doomed `try_push` inside `flush_wfq` before the - // failure is observed, which is exactly the defect being covered. - dispatcher.dispatch_to_core(dead_core, request).unwrap(); - - let responses = dispatcher.poll_responses(); - assert_eq!(responses.len(), 1, "expected one synthesized response"); - let resp = &responses[0]; - assert_eq!(resp.request_id.as_u64(), request_id); - assert_eq!(resp.status, Status::Error); - match resp.error_code.as_deref() { - Some(ErrorCode::Internal { detail }) => { - assert!( - detail.contains(&dead_core.to_string()), - "error detail should name the dead core, got: {detail}" - ); - } - other => panic!("expected ErrorCode::Internal naming the core, got: {other:?}"), - } - } - - #[test] - fn dead_core_synthesized_response_resets_tenant_inflight() { - // The ratchet: `tenant_inflight` is incremented on dispatch and must - // return to its pre-dispatch value once the synthesized response for - // the lost request is drained through `poll_responses` — otherwise it - // climbs forever and eventually starves the tenant on healthy cores. - let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); - let dead_core = 0; - drop(data_sides.remove(dead_core)); - - let request = make_request_for_db(0, 1, 1); - let tenant_id = request.tenant_id.as_u64(); - - let before = dispatcher - .tenant_inflight - .get(&tenant_id) - .copied() - .unwrap_or(0); - - dispatcher.dispatch_to_core(dead_core, request).unwrap(); - assert_eq!( - dispatcher.tenant_inflight.get(&tenant_id).copied(), - Some(before + 1), - "dispatch must still increment tenant_inflight even though the core is dead" - ); - - let responses = dispatcher.poll_responses(); - assert_eq!(responses.len(), 1); - assert_eq!( - dispatcher - .tenant_inflight - .get(&tenant_id) - .copied() - .unwrap_or(0), - before, - "tenant_inflight must return to its pre-dispatch value, not ratchet upward" - ); - assert!(!dispatcher.request_tenant.contains_key(&1)); - } - - #[test] - fn dead_core_does_not_affect_live_core() { - let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); - let dead_core = 0; - let live_core = 1; - drop(data_sides.remove(dead_core)); - // Removing index 0 shifted core 1's data side down to index 0. - let live_data_side = &mut data_sides[0]; - - let dead_request = make_request_for_db(0, 1, 1); - let live_request = make_request_for_db(0, 2, 2); - let live_request_id = live_request.request_id.as_u64(); - - dispatcher - .dispatch_to_core(dead_core, dead_request) - .unwrap(); - dispatcher - .dispatch_to_core(live_core, live_request) - .unwrap(); - - // The live core answers normally, through the real ring buffer. - let _req = live_data_side.request_rx.try_pop().unwrap(); - live_data_side - .response_tx - .try_push(BridgeResponse { - inner: envelope::Response { - request_id: RequestId::new(live_request_id), - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }, - }) - .unwrap(); - - let responses = dispatcher.poll_responses(); - assert_eq!( - responses.len(), - 2, - "one synthesized error from the dead core, one real Ok from the live core" - ); - - let live_resp = responses - .iter() - .find(|r| r.request_id.as_u64() == live_request_id) - .expect("live core's real response must be present"); - assert_eq!(live_resp.status, Status::Ok); - assert!(live_resp.error_code.is_none()); - - let dead_resp = responses - .iter() - .find(|r| r.request_id.as_u64() != live_request_id) - .expect("dead core's synthesized response must be present"); - assert_eq!(dead_resp.status, Status::Error); - assert!(dead_resp.error_code.is_some()); - } - - #[test] - fn dead_core_fails_requests_still_queued_in_wfq() { - // Fill the physical ring to capacity while the core is alive, so a - // request dispatched afterward parks in the WFQ without ever - // attempting a push (flush_wfq's utilization check breaks before it - // reaches the doomed try_push). Then kill the core and confirm the - // WFQ-queued request is failed too, not left sitting in the queue - // forever. - let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 4); - - for i in 0..4u64 { - dispatcher - .dispatch_to_core(0, make_request_for_db(0, i + 1, i + 1)) - .unwrap(); - } - assert_eq!(data_sides[0].request_rx.len(), 4); - - // Core 0's thread dies with 4 unanswered requests sitting in its ring. - drop(data_sides.remove(0)); - - // This request cannot reach the (full, dead) physical ring — it stays - // parked in the WFQ. - let parked_request_id = 99u64; - dispatcher - .dispatch_to_core(0, make_request_for_db(0, 99, parked_request_id)) - .unwrap(); - - let responses = dispatcher.poll_responses(); - let ids: std::collections::HashSet = - responses.iter().map(|r| r.request_id.as_u64()).collect(); - - // The 4 previously-dispatched-but-unanswered requests, plus the one - // still parked in the WFQ, must all be failed. - assert_eq!( - responses.len(), - 5, - "expected all 5 outstanding requests failed" - ); - for id in 1..=4u64 { - assert!( - ids.contains(&id), - "request {id} in the dead ring must be failed" - ); - } - assert!( - ids.contains(&parked_request_id), - "request parked in the WFQ must be failed, not left queued" - ); - for r in &responses { - assert_eq!(r.status, Status::Error); - assert!(r.error_code.is_some()); - } - } -} diff --git a/nodedb/src/bridge/dispatch/drain.rs b/nodedb/src/bridge/dispatch/drain.rs index 37a89d9c4..c05cf3ed5 100644 --- a/nodedb/src/bridge/dispatch/drain.rs +++ b/nodedb/src/bridge/dispatch/drain.rs @@ -27,6 +27,7 @@ use crate::bridge::envelope::{ErrorCode, Payload, Status}; use crate::types::{Lsn, RequestId}; use super::dispatcher::Dispatcher; +use super::enqueue::release_inflight_slot; /// Work one core still owes the Control Plane. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -128,6 +129,7 @@ impl Dispatcher { /// core, and outstanding work reached one that did not answer in time. pub fn abandon_data_plane_work(&mut self) -> Vec { let mut abandoned = Vec::new(); + let mut freed = false; for (core_id, channel) in self.cores.iter_mut().enumerate() { let mut ids: Vec = channel .wfq @@ -150,11 +152,8 @@ impl Dispatcher { "data plane drain deadline expired — failing the requests this core still holds" ); for rid in ids { - if let Some(tid) = self.request_tenant.remove(&rid) - && let Some(count) = self.tenant_inflight.get_mut(&tid) - { - *count = count.saturating_sub(1); - } + freed |= + release_inflight_slot(&mut self.request_tenant, &mut self.tenant_inflight, rid); abandoned.push(envelope::Response { request_id: RequestId::new(rid), status: Status::Error, @@ -174,6 +173,9 @@ impl Dispatcher { }); } } + if freed { + self.capacity_freed.notify_waiters(); + } abandoned } } @@ -181,42 +183,7 @@ impl Dispatcher { #[cfg(test)] mod tests { use super::*; - use crate::bridge::envelope::{Admission, ExemptReason, PhysicalPlan, Priority, Request}; - use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; - use nodedb_physical::physical_plan::DocumentOp; - use nodedb_types::QualifiedCollection; - use std::time::{Duration, Instant}; - - fn make_request(id: u64, vshard: u32) -> Request { - Request { - request_id: RequestId::new(id), - tenant_id: TenantId::new(1), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(vshard), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - document_id: "d".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: Admission::Exempt(ExemptReason::Read), - } - } + use crate::bridge::dispatch::test_requests::make_request_for_db; #[test] fn a_fresh_dispatcher_accepts_work_and_reports_it_pending() { @@ -225,7 +192,7 @@ mod tests { assert!(dispatcher.data_plane_pending().is_empty()); dispatcher - .dispatch(make_request(1, 0)) + .dispatch(make_request_for_db(0, 0, 1)) .expect("a running dispatcher accepts work"); let pending = dispatcher.data_plane_pending(); assert_eq!(pending.len(), 1, "one core holds the request"); @@ -238,7 +205,7 @@ mod tests { dispatcher.begin_data_plane_drain(); let err = dispatcher - .dispatch(make_request(1, 0)) + .dispatch(make_request_for_db(0, 0, 1)) .expect_err("a draining dispatcher must refuse new work"); assert!( matches!(err, crate::Error::Dispatch { .. }), @@ -256,7 +223,7 @@ mod tests { dispatcher.begin_data_plane_drain(); let err = dispatcher - .dispatch_to_core(0, make_request(1, 0)) + .dispatch_to_core(0, make_request_for_db(0, 0, 1)) .expect_err("the direct-to-core path uses the same gate"); assert!(matches!(err, crate::Error::Dispatch { .. })); } @@ -268,7 +235,7 @@ mod tests { dispatcher.begin_data_plane_drain(); assert!(dispatcher.is_data_plane_draining()); assert!( - dispatcher.dispatch(make_request(1, 0)).is_err(), + dispatcher.dispatch(make_request_for_db(0, 0, 1)).is_err(), "a second drain start changes nothing" ); } @@ -277,7 +244,7 @@ mod tests { fn abandoned_work_is_answered_and_cleared() { let (mut dispatcher, _data_sides) = Dispatcher::new(1, 64); dispatcher - .dispatch(make_request(7, 0)) + .dispatch(make_request_for_db(0, 0, 7)) .expect("accept before the drain"); dispatcher.begin_data_plane_drain(); diff --git a/nodedb/src/bridge/dispatch/enqueue.rs b/nodedb/src/bridge/dispatch/enqueue.rs new file mode 100644 index 000000000..15992c6bf --- /dev/null +++ b/nodedb/src/bridge/dispatch/enqueue.rs @@ -0,0 +1,351 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Admission, weighted-fair enqueue, and per-tenant in-flight accounting for +//! the bridge [`Dispatcher`]. +//! +//! A capacity limit refuses with [`crate::Error::DispatchCapacity`] and hands +//! the request back. Every other refusal is terminal. + +use std::collections::HashMap; + +use tracing::warn; + +use crate::DispatchCapacityScope; +use crate::bridge::admission_chokepoint::{assert_write_admitted, reject_uninjected_write}; +use crate::bridge::envelope; + +use super::dispatcher::Dispatcher; +use super::refusal::DispatchRefusal; + +impl Dispatcher { + /// Dispatch a request to the correct Data Plane core. + /// + /// Enqueues into the per-core weighted-fair queue keyed by `DatabaseId`, + /// then flushes WFQ → physical ring. A capacity limit refuses with + /// [`crate::Error::DispatchCapacity`]. Every other refusal is terminal. + pub fn dispatch(&mut self, request: envelope::Request) -> crate::Result<()> { + self.try_dispatch(request).map_err(|refusal| refusal.error) + } + + /// Dispatch like [`Self::dispatch`], handing the request back on refusal. + /// + /// A caller that must retry a capacity refusal re-sends the returned + /// request. The dispatcher tracks nothing for a refused request. + pub fn try_dispatch(&mut self, request: envelope::Request) -> Result<(), Box> { + if let Err(error) = reject_uninjected_write(&request) { + return Err(DispatchRefusal::boxed(error, request)); + } + assert_write_admitted(&request); + if let Err(error) = self.reject_if_draining() { + return Err(DispatchRefusal::boxed(error, request)); + } + let tenant_id = request.tenant_id.as_u64(); + let req_id = request.request_id.as_u64(); + let database_id = request.database_id.as_u64(); + + // Per-tenant fairness: refuse while the tenant holds its in-flight cap. + if self.max_per_tenant_inflight > 0 { + let inflight = self.tenant_inflight.get(&tenant_id).copied().unwrap_or(0); + if inflight >= self.max_per_tenant_inflight { + let scope = DispatchCapacityScope::TenantInflight { + tenant_id: request.tenant_id, + inflight, + cap: self.max_per_tenant_inflight, + }; + return Err(DispatchRefusal::boxed( + crate::Error::DispatchCapacity { scope }, + request, + )); + } + } + + let Some(core_id) = self.router.resolve(request.vshard_id) else { + let error = crate::Error::Dispatch { + detail: format!("no core for vshard {}", request.vshard_id), + }; + return Err(DispatchRefusal::boxed(error, request)); + }; + + let channel = &mut self.cores[core_id]; + + // Refresh priority for this DB in the WFQ. + let cls = self.priority_resolver.priority_for(database_id); + channel.wfq.set_priority(database_id, cls); + + // Check per-DB suspended state (≥95% of fair share). + if channel.wfq.is_suspended_for(database_id) { + let scope = DispatchCapacityScope::DatabaseSuspended { + database_id: request.database_id, + core_id, + }; + return Err(DispatchRefusal::boxed( + crate::Error::DispatchCapacity { scope }, + request, + )); + } + + // Enqueue into the WFQ. A full queue hands the request back. + if let Err(request) = channel.wfq.try_enqueue(database_id, request) { + let scope = DispatchCapacityScope::QueueFull { + core_id, + capacity: self.per_core_capacity, + }; + return Err(DispatchRefusal::boxed( + crate::Error::DispatchCapacity { scope }, + request, + )); + } + + self.commit_enqueued(core_id, database_id, tenant_id, req_id); + Ok(()) + } + + /// Dispatch a request directly to a specific core by index. + /// + /// Bypasses vShard routing. Used by the checkpoint manager to send + /// checkpoint requests to every core regardless of vShard assignment. + pub fn dispatch_to_core( + &mut self, + core_id: usize, + request: envelope::Request, + ) -> crate::Result<()> { + reject_uninjected_write(&request)?; + assert_write_admitted(&request); + self.reject_if_draining()?; + if core_id >= self.cores.len() { + return Err(crate::Error::Dispatch { + detail: format!("core {core_id} out of range (have {})", self.cores.len()), + }); + } + + let tenant_id = request.tenant_id.as_u64(); + let req_id = request.request_id.as_u64(); + let database_id = request.database_id.as_u64(); + let channel = &mut self.cores[core_id]; + + let cls = self.priority_resolver.priority_for(database_id); + channel.wfq.set_priority(database_id, cls); + + channel.wfq.try_enqueue(database_id, request).map_err(|_| { + crate::Error::DispatchCapacity { + scope: DispatchCapacityScope::QueueFull { + core_id, + capacity: self.per_core_capacity, + }, + } + })?; + + self.commit_enqueued(core_id, database_id, tenant_id, req_id); + Ok(()) + } + + /// Recalculate the per-tenant in-flight limit based on active tenants. + pub fn recalculate_tenant_limits(&mut self) { + let active = self.tenant_inflight.len().max(1) as u32; + let total_capacity: u32 = self.cores.len() as u32 * self.per_core_capacity; + self.max_per_tenant_inflight = (total_capacity / active).max(2); + self.tenant_inflight.retain(|_, count| *count > 0); + } + + /// Bookkeeping once a request sits in `core_id`'s weighted-fair queue: + /// flush it toward the ring, record pressure, track it as outstanding and + /// in flight for its tenant, and wake the core. + fn commit_enqueued(&mut self, core_id: usize, database_id: u64, tenant_id: u64, req_id: u64) { + let channel = &mut self.cores[core_id]; + + // Update per-DB pressure. + channel.update_db_pressure(database_id); + + // Flush WFQ → physical ring. + channel.flush_wfq(); + + // Update global backpressure based on ring utilization. + let util = channel.request_tx.utilization(); + if let Some(new_state) = channel.backpressure.update(util) { + warn!( + core_id, + utilization = util, + state = ?new_state, + "backpressure transition" + ); + } + + // Track the request as outstanding on this core, so a later core death + // can fail it instead of stranding the caller's waiter. + channel.outstanding.insert(req_id); + + // Track per-tenant in-flight + request→tenant mapping for response routing. + *self.tenant_inflight.entry(tenant_id).or_insert(0) += 1; + self.request_tenant.insert(req_id, tenant_id); + + // Wake the Data Plane core via eventfd. + if let Some(ref notifier) = self.cores[core_id].wake_notifier { + notifier.notify(); + } + } +} + +/// Release the in-flight slot `request_id` holds for its tenant. +/// +/// Returns `true` when a slot was freed. A free function over the two maps, +/// so a caller can release while it holds a borrow of one core's channel. +pub(super) fn release_inflight_slot( + request_tenant: &mut HashMap, + tenant_inflight: &mut HashMap, + request_id: u64, +) -> bool { + let Some(tenant_id) = request_tenant.remove(&request_id) else { + return false; + }; + match tenant_inflight.get_mut(&tenant_id) { + Some(count) if *count > 0 => { + *count -= 1; + true + } + _ => false, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::dispatch::BridgeResponse; + use crate::bridge::dispatch::test_requests::{make_request, make_request_for_db}; + use crate::bridge::envelope::*; + use crate::types::*; + + #[test] + fn dispatch_routes_to_correct_core() { + let (mut dispatcher, data_sides) = Dispatcher::new(4, 64); + + dispatcher.dispatch(make_request(0)).unwrap(); + dispatcher.dispatch(make_request(1)).unwrap(); + dispatcher.dispatch(make_request(4)).unwrap(); // Wraps to core 0. + + assert_eq!(data_sides[0].request_rx.len(), 2); + assert_eq!(data_sides[1].request_rx.len(), 1); + assert_eq!(data_sides[2].request_rx.len(), 0); + } + + #[test] + fn tenant_at_inflight_cap_is_refused_with_tenant_scope() { + // One core with capacity 4 caps each tenant at 4 in-flight requests. + let (mut dispatcher, _data_sides) = Dispatcher::new(1, 4); + + for i in 0..4u64 { + dispatcher + .dispatch(make_request_for_db(0, i + 1, i + 1)) + .unwrap(); + } + + let refusal = dispatcher + .try_dispatch(make_request_for_db(0, 99, 99)) + .expect_err("the fifth request exceeds the tenant cap"); + let DispatchRefusal { error, request } = *refusal; + match error { + crate::Error::DispatchCapacity { + scope: + DispatchCapacityScope::TenantInflight { + tenant_id, + inflight, + cap, + }, + } => { + assert_eq!(tenant_id, TenantId::new(1)); + assert_eq!(inflight, 4); + assert_eq!(cap, 4); + } + other => panic!("expected a tenant-cap refusal, got: {other}"), + } + assert_eq!( + request.request_id, + RequestId::new(99), + "the refused request is handed back unsent" + ); + } + + #[test] + fn full_weighted_fair_queue_is_refused_with_queue_full_scope() { + // Distinct tenants and databases keep the tenant cap and the per-DB + // suspension out of play, so only the queue total can refuse. + let (mut dispatcher, _data_sides) = Dispatcher::new(1, 4); + + for i in 1..=64u64 { + let mut request = make_request_for_db(0, i, i); + request.tenant_id = TenantId::new(i); + match dispatcher.dispatch(request) { + Ok(()) => continue, + Err(crate::Error::DispatchCapacity { + scope: DispatchCapacityScope::QueueFull { core_id, capacity }, + }) => { + assert_eq!(core_id, 0); + assert_eq!(capacity, 4); + return; + } + Err(other) => panic!("expected a queue-full refusal, got: {other}"), + } + } + panic!("the weighted-fair queue never filled"); + } + + #[test] + fn dispatch_to_core_tracks_request_lifecycle() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); + let request = make_request(0); + let tenant_id = request.tenant_id.as_u64(); + let request_id = request.request_id.as_u64(); + + dispatcher.dispatch_to_core(1, request).unwrap(); + + assert_eq!(dispatcher.tenant_inflight.get(&tenant_id), Some(&1)); + assert_eq!(dispatcher.request_tenant.get(&request_id), Some(&tenant_id)); + assert_eq!(data_sides[1].request_rx.len(), 1); + + let _req = data_sides[1].request_rx.try_pop().unwrap(); + data_sides[1] + .response_tx + .try_push(BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(request_id), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .unwrap(); + + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1); + assert_eq!(dispatcher.tenant_inflight.get(&tenant_id), Some(&0)); + assert!(!dispatcher.request_tenant.contains_key(&request_id)); + } + + #[test] + fn per_db_pressure_reported() { + let (mut dispatcher, _) = Dispatcher::new(1, 8); + // Fill fair share for DB 1 using 4 of 8 slots. + // With one DB initially, fair share = 8. With two DBs = 4 each. + // First enqueue DB1 + DB2, so fair_share = 4. + for i in 0..4u64 { + dispatcher + .dispatch(make_request_for_db(0, 1, i + 10)) + .unwrap(); + } + for i in 0..4u64 { + dispatcher + .dispatch(make_request_for_db(0, 2, i + 20)) + .unwrap(); + } + // After filling DB1's fair share, it should be suspended on core 0. + // (exact state depends on WFQ flush draining items to ring first) + // The test confirms per-DB pressure is being tracked without panic. + let _ = dispatcher.db_pressure_on_core(0, 1); + let _ = dispatcher.db_pressure_on_core(0, 2); + } +} diff --git a/nodedb/src/bridge/dispatch/mod.rs b/nodedb/src/bridge/dispatch/mod.rs index c261c0805..f7f192227 100644 --- a/nodedb/src/bridge/dispatch/mod.rs +++ b/nodedb/src/bridge/dispatch/mod.rs @@ -3,9 +3,15 @@ mod core_channel; mod dispatcher; mod drain; +mod enqueue; +mod refusal; +mod response_poll; +#[cfg(test)] +mod test_requests; pub use core_channel::{CoreChannel, CoreChannelDataSide}; pub use dispatcher::{ BridgeRequest, BridgeResponse, DatabasePriorityResolver, DefaultPriorityResolver, Dispatcher, }; pub use drain::CorePending; +pub use refusal::DispatchRefusal; diff --git a/nodedb/src/bridge/dispatch/refusal.rs b/nodedb/src/bridge/dispatch/refusal.rs new file mode 100644 index 000000000..5259d129d --- /dev/null +++ b/nodedb/src/bridge/dispatch/refusal.rs @@ -0,0 +1,24 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A refused dispatch that hands the unsent request back to the caller. + +use crate::bridge::envelope; + +/// A request the dispatcher refused, returned unsent with the reason. +/// +/// A caller that retries a capacity refusal re-dispatches `request` as is, +/// with no clone and no rebuild. +#[derive(Debug)] +pub struct DispatchRefusal { + /// Why the dispatcher refused the request. + pub error: crate::Error, + /// The refused request. The dispatcher tracked nothing for it. + pub request: envelope::Request, +} + +impl DispatchRefusal { + /// Box a refusal of `request` for `error`. + pub(super) fn boxed(error: crate::Error, request: envelope::Request) -> Box { + Box::new(Self { error, request }) + } +} diff --git a/nodedb/src/bridge/dispatch/response_poll.rs b/nodedb/src/bridge/dispatch/response_poll.rs new file mode 100644 index 000000000..8539eaec1 --- /dev/null +++ b/nodedb/src/bridge/dispatch/response_poll.rs @@ -0,0 +1,389 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Response polling for the bridge [`Dispatcher`], including the error +//! responses it synthesizes for a core that died. + +use std::collections::HashSet; + +use crate::bridge::envelope; +use crate::bridge::envelope::{ErrorCode, Payload, Status}; +use crate::types::{Lsn, RequestId}; + +use super::dispatcher::Dispatcher; +use super::enqueue::release_inflight_slot; + +impl Dispatcher { + /// Poll responses from all Data Plane cores. + /// + /// A core whose channel has been observed dead contributes a synthesized + /// error `Response` for every request still outstanding on it: the one a + /// failed `try_push` consumed, everything still staged in its WFQ, and + /// everything dispatched earlier that it never answered. Those travel back + /// with the real responses so the single completion loop in the caller + /// finishes each waiter, and the loop below releases each request's + /// `tenant_inflight` slot exactly as a real response would — without which + /// one dead core ratchets the tenant's in-flight count until the tenant is + /// rejected on healthy cores too. + /// + /// Fires [`Dispatcher::capacity_freed`] once when the poll released at + /// least one in-flight slot. + pub fn poll_responses(&mut self) -> Vec { + let mut responses = Vec::new(); + let mut freed = false; + for (core_id, channel) in self.cores.iter_mut().enumerate() { + let mut batch = Vec::new(); + let (_drained, producer_gone) = channel.response_rx.drain_into(&mut batch, 64); + for br in batch { + let rid = br.inner.request_id.as_u64(); + // A streaming scan answers with many partials before its final + // response. The request is still executing on the core until + // that final one arrives, so releasing it here would let the + // shutdown drain call a live scan finished and would drop the + // tenant's in-flight slot mid-stream. + if !br.inner.partial { + channel.outstanding.remove(&rid); + freed |= release_inflight_slot( + &mut self.request_tenant, + &mut self.tenant_inflight, + rid, + ); + } + responses.push(br.inner); + } + + if !(producer_gone || channel.request_tx.is_disconnected()) { + // Opportunistically flush WFQ after draining responses to fill headroom. + channel.flush_wfq(); + continue; + } + + // The core is gone. Collect every request it can no longer answer: + // items still staged in the WFQ first (dispatch order), then the + // rest of the outstanding set. A staged item is also in + // `outstanding`, so `seen` keeps each id to a single response. + let mut seen = HashSet::new(); + let mut lost = Vec::new(); + for staged in channel.wfq.drain() { + let rid = staged.request_id.as_u64(); + if seen.insert(rid) { + lost.push(rid); + } + } + for rid in channel.outstanding.drain() { + if seen.insert(rid) { + lost.push(rid); + } + } + + // Idempotence: both sources are emptied here — `wfq.drain` leaves + // the staging queue empty and `outstanding.drain` clears the set — + // and `flush_wfq` refuses to stage anything new onto a + // disconnected producer. A later poll therefore finds both empty + // and emits nothing, so a permanently dead core costs one pass + // over two empty containers rather than a repeating failure storm. + for rid in lost { + freed |= + release_inflight_slot(&mut self.request_tenant, &mut self.tenant_inflight, rid); + responses.push(envelope::Response { + request_id: RequestId::new(rid), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: Some(Box::new(ErrorCode::Internal { + detail: format!( + "core-{core_id} is gone; the request can never be executed" + ), + })), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }); + } + } + if freed { + self.capacity_freed.notify_waiters(); + } + responses + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + use crate::bridge::dispatch::BridgeResponse; + use crate::bridge::dispatch::test_requests::{make_request, make_request_for_db}; + use crate::types::*; + + #[test] + fn response_roundtrip() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); + + dispatcher.dispatch(make_request(0)).unwrap(); + + let _req = data_sides[0].request_rx.try_pop().unwrap(); + data_sides[0] + .response_tx + .try_push(BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(1), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::from_vec(b"result".to_vec()), + watermark_lsn: Lsn::new(42), + error_code: None, + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .unwrap(); + + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1); + assert_eq!(responses[0].status, Status::Ok); + assert_eq!(&*responses[0].payload, b"result"); + } + + /// A routed final response frees its tenant's slot and fires the + /// capacity-freed signal. + #[tokio::test] + async fn final_response_fires_capacity_freed() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 64); + dispatcher.dispatch(make_request(0)).unwrap(); + let _req = data_sides[0].request_rx.try_pop().unwrap(); + data_sides[0] + .response_tx + .try_push(BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(1), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .unwrap(); + + let signal = dispatcher.capacity_freed(); + let notified = signal.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + assert_eq!(dispatcher.poll_responses().len(), 1); + + assert!( + tokio::time::timeout(Duration::from_millis(100), notified) + .await + .is_ok(), + "a freed in-flight slot must fire the capacity-freed signal" + ); + } + + // --- Dead-core request loss --- + // + // When a Data Plane core's consumer/producer is dropped (the core thread + // died), `Dispatcher` must synthesize an error `Response` for every + // request it knows is outstanding on that core, rather than dropping the + // request silently and leaking the caller's waiter + `tenant_inflight` + // slot forever. Dropping one element of the `data_sides` vector handed + // back by `Dispatcher::new`/`with_resolver` simulates that core thread + // dying, matching how `dispatch_routes_to_correct_core` and + // `response_roundtrip` above obtain the data-plane side of the channel. + + #[test] + fn dead_core_synthesizes_error_response_for_lost_request() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(3, 64); + + // Core 2's thread has died: both halves of its data-plane side are gone. + let dead_core = 2; + drop(data_sides.remove(dead_core)); + + let request = make_request_for_db(0, 1, 7); + let request_id = request.request_id.as_u64(); + + // `dispatch_to_core` still reports success: the request was already + // moved into the doomed `try_push` inside `flush_wfq` before the + // failure is observed, which is exactly the defect being covered. + dispatcher.dispatch_to_core(dead_core, request).unwrap(); + + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1, "expected one synthesized response"); + let resp = &responses[0]; + assert_eq!(resp.request_id.as_u64(), request_id); + assert_eq!(resp.status, Status::Error); + match resp.error_code.as_deref() { + Some(ErrorCode::Internal { detail }) => { + assert!( + detail.contains(&dead_core.to_string()), + "error detail should name the dead core, got: {detail}" + ); + } + other => panic!("expected ErrorCode::Internal naming the core, got: {other:?}"), + } + } + + #[test] + fn dead_core_synthesized_response_resets_tenant_inflight() { + // The ratchet: `tenant_inflight` is incremented on dispatch and must + // return to its pre-dispatch value once the synthesized response for + // the lost request is drained through `poll_responses` — otherwise it + // climbs forever and eventually starves the tenant on healthy cores. + let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); + let dead_core = 0; + drop(data_sides.remove(dead_core)); + + let request = make_request_for_db(0, 1, 1); + let tenant_id = request.tenant_id.as_u64(); + + let before = dispatcher + .tenant_inflight + .get(&tenant_id) + .copied() + .unwrap_or(0); + + dispatcher.dispatch_to_core(dead_core, request).unwrap(); + assert_eq!( + dispatcher.tenant_inflight.get(&tenant_id).copied(), + Some(before + 1), + "dispatch must still increment tenant_inflight even though the core is dead" + ); + + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1); + assert_eq!( + dispatcher + .tenant_inflight + .get(&tenant_id) + .copied() + .unwrap_or(0), + before, + "tenant_inflight must return to its pre-dispatch value, not ratchet upward" + ); + assert!(!dispatcher.request_tenant.contains_key(&1)); + } + + #[test] + fn dead_core_does_not_affect_live_core() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(2, 64); + let dead_core = 0; + let live_core = 1; + drop(data_sides.remove(dead_core)); + // Removing index 0 shifted core 1's data side down to index 0. + let live_data_side = &mut data_sides[0]; + + let dead_request = make_request_for_db(0, 1, 1); + let live_request = make_request_for_db(0, 2, 2); + let live_request_id = live_request.request_id.as_u64(); + + dispatcher + .dispatch_to_core(dead_core, dead_request) + .unwrap(); + dispatcher + .dispatch_to_core(live_core, live_request) + .unwrap(); + + // The live core answers normally, through the real ring buffer. + let _req = live_data_side.request_rx.try_pop().unwrap(); + live_data_side + .response_tx + .try_push(BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(live_request_id), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .unwrap(); + + let responses = dispatcher.poll_responses(); + assert_eq!( + responses.len(), + 2, + "one synthesized error from the dead core, one real Ok from the live core" + ); + + let live_resp = responses + .iter() + .find(|r| r.request_id.as_u64() == live_request_id) + .expect("live core's real response must be present"); + assert_eq!(live_resp.status, Status::Ok); + assert!(live_resp.error_code.is_none()); + + let dead_resp = responses + .iter() + .find(|r| r.request_id.as_u64() != live_request_id) + .expect("dead core's synthesized response must be present"); + assert_eq!(dead_resp.status, Status::Error); + assert!(dead_resp.error_code.is_some()); + } + + #[test] + fn dead_core_fails_requests_still_queued_in_wfq() { + // Fill the physical ring to capacity while the core is alive, so a + // request dispatched afterward parks in the WFQ without ever + // attempting a push (flush_wfq's utilization check breaks before it + // reaches the doomed try_push). Then kill the core and confirm the + // WFQ-queued request is failed too, not left sitting in the queue + // forever. + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 4); + + for i in 0..4u64 { + dispatcher + .dispatch_to_core(0, make_request_for_db(0, i + 1, i + 1)) + .unwrap(); + } + assert_eq!(data_sides[0].request_rx.len(), 4); + + // Core 0's thread dies with 4 unanswered requests sitting in its ring. + drop(data_sides.remove(0)); + + // This request cannot reach the (full, dead) physical ring — it stays + // parked in the WFQ. + let parked_request_id = 99u64; + dispatcher + .dispatch_to_core(0, make_request_for_db(0, 99, parked_request_id)) + .unwrap(); + + let responses = dispatcher.poll_responses(); + let ids: std::collections::HashSet = + responses.iter().map(|r| r.request_id.as_u64()).collect(); + + // The 4 previously-dispatched-but-unanswered requests, plus the one + // still parked in the WFQ, must all be failed. + assert_eq!( + responses.len(), + 5, + "expected all 5 outstanding requests failed" + ); + for id in 1..=4u64 { + assert!( + ids.contains(&id), + "request {id} in the dead ring must be failed" + ); + } + assert!( + ids.contains(&parked_request_id), + "request parked in the WFQ must be failed, not left queued" + ); + for r in &responses { + assert_eq!(r.status, Status::Error); + assert!(r.error_code.is_some()); + } + } +} diff --git a/nodedb/src/bridge/dispatch/test_requests.rs b/nodedb/src/bridge/dispatch/test_requests.rs new file mode 100644 index 000000000..442aae21a --- /dev/null +++ b/nodedb/src/bridge/dispatch/test_requests.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Request builders shared by the dispatcher's unit tests. + +use std::time::{Duration, Instant}; + +use nodedb_physical::physical_plan::DocumentOp; + +use crate::bridge::envelope; +use crate::bridge::envelope::*; +use crate::types::*; + +/// A read request with id 1 for tenant 1 in the default database on `vshard`. +pub(super) fn make_request(vshard: u32) -> envelope::Request { + envelope::Request { + request_id: RequestId::new(1), + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(vshard), + plan: PhysicalPlan::Document(DocumentOp::PointGet { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + document_id: "u1".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }), + deadline: Instant::now() + Duration::from_secs(5), + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + } +} + +/// A read request with id `req_id` for tenant 1 in database `db` on `vshard`. +pub(super) fn make_request_for_db(vshard: u32, db: u64, req_id: u64) -> envelope::Request { + envelope::Request { + request_id: RequestId::new(req_id), + tenant_id: TenantId::new(1), + database_id: DatabaseId::new(db), + vshard_id: VShardId::new(vshard), + plan: PhysicalPlan::Document(DocumentOp::PointGet { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::new(db), "c"), + document_id: "d".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }), + deadline: Instant::now() + Duration::from_secs(5), + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + } +} diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index b0a115121..66f706df7 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -132,6 +132,10 @@ pub enum ErrorCode { /// special-cases `NotFound`) and reaches the client as SQLSTATE `22012` /// rather than the generic `XX000` every `Internal` maps to. DivisionByZero, + /// The bridge dispatcher refused the request at a capacity limit, so + /// nothing was enqueued or applied. Transient: the same request succeeds + /// once capacity frees. `reason` names the limit and its counts. + DispatchCapacity { reason: String }, } impl From for ErrorCode { @@ -218,6 +222,11 @@ impl From for ErrorCode { gate, retry_after_ms, }, + // A capacity refusal enqueued nothing, and the same request + // succeeds once capacity frees. + capacity @ crate::Error::DispatchCapacity { .. } => Self::DispatchCapacity { + reason: capacity.to_string(), + }, crate::Error::TxnOverlayMemoryExceeded { limit } => { Self::TxnOverlayMemoryExceeded { limit } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs index 623313670..81d0811e9 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs @@ -18,6 +18,7 @@ //! Txn into a no-op, and Reserve/Release re-application is a lock-manager no-op. use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::calvin::types::SchedulerInput; use super::scheduler::Scheduler; @@ -113,32 +114,72 @@ impl Scheduler { // 3. SM-lock scope: decode the raw log entries into this vShard's // `SchedulerInput` stream (a pure `&self` read — no side effects). - let inputs = { + // Each entry decodes on its own, so every input keeps the Raft index + // it came from. Decoding holds no cross-entry state, so the stream is + // identical to a whole-range decode. + let inputs: Vec<(u64, SchedulerInput)> = { let sm = self .sequencer_state_machine .lock() .unwrap_or_else(|p| p.into_inner()); - sm.replay_epochs_for_vshard(&entries, self.vshard_id, 0, u64::MAX) + entries + .iter() + .flat_map(|entry| { + sm.replay_epochs_for_vshard( + std::slice::from_ref(entry), + self.vshard_id, + 0, + u64::MAX, + ) + .into_iter() + .map(move |input| (entry.index, input)) + }) + .collect() }; // 4. Feed each replayed input through the SAME live processing path — no // lock held. Determinism: identical inputs through identical code. // The in-flight guard makes an overlapping already-in-flight Txn a // no-op; Reserve/Release re-application is idempotent. - let replayed = inputs.len() as u64; - for input in inputs { + // + // A dispatch refused at capacity stops the feed. The refused txn is + // parked in flight, and the next drain resumes at the first input + // not yet processed. + let mut replayed: u64 = 0; + let mut resume_from: Option = None; + let mut feed = inputs.into_iter().peekable(); + while let Some((_, input)) = feed.next() { + let deferred_before = self.deferred_dispatch_len(); self.process_scheduler_input(input); + replayed += 1; + if self.deferred_dispatch_len() > deferred_before { + resume_from = feed.peek().map(|(index, _)| *index); + break; + } } - // Replay of `lo ..= hi` is complete: clear the armed catch-up, but only - // up to `hi` — a concurrent drop recorded at an index `> hi` while this - // replay ran is preserved for the next drain. This is the CONFIRM step - // the peek-not-take at the top defers to; a transient failure above - // returned early and left the entry armed. - self.sequencer_state_machine - .lock() - .unwrap_or_else(|p| p.into_inner()) - .clear_catch_up_up_to(self.vshard_id, hi); + { + let sm = self + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()); + match resume_from { + // Stopped early: re-arm exactly at the first unprocessed input's + // index. Clearing below it first lets the min-collapse arm move + // the entry forward. Both run under one SM lock. + Some(next) => { + sm.clear_catch_up_up_to(self.vshard_id, next.saturating_sub(1)); + sm.arm_catch_up_from(self.vshard_id, next); + } + // Replay of `lo ..= hi` is complete: clear the armed catch-up, + // but only up to `hi` — a concurrent drop recorded at an index + // `> hi` while this replay ran is preserved for the next drain. + // This is the CONFIRM step the peek-not-take at the top defers + // to; a transient failure above returned early and left the + // entry armed. + None => sm.clear_catch_up_up_to(self.vshard_id, hi), + } + } if replayed > 0 { self.metrics.record_catch_up_replayed(replayed); @@ -160,12 +201,14 @@ mod tests { use std::sync::atomic::Ordering; use std::time::{Duration, Instant}; - use nodedb_cluster::calvin::SequencerEntry; use nodedb_cluster::calvin::types::{EpochBatch, SchedulerInput, SequencedTxn}; + use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerEntry}; + use nodedb_types::TenantId; use nodedb_types::id::{DatabaseId, VShardId}; use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ - build_test_scheduler, make_sequenced_txn, + build_test_scheduler, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_sequenced_txn, make_validate_only_txn, test_coll_vshard, }; use crate::control::cluster::calvin::scheduler::lock_manager::{AcquireOutcome, TxnId}; @@ -457,4 +500,68 @@ mod tests { "the guarded overlap must not cause a second dispatch" ); } + + /// Commit two single-txn batches at epochs 0 and 1, each a validate-only + /// txn that stages on this scheduler, and drop both through a full + /// fan-out channel. Returns the two committed Raft indexes. + fn arm_two_dropped_stage_batches(scheduler: &Scheduler) -> (u64, u64) { + ensure_sequencer_leader(scheduler); + let txn0 = make_validate_only_txn(0, 0); + let txn1 = make_validate_only_txn(1, 0); + let (idx0, bytes0) = commit_epoch_batch(scheduler, make_batch(0, &txn0)); + let (idx1, bytes1) = commit_epoch_batch(scheduler, make_batch(1, &txn1)); + assert!(idx1 > idx0, "second batch commits at a later Raft index"); + apply_with_full_channel(scheduler, scheduler.vshard_id, idx0, &bytes0, &txn0); + apply_with_full_channel(scheduler, scheduler.vshard_id, idx1, &bytes1, &txn1); + (idx0, idx1) + } + + /// Draining a replay range against a dispatcher at tenant capacity marks + /// no replayed position applied. + #[tokio::test] + async fn drain_against_full_dispatcher_marks_no_replayed_position_applied() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + arm_two_dropped_stage_batches(&scheduler); + let shared = std::sync::Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.drain_catch_up(); + + assert!( + !scheduler.applied.is_applied(0, 0), + "the refused replayed txn must stay unapplied" + ); + assert!( + !scheduler.applied.is_applied(1, 0), + "a replayed txn after the refusal must stay unapplied" + ); + } + + /// Draining against a dispatcher at tenant capacity stops at the first + /// refusal and leaves catch-up armed from the first input it did not + /// process. + #[tokio::test] + async fn drain_against_full_dispatcher_stays_armed_from_first_unprocessed_input() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let (_idx0, idx1) = arm_two_dropped_stage_batches(&scheduler); + let shared = std::sync::Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.drain_catch_up(); + + let armed = scheduler + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .peek_catch_up_from(scheduler.vshard_id); + assert_eq!( + armed, + Some(idx1), + "catch-up must stay armed from the input after the refused one" + ); + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index d9181f9bb..4b97d6d3c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -12,6 +12,7 @@ //! `finish_resolved_commit` / `commit_apply_tail` complete. use super::super::types::CommitState; +use super::deferred::{DispatchOutcome, DispatchStep}; use super::scheduler::Scheduler; use crate::bridge::envelope::{Response, Status}; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; @@ -40,7 +41,7 @@ impl Scheduler { position = txn_id.position, "calvin: CalvinResolve response was not Ok; locks NOT released (shard degraded)" ); - self.abort_redo_resolve_infra_error(txn_id); + self.complete_infra_abort(txn_id); return; } @@ -54,7 +55,7 @@ impl Scheduler { error = %e, "calvin: CalvinResolve redo record decode failed" ); - self.abort_redo_resolve_infra_error(txn_id); + self.complete_infra_abort(txn_id); return; } }; @@ -92,15 +93,18 @@ impl Scheduler { error = %e, "calvin: TransactionRedo WAL append failed" ); - self.abort_redo_resolve_infra_error(txn_id); + self.complete_infra_abort(txn_id); return; } } }; - if !self.dispatch_commit_resolution(txn_id, true, redo_lsn) { - // `dispatch_commit_resolution` already logged the dispatch failure. - self.abort_redo_resolve_infra_error(txn_id); + // A flush refused at capacity is parked for re-send. The txn awaits its + // flush response either way, so the state below is the same. + if let DispatchOutcome::Failed(error) = + self.dispatch_commit_resolution(txn_id, true, redo_lsn) + { + self.fail_dispatch_step(txn_id, DispatchStep::Flush, error); return; } @@ -114,8 +118,12 @@ impl Scheduler { /// Complete `txn_id` as an infra error: releases its locks so the epoch /// advances rather than stalling. Shared by every `finish_redo_resolve` - /// failure branch. - fn abort_redo_resolve_infra_error(&mut self, txn_id: TxnId) { + /// failure branch and every terminal resolve, flush, or drop dispatch + /// refusal. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn complete_infra_abort( + &mut self, + txn_id: TxnId, + ) { self.metrics.record_executor_error(); self.metrics .record_infra_abort(infra_abort_reason::IO_ERROR); @@ -130,14 +138,15 @@ impl Scheduler { /// Mirrors `dispatch_commit_resolution`'s exempt, no-WAL-LSN dispatch /// shape — a resolve reads the staged overlay and writes nothing. /// - /// Returns `false` if the dispatch failed (the caller then completes the - /// txn as an infra error). + /// A capacity refusal returns [`DispatchOutcome::Deferred`]: the resolve + /// is parked for re-send and the txn stays in flight. A txn with no + /// `pending` entry returns [`DispatchOutcome::Failed`]. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_calvin_resolve( &mut self, txn_id: TxnId, - ) -> bool { + ) -> DispatchOutcome { let Some(pending) = self.pending.get(&txn_id) else { - return false; + return DispatchOutcome::Failed(missing_pending_error(txn_id)); }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; @@ -150,26 +159,21 @@ impl Scheduler { // itself, so no committed LSN rides on this envelope. let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, None); - let resp_rx = self.shared.tracker.register(request_id); - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - if let Err(e) = dispatch_result { - self.shared.tracker.cancel(&request_id); - tracing::error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin: CalvinResolve dispatch failed" - ); - return false; - } - // The resolve response re-enters the completion loop under the SAME // txn_id, now in `AwaitingRedoResolve`, where `finish_redo_resolve` runs. - self.spawn_response_bridge(txn_id, request_id, resp_rx); - true + self.dispatch_sequenced(txn_id, DispatchStep::Resolve, request) + } +} + +/// The terminal error for a commit-resolution dispatch whose txn has no +/// `pending` entry to build the request from. +pub(in crate::control::cluster::calvin::scheduler::driver::core) fn missing_pending_error( + txn_id: TxnId, +) -> crate::Error { + crate::Error::Internal { + detail: format!( + "calvin txn {}/{} has no pending entry to dispatch from", + txn_id.epoch, txn_id.position + ), } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs index 42dc11229..362cef90b 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs @@ -5,49 +5,43 @@ use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; +use super::commit_redo::missing_pending_error; +use super::deferred::{DispatchOutcome, DispatchStep}; use super::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; impl Scheduler { /// Dispatch a flush or drop of a staged transaction's commit-pending buffer. + /// + /// A capacity refusal returns [`DispatchOutcome::Deferred`]: the flush or + /// drop is parked for re-send and the txn stays in flight. A txn with no + /// `pending` entry returns [`DispatchOutcome::Failed`]. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_commit_resolution( &mut self, txn_id: TxnId, committed: bool, wal_lsn: Option, - ) -> bool { + ) -> DispatchOutcome { let Some(pending) = self.pending.get(&txn_id) else { - return false; + return DispatchOutcome::Failed(missing_pending_error(txn_id)); }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; let epoch = txn_id.epoch; let position = txn_id.position; - let plan = if committed { - PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch, position }) + let (plan, step) = if committed { + ( + PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch, position }), + DispatchStep::Flush, + ) } else { - PhysicalPlan::Meta(MetaOp::CalvinDrop { epoch, position }) + ( + PhysicalPlan::Meta(MetaOp::CalvinDrop { epoch, position }), + DispatchStep::Drop, + ) }; let request_id = self.next_request_id(); let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, wal_lsn); - let resp_rx = self.shared.tracker.register(request_id); - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut dispatcher) => dispatcher.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - if let Err(error) = dispatch_result { - self.shared.tracker.cancel(&request_id); - tracing::error!( - vshard_id = self.vshard_id, - epoch, - position, - committed, - %error, - "calvin: commit resolution dispatch failed" - ); - return false; - } - self.spawn_response_bridge(txn_id, request_id, resp_rx); - true + self.dispatch_sequenced(txn_id, step, request) } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs index da27702da..cee99eae4 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs @@ -8,10 +8,12 @@ use std::time::Instant; use nodedb_cluster::calvin::VerdictSignal; +use crate::control::cluster::calvin::scheduler::driver::core::deferred::{ + DispatchOutcome, DispatchStep, +}; use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::driver::types::CommitState; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; -use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; impl Scheduler { /// Resume a txn parked in [`CommitState::AwaitingVerdict`] once the durable @@ -45,27 +47,29 @@ impl Scheduler { return; } - let dispatched = if committed { + let (outcome, step) = if committed { // Resolve the staged post-images into a replayable `RedoRecord` // first; the redo is WAL-appended (in `finish_redo_resolve`) before // the flush is dispatched, restoring restart durability for this // vShard's slice of a multi-shard Calvin commit. - self.dispatch_calvin_resolve(txn_id) + (self.dispatch_calvin_resolve(txn_id), DispatchStep::Resolve) } else { - self.dispatch_commit_resolution(txn_id, false, None) + ( + self.dispatch_commit_resolution(txn_id, false, None), + DispatchStep::Drop, + ) }; - if !dispatched { - // Resolve/drop dispatch failed: complete the txn as an infra error so - // its locks release and the epoch advances rather than stalling. The - // staged buffer is reclaimed by a later drop or on core teardown. - self.metrics.record_executor_error(); - self.metrics - .record_infra_abort(infra_abort_reason::IO_ERROR); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); + if let DispatchOutcome::Failed(error) = outcome { + // Terminal resolve/drop refusal: complete the txn as an infra error + // so its locks release and the epoch advances rather than stalling. + // The staged buffer is reclaimed by a later drop or on core teardown. + self.fail_dispatch_step(txn_id, step, error); return; } + // Sent or parked for re-send at capacity: either way the txn awaits + // this step's response, so a duplicate verdict push or probe is a + // no-op under the guard above. if let Some(pending) = self.pending.get_mut(&txn_id) { pending.commit_state = Some(if committed { CommitState::AwaitingRedoResolve @@ -165,12 +169,189 @@ mod tests { }; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; + use nodedb_types::TenantId; use super::*; - use crate::bridge::envelope::Status; + use crate::bridge::dispatch::CoreChannelDataSide; + use crate::bridge::envelope::{Payload, Status}; use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ - build_test_scheduler_with_data_side, make_sequenced_txn, staged_pending, staged_response, + await_data_plane_request, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_sequenced_txn, release_filler, spawn_scheduler_loop, staged_pending, staged_response, }; + use crate::control::state::SharedState; + use crate::types::RequestId; + use crate::wal::RedoRecord; + + /// A scheduler with one txn parked in a commit state while its tenant sits + /// at the dispatcher's in-flight cap. + struct ParkedAtCapacity { + scheduler: Scheduler, + _dir: tempfile::TempDir, + data_side: CoreChannelDataSide, + shared: Arc, + fillers: Vec, + } + + /// Park `txn_id` in `state`, then fill its tenant to the in-flight cap. + fn parked_at_capacity(txn_id: TxnId, state: CommitState) -> ParkedAtCapacity { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, dir, mut data_side) = build_test_scheduler_with_data_side(7, registry); + let mut pending = staged_pending(make_sequenced_txn(txn_id.epoch, txn_id.position), txn_id); + pending.commit_state = Some(state); + scheduler.pending.insert(txn_id, pending); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + ParkedAtCapacity { + scheduler, + _dir: dir, + data_side, + shared, + fillers, + } + } + + /// An Ok resolve response whose redo record carries no ops, so the flush + /// dispatch follows at once with no WAL append. + fn empty_redo_response() -> crate::bridge::envelope::Response { + let redo = RedoRecord { + version: 1, + ops: Vec::new(), + calvin_stamp: None, + }; + let mut response = staged_response(Status::Ok, None); + response.payload = Payload::from_vec(redo.to_bytes().expect("encode empty redo record")); + response + } + + /// Under a COMMIT verdict, a refused resolve dispatch does not complete + /// the txn: its position stays unapplied and its pending entry stays. + #[tokio::test] + async fn commit_verdict_resolve_refused_at_capacity_does_not_complete_txn() { + let txn_id = TxnId::new(14, 2); + let mut parked = parked_at_capacity(txn_id, CommitState::AwaitingVerdict); + let scheduler = &mut parked.scheduler; + + scheduler.resume_on_verdict(txn_id, true); + + assert!( + !scheduler.applied.is_applied(14, 2), + "a refused resolve must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "a refused resolve must keep the txn's pending entry" + ); + } + + /// A refused flush dispatch after the redo resolves does not complete the + /// txn: its position stays unapplied and its pending entry stays. + #[tokio::test] + async fn resolved_redo_flush_refused_at_capacity_does_not_complete_txn() { + let txn_id = TxnId::new(14, 2); + let mut parked = parked_at_capacity(txn_id, CommitState::AwaitingRedoResolve); + let scheduler = &mut parked.scheduler; + + scheduler.finish_redo_resolve(txn_id, empty_redo_response()); + + assert!( + !scheduler.applied.is_applied(14, 2), + "a refused flush must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "a refused flush must keep the txn's pending entry" + ); + } + + /// Under an ABORT verdict, a refused drop dispatch does not complete the + /// txn: its position stays unapplied and its pending entry stays. + #[tokio::test] + async fn abort_verdict_drop_refused_at_capacity_does_not_complete_txn() { + let txn_id = TxnId::new(14, 2); + let mut parked = parked_at_capacity(txn_id, CommitState::AwaitingVerdict); + let scheduler = &mut parked.scheduler; + + scheduler.resume_on_verdict(txn_id, false); + + assert!( + !scheduler.applied.is_applied(14, 2), + "a refused drop must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "a refused drop must keep the txn's pending entry" + ); + } + + /// Once a Data Plane response frees tenant capacity, a refused resolve + /// reaches the Data Plane. + #[tokio::test] + async fn refused_resolve_reaches_data_plane_after_capacity_frees() { + let txn_id = TxnId::new(14, 2); + let ParkedAtCapacity { + mut scheduler, + _dir, + mut data_side, + shared, + fillers, + } = parked_at_capacity(txn_id, CommitState::AwaitingVerdict); + + scheduler.resume_on_verdict(txn_id, true); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinResolve { + epoch: 14, + position: 2 + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused resolve must reach the Data Plane once capacity frees" + ); + } + + /// Once a Data Plane response frees tenant capacity, a refused flush + /// reaches the Data Plane. + #[tokio::test] + async fn refused_flush_reaches_data_plane_after_capacity_frees() { + let txn_id = TxnId::new(14, 2); + let ParkedAtCapacity { + mut scheduler, + _dir, + mut data_side, + shared, + fillers, + } = parked_at_capacity(txn_id, CommitState::AwaitingRedoResolve); + + scheduler.finish_redo_resolve(txn_id, empty_redo_response()); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch: 14, + position: 2 + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused flush must reach the Data Plane once capacity frees" + ); + } /// A false vote from either participant makes the only global verdict abort; /// applying that durable verdict broadcasts the abort to every parked local diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs new file mode 100644 index 000000000..3fc7aea53 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs @@ -0,0 +1,291 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capacity-safe Data-Plane dispatch for sequenced Calvin work. +//! +//! A sequenced txn has no refusal outcome: every replica must apply it. So +//! every scheduler dispatch goes through [`Scheduler::dispatch_sequenced`]. +//! A capacity refusal parks the request in a FIFO. The txn keeps its locks +//! and its `pending` entry, and never reaches `on_txn_complete`. The run loop +//! re-sends parked requests once a routed response frees capacity. + +use std::collections::VecDeque; +use std::sync::atomic::Ordering; +use std::time::Instant; + +use nodedb_physical::physical_plan::PhysicalPlan; +use nodedb_physical::physical_plan::meta::MetaOp; +use tokio::sync::mpsc; + +use super::scheduler::Scheduler; +use crate::bridge::dispatch::DispatchRefusal; +use crate::bridge::envelope::{Request, Response}; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::types::RequestId; + +/// The Calvin sub-operation one scheduler dispatch carries. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum DispatchStep { + /// `CalvinExecuteStatic` stage of a static txn. + StageStatic, + /// `CalvinExecuteActive` stage of a dependent-read txn. + StageActive, + /// `CalvinResolve` of a committed staged txn. + Resolve, + /// `CalvinFlush` of a committed staged txn. + Flush, + /// `CalvinDrop` of an aborted staged txn. + Drop, + /// One-way `RecordCalvinWriteVersions` of a committed txn. + WriteVersionRecord, +} + +/// Result of [`Scheduler::dispatch_sequenced`]. +#[derive(Debug)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum DispatchOutcome { + /// The dispatcher accepted the request. + Sent, + /// The dispatcher refused at capacity. The request waits in the deferred + /// FIFO, and the txn stays in flight. + Deferred, + /// The dispatcher refused terminally. Nothing is parked. + Failed(crate::Error), +} + +/// A request refused at capacity, waiting to be re-sent. +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct DeferredDispatch { + txn_id: TxnId, + step: DispatchStep, + request: Request, +} + +/// FIFO of requests refused at capacity, in refusal order. +pub(in crate::control::cluster::calvin::scheduler::driver::core) type DeferredQueue = + VecDeque; + +/// Result of one send attempt, before the caller decides where a refused +/// request goes in the FIFO. +enum Attempt { + Sent, + Capacity(Box), + Failed(crate::Error), +} + +impl Scheduler { + /// Register `request` in the tracker, dispatch it, and cancel the + /// registration on refusal. + /// + /// A capacity refusal appends the request to the deferred FIFO and + /// returns [`DispatchOutcome::Deferred`]. The caller keeps the txn in + /// flight. Any other refusal returns [`DispatchOutcome::Failed`], and the + /// caller runs its terminal handling. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_sequenced( + &mut self, + txn_id: TxnId, + step: DispatchStep, + request: Request, + ) -> DispatchOutcome { + match self.send_once(txn_id, step, request) { + Attempt::Sent => DispatchOutcome::Sent, + Attempt::Capacity(parked) => { + self.deferred.push_back(*parked); + self.metrics + .set_dispatch_deferred_depth(self.deferred.len()); + DispatchOutcome::Deferred + } + Attempt::Failed(error) => DispatchOutcome::Failed(error), + } + } + + /// Whether any refused request waits for capacity. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn has_deferred_dispatch( + &self, + ) -> bool { + !self.deferred.is_empty() + } + + /// Number of refused requests waiting for capacity. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn deferred_dispatch_len( + &self, + ) -> usize { + self.deferred.len() + } + + /// Re-send parked requests in FIFO order. + /// + /// Stops at the first capacity refusal, which goes back to the FIFO + /// front. A terminal refusal runs the step's terminal handling. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn redispatch_deferred( + &mut self, + ) { + while let Some(parked) = self.deferred.pop_front() { + let DeferredDispatch { + txn_id, + step, + mut request, + } = parked; + if step != DispatchStep::WriteVersionRecord && !self.pending.contains_key(&txn_id) { + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + ?step, + "calvin: parked dispatch for a txn no longer pending; one step per txn is broken, discarding" + ); + continue; + } + self.refresh_deferred_request(step, &mut request); + match self.send_once(txn_id, step, request) { + Attempt::Sent => {} + Attempt::Capacity(parked) => { + self.deferred.push_front(*parked); + break; + } + Attempt::Failed(error) => self.fail_dispatch_step(txn_id, step, error), + } + } + self.metrics + .set_dispatch_deferred_depth(self.deferred.len()); + } + + /// Run the terminal handling of a dispatch the dispatcher refused for a + /// reason other than capacity. + /// + /// Stage steps release the txn's locks. Resolve, flush, and drop steps + /// complete the txn as an infra abort. A write-version record is dropped + /// with a warning, because the commit does not depend on it. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn fail_dispatch_step( + &mut self, + txn_id: TxnId, + step: DispatchStep, + error: crate::Error, + ) { + match step { + DispatchStep::StageStatic | DispatchStep::StageActive => { + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + ?step, + %error, + "calvin scheduler: dispatch failed; releasing locks" + ); + self.on_txn_complete(txn_id); + } + DispatchStep::Resolve | DispatchStep::Flush | DispatchStep::Drop => { + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + ?step, + %error, + "calvin: commit resolution dispatch failed" + ); + self.complete_infra_abort(txn_id); + } + DispatchStep::WriteVersionRecord => { + tracing::warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + %error, + "calvin: write-version record dispatch failed" + ); + } + } + } + + /// One send attempt: register, dispatch, and cancel on refusal. + fn send_once(&mut self, txn_id: TxnId, step: DispatchStep, request: Request) -> Attempt { + let request_id = request.request_id; + let resp_rx = self.shared.tracker.register(request_id); + let result = match self.shared.dispatcher.lock() { + Ok(mut dispatcher) => dispatcher.try_dispatch(request), + Err(poisoned) => poisoned.into_inner().try_dispatch(request), + }; + let refusal = match result { + Ok(()) => { + self.on_dispatch_sent(txn_id, step, request_id, resp_rx); + return Attempt::Sent; + } + Err(refusal) => refusal, + }; + self.shared.tracker.cancel(&request_id); + let DispatchRefusal { error, request } = *refusal; + match error { + crate::Error::DispatchCapacity { scope } => { + self.metrics.record_dispatch_deferred(); + tracing::debug!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + ?step, + %scope, + "calvin: dispatch refused at capacity; deferred until capacity frees" + ); + Attempt::Capacity(Box::new(DeferredDispatch { + txn_id, + step, + request, + })) + } + other => Attempt::Failed(other), + } + } + + /// Per-step bookkeeping once the dispatcher accepts a request. + fn on_dispatch_sent( + &mut self, + txn_id: TxnId, + step: DispatchStep, + request_id: RequestId, + resp_rx: mpsc::Receiver, + ) { + match step { + DispatchStep::StageStatic | DispatchStep::StageActive => { + self.metrics.record_dispatch(); + if let Some(pending) = self.pending.get_mut(&txn_id) { + // no-determinism: dispatch_time is executor-latency observability, off-WAL + pending.dispatch_time = Instant::now(); + } + self.spawn_response_bridge(txn_id, request_id, resp_rx); + } + DispatchStep::Resolve | DispatchStep::Flush | DispatchStep::Drop => { + self.spawn_response_bridge(txn_id, request_id, resp_rx); + } + DispatchStep::WriteVersionRecord => { + // One-way record: drain the response so it routes to a live + // receiver, then discard it. + tokio::spawn(async move { + let mut rx = resp_rx; + let _ = rx.recv().await; + }); + self.shared + .calvin_counters + .write_versions_recorded + .fetch_add(1, Ordering::Relaxed); + } + } + } + + /// Refresh the dispatch-time fields of a parked request before a re-send. + /// + /// The deadline restarts from now. A stage request re-reads group + /// leadership, which the Data Plane uses to gate OLLP verification. + fn refresh_deferred_request(&self, step: DispatchStep, request: &mut Request) { + request.deadline = self.request_deadline(); + if !matches!(step, DispatchStep::StageStatic | DispatchStep::StageActive) { + return; + } + if let PhysicalPlan::Meta( + MetaOp::CalvinExecuteStatic { + is_group_leader, .. + } + | MetaOp::CalvinExecuteActive { + is_group_leader, .. + }, + ) = &mut request.plan + { + *is_group_leader = self.is_group_leader(); + } + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs index 206d24982..13a93c247 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs @@ -11,6 +11,7 @@ use nodedb_cluster::calvin::types::SequencedTxn; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; +use super::super::deferred::{DispatchOutcome, DispatchStep}; use super::super::scheduler::Scheduler; use super::primary_write::{ participant_change_sets, plans_have_primary_write, plans_have_returning, @@ -45,7 +46,7 @@ impl Scheduler { error = %e, "calvin scheduler: active plan decode failed; releasing locks" ); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } }; @@ -75,7 +76,7 @@ impl Scheduler { "calvin scheduler: active txn homes no local writes; releasing locks" ); self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } Err(e) => { @@ -87,11 +88,17 @@ impl Scheduler { "calvin scheduler: active txn routing failed; releasing locks" ); self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } }; - if !self.bind_local_identities(&mut plans, txn.tx_class.database_id, tenant_id, txn_id) { + if !self.bind_local_identities( + &mut plans, + txn.tx_class.database_id, + tenant_id, + txn_id, + lock_owner, + ) { return; } let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); @@ -110,42 +117,18 @@ impl Scheduler { // Calvin allocates the CalvinApplied WAL LSN post-apply (in the // scheduler's response handler), so no committed LSN is known at // dispatch time to stamp here. - let request = - self.build_exempt_request(request_id, tenant_id, txn.tx_class.database_id, plan, None); - - let resp_rx = self.shared.tracker.register(request_id); - - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - - if let Err(e) = dispatch_result { - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: active dispatch failed; releasing locks" - ); - self.on_txn_complete(txn_id); - return; - } - - self.metrics.record_dispatch(); - - // no-determinism: executor latency observability, off-WAL path - let dispatch_instant = Instant::now(); - - self.spawn_response_bridge(txn_id, request_id, resp_rx); + let database_id = txn.tx_class.database_id; + let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, None); + // The txn enters `pending` before the dispatch, so a stage refused at + // capacity stays in flight with its locks until the re-send. self.pending.insert( txn_id, super::super::super::types::PendingTxn { txn, lock_owner, // no-determinism: dispatch_time is scheduler observability, not Calvin WAL data - dispatch_time: dispatch_instant, + dispatch_time: Instant::now(), has_primary_write, has_returning, change_sets, @@ -159,5 +142,83 @@ impl Scheduler { verdict_deadline: None, }, ); + + if let DispatchOutcome::Failed(error) = + self.dispatch_sequenced(txn_id, DispatchStep::StageActive, request) + { + self.fail_dispatch_step(txn_id, DispatchStep::StageActive, error); + } + } +} + +#[cfg(test)] +mod tests { + use std::collections::BTreeMap; + use std::sync::Arc; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_types::TenantId; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + await_data_plane_request, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_local_write_txn, release_filler, spawn_scheduler_loop, test_coll_vshard, + }; + + /// A refused active dispatch keeps the txn in flight: its position stays + /// unapplied and its pending entry stays. + #[tokio::test] + async fn active_dispatch_refused_at_capacity_leaves_txn_unapplied() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let txn_id = TxnId::new(5, 0); + + scheduler.dispatch_active_txn(make_local_write_txn(5, 0), txn_id, txn_id, BTreeMap::new()); + + assert!( + !scheduler.applied.is_applied(5, 0), + "a refused active dispatch must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "a refused active dispatch must keep the txn's pending entry" + ); + } + + /// Once a Data Plane response frees tenant capacity, the refused active + /// request reaches the Data Plane. + #[tokio::test] + async fn active_dispatch_refused_at_capacity_is_retried_after_capacity_frees() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let txn_id = TxnId::new(5, 0); + + scheduler.dispatch_active_txn(make_local_write_txn(5, 0), txn_id, txn_id, BTreeMap::new()); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinExecuteActive { + epoch: 5, + position: 0, + .. + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused active request must reach the Data Plane once capacity frees" + ); } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs index e8ffd9451..e00a6202d 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs @@ -22,12 +22,15 @@ impl Scheduler { /// failure: applying rows nobody can resolve by key is worse than aborting. /// /// Returns `false` after terminating the txn; the caller returns at once. + /// The txn is not yet in `pending`, so its locks release under + /// `lock_owner`. pub(super) fn bind_local_identities( &mut self, plans: &mut [PhysicalPlan], database_id: DatabaseId, tenant_id: TenantId, txn_id: TxnId, + lock_owner: TxnId, ) -> bool { let assigner = &self.shared.surrogate_assigner; let bound = plans @@ -46,7 +49,7 @@ impl Scheduler { "calvin scheduler: surrogate binding failed; releasing locks" ); self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); false } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs index e8443cf30..df973a1a9 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs @@ -11,6 +11,7 @@ use nodedb_cluster::calvin::types::SequencedTxn; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; +use super::super::deferred::{DispatchOutcome, DispatchStep}; use super::super::routing::PlanRouting; use super::super::scheduler::Scheduler; use super::primary_write::{ @@ -148,7 +149,7 @@ impl Scheduler { error = %e, "calvin scheduler: plan decode failed; releasing locks and skipping txn" ); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } }; @@ -165,11 +166,17 @@ impl Scheduler { "calvin scheduler: static txn routing failed; releasing locks" ); self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } }; - if !self.bind_local_identities(&mut local, txn.tx_class.database_id, tenant_id, txn_id) { + if !self.bind_local_identities( + &mut local, + txn.tx_class.database_id, + tenant_id, + txn_id, + lock_owner, + ) { return; } @@ -200,7 +207,7 @@ impl Scheduler { "calvin scheduler: static txn homes no local work; releasing locks" ); self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_txn_complete(txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); return; } @@ -219,8 +226,8 @@ impl Scheduler { ); } - /// Build and dispatch a `CalvinExecuteStatic` task, then park the txn in - /// `pending` as `Staged`. + /// Park the txn in `pending` as `Staged`, then build and dispatch its + /// `CalvinExecuteStatic` task. /// /// Shared by the write path (`plans` = this vShard's local write slice) and /// the validate-only read path (`plans` empty). Both carry the txn's FULL @@ -228,6 +235,9 @@ impl Scheduler { /// the read-set — whether or not `plans` is empty — and returns the commit /// vote on `read_set_valid`. A validate-only task has `has_primary_write == /// false`, so it deposits no result sidecar entry, exactly as intended. + /// + /// The txn enters `pending` before the dispatch, so a stage refused at + /// capacity stays in flight with its locks until the re-send. fn dispatch_calvin_static( &mut self, txn: SequencedTxn, @@ -246,6 +256,7 @@ impl Scheduler { let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); let has_returning = plans_have_returning(&plans); let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); + let database_id = txn.tx_class.database_id; let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteStatic { epoch, position, @@ -262,34 +273,7 @@ impl Scheduler { // Calvin allocates the CalvinApplied WAL LSN post-apply (in the // scheduler's response handler), so no committed LSN is known at // dispatch time to stamp here. - let request = - self.build_exempt_request(request_id, tenant_id, txn.tx_class.database_id, plan, None); - - let resp_rx = self.shared.tracker.register(request_id); - - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - - if let Err(e) = dispatch_result { - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: dispatch failed; releasing locks" - ); - self.on_txn_complete(txn_id); - return; - } - - self.metrics.record_dispatch(); - - // no-determinism: executor latency observability, off-WAL path - let dispatch_instant = Instant::now(); - - self.spawn_response_bridge(txn_id, request_id, resp_rx); + let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, None); self.pending.insert( txn_id, @@ -297,11 +281,11 @@ impl Scheduler { txn, lock_owner, // no-determinism: dispatch_time is scheduler observability, not Calvin WAL data - dispatch_time: dispatch_instant, + dispatch_time: Instant::now(), has_primary_write, has_returning, change_sets, - // This dispatch STAGED the txn (validate + buffer, no apply); + // This dispatch STAGES the txn (validate + buffer, no apply); // its response carries the local commit vote that drives the // subsequent flush-or-drop. commit_state: Some(super::super::super::types::CommitState::Staged), @@ -309,5 +293,11 @@ impl Scheduler { verdict_deadline: None, }, ); + + if let DispatchOutcome::Failed(error) = + self.dispatch_sequenced(txn_id, DispatchStep::StageStatic, request) + { + self.fail_dispatch_step(txn_id, DispatchStep::StageStatic, error); + } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index 52cfb1c8e..a27e43e39 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -17,6 +17,8 @@ //! - [`catch_up`] — sequencer-fan-out catch-up drain: replays inputs dropped on //! this replica (channel Full/Closed) from the committed sequencer Raft log. //! - [`dispatch`] — static / active dispatch to the Data Plane executor. +//! - [`deferred`] — capacity-safe dispatch: parks a request the bridge refuses +//! at capacity and re-sends it once capacity frees. //! - [`routing`] — exhaustive `PhysicalPlan` → vshard routing oracle used by //! `dispatch`'s local-plan filtering. //! - [`commit_resolve`] — verdict-driven flush-or-drop of a staged static @@ -50,6 +52,7 @@ pub mod commit_redo; pub mod commit_resolution_dispatch; pub mod commit_resolve; pub mod completion_route; +pub mod deferred; pub mod dispatch; pub mod process; pub mod propose; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs index 7f20cef25..23bf6c66f 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs @@ -243,21 +243,41 @@ impl Scheduler { self.dependent_barrier.insert(txn_id, barrier); } - /// Called when a transaction completes (success or infrastructure error). + /// Complete an in-flight txn (success or infrastructure error). + /// + /// Releases the lock-table owner recorded in its `pending` entry. A txn + /// with no `pending` entry has already completed, so this logs and + /// releases nothing. A txn that fails before it enters `pending` uses + /// [`Self::on_unpending_txn_complete`] instead. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn on_txn_complete( &mut self, txn_id: TxnId, ) { - let epoch = txn_id.epoch; - // Recover the lock-table owner (equals `txn_id` unless a reservation - // owned the lock). Blocked txns never reach here, so `pending` always - // holds the entry by the time a txn completes. - let lock_owner = self - .pending - .get(&txn_id) - .map(|p| p.lock_owner) - .unwrap_or(txn_id); + let Some(pending) = self.pending.remove(&txn_id) else { + tracing::error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + "calvin: completion for a txn with no pending entry; nothing to release" + ); + return; + }; + self.release_and_mark_applied(txn_id, pending.lock_owner); + } + /// Complete a txn that failed before it entered `pending`, releasing the + /// locks held under `lock_owner`. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn on_unpending_txn_complete( + &mut self, + txn_id: TxnId, + lock_owner: TxnId, + ) { + self.release_and_mark_applied(txn_id, lock_owner); + } + + /// Release `lock_owner`'s locks, dispatch the promoted waiters, and mark + /// `txn_id`'s position applied. + fn release_and_mark_applied(&mut self, txn_id: TxnId, lock_owner: TxnId) { // Release this txn's locks. `release` promotes any waiter queued behind // each freed key to holder (moving it pending -> held) and returns the // fully-promoted ids. Those ids are already holders in the table the @@ -275,11 +295,9 @@ impl Scheduler { // once ALL of its positions for this vShard have terminally completed, // so any advertised watermark reflects a FULLY-applied epoch — the value // `BEGIN` needs for a torn-free cross-shard snapshot anchor. - if let Some(watermark) = self.applied.mark_applied(epoch, txn_id.position) { + if let Some(watermark) = self.applied.mark_applied(txn_id.epoch, txn_id.position) { self.publish_watermark(watermark); } - - self.pending.remove(&txn_id); } /// Dispatch transactions that a `LockManager::release` promoted to holder. @@ -364,10 +382,112 @@ mod tests { use std::collections::BTreeSet; use std::sync::atomic::Ordering; + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_physical::physical_plan::PhysicalPlan; + use nodedb_physical::physical_plan::meta::MetaOp; + use nodedb_types::TenantId; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ - build_test_scheduler, make_sequenced_txn, + await_data_plane_request, build_test_scheduler, build_test_scheduler_with_data_side, + fill_tenant_inflight, make_sequenced_txn, make_validate_only_txn, release_filler, + spawn_scheduler_loop, test_coll_vshard, }; + /// A refused stage dispatch leaves the txn unapplied and publishes no + /// watermark for its epoch. + #[tokio::test] + async fn stage_dispatch_refused_at_capacity_leaves_txn_unapplied() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let watermark_before = shared.last_applied_calvin_epoch.load(Ordering::Acquire); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + + assert!( + !scheduler.applied.is_applied(3, 0), + "a capacity refusal must not mark the position applied" + ); + assert_eq!( + shared.last_applied_calvin_epoch.load(Ordering::Acquire), + watermark_before, + "a capacity refusal must not publish a watermark for the txn's epoch" + ); + } + + /// A refused stage dispatch keeps the txn's key locks: a later txn on the + /// same key queues behind it. + #[tokio::test] + async fn stage_dispatch_refused_at_capacity_keeps_key_locks() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(4, 0))); + + assert!( + scheduler.blocked.contains_key(&TxnId::new(4, 0)), + "a txn on the same key must block behind the refused txn's held locks" + ); + } + + /// A refused stage dispatch leaves no request-tracker entry behind. + #[tokio::test] + async fn stage_dispatch_refused_at_capacity_leaves_no_tracker_entry() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let tracked_before = shared.tracker.in_flight(); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + + assert_eq!( + shared.tracker.in_flight(), + tracked_before, + "a refused dispatch must not leave a registered tracker entry" + ); + } + + /// Once a Data Plane response frees tenant capacity, the refused txn's + /// stage request reaches the Data Plane. + #[tokio::test] + async fn stage_dispatch_refused_at_capacity_is_retried_after_capacity_frees() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinExecuteStatic { + epoch: 3, + position: 0, + .. + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused stage request must reach the Data Plane once capacity frees" + ); + } + #[tokio::test] async fn in_flight_guard_skips_replayed_txn_already_in_flight() { let (mut scheduler, _dir) = build_test_scheduler(0); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs index 55bcfd619..4a5a11539 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs @@ -77,7 +77,9 @@ impl Scheduler { self.metrics.record_infra_abort( crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason::PASSIVE_PARTICIPANT_TIMEOUT, ); - self.on_txn_complete(txn_id); + // A barrier txn never entered `pending`: release its locks + // under the owner the barrier recorded. + self.on_unpending_txn_complete(txn_id, barrier.lock_owner); } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs index 7648c3586..f967e7da4 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs @@ -32,11 +32,7 @@ impl Scheduler { database_id, vshard_id: VShardId::new(self.vshard_id), plan, - // no-determinism: scheduler deadline controls waiting, not ordered state. - deadline: Instant::now() - + Duration::from_millis( - self.config.epoch_duration_ms * u64::from(self.config.txn_deadline_multiplier), - ), + deadline: self.request_deadline(), priority: Priority::Normal, trace_id: nodedb_types::TraceId([0u8; 16]), consistency: ReadConsistency::Strong, @@ -51,4 +47,16 @@ impl Scheduler { admission: Admission::Exempt(ExemptReason::AlreadyOrdered), } } + + /// The deadline for a Calvin sub-operation sent now: one epoch duration + /// times the configured deadline multiplier. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn request_deadline( + &self, + ) -> Instant { + // no-determinism: scheduler deadline controls waiting, not ordered state. + Instant::now() + + Duration::from_millis( + self.config.epoch_duration_ms * u64::from(self.config.txn_deadline_multiplier), + ) + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index b9656b352..2155c55bc 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -5,7 +5,7 @@ use std::collections::BTreeMap; use std::sync::{Arc, Mutex}; -use tokio::sync::mpsc; +use tokio::sync::{Notify, mpsc}; use tracing::info; use nodedb_cluster::MultiRaft; @@ -18,6 +18,7 @@ use nodedb_cluster::calvin::{ use super::super::barrier::{PendingDependentBarrier, ReadResultEvent}; use super::super::config::SchedulerConfig; use super::super::types::{BlockedTxn, PendingTxn}; +use super::deferred::DeferredQueue; use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; @@ -72,7 +73,8 @@ pub struct Scheduler { /// for the brief probe the gate takes. pub(in crate::control::cluster::calvin::scheduler::driver::core) lock_manager: Arc>, - /// In-flight static/active transactions awaiting executor response. + /// In-flight static/active transactions awaiting executor response, + /// including those whose request waits in `deferred` for capacity. /// `BTreeMap` ensures deterministic iteration order. pub(in crate::control::cluster::calvin::scheduler::driver::core) pending: BTreeMap, @@ -143,6 +145,14 @@ pub struct Scheduler { /// push, so a full/closed channel is never a correctness hazard. pub(in crate::control::cluster::calvin::scheduler::driver::core) verdict_rx: mpsc::Receiver, + /// Requests the bridge dispatcher refused at capacity, in refusal order. + /// Each txn stays in flight and holds its locks until its request is + /// re-sent. Holds at most one step per in-flight txn, plus one + /// write-version record per committed txn. + pub(in crate::control::cluster::calvin::scheduler::driver::core) deferred: DeferredQueue, + /// The bridge dispatcher's capacity-freed signal, cloned once at + /// construction. The run loop waits on it while requests are deferred. + pub(in crate::control::cluster::calvin::scheduler::driver::core) capacity_freed: Arc, } /// Parameters for [`Scheduler::new`]. @@ -205,6 +215,12 @@ impl Scheduler { let completion_cap = config.channel_capacity; let (completion_tx, completion_rx) = mpsc::channel(completion_cap); + let capacity_freed = shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .capacity_freed(); + Self { vshard_id, receiver, @@ -226,6 +242,8 @@ impl Scheduler { promotion_rx, registry, verdict_rx, + deferred: DeferredQueue::new(), + capacity_freed, } } @@ -314,7 +332,20 @@ impl Scheduler { let mut stall_tick = tokio::time::interval(self.config.verdict_stall_warn() / 4); stall_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + // Woken when a routed Data-Plane response frees dispatcher capacity. + let capacity_freed = Arc::clone(&self.capacity_freed); + loop { + // Register for the capacity wake BEFORE the re-send pass. A + // response routed after a refusal but before this point is + // covered by the pass itself. One routed after it wakes the arm. + let capacity_notified = capacity_freed.notified(); + tokio::pin!(capacity_notified); + capacity_notified.as_mut().enable(); + if self.has_deferred_dispatch() { + self.redispatch_deferred(); + } + self.check_dependent_barrier_timeouts(); self.check_awaiting_verdict_stalls(); @@ -357,6 +388,11 @@ impl Scheduler { } } + _ = &mut capacity_notified, if self.has_deferred_dispatch() => { + // Capacity freed: the next loop pass re-sends deferred + // requests in FIFO order. + } + maybe_txn = self.receiver.recv() => { match maybe_txn { Some(input) => self.process_scheduler_input(input), @@ -377,9 +413,9 @@ impl Scheduler { // O(1) common case (no pending catch-up). See `drain_catch_up`. self.drain_catch_up(); // The top-of-loop check_awaiting_verdict_stalls / - // check_dependent_barrier_timeouts do the stall work on every - // wake; this arm guarantees the loop wakes to run them (and the - // drain) when no other event arrives. + // check_dependent_barrier_timeouts and the deferred re-send + // pass run on every wake; this arm guarantees the loop wakes + // to run them (and the drain) when no other event arrives. } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs index 00ebba4fa..a15336225 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -4,18 +4,25 @@ use std::collections::{BTreeSet, HashMap}; use std::sync::{Arc, Mutex}; -use std::time::Instant; +use std::time::{Duration, Instant}; use nodedb_cluster::MultiRaft; use nodedb_cluster::RoutingTable; use nodedb_cluster::calvin::types::{ - EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet, + EngineKeySet, EngineTag, ReadKeyIdent, ReadWriteSet, SchedulerInput, SequencedTxn, SortedVec, + TxClass, VersionedReadEntry, VersionedReadSet, }; use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerStateMachine}; -use nodedb_types::TenantId; +use nodedb_physical::physical_plan::wire as plan_wire; +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; +use nodedb_types::{KeyRepr, QualifiedCollection, TenantId}; +use tokio::sync::mpsc; -use crate::bridge::dispatch::{CoreChannelDataSide, Dispatcher}; -use crate::bridge::envelope::{Payload, Response, Status}; +use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; +use crate::bridge::envelope::{ + Admission, ExemptReason, Payload, Priority, Request, Response, Status, +}; +use crate::control::cluster::calvin::scheduler::driver::barrier::ReadResultEvent; use crate::control::cluster::calvin::scheduler::driver::core::scheduler::{ Scheduler, SchedulerParams, }; @@ -23,8 +30,9 @@ use crate::control::cluster::calvin::scheduler::driver::types::{CommitState, Pen use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; use crate::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, SchedulerConfig}; +use crate::control::shutdown::ShutdownWatch; use crate::control::state::SharedState; -use crate::types::{Lsn, RequestId}; +use crate::types::{DatabaseId, Lsn, ReadConsistency, RequestId, VShardId}; use crate::wal::WalManager; /// Build a minimally-wired `Scheduler` for driver-level unit tests. The Data @@ -161,11 +169,250 @@ pub(super) fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { } } +/// The vShard that `"test_coll"` homes to in the default database. A +/// scheduler built on this vShard owns the reads of [`make_validate_only_txn`]. +pub(super) fn test_coll_vshard() -> u32 { + VShardId::from_collection_in_database(DatabaseId::DEFAULT, "test_coll").as_u32() +} + +/// Build a static `SequencedTxn` at `(epoch, position)` that reaches the +/// `CalvinExecuteStatic` stage dispatch on the [`test_coll_vshard`] scheduler. +/// +/// It carries an encoded empty plan batch and one versioned read on +/// `"test_coll"`, so that scheduler stages it as a validate-only read +/// participant. Its write set locks `"test_coll"` surrogate 1, the same key +/// as [`make_sequenced_txn`]. +pub(super) fn make_validate_only_txn(epoch: u64, position: u32) -> SequencedTxn { + let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { + collection: "test_coll".to_string(), + surrogates: SortedVec::new(vec![1]), + }]); + let plans = plan_wire::encode_batch(&Vec::new()).expect("encode empty plan batch"); + let versioned_reads = VersionedReadSet::new(vec![VersionedReadEntry { + engine: EngineTag::Document, + collection: "test_coll".to_string(), + key: ReadKeyIdent::Point(KeyRepr::Surrogate(1)), + read_lsn: Lsn::ZERO, + }]); + let tx_class = TxClass::new_single_vshard( + ReadWriteSet::new(vec![]), + write_set, + plans, + TenantId::new(1), + None, + versioned_reads, + ) + .expect("valid TxClass"); + SequencedTxn { + epoch, + position, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + } +} + +/// Build a `SequencedTxn` at `(epoch, position)` whose one write plan, a +/// truncate of `"test_coll"`, homes to [`test_coll_vshard`]. +/// +/// The plan carries no identity to bind, so it reaches the stage dispatch of +/// either path unchanged. Its write set locks the same key as +/// [`make_sequenced_txn`]. +pub(super) fn make_local_write_txn(epoch: u64, position: u32) -> SequencedTxn { + let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { + collection: "test_coll".to_string(), + surrogates: SortedVec::new(vec![1]), + }]); + let batch = vec![PhysicalPlan::Document(DocumentOp::Truncate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "test_coll"), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + })]; + let plans = plan_wire::encode_batch(&batch).expect("encode one truncate plan"); + let tx_class = TxClass::new_single_vshard( + ReadWriteSet::new(vec![]), + write_set, + plans, + TenantId::new(1), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + SequencedTxn { + epoch, + position, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + } +} + +/// Upper bound on filler dispatches. The fixture dispatcher caps a tenant at +/// 64 in-flight requests, so the cap is hit long before this bound. +const MAX_FILLERS: usize = 4096; + +/// How long a test waits for a request to reach the Data Plane side. +const DATA_PLANE_WAIT: Duration = Duration::from_secs(5); + +/// A read request for `tenant_id` that holds one in-flight slot until the +/// Data Plane answers it. +fn filler_request(request_id: RequestId, tenant_id: TenantId) -> Request { + Request { + request_id, + tenant_id, + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan: PhysicalPlan::Document(DocumentOp::PointGet { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "filler"), + document_id: "d".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }), + // no-determinism: test-only filler deadline, never Calvin WAL data. + deadline: Instant::now() + Duration::from_secs(60), + priority: Priority::Normal, + trace_id: nodedb_types::TraceId([0u8; 16]), + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + } +} + +/// Dispatch filler reads for `tenant_id` until the dispatcher refuses the +/// tenant at its in-flight cap. Returns the filler request ids. +/// +/// Each accepted filler is popped off the request ring at once, so the ring +/// and the weighted-fair queue stay empty. The only refusal left is the +/// per-tenant in-flight cap, and this function panics on any other refusal. +pub(super) fn fill_tenant_inflight( + shared: &SharedState, + data_side: &mut CoreChannelDataSide, + tenant_id: TenantId, +) -> Vec { + let mut fillers = Vec::new(); + let mut dispatcher = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); + for _ in 0..MAX_FILLERS { + let request_id = shared.next_request_id(); + match dispatcher.dispatch(filler_request(request_id, tenant_id)) { + Ok(()) => { + fillers.push(request_id); + while data_side.request_rx.try_pop().is_ok() {} + } + Err(crate::Error::DispatchCapacity { + scope: crate::DispatchCapacityScope::TenantInflight { .. }, + }) => { + assert!( + !fillers.is_empty(), + "the cap must admit at least one filler" + ); + return fillers; + } + Err(other) => panic!("unexpected filler dispatch error: {other}"), + } + } + panic!("tenant in-flight cap not reached after {MAX_FILLERS} fillers"); +} + +/// Answer one filler request on the Data Plane side and poll it back, which +/// frees one in-flight slot for its tenant. +pub(super) fn release_filler( + shared: &SharedState, + data_side: &mut CoreChannelDataSide, + request_id: RequestId, +) { + let mut response = staged_response(Status::Ok, None); + response.request_id = request_id; + data_side + .response_tx + .try_push(BridgeResponse { inner: response }) + .expect("response ring has room for one filler response"); + let polled = shared.poll_and_route_responses(); + assert!(polled >= 1, "the filler response must be polled"); +} + +/// Wait until a request whose plan satisfies `wanted` reaches the Data Plane +/// side. Returns `false` if none arrives within [`DATA_PLANE_WAIT`]. +pub(super) async fn await_data_plane_request( + data_side: &mut CoreChannelDataSide, + wanted: impl Fn(&PhysicalPlan) -> bool, +) -> bool { + let wait = async { + loop { + while let Ok(request) = data_side.request_rx.try_pop() { + if wanted(&request.inner.plan) { + return; + } + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }; + tokio::time::timeout(DATA_PLANE_WAIT, wait).await.is_ok() +} + +/// A scheduler run loop spawned on the test runtime. +/// +/// Holds every input sender so no loop channel reports closed. +pub(super) struct RunningScheduler { + shutdown: ShutdownWatch, + handle: tokio::task::JoinHandle<()>, + _input_tx: mpsc::Sender, + _read_result_tx: mpsc::Sender, + _promotion_tx: mpsc::UnboundedSender>, +} + +impl RunningScheduler { + /// Signal shutdown and wait for the loop to exit. + pub(super) async fn stop(self) { + self.shutdown.signal(); + tokio::time::timeout(DATA_PLANE_WAIT, self.handle) + .await + .expect("scheduler loop exits after shutdown") + .expect("scheduler loop does not panic"); + } +} + +/// Spawn `scheduler`'s run loop with open input channels and a short +/// liveness tick. +pub(super) fn spawn_scheduler_loop(mut scheduler: Scheduler) -> RunningScheduler { + let (input_tx, input_rx) = mpsc::channel(16); + let (read_result_tx, read_result_rx) = mpsc::channel(16); + let (promotion_tx, promotion_rx) = mpsc::unbounded_channel(); + scheduler.receiver = input_rx; + scheduler.read_result_rx = read_result_rx; + scheduler.promotion_rx = promotion_rx; + // The loop's liveness tick fires every quarter of this interval. + scheduler.config.verdict_stall_warn_ms = 200; + let shutdown = ShutdownWatch::new(); + let receiver = shutdown.subscribe(); + let handle = tokio::spawn(scheduler.run(receiver)); + RunningScheduler { + shutdown, + handle, + _input_tx: input_tx, + _read_result_tx: read_result_tx, + _promotion_tx: promotion_tx, + } +} + /// A `PendingTxn` staged and parked awaiting the cross-shard commit verdict. pub(super) fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { PendingTxn { txn, lock_owner: txn_id, + // no-determinism: test-only dispatch timestamp for a fabricated PendingTxn fixture. dispatch_time: Instant::now(), has_primary_write: true, has_returning: false, diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs index 497cc8111..42b314570 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs @@ -12,8 +12,7 @@ //! write-version recorder at that LSN — the same shard-local WAL-LSN space the //! single-shard fast path and read watermarks use. -use std::sync::atomic::Ordering; - +use super::deferred::{DispatchOutcome, DispatchStep}; use super::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::types::Lsn; @@ -35,11 +34,12 @@ impl Scheduler { /// Fire-and-forget: the recorded version is not needed to complete the /// transaction, so the response is drained and discarded. A brief index-lag /// window before the record op lands is harmless — nothing enforces read-set - /// validation against these versions yet. A dropped record (decode failure, - /// no local write plan, or dispatch backpressure) simply leaves the version + /// validation against these versions yet. A record refused at capacity is + /// parked and re-sent once capacity frees, never dropped. A decode or + /// routing failure, or a terminal dispatch refusal, leaves the version /// unrecorded and never blocks the commit. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn record_calvin_write_versions( - &self, + &mut self, txn_id: TxnId, applied_lsn: Lsn, ) { @@ -96,32 +96,65 @@ impl Scheduler { let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, Some(applied_lsn)); - // Register so the response routes to a real receiver (not the - // unknown-request warning path), then discard it — the recording is - // one-way. - let resp_rx = self.shared.tracker.register(request_id); - let dispatch_result = match self.shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - if let Err(e) = dispatch_result { - self.shared.tracker.cancel(&request_id); - tracing::warn!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin: write-version record dispatch failed" - ); - return; + // The request carries everything a re-send needs, so a parked record + // outlives the txn's `pending` entry. + if let DispatchOutcome::Failed(error) = + self.dispatch_sequenced(txn_id, DispatchStep::WriteVersionRecord, request) + { + self.fail_dispatch_step(txn_id, DispatchStep::WriteVersionRecord, error); } - tokio::spawn(async move { - let mut rx = resp_rx; - let _ = rx.recv().await; - }); - self.shared - .calvin_counters - .write_versions_recorded - .fetch_add(1, Ordering::Relaxed); + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_types::TenantId; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + await_data_plane_request, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_validate_only_txn, release_filler, spawn_scheduler_loop, staged_pending, + test_coll_vshard, + }; + + /// Once a Data Plane response frees tenant capacity, a write-version + /// record refused at capacity reaches the Data Plane. + #[tokio::test] + async fn refused_write_version_record_reaches_data_plane_after_capacity_frees() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let txn_id = TxnId::new(21, 0); + scheduler.pending.insert( + txn_id, + staged_pending(make_validate_only_txn(21, 0), txn_id), + ); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + + scheduler.record_calvin_write_versions(txn_id, Lsn::new(42)); + let running = spawn_scheduler_loop(scheduler); + release_filler(&shared, &mut data_side, fillers[0]); + + let arrived = await_data_plane_request(&mut data_side, |plan| { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::RecordCalvinWriteVersions { + epoch: 21, + position: 0, + .. + }) + ) + }) + .await; + running.stop().await; + + assert!( + arrived, + "the refused write-version record must reach the Data Plane once capacity frees" + ); } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs index 9af0d5bef..64c65d63d 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs @@ -23,7 +23,8 @@ pub(super) struct PendingTxn { /// reservation owns the lock). Used by `on_txn_complete` to `release` the /// correct lock-manager identity. pub lock_owner: TxnId, - /// Wall-clock time at dispatch (for lock-wait latency metrics). + /// Wall-clock time of the last stage dispatch attempt (for executor + /// latency metrics). A re-send after a capacity refusal resets it. /// /// `Instant::now()` is used here for observability only; never /// influences WAL bytes. @@ -47,8 +48,8 @@ pub(super) struct PendingTxn { pub change_sets: Vec, /// Commit-resolution state for a static-set Calvin txn. /// - /// `Some(CommitState::Staged)` for a static txn dispatched via the - /// validate-and-stage path: its first executor response carries the local + /// `Some(CommitState::Staged)` for a txn dispatched, or parked for re-send, + /// via the validate-and-stage path: its first executor response carries the local /// commit vote and drives a flush-or-drop before the commit tail runs. /// `None` for dependent/active txns, which apply directly. pub commit_state: Option, @@ -82,12 +83,14 @@ pub(in crate::control::cluster::calvin::scheduler::driver) enum CommitState { /// vote, so a torn commit (one shard flushes while a peer drops) is /// impossible. AwaitingVerdict, - /// The txn committed and a `MetaOp::CalvinResolve` has been dispatched to - /// resolve its staged post-images into a replayable `RedoRecord`; awaiting - /// that response before the redo is WAL-appended and the flush dispatched. + /// The txn committed and a `MetaOp::CalvinResolve` has been dispatched, or + /// parked for re-send at dispatcher capacity, to resolve its staged + /// post-images into a replayable `RedoRecord`; awaiting that response + /// before the redo is WAL-appended and the flush dispatched. AwaitingRedoResolve, /// A flush (`committed = true`) or drop (`committed = false`) has been - /// dispatched; awaiting its response before the commit tail runs. + /// dispatched, or parked for re-send at dispatcher capacity; awaiting its + /// response before the commit tail runs. /// /// `redo_lsn` is `Some(lsn)` when a `TransactionRedo` record was appended /// for this commit's non-empty write set — `commit_apply_tail` then only diff --git a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs index d8ac0d837..46d50c57b 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs @@ -58,6 +58,11 @@ pub struct SchedulerMetrics { /// flags that catch-up is relying on snapshot coverage rather than log replay /// and warrants operator attention. pub catch_up_log_compacted: AtomicU64, + /// Scheduler dispatches the bridge dispatcher refused at capacity. Each + /// refusal parks the request for re-send; none is an abort. + pub dispatch_deferred_count: AtomicU64, + /// Requests parked for re-send right now, waiting for dispatcher capacity. + pub dispatch_deferred_depth: AtomicU64, } /// Reason codes for `nodedb_calvin_infra_abort_total`. @@ -135,6 +140,17 @@ impl SchedulerMetrics { self.catch_up_log_compacted.fetch_add(1, Ordering::Relaxed); } + /// Record that the dispatcher refused a scheduler dispatch at capacity. + pub fn record_dispatch_deferred(&self) { + self.dispatch_deferred_count.fetch_add(1, Ordering::Relaxed); + } + + /// Set the number of requests parked for re-send. + pub fn set_dispatch_deferred_depth(&self, depth: usize) { + self.dispatch_deferred_depth + .store(depth as u64, Ordering::Relaxed); + } + /// Record the end-to-end executor txn duration (dispatch → response). /// /// Increments the appropriate histogram bucket and the running sum. @@ -290,6 +306,30 @@ impl SchedulerMetrics { self.catch_up_log_compacted.load(Ordering::Relaxed) ); + let _ = writeln!( + out, + "# HELP nodedb_calvin_dispatch_deferred_total \ + Scheduler dispatches refused at dispatcher capacity and parked for re-send." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_dispatch_deferred_total counter"); + let _ = writeln!( + out, + "nodedb_calvin_dispatch_deferred_total{{{label}}} {}", + self.dispatch_deferred_count.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_dispatch_deferred_depth \ + Scheduler requests parked for re-send, waiting for dispatcher capacity." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_dispatch_deferred_depth gauge"); + let _ = writeln!( + out, + "nodedb_calvin_dispatch_deferred_depth{{{label}}} {}", + self.dispatch_deferred_depth.load(Ordering::Relaxed) + ); + out } } @@ -309,6 +349,8 @@ impl Default for SchedulerMetrics { verdict_stall_count: AtomicU64::new(0), catch_up_replayed: AtomicU64::new(0), catch_up_log_compacted: AtomicU64::new(0), + dispatch_deferred_count: AtomicU64::new(0), + dispatch_deferred_depth: AtomicU64::new(0), } } } diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 2ad97269b..85b577301 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -42,6 +42,13 @@ pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { constraint, detail, }, + // A capacity refusal crosses as its own verdict, so the coordinator + // answers the retryable overload class. + capacity @ crate::Error::DispatchCapacity { .. } => TypedClusterError::DataPlane { + code: DataPlaneErrorCode::DispatchCapacity { + reason: capacity.to_string(), + }, + }, other => { let message = other.to_string(); let code = u32::from(nodedb_types::error::NodeDbError::from(other).code().0); @@ -148,6 +155,7 @@ impl From for DataPlaneErrorCode { limit: to_wire_count(limit), }, ErrorCode::DivisionByZero => Self::DivisionByZero, + ErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, } } } @@ -249,6 +257,7 @@ impl From for ErrorCode { } } DataPlaneErrorCode::DivisionByZero => Self::DivisionByZero, + DataPlaneErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, } } } @@ -300,6 +309,21 @@ mod tests { } } + #[test] + fn dispatch_capacity_code_roundtrips_verbatim() { + let original = ErrorCode::DispatchCapacity { + reason: "tenant 1 holds 64/64 in-flight requests".into(), + }; + let wire = DataPlaneErrorCode::from(original.clone()); + assert_eq!( + wire, + DataPlaneErrorCode::DispatchCapacity { + reason: "tenant 1 holds 64/64 in-flight requests".into(), + } + ); + assert_eq!(ErrorCode::from(wire), original); + } + #[test] fn counted_code_roundtrips_across_the_u64_wire_field() { let original = ErrorCode::RecursionDepthExceeded { diff --git a/nodedb/src/control/gateway/error_map/http.rs b/nodedb/src/control/gateway/error_map/http.rs index 281e31435..231f3375a 100644 --- a/nodedb/src/control/gateway/error_map/http.rs +++ b/nodedb/src/control/gateway/error_map/http.rs @@ -13,7 +13,7 @@ impl GatewayErrorMap { /// - 400 Bad Request for client-side errors (bad SQL, not found) /// - 403 Forbidden for authz errors /// - 409 Conflict for write-conflict / constraint violations - /// - 503 Service Unavailable for routing/leader errors + /// - 503 Service Unavailable for routing/leader errors and dispatch overload /// - 504 Gateway Timeout for deadline exceeded /// - 500 Internal Server Error as the default fallback pub fn to_http(err: &Error) -> (u16, String) { @@ -32,6 +32,7 @@ impl GatewayErrorMap { Error::PlanError { detail } => (400, detail.clone()), Error::RejectedConstraint { detail, .. } => (409, detail.clone()), Error::NoLeader { .. } => (503, err.to_string()), + Error::DispatchCapacity { .. } => (503, err.to_string()), Error::Serialization { .. } | Error::Codec { .. } => (500, err.to_string()), Error::Internal { .. } => (500, err.to_string()), // 501 Not Implemented: a valid op refused because cross-core diff --git a/nodedb/src/control/gateway/error_map/pgwire.rs b/nodedb/src/control/gateway/error_map/pgwire.rs index 6c69cb6dc..cbae262d0 100644 --- a/nodedb/src/control/gateway/error_map/pgwire.rs +++ b/nodedb/src/control/gateway/error_map/pgwire.rs @@ -159,6 +159,12 @@ mod tests { actual: 2, }, Error::BackupKeyMismatch, + Error::DispatchCapacity { + scope: crate::DispatchCapacityScope::QueueFull { + core_id: 0, + capacity: 4, + }, + }, ]; for err in samples { diff --git a/nodedb/src/control/gateway/error_map/remote_code.rs b/nodedb/src/control/gateway/error_map/remote_code.rs index 90e6a5987..6deb5287b 100644 --- a/nodedb/src/control/gateway/error_map/remote_code.rs +++ b/nodedb/src/control/gateway/error_map/remote_code.rs @@ -11,7 +11,7 @@ pub(super) fn remote_code_to_http_status(code: nodedb_types::error::ErrorCode) -> u16 { use nodedb_types::error::ErrorCode as Ec; match code { - Ec::NOT_LEADER | Ec::NO_LEADER => 503, + Ec::NOT_LEADER | Ec::NO_LEADER | Ec::SERVER_OVERLOAD => 503, Ec::DEADLINE_EXCEEDED => 504, Ec::COLLECTION_NOT_FOUND => 404, Ec::AUTHORIZATION_DENIED => 403, @@ -31,6 +31,7 @@ pub(super) fn remote_code_to_resp_prefix(code: nodedb_types::error::ErrorCode) - Ec::AUTHORIZATION_DENIED => "NOPERM", Ec::CONSTRAINT_VIOLATION => "CONSTRAINT", Ec::TYPE_MISMATCH => "WRONGTYPE", + Ec::SERVER_OVERLOAD => "BUSY", _ => "ERR", } } diff --git a/nodedb/src/control/gateway/error_map/resp.rs b/nodedb/src/control/gateway/error_map/resp.rs index 0e30c1a0b..1bb64df9d 100644 --- a/nodedb/src/control/gateway/error_map/resp.rs +++ b/nodedb/src/control/gateway/error_map/resp.rs @@ -27,6 +27,7 @@ impl GatewayErrorMap { Error::RejectedConstraint { detail, .. } => format!("CONSTRAINT {detail}"), Error::TypeMismatch { detail, .. } => format!("WRONGTYPE {detail}"), Error::RetryableSchemaChanged { .. } => format!("ERR {err}"), + Error::DispatchCapacity { .. } => format!("BUSY {err}"), Error::RemoteTyped { code, message } => { format!("{} {message}", remote_code_to_resp_prefix(*code)) } diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index 373523bef..353092846 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -131,6 +131,7 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { // Admission verdicts: the request never reached the mutation at all. | ErrorCode::RateExceeded { .. } | ErrorCode::CollectionDraining { .. } + | ErrorCode::DispatchCapacity { .. } | ErrorCode::Unsupported { .. } // The target row or collection did not exist, so the write had nothing // to mutate. @@ -205,6 +206,14 @@ mod tests { )); } + /// A dispatcher capacity refusal enqueued nothing, so the record aborts. + #[test] + fn dispatch_capacity_refusal_aborts_the_record() { + assert!(write_definitely_not_applied(&ErrorCode::DispatchCapacity { + reason: "core 0 queue is full at 64 requests".into(), + })); + } + /// The asymmetry that keeps this safe: an ambiguous outcome must never /// produce an abort, because the write it would erase may have landed. #[test] diff --git a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs index 8a32fbc2d..5ed040d5d 100644 --- a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs @@ -69,9 +69,14 @@ pub(super) async fn dispatch_authorized_single_task( .execute(&query, checked) .await .map(gateway_payloads_to_response) - .map_err(|error| { - let (_, detail) = GatewayErrorMap::to_native(&error); - crate::Error::Dispatch { detail } + .map_err(|error| match error { + // A capacity refusal keeps its type so the client sees + // the retryable overload class. + capacity @ crate::Error::DispatchCapacity { .. } => capacity, + other => { + let (_, detail) = GatewayErrorMap::to_native(&other); + crate::Error::Dispatch { detail } + } }) } None => dispatch_without_gateway(ctx, checked).await, diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index fbd48e481..fb0793152 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -160,6 +160,11 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::RateExceeded { .. } => { ("ERROR", sqlstate::TOO_MANY_CONNECTIONS, err.to_string()) } + // A dispatcher capacity refusal enqueued nothing. SERVER_OVERLOAD + // (57P03) is transient: the client retries after a backoff. + crate::Error::DispatchCapacity { .. } => { + ("ERROR", sqlstate::SERVER_OVERLOAD, err.to_string()) + } crate::Error::MemoryExhausted { .. } => ("ERROR", sqlstate::OUT_OF_MEMORY, err.to_string()), crate::Error::Backpressure { .. } => ("ERROR", sqlstate::OUT_OF_MEMORY, err.to_string()), crate::Error::FanOutExceeded { .. } => { @@ -267,6 +272,8 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> Ec::RATE_EXCEEDED => sqlstate::TOO_MANY_CONNECTIONS, // Mirrors the `MemoryExhausted` / `Backpressure` arms. Ec::MEMORY_EXHAUSTED => sqlstate::OUT_OF_MEMORY, + // Mirrors the `DispatchCapacity` arm. + Ec::SERVER_OVERLOAD => sqlstate::SERVER_OVERLOAD, // Mirrors the `NoLeader` arm. Ec::NO_LEADER => sqlstate::LOCK_NOT_AVAILABLE, // Mirrors the `NotLeader` arm. diff --git a/nodedb/src/control/server/resp/gateway_dispatch.rs b/nodedb/src/control/server/resp/gateway_dispatch.rs index cc33c82a5..e930ba858 100644 --- a/nodedb/src/control/server/resp/gateway_dispatch.rs +++ b/nodedb/src/control/server/resp/gateway_dispatch.rs @@ -373,7 +373,9 @@ fn gateway_payloads_to_response(payloads: Vec>) -> Response { /// which Redis clients handle with automatic retry (same as Redis Cluster BUSY). fn map_busy_error(e: crate::Error) -> crate::Error { match &e { - crate::Error::Bridge { .. } | crate::Error::Dispatch { .. } => crate::Error::Bridge { + crate::Error::Bridge { .. } + | crate::Error::Dispatch { .. } + | crate::Error::DispatchCapacity { .. } => crate::Error::Bridge { detail: "BUSY NodeDB is processing requests, retry later".into(), }, _ => e, diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs index 1e627cfdd..bf29d2128 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs @@ -24,6 +24,7 @@ //! vShard set records both homes so ROLLBACK tears down both. use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::pgwire::types::error_to_sqlstate; use crate::control::server::shared::session::DmlTxnCtx; use crate::control::server::shared::session::staging_gate::{ InTxnRoute, StagingGateError, route_in_tx_write, @@ -155,7 +156,10 @@ pub(super) async fn stage_edge_write_in_txn( // caller that already checked `InBlock`; there is no affected count to // report for either, so treat them as a no-op tag rather than panicking. Ok(InTxnRoute::Read(_)) | Ok(InTxnRoute::Buffered) => Ok(0), - Err(StagingGateError::Dispatch(e)) => Err(ddl_err("XX000", e.to_string())), + Err(StagingGateError::Dispatch(e)) => { + let (_, sqlstate, message) = error_to_sqlstate(&e); + Err(ddl_err(sqlstate, message)) + } Err(StagingGateError::Rejected { code }) => { let (_, sqlstate, message) = match code { Some(code) => { diff --git a/nodedb/src/control/server/shared/ddl/neutral/topic/publish.rs b/nodedb/src/control/server/shared/ddl/neutral/topic/publish.rs index 289efcc79..dc6cd4f27 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/topic/publish.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/topic/publish.rs @@ -37,6 +37,9 @@ pub async fn handle_publish( crate::Error::CollectionNotFound { .. } => "42704", crate::Error::BadRequest { .. } => "42601", crate::Error::Dispatch { .. } => "58000", + crate::Error::DispatchCapacity { .. } => { + nodedb_types::error::sqlstate::SERVER_OVERLOAD + } _ => "XX000", }; Err(DdlError::new(sqlstate.to_string(), e.to_string())) diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index 4a3a4c0da..a9e2e540f 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -187,6 +187,10 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, sqlstate::DIVISION_BY_ZERO, "division by zero".into(), ), + // Transient: the client retries after a backoff. + ErrorCode::DispatchCapacity { reason } => { + ("ERROR", sqlstate::SERVER_OVERLOAD, reason.clone()) + } ErrorCode::Unsupported { detail } => { ("ERROR", sqlstate::FEATURE_NOT_SUPPORTED, detail.clone()) } diff --git a/nodedb/src/control/server/sync/refusal.rs b/nodedb/src/control/server/sync/refusal.rs index b009280e7..72be91880 100644 --- a/nodedb/src/control/server/sync/refusal.rs +++ b/nodedb/src/control/server/sync/refusal.rs @@ -32,9 +32,10 @@ pub(super) fn retryable_refusal_reason(error: &crate::Error) -> Option<&str> { /// refused on its merits. /// /// These are the failures where the cluster never judged the write at all — it -/// timed out, the leader moved, the sequencer was absent, or memory pressure -/// shed it. Nothing about the write itself is wrong, so the same bytes are -/// expected to land once the condition clears. +/// timed out, the leader moved, the sequencer was absent, memory pressure +/// shed it, or the dispatcher refused it at capacity. Nothing about the write +/// itself is wrong, so the same bytes are expected to land once the condition +/// clears. fn is_indeterminate(error: &crate::Error) -> bool { use crate::bridge::envelope::ErrorCode; matches!( @@ -46,10 +47,12 @@ fn is_indeterminate(error: &crate::Error) -> bool { | crate::Error::StaleReadNotLeader { .. } | crate::Error::SequencerUnavailable | crate::Error::Backpressure { .. } + | crate::Error::DispatchCapacity { .. } | crate::Error::ConflictRetry { .. } | crate::Error::DataPlane( ErrorCode::DeadlineExceeded | ErrorCode::ResourcesExhausted + | ErrorCode::DispatchCapacity { .. } | ErrorCode::ConflictRetry ) ) @@ -168,6 +171,21 @@ mod tests { ); } + #[test] + fn a_dispatch_refused_at_capacity_is_retryable_not_a_refusal_of_the_write() { + let error = crate::Error::DispatchCapacity { + scope: crate::DispatchCapacityScope::TenantInflight { + tenant_id: crate::types::TenantId::new(1), + inflight: 64, + cap: 64, + }, + }; + assert_eq!( + ack_status_for_dispatch_error(&error, 4), + AckStatus::Gap { expected: 4 } + ); + } + #[test] fn a_retryable_status_carries_no_reject_reason() { // A reason beside a retryable status reads as "give up" to a receiver diff --git a/nodedb/src/error/dispatch_capacity.rs b/nodedb/src/error/dispatch_capacity.rs new file mode 100644 index 000000000..5042b17f6 --- /dev/null +++ b/nodedb/src/error/dispatch_capacity.rs @@ -0,0 +1,66 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which bridge-dispatcher capacity limit refused a request. +//! +//! Carried by [`Error::DispatchCapacity`](super::types::Error::DispatchCapacity). +//! Each scope is transient: the dispatcher enqueued nothing, and the same +//! request succeeds once in-flight responses free capacity. + +use std::fmt; + +use crate::types::{DatabaseId, TenantId}; + +/// The capacity limit that refused a dispatch, with the facts of the refusal. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DispatchCapacityScope { + /// The tenant already holds its cap of in-flight requests. + TenantInflight { + /// The refused tenant. + tenant_id: TenantId, + /// The tenant's in-flight requests at the refusal. + inflight: u32, + /// The per-tenant in-flight cap. + cap: u32, + }, + /// The database's virtual queue on the target core is suspended at its + /// fair share. + DatabaseSuspended { + /// The refused database. + database_id: DatabaseId, + /// The target core. + core_id: usize, + }, + /// The target core's weighted-fair queue is full. + QueueFull { + /// The target core. + core_id: usize, + /// The queue's total capacity in requests. + capacity: u32, + }, +} + +impl fmt::Display for DispatchCapacityScope { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::TenantInflight { + tenant_id, + inflight, + cap, + } => write!( + f, + "tenant {tenant_id} holds {inflight}/{cap} in-flight requests" + ), + Self::DatabaseSuspended { + database_id, + core_id, + } => write!( + f, + "database {database_id} virtual queue is suspended at its fair share on core \ + {core_id}" + ), + Self::QueueFull { core_id, capacity } => { + write!(f, "core {core_id} queue is full at {capacity} requests") + } + } + } +} diff --git a/nodedb/src/error/mod.rs b/nodedb/src/error/mod.rs index 1ae270477..2fa48864e 100644 --- a/nodedb/src/error/mod.rs +++ b/nodedb/src/error/mod.rs @@ -6,6 +6,7 @@ //! every subsystem (write path, read path, routing, client input, //! infrastructure) — plus the `Result` alias built on it. `conversions` //! owns `From` impls that turn external-crate error types into `Error`. +//! `dispatch_capacity` and `ollp` own payload types carried by variants. //! //! Conversions into the crate's *public* error type (`NodeDbError`) and //! cluster wire-error conversions live in `crate::error_from` rather than @@ -16,8 +17,10 @@ //! `&Error` classify through the same table. mod conversions; +mod dispatch_capacity; mod ollp; mod types; +pub use dispatch_capacity::DispatchCapacityScope; pub use ollp::OllpExhaustedCause; pub use types::{Error, Result}; diff --git a/nodedb/src/error/types.rs b/nodedb/src/error/types.rs index 2d385b35b..45f820daf 100644 --- a/nodedb/src/error/types.rs +++ b/nodedb/src/error/types.rs @@ -378,6 +378,13 @@ pub enum Error { #[error("dispatch error: {detail}")] Dispatch { detail: String }, + /// The bridge dispatcher refused a request at a capacity limit. Nothing + /// was enqueued, so the caller retries once capacity frees. + #[error("dispatch refused at capacity: {scope}; the request was not enqueued and is retryable")] + DispatchCapacity { + scope: super::dispatch_capacity::DispatchCapacityScope, + }, + #[error("storage error ({engine}): {detail}")] Storage { engine: String, detail: String }, diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index 11d5fe5f5..66ddc30ab 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -196,6 +196,8 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { Error::Wal(wal_err) => NodeDbError::wal(wal_err), Error::Dispatch { detail } => NodeDbError::dispatch(detail), + // A capacity refusal enqueued nothing, so it is the retryable overload class. + Error::DispatchCapacity { .. } => NodeDbError::server_overload(e), Error::Storage { detail, .. } => NodeDbError::storage(detail), Error::ColdStorage { detail } => NodeDbError::cold_storage(detail), Error::Serialization { format, detail } => { diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index 46ac98e1f..dfeab5e45 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -130,6 +130,9 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { ErrorCode::UndefinedColumn { column } => NodeDbError::undefined_column(column), ErrorCode::Unsupported { detail } => NodeDbError::bad_request(detail), ErrorCode::DivisionByZero => NodeDbError::division_by_zero(), + // Nothing was enqueued, and the same request succeeds once capacity + // frees: the retryable overload class. + ErrorCode::DispatchCapacity { reason } => NodeDbError::server_overload(reason), ErrorCode::TxnOverlayMemoryExceeded { limit } => NodeDbError::bad_request(format!( "transaction staging overlay exceeded its {limit}-byte per-core budget; \ split the transaction into smaller batches" diff --git a/nodedb/src/lib.rs b/nodedb/src/lib.rs index 1ccfa4f78..abc744536 100644 --- a/nodedb/src/lib.rs +++ b/nodedb/src/lib.rs @@ -39,6 +39,6 @@ pub mod version; pub mod wal; pub use config::{EngineConfig, ServerConfig}; -pub use error::{Error, OllpExhaustedCause, Result}; +pub use error::{DispatchCapacityScope, Error, OllpExhaustedCause, Result}; pub use nodedb_types::error::{ErrorCode, NodeDbError, NodeDbResult}; pub use types::{DocumentId, Lsn, ReadConsistency, RequestId, TenantId, VShardId}; From dbdf41cbad17c37f8e8bdc85b42a9557ecb67fbf Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 13:02:01 +0800 Subject: [PATCH 03/64] refactor(bridge): name the per-core dispatch queue capacity Replace the magic 1024 literal used at every Dispatcher::new call site with a documented DATA_PLANE_QUEUE_CAPACITY constant, so downstream code (scheduler backpressure tuning) can reference the same bound. --- .../src/cluster_harness/node/lifecycle/spawn_full.rs | 4 ++-- nodedb/src/bridge/dispatch/dispatcher.rs | 6 ++++++ nodedb/src/bridge/dispatch/mod.rs | 3 ++- nodedb/src/main_boot/data_plane.rs | 4 ++-- 4 files changed, 12 insertions(+), 5 deletions(-) diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs index 64cb89d60..06f6d6e6a 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs @@ -11,7 +11,7 @@ use std::path::PathBuf; use std::sync::Arc; use std::time::Duration; -use nodedb::bridge::dispatch::Dispatcher; +use nodedb::bridge::dispatch::{DATA_PLANE_QUEUE_CAPACITY, Dispatcher}; use nodedb::config::auth::AuthMode; use nodedb::config::server::ClusterSettings; use nodedb::control::server::pgwire::listener::PgListener; @@ -101,7 +101,7 @@ impl TestClusterNode { )?); let wal_records: Arc<[nodedb_wal::WalRecord]> = Arc::from(wal.replay()?.into_boxed_slice()); let replay_tombstones = nodedb_wal::extract_tombstones(&wal_records).unwrap(); - let (dispatcher, data_sides) = Dispatcher::new(num_cores, 1024); + let (dispatcher, data_sides) = Dispatcher::new(num_cores, DATA_PLANE_QUEUE_CAPACITY); let (event_producers, event_consumers) = create_event_bus(num_cores); // Credential store backed by the system catalog — required for diff --git a/nodedb/src/bridge/dispatch/dispatcher.rs b/nodedb/src/bridge/dispatch/dispatcher.rs index 728d0ca8b..44b1766b7 100644 --- a/nodedb/src/bridge/dispatch/dispatcher.rs +++ b/nodedb/src/bridge/dispatch/dispatcher.rs @@ -21,6 +21,12 @@ use crate::data::eventfd::EventFdNotifier; use super::core_channel::{CoreChannel, CoreChannelDataSide}; +/// Per-core request queue capacity of the server's bridge dispatcher. +/// +/// Each core's weighted-fair queue and SPSC rings hold at most this many +/// requests. Every request for one vShard routes to the same core. +pub const DATA_PLANE_QUEUE_CAPACITY: usize = 1024; + /// Serialized form of a request that goes through the SPSC ring buffer. /// /// The bridge crate is generic over `T` — we serialize our typed `Request` diff --git a/nodedb/src/bridge/dispatch/mod.rs b/nodedb/src/bridge/dispatch/mod.rs index f7f192227..ee2bfb1cc 100644 --- a/nodedb/src/bridge/dispatch/mod.rs +++ b/nodedb/src/bridge/dispatch/mod.rs @@ -11,7 +11,8 @@ mod test_requests; pub use core_channel::{CoreChannel, CoreChannelDataSide}; pub use dispatcher::{ - BridgeRequest, BridgeResponse, DatabasePriorityResolver, DefaultPriorityResolver, Dispatcher, + BridgeRequest, BridgeResponse, DATA_PLANE_QUEUE_CAPACITY, DatabasePriorityResolver, + DefaultPriorityResolver, Dispatcher, }; pub use drain::CorePending; pub use refusal::DispatchRefusal; diff --git a/nodedb/src/main_boot/data_plane.rs b/nodedb/src/main_boot/data_plane.rs index 1540aca89..77b5db3a5 100644 --- a/nodedb/src/main_boot/data_plane.rs +++ b/nodedb/src/main_boot/data_plane.rs @@ -8,7 +8,7 @@ use std::sync::Arc; use nodedb::ServerConfig; use nodedb::bootstrap; -use nodedb::bridge::dispatch::Dispatcher; +use nodedb::bridge::dispatch::{DATA_PLANE_QUEUE_CAPACITY, Dispatcher}; /// Everything downstream boot phases need from Data Plane bootstrap, /// bundled so the call site doesn't juggle 15 separate `let`s. @@ -55,7 +55,7 @@ pub(crate) async fn bootstrap_data_plane( // Create SPSC bridge: Dispatcher (Control Plane) + CoreChannelDataSide (Data Plane). let num_cores = config.server.data_plane_cores; - let (mut dispatcher, data_sides) = Dispatcher::new(num_cores, 1024); + let (mut dispatcher, data_sides) = Dispatcher::new(num_cores, DATA_PLANE_QUEUE_CAPACITY); // Create Event Bus: per-core ring buffers (Data Plane → Event Plane). let (event_producers, event_consumers) = nodedb::event::bus::create_event_bus(num_cores); From 69d0ed433b7bbb3593aa098a6b8151060e1621df Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 13:02:12 +0800 Subject: [PATCH 04/64] feat(scheduler): gate Calvin intake on dispatch and backlog capacity Stop the run loop from reading new sequenced input once a dispatch is deferred at capacity, or once the in-flight backlog (pending, blocked, dependent-barrier txns) sits at the dispatcher queue bound while some of it can only drain on executor responses. A backlog of blocked txns alone keeps intake open, since the reservation release they wait on arrives as input. Cap each catch-up drain to a bounded window of committed sequencer log entries instead of the whole armed range, and resume from the first unreplayed index on the next drain, so a closed intake gate cannot make one drain read unbounded history. Expose the gate state, backlog depth, and closure reasons as scheduler metrics. --- .../cluster/calvin/scheduler/driver/config.rs | 24 +- .../calvin/scheduler/driver/core/catch_up.rs | 139 +++++++-- .../calvin/scheduler/driver/core/intake.rs | 285 ++++++++++++++++++ .../calvin/scheduler/driver/core/mod.rs | 3 + .../calvin/scheduler/driver/core/scheduler.rs | 20 +- .../scheduler/driver/core/test_support.rs | 9 +- .../cluster/calvin/scheduler/metrics.rs | 77 +++++ 7 files changed, 531 insertions(+), 26 deletions(-) create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/config.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/config.rs index 3b28a1874..a71084d62 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/config.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/config.rs @@ -4,6 +4,8 @@ use std::time::Duration; +use crate::bridge::dispatch::DATA_PLANE_QUEUE_CAPACITY; + /// Tuning parameters for a [`super::core::Scheduler`] instance. #[derive(Debug, Clone)] pub struct SchedulerConfig { @@ -29,17 +31,37 @@ pub struct SchedulerConfig { /// /// Default: `epoch_duration_ms * 250` milliseconds. pub verdict_stall_warn_ms: u64, + /// In-flight backlog at which the scheduler stops taking new sequenced + /// input. The backlog counts pending, blocked, and dependent-barrier txns. + /// + /// The bound applies only while some backlog txn progresses without new + /// input. A backlog of blocked txns alone keeps intake open, because a + /// reservation release they wait on arrives as input. + /// + /// Default: [`DATA_PLANE_QUEUE_CAPACITY`]. Every dispatch of one vShard + /// goes to one Data Plane core, whose queue holds that many requests. A + /// larger backlog cannot hold a dispatch slot per txn at once. + pub max_inflight_backlog: usize, + /// Most sequencer log entries one catch-up drain reads and replays. + /// + /// Default: `channel_capacity`. A drop happens only when the fan-out + /// channel is full, so one window covers about one channel of missed + /// entries. Windows run back to back while intake is open. + pub catch_up_window: u64, } impl Default for SchedulerConfig { fn default() -> Self { let epoch_duration_ms = 20u64; + let channel_capacity = 512usize; Self { - channel_capacity: 512, + channel_capacity, txn_deadline_multiplier: 3, epoch_duration_ms, dependent_read_passive_timeout_ms: epoch_duration_ms * 3, verdict_stall_warn_ms: epoch_duration_ms * 250, + max_inflight_backlog: DATA_PLANE_QUEUE_CAPACITY, + catch_up_window: channel_capacity as u64, } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs index 81d0811e9..3c50aac69 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs @@ -16,18 +16,39 @@ //! and thereby reconstructs the missed input deterministically. Replay is //! idempotent — `process_new_txn`'s in-flight guard turns an already-in-flight //! Txn into a no-op, and Reserve/Release re-application is a lock-manager no-op. +//! +//! One drain reads at most [`SchedulerConfig::catch_up_window`] log entries. +//! +//! [`SchedulerConfig::catch_up_window`]: super::super::config::SchedulerConfig::catch_up_window use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; use nodedb_cluster::calvin::types::SchedulerInput; use super::scheduler::Scheduler; +/// Result of one [`Scheduler::drain_catch_up`] call. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum CatchUpDrain { + /// Nothing is left to replay now. No catch-up is armed, nothing is + /// committed at the armed index yet, the log read failed and the entry + /// stays armed for a later tick, or the armed range is fully replayed. + Settled, + /// The drain stopped before the committed index: the window filled or + /// the intake gate closed. Catch-up stays armed at the first unprocessed + /// index, and the next drain can resume at once. + Remaining, +} + impl Scheduler { /// Replay any sequencer-fan-out inputs dropped on this replica. /// /// Run on the periodic stall tick. O(1) in the common case (no pending /// catch-up → one map probe and return). /// + /// Reads and replays at most `catch_up_window` log entries from the armed + /// index. Stops feeding at the first input after which the intake gate is + /// closed. + /// /// # Lock discipline (deadlock-safety) /// /// The two shared mutexes — the sequencer state machine and MultiRaft — are @@ -35,7 +56,9 @@ impl Scheduler { /// holds the SM lock while fanning out but never takes MultiRaft underneath /// it; this drain takes them strictly one-at-a-time (SM → release → MultiRaft /// → release → SM → release), so the two paths can never form a lock cycle. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn drain_catch_up(&mut self) { + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn drain_catch_up( + &mut self, + ) -> CatchUpDrain { // 1. SM-lock scope: PEEK the earliest armed index for this vShard. // `None` (the common case) means no catch-up is pending — return O(1). // Otherwise pair it with the committed-index watermark as the replay @@ -50,27 +73,30 @@ impl Scheduler { .lock() .unwrap_or_else(|p| p.into_inner()); let Some(lo) = sm.peek_catch_up_from(self.vshard_id) else { - return; + return CatchUpDrain::Settled; }; let Some(hi) = sm.current_committed_index() else { // Armed but nothing applied yet — leave it armed and retry once // an entry is applied and `hi` is known. - return; + return CatchUpDrain::Settled; }; if lo > hi { // Armed ahead of the committed watermark (e.g. spawn-armed from // the first available index before any entry applied on this // replica). Nothing to replay yet; stay armed. - return; + return CatchUpDrain::Settled; } (lo, hi) }; + // Last index this drain reads: the window end, capped at `hi`. + let window = self.config.catch_up_window.max(1); + let end = lo.saturating_add(window - 1).min(hi); - // 2. MultiRaft-lock scope: read the committed sequencer log range. No SM - // lock is held here (see the lock-discipline note above). + // 2. MultiRaft-lock scope: read the committed sequencer log window. No + // SM lock is held here (see the lock-discipline note above). let entries = { let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - match mr.read_committed_entries(SEQUENCER_GROUP_ID, lo, hi) { + match mr.read_committed_entries(SEQUENCER_GROUP_ID, lo, end) { Ok(entries) => entries, Err(nodedb_cluster::error::ClusterError::Raft( nodedb_raft::RaftError::LogCompacted { .. }, @@ -94,7 +120,7 @@ impl Scheduler { .lock() .unwrap_or_else(|p| p.into_inner()) .clear_catch_up_up_to(self.vshard_id, hi); - return; + return CatchUpDrain::Settled; } Err(e) => { // Transient infra fault (e.g. group transiently absent). @@ -103,11 +129,11 @@ impl Scheduler { tracing::warn!( vshard = self.vshard_id, lo, - hi, + end, error = %e, "calvin catch-up: failed to read committed sequencer entries" ); - return; + return CatchUpDrain::Settled; } } }; @@ -142,21 +168,22 @@ impl Scheduler { // The in-flight guard makes an overlapping already-in-flight Txn a // no-op; Reserve/Release re-application is idempotent. // - // A dispatch refused at capacity stops the feed. The refused txn is - // parked in flight, and the next drain resumes at the first input - // not yet processed. + // An input after which the intake gate is closed stops the feed: a + // dispatch deferred at capacity, or a full in-flight backlog. The + // next drain resumes at the first input not yet processed. let mut replayed: u64 = 0; let mut resume_from: Option = None; let mut feed = inputs.into_iter().peekable(); while let Some((_, input)) = feed.next() { - let deferred_before = self.deferred_dispatch_len(); self.process_scheduler_input(input); replayed += 1; - if self.deferred_dispatch_len() > deferred_before { + if self.intake_closure().is_some() { resume_from = feed.peek().map(|(index, _)| *index); break; } } + // A window that ends before `hi` resumes at the first index past it. + let resume_from = resume_from.or(end.checked_add(1).filter(|&next| next <= hi)); { let sm = self @@ -171,12 +198,12 @@ impl Scheduler { sm.clear_catch_up_up_to(self.vshard_id, next.saturating_sub(1)); sm.arm_catch_up_from(self.vshard_id, next); } - // Replay of `lo ..= hi` is complete: clear the armed catch-up, - // but only up to `hi` — a concurrent drop recorded at an index - // `> hi` while this replay ran is preserved for the next drain. - // This is the CONFIRM step the peek-not-take at the top defers - // to; a transient failure above returned early and left the - // entry armed. + // Replay of `lo ..= hi` is complete (`end == hi` here): clear + // the armed catch-up, but only up to `hi` — a concurrent drop + // recorded at an index `> hi` while this replay ran is + // preserved for the next drain. This is the CONFIRM step the + // peek-not-take at the top defers to; a transient failure + // above returned early and left the entry armed. None => sm.clear_catch_up_up_to(self.vshard_id, hi), } } @@ -186,11 +213,17 @@ impl Scheduler { tracing::info!( vshard = self.vshard_id, lo, + end, hi, replayed, "calvin catch-up: replayed dropped sequencer inputs from committed log" ); } + if resume_from.is_some() { + CatchUpDrain::Remaining + } else { + CatchUpDrain::Settled + } } } @@ -564,4 +597,68 @@ mod tests { "catch-up must stay armed from the input after the refused one" ); } + + /// A drain over a range longer than its window replays exactly the + /// window and stays armed at the first index past it. The next drain + /// continues from there. + #[tokio::test] + async fn drain_replays_one_window_then_resumes_past_it() { + let vshard = test_coll_vshard(); + let (mut scheduler, _dir) = build_test_scheduler(vshard); + scheduler.config.catch_up_window = 1; + ensure_sequencer_leader(&scheduler); + + let txn0 = make_sequenced_txn(0, 0); + let txn1 = make_sequenced_txn(1, 0); + let (idx0, bytes0) = commit_epoch_batch(&scheduler, make_batch(0, &txn0)); + let (idx1, bytes1) = commit_epoch_batch(&scheduler, make_batch(1, &txn1)); + assert_eq!(idx1, idx0 + 1, "the two batches commit at adjacent indexes"); + apply_with_full_channel(&scheduler, vshard, idx0, &bytes0, &txn0); + apply_with_full_channel(&scheduler, vshard, idx1, &bytes1, &txn1); + + // A conflicting holder on the shared key makes each replayed txn + // block, so nothing dispatches. + let keys = + crate::control::cluster::calvin::scheduler::driver::helpers::expand_rw_set(&txn0); + { + let mut lm = scheduler + .lock_manager + .lock() + .unwrap_or_else(|p| p.into_inner()); + assert_eq!( + lm.acquire(TxnId::new(u64::MAX, 0), keys), + AcquireOutcome::Ready + ); + } + let armed = |scheduler: &Scheduler| { + scheduler + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .peek_catch_up_from(vshard) + }; + + let first = scheduler.drain_catch_up(); + + assert_eq!(first, CatchUpDrain::Remaining); + assert!(scheduler.blocked.contains_key(&TxnId::new(0, 0))); + assert!( + !scheduler.blocked.contains_key(&TxnId::new(1, 0)), + "the entry past the window must not be replayed" + ); + assert_eq!( + armed(&scheduler), + Some(idx1), + "catch-up stays armed at the first index past the window" + ); + + let second = scheduler.drain_catch_up(); + + assert_eq!(second, CatchUpDrain::Settled); + assert!( + scheduler.blocked.contains_key(&TxnId::new(1, 0)), + "the next drain replays the entry past the first window" + ); + assert_eq!(armed(&scheduler), None, "the armed range is fully replayed"); + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs new file mode 100644 index 000000000..a5c1c1c60 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs @@ -0,0 +1,285 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Intake gate for the Calvin scheduler. +//! +//! The scheduler takes new sequenced input only while it can make progress. +//! The gate closes while a capacity-refused dispatch waits for re-send, or +//! while the in-flight backlog sits at [`SchedulerConfig::max_inflight_backlog`]. +//! A closed gate disables the run loop's receiver arm and skips the catch-up +//! drain. Completions, verdicts, read results, promotions, and capacity +//! wakeups stay active, because they drain the backlog. +//! +//! Input left unread fills the bounded fan-out channel. The sequencer apply +//! path then drops the next input and arms catch-up for this vShard, and the +//! drain replays it from the committed log once the gate opens. +//! +//! The gate changes only when inputs are read, never the order they are +//! processed in, so replicas stay deterministic. +//! +//! [`SchedulerConfig::max_inflight_backlog`]: super::super::config::SchedulerConfig::max_inflight_backlog + +use tracing::debug; + +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::metrics::intake_closure_reason; + +/// Why the intake gate is closed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum IntakeClosure { + /// A capacity-refused dispatch waits in the deferred FIFO. + DeferredDispatch, + /// The in-flight backlog is at its bound, and some of it progresses + /// without new input. + BacklogFull, +} + +impl IntakeClosure { + /// The `nodedb_calvin_intake_gate_closed_total` reason index. + fn metric_reason(self) -> usize { + match self { + Self::DeferredDispatch => intake_closure_reason::DEFERRED_DISPATCH, + Self::BacklogFull => intake_closure_reason::BACKLOG_FULL, + } + } +} + +/// Last observed intake gate state, kept to detect transitions. +#[derive(Debug, Default)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct IntakeGate { + closed: Option, +} + +impl Scheduler { + /// Pending, blocked, and dependent-barrier txns. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn inflight_backlog( + &self, + ) -> usize { + self.pending.len() + self.blocked.len() + self.dependent_barrier.len() + } + + /// Why intake must stay closed, or `None` when the gate is open. + /// + /// The backlog bound applies only while a pending txn or a dependent + /// barrier exists. Those finish on executor responses, verdicts, and read + /// results. A backlog of blocked txns alone can wait on a reservation + /// release, which arrives as input, so it keeps the gate open. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn intake_closure( + &self, + ) -> Option { + if self.has_deferred_dispatch() { + return Some(IntakeClosure::DeferredDispatch); + } + let drains_without_input = !self.pending.is_empty() || !self.dependent_barrier.is_empty(); + if drains_without_input && self.inflight_backlog() >= self.config.max_inflight_backlog { + return Some(IntakeClosure::BacklogFull); + } + None + } + + /// Re-evaluate the intake gate and return whether it is open. + /// + /// Publishes the backlog gauge on every call. Logs each transition at + /// debug, and counts each open-to-closed transition by reason. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn refresh_intake_gate( + &mut self, + ) -> bool { + let backlog = self.inflight_backlog(); + let closure = self.intake_closure(); + self.metrics.set_intake_backlog(backlog); + let previous = self.intake.closed; + if closure == previous { + return closure.is_none(); + } + match closure { + Some(reason) => { + if previous.is_none() { + self.metrics + .record_intake_gate_closed(reason.metric_reason()); + } + debug!( + vshard_id = self.vshard_id, + ?reason, + backlog, + deferred = self.deferred_dispatch_len(), + "calvin scheduler: intake gate closed" + ); + } + None => { + debug!( + vshard_id = self.vshard_id, + backlog, "calvin scheduler: intake gate opened" + ); + } + } + self.metrics.set_intake_gate_closed(closure.is_some()); + self.intake.closed = closure; + closure.is_none() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::BTreeSet; + use std::sync::Arc; + use std::sync::atomic::Ordering; + use std::time::{Duration, Instant}; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_cluster::calvin::types::SchedulerInput; + use nodedb_types::TenantId; + use tokio::sync::mpsc; + + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_sequenced_txn, make_validate_only_txn, release_filler, spawn_scheduler_loop, + staged_pending, test_coll_vshard, + }; + use crate::control::cluster::calvin::scheduler::driver::types::BlockedTxn; + use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + + /// How long a held input must stay unread. Several liveness ticks of the + /// spawned loop fit in it. + const HOLD_WAIT: Duration = Duration::from_millis(300); + + /// How long a test waits for the loop to read an input. + const CONSUME_WAIT: Duration = Duration::from_secs(5); + + /// Whether the loop reads every input sent on `input_tx` within `wait`. + async fn inputs_consumed_within( + input_tx: &mpsc::Sender, + wait: Duration, + ) -> bool { + let drained = async { + while input_tx.capacity() < input_tx.max_capacity() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }; + tokio::time::timeout(wait, drained).await.is_ok() + } + + fn blocked_fixture(epoch: u64) -> BlockedTxn { + BlockedTxn { + txn: make_sequenced_txn(epoch, 0), + keys: BTreeSet::new(), + // no-determinism: test-only blocked_at timestamp for a fabricated BlockedTxn fixture. + blocked_at: Instant::now(), + } + } + + /// A backlog of blocked txns alone keeps intake open at the bound: a + /// reservation release they wait on arrives as input. + #[tokio::test] + async fn blocked_only_backlog_at_bound_keeps_intake_open() { + let (mut scheduler, _dir) = build_test_scheduler(0); + scheduler.config.max_inflight_backlog = 1; + scheduler + .blocked + .insert(TxnId::new(3, 0), blocked_fixture(3)); + + assert_eq!(scheduler.intake_closure(), None); + } + + /// A pending txn that fills the backlog bound closes intake. + #[tokio::test] + async fn pending_backlog_at_bound_closes_intake() { + let (mut scheduler, _dir) = build_test_scheduler(0); + scheduler.config.max_inflight_backlog = 2; + let held = TxnId::new(3, 0); + scheduler + .pending + .insert(held, staged_pending(make_sequenced_txn(3, 0), held)); + assert_eq!(scheduler.intake_closure(), None, "below the bound"); + + scheduler + .blocked + .insert(TxnId::new(4, 0), blocked_fixture(4)); + assert_eq!( + scheduler.intake_closure(), + Some(IntakeClosure::BacklogFull), + "at the bound with a pending txn" + ); + } + + /// With a dispatch deferred at capacity, the run loop leaves a new input + /// unread. It reads the input once capacity frees and the deferral drains. + #[tokio::test] + async fn deferred_dispatch_holds_new_input_until_capacity_frees() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + assert!( + scheduler.has_deferred_dispatch(), + "the stage dispatch defers" + ); + let metrics = Arc::clone(&scheduler.metrics); + + let running = spawn_scheduler_loop(scheduler); + running + .input_tx() + .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .await + .expect("the loop's input channel is open"); + + let read_while_deferred = inputs_consumed_within(running.input_tx(), HOLD_WAIT).await; + let gate_closed = metrics.intake_gate_closed.load(Ordering::Relaxed); + let deferred_closures = metrics.intake_gate_closed_counts + [intake_closure_reason::DEFERRED_DISPATCH] + .load(Ordering::Relaxed); + + release_filler(&shared, &mut data_side, fillers[0]); + let read_after_capacity = inputs_consumed_within(running.input_tx(), CONSUME_WAIT).await; + running.stop().await; + + assert!( + !read_while_deferred, + "the loop must not read input while a dispatch is deferred" + ); + assert_eq!(gate_closed, 1, "the gate gauge reports closed"); + assert_eq!(deferred_closures, 1, "one closure for a deferred dispatch"); + assert!( + read_after_capacity, + "the loop must read the input once the deferral drains" + ); + } + + /// With the backlog at the configured bound, the run loop leaves a new + /// input unread. + #[tokio::test] + async fn full_backlog_holds_new_input() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, _data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + scheduler.config.max_inflight_backlog = 1; + let held = TxnId::new(3, 0); + scheduler + .pending + .insert(held, staged_pending(make_validate_only_txn(3, 0), held)); + let metrics = Arc::clone(&scheduler.metrics); + + let running = spawn_scheduler_loop(scheduler); + running + .input_tx() + .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .await + .expect("the loop's input channel is open"); + + let read_at_bound = inputs_consumed_within(running.input_tx(), HOLD_WAIT).await; + running.stop().await; + + assert!( + !read_at_bound, + "the loop must not read input while the backlog is at its bound" + ); + assert_eq!(metrics.intake_gate_closed.load(Ordering::Relaxed), 1); + assert_eq!(metrics.intake_backlog.load(Ordering::Relaxed), 1); + assert_eq!( + metrics.intake_gate_closed_counts[intake_closure_reason::BACKLOG_FULL] + .load(Ordering::Relaxed), + 1 + ); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index a27e43e39..140908580 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -16,6 +16,8 @@ //! txn-completion bookkeeping. //! - [`catch_up`] — sequencer-fan-out catch-up drain: replays inputs dropped on //! this replica (channel Full/Closed) from the committed sequencer Raft log. +//! - [`intake`] — intake gate: stops reading new sequenced input while a +//! dispatch is deferred or the in-flight backlog is at its bound. //! - [`dispatch`] — static / active dispatch to the Data Plane executor. //! - [`deferred`] — capacity-safe dispatch: parks a request the bridge refuses //! at capacity and re-sends it once capacity frees. @@ -54,6 +56,7 @@ pub mod commit_resolve; pub mod completion_route; pub mod deferred; pub mod dispatch; +pub mod intake; pub mod process; pub mod propose; pub mod read_result; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index 2155c55bc..5bf93087f 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -18,7 +18,9 @@ use nodedb_cluster::calvin::{ use super::super::barrier::{PendingDependentBarrier, ReadResultEvent}; use super::super::config::SchedulerConfig; use super::super::types::{BlockedTxn, PendingTxn}; +use super::catch_up::CatchUpDrain; use super::deferred::DeferredQueue; +use super::intake::IntakeGate; use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; @@ -153,6 +155,8 @@ pub struct Scheduler { /// The bridge dispatcher's capacity-freed signal, cloned once at /// construction. The run loop waits on it while requests are deferred. pub(in crate::control::cluster::calvin::scheduler::driver::core) capacity_freed: Arc, + /// Last observed intake gate state. See [`super::intake`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) intake: IntakeGate, } /// Parameters for [`Scheduler::new`]. @@ -244,6 +248,7 @@ impl Scheduler { verdict_rx, deferred: DeferredQueue::new(), capacity_freed, + intake: IntakeGate::default(), } } @@ -334,6 +339,9 @@ impl Scheduler { // Woken when a routed Data-Plane response frees dispatcher capacity. let capacity_freed = Arc::clone(&self.capacity_freed); + // Set when a tick left armed catch-up unreplayed. The next open-gate + // pass fires the tick at once to resume it. + let mut catch_up_resume = false; loop { // Register for the capacity wake BEFORE the re-send pass. A @@ -349,6 +357,12 @@ impl Scheduler { self.check_dependent_barrier_timeouts(); self.check_awaiting_verdict_stalls(); + let intake_open = self.refresh_intake_gate(); + if intake_open && catch_up_resume { + catch_up_resume = false; + stall_tick.reset_immediately(); + } + tokio::select! { biased; @@ -393,7 +407,7 @@ impl Scheduler { // requests in FIFO order. } - maybe_txn = self.receiver.recv() => { + maybe_txn = self.receiver.recv(), if intake_open => { match maybe_txn { Some(input) => self.process_scheduler_input(input), None => { @@ -411,7 +425,9 @@ impl Scheduler { // (channel Full/Closed) so a missed `SchedulerInput` never // permanently diverges this vShard's lock table from its peers. // O(1) common case (no pending catch-up). See `drain_catch_up`. - self.drain_catch_up(); + // A closed intake gate skips the drain until it opens. + catch_up_resume = + !intake_open || self.drain_catch_up() == CatchUpDrain::Remaining; // The top-of-loop check_awaiting_verdict_stalls / // check_dependent_barrier_timeouts and the deferred re-send // pass run on every wake; this arm guarantees the loop wakes diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs index a15336225..123330f76 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -368,12 +368,17 @@ pub(super) async fn await_data_plane_request( pub(super) struct RunningScheduler { shutdown: ShutdownWatch, handle: tokio::task::JoinHandle<()>, - _input_tx: mpsc::Sender, + input_tx: mpsc::Sender, _read_result_tx: mpsc::Sender, _promotion_tx: mpsc::UnboundedSender>, } impl RunningScheduler { + /// The sender feeding the loop's sequenced-input receiver. + pub(super) fn input_tx(&self) -> &mpsc::Sender { + &self.input_tx + } + /// Signal shutdown and wait for the loop to exit. pub(super) async fn stop(self) { self.shutdown.signal(); @@ -401,7 +406,7 @@ pub(super) fn spawn_scheduler_loop(mut scheduler: Scheduler) -> RunningScheduler RunningScheduler { shutdown, handle, - _input_tx: input_tx, + input_tx, _read_result_tx: read_result_tx, _promotion_tx: promotion_tx, } diff --git a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs index 46d50c57b..1bc1e68dd 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs @@ -63,6 +63,14 @@ pub struct SchedulerMetrics { pub dispatch_deferred_count: AtomicU64, /// Requests parked for re-send right now, waiting for dispatcher capacity. pub dispatch_deferred_depth: AtomicU64, + /// Intake gate state: 1 while the scheduler takes no new sequenced input, + /// 0 while it does. + pub intake_gate_closed: AtomicU64, + /// In-flight backlog: pending, blocked, and dependent-barrier txns. + pub intake_backlog: AtomicU64, + /// Intake gate closures by reason. Indexes are the constants in + /// [`intake_closure_reason`]. + pub intake_gate_closed_counts: [AtomicU64; 2], } /// Reason codes for `nodedb_calvin_infra_abort_total`. @@ -84,6 +92,14 @@ pub mod infra_abort_reason { ]; } +/// Reason codes for `nodedb_calvin_intake_gate_closed_total`. +pub mod intake_closure_reason { + pub const DEFERRED_DISPATCH: usize = 0; + pub const BACKLOG_FULL: usize = 1; + + pub const LABELS: &[&str] = &["deferred_dispatch", "backlog_full"]; +} + impl SchedulerMetrics { pub fn new() -> Arc { Arc::new(Self::default()) @@ -151,6 +167,26 @@ impl SchedulerMetrics { .store(depth as u64, Ordering::Relaxed); } + /// Record that the intake gate closed for `reason`. + /// + /// `reason` must be one of the constants in [`intake_closure_reason`]. + pub fn record_intake_gate_closed(&self, reason: usize) { + if let Some(counter) = self.intake_gate_closed_counts.get(reason) { + counter.fetch_add(1, Ordering::Relaxed); + } + } + + /// Set the intake gate state gauge. + pub fn set_intake_gate_closed(&self, closed: bool) { + self.intake_gate_closed + .store(u64::from(closed), Ordering::Relaxed); + } + + /// Set the in-flight backlog gauge. + pub fn set_intake_backlog(&self, backlog: usize) { + self.intake_backlog.store(backlog as u64, Ordering::Relaxed); + } + /// Record the end-to-end executor txn duration (dispatch → response). /// /// Increments the appropriate histogram bucket and the running sum. @@ -330,6 +366,44 @@ impl SchedulerMetrics { self.dispatch_deferred_depth.load(Ordering::Relaxed) ); + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_gate_closed \ + 1 while the scheduler takes no new sequenced input." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_gate_closed gauge"); + let _ = writeln!( + out, + "nodedb_calvin_intake_gate_closed{{{label}}} {}", + self.intake_gate_closed.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_backlog \ + Pending, blocked, and dependent-barrier txns in the scheduler." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_backlog gauge"); + let _ = writeln!( + out, + "nodedb_calvin_intake_backlog{{{label}}} {}", + self.intake_backlog.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_gate_closed_total \ + Times the scheduler stopped taking new sequenced input, by reason." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_gate_closed_total counter"); + for (i, &reason_label) in intake_closure_reason::LABELS.iter().enumerate() { + let _ = writeln!( + out, + "nodedb_calvin_intake_gate_closed_total{{{label},reason=\"{reason_label}\"}} {}", + self.intake_gate_closed_counts[i].load(Ordering::Relaxed) + ); + } + out } } @@ -351,6 +425,9 @@ impl Default for SchedulerMetrics { catch_up_log_compacted: AtomicU64::new(0), dispatch_deferred_count: AtomicU64::new(0), dispatch_deferred_depth: AtomicU64::new(0), + intake_gate_closed: AtomicU64::new(0), + intake_backlog: AtomicU64::new(0), + intake_gate_closed_counts: std::array::from_fn(|_| AtomicU64::new(0)), } } } From 1885c7a1853ac3b7c5e32738c6300d639441a2ad Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 13:43:42 +0800 Subject: [PATCH 05/64] fix(executor): never expire already-ordered work on deadline Request::deadline was read directly by DeadlineCheck and ExecutionTask::is_expired, so Calvin applies, replicated applies, replay, clone, and checkpoint work could be dropped as DeadlineExceeded once their envelope deadline passed. That work is already ordered and every replica must run it to completion, or replicas diverge. Add Request::execution_deadline, which returns None for Admission::Exempt(ExemptReason::AlreadyOrdered) and Some(deadline) otherwise. Route every deadline check through it instead of the raw field, and carry the parent's admission (not just its deadline) into transaction sub-plan tasks and exec_tx_passthrough so a sub-plan inherits its parent's exemption. --- nodedb/src/bridge/envelope/request.rs | 43 +++++++++- nodedb/src/data/executor/deadline.rs | 80 ++++++++++++++++--- .../executor/handlers/transaction/sub_plan.rs | 55 +++++++------ .../handlers/transaction/sub_plan_columnar.rs | 2 +- .../handlers/transaction/sub_plan_kv_ops.rs | 2 +- .../handlers/transaction/sub_plan_write.rs | 20 ++--- nodedb/src/data/executor/task.rs | 32 +++++++- .../inproc/cases/calvin_two_phase_apply.rs | 67 ++++++++++++++++ 8 files changed, 252 insertions(+), 49 deletions(-) diff --git a/nodedb/src/bridge/envelope/request.rs b/nodedb/src/bridge/envelope/request.rs index 791d1d887..cfc21210e 100644 --- a/nodedb/src/bridge/envelope/request.rs +++ b/nodedb/src/bridge/envelope/request.rs @@ -34,7 +34,9 @@ pub struct Request { /// Opaque plan digest identifying the physical operation to execute. pub plan: PhysicalPlan, - /// Absolute deadline. Data Plane MUST stop at next safe point after expiry. + /// Absolute deadline. The Data Plane reads it only through + /// [`Request::execution_deadline`], and stops at the next safe point after + /// that deadline passes. Already-ordered work has no execution deadline. pub deadline: Instant, /// Request priority for scheduling on the Data Plane. @@ -119,6 +121,25 @@ pub struct Request { pub admission: Admission, } +impl Request { + /// The deadline the Data Plane enforces while it runs this request. + /// + /// Returns `None` for [`ExemptReason::AlreadyOrdered`] work: Calvin + /// applies, replicated applies, replay, clone, and checkpoint. Their order + /// is already fixed and other replicas apply the same work. Such work has + /// no refusal outcome, so every replica must run it to completion. A + /// replica that drops it on a deadline diverges from the others. + /// + /// Returns `Some(self.deadline)` for every other request. Every Data-Plane + /// deadline check reads this method, never the raw field. + pub fn execution_deadline(&self) -> Option { + match self.admission { + Admission::Exempt(ExemptReason::AlreadyOrdered) => None, + Admission::Admitted | Admission::Exempt(ExemptReason::Read) => Some(self.deadline), + } + } +} + /// Write-admission marker carried by every [`Request`]. /// /// A write-class plan becomes [`Admission::Admitted`] only by passing the @@ -214,6 +235,26 @@ mod tests { assert_ne!(req.trace_id, TraceId::ZERO); } + #[test] + fn already_ordered_request_has_no_execution_deadline() { + let req = Request { + admission: Admission::Exempt(ExemptReason::AlreadyOrdered), + ..sample_request() + }; + assert_eq!(req.execution_deadline(), None); + } + + #[test] + fn admitted_and_read_requests_keep_their_deadline() { + for admission in [Admission::Admitted, Admission::Exempt(ExemptReason::Read)] { + let req = Request { + admission, + ..sample_request() + }; + assert_eq!(req.execution_deadline(), Some(req.deadline)); + } + } + #[test] fn cancel_plan() { let req = Request { diff --git a/nodedb/src/data/executor/deadline.rs b/nodedb/src/data/executor/deadline.rs index c9bbb6c34..472e2da3b 100644 --- a/nodedb/src/data/executor/deadline.rs +++ b/nodedb/src/data/executor/deadline.rs @@ -2,12 +2,13 @@ //! Cooperative deadline enforcement inside Data-Plane execution. //! -//! The Control -> Data request envelope carries one absolute deadline -//! ([`Request::deadline`](crate::bridge::envelope::Request)). The core loop -//! refuses a task that is already past it before execution starts; this type -//! carries the same deadline INTO execution so a statement that goes over -//! while it runs stops at its next safe point instead of running to -//! completion. +//! The Control -> Data request envelope carries one execution deadline +//! ([`Request::execution_deadline`](crate::bridge::envelope::Request::execution_deadline)). +//! The core loop refuses a task that is already past it before execution +//! starts. This type carries the same deadline INTO execution, so a statement +//! that goes over while it runs stops at its next safe point instead of +//! running to completion. Already-ordered work has no execution deadline, and +//! this check never trips for it. //! //! The clock is [`Instant`], matching the envelope field. `Instant` is //! monotonic, so an NTP step or a leap second cannot stretch or shrink a @@ -39,7 +40,9 @@ const STRIDE: u32 = 1024; /// to read a `statement_timeout` from. A statement budget would also be the /// wrong shape: abandoning replay partway leaves engine state short of the WAL, /// which is data loss rather than a cancelled query. The field exists on the -/// envelope, so replay fills it; no safe point on the replay path consults it. +/// envelope, so replay fills it. Replay tasks are already ordered, so +/// [`Request::execution_deadline`](crate::bridge::envelope::Request::execution_deadline) +/// returns `None` for them and no check consults this value. pub(in crate::data::executor) const REPLAY_DEADLINE: Duration = Duration::from_secs(60); /// A task's deadline, checkable from a row loop. @@ -47,16 +50,17 @@ pub(in crate::data::executor) const REPLAY_DEADLINE: Duration = Duration::from_s /// `!Send` by construction (it holds [`Cell`]s), which is what the Data Plane /// requires. Interior mutability lets an `Fn` predicate closure consult it. pub(in crate::data::executor) struct DeadlineCheck { - deadline: Instant, + /// `None` for already-ordered work, which runs to completion. + deadline: Option, countdown: Cell, tripped: Cell, } impl DeadlineCheck { - /// Take the deadline off the request envelope this task arrived on. + /// Take the execution deadline off the request envelope this task arrived on. pub(in crate::data::executor) fn for_task(task: &ExecutionTask) -> Self { Self { - deadline: task.request.deadline, + deadline: task.request.execution_deadline(), // The first call reads the clock: a task that entered execution // barely inside its deadline must stop on its first row, not after // a full stride. @@ -96,7 +100,10 @@ impl DeadlineCheck { } fn read_clock(&self) -> bool { - let over = Instant::now() > self.deadline; + let Some(deadline) = self.deadline else { + return false; + }; + let over = Instant::now() > deadline; if over { self.tripped.set(true); } @@ -111,7 +118,7 @@ mod tests { fn check_at(deadline: Instant) -> DeadlineCheck { DeadlineCheck { - deadline, + deadline: Some(deadline), countdown: Cell::new(1), tripped: Cell::new(false), } @@ -140,6 +147,55 @@ mod tests { assert!(check.expired()); } + fn past_deadline_task(admission: crate::bridge::envelope::Admission) -> ExecutionTask { + use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; + use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; + use nodedb_physical::physical_plan::MetaOp; + ExecutionTask::new(Request { + request_id: RequestId::new(1), + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan: PhysicalPlan::Meta(MetaOp::Cancel { + target_request_id: RequestId::new(7), + }), + deadline: Instant::now() - Duration::from_secs(1), + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission, + }) + } + + #[test] + fn already_ordered_work_never_trips_past_its_deadline() { + let task = past_deadline_task(crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::AlreadyOrdered, + )); + let check = DeadlineCheck::for_task(&task); + assert!(!check.expired_now()); + for _ in 0..(STRIDE * 3) { + assert!(!check.expired()); + } + assert!(!check.tripped()); + } + + #[test] + fn admitted_work_trips_past_its_deadline() { + let task = past_deadline_task(crate::bridge::envelope::Admission::Admitted); + let check = DeadlineCheck::for_task(&task); + assert!(check.expired()); + assert!(check.tripped()); + } + #[test] fn stride_bounds_clock_reads() { // A live deadline leaves the countdown mid-stride after one call, which diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan.rs index 1a12a202c..03c544a6b 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan.rs @@ -6,7 +6,7 @@ //! and record undo entries) live in `sub_plan_write.rs`; this file only //! routes each `PhysicalPlan` variant to its engine-specific handler. -use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; +use crate::bridge::envelope::{Admission, ErrorCode, PhysicalPlan, Request, Response, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; use crate::types::{DatabaseId, TenantId, TraceId}; @@ -62,14 +62,9 @@ impl CoreLoop { return self.exec_tx_timeseries(parent, tid, plan, op, undo_log); } - let task = Self::build_dummy_task_at( - tid, - parent.request.database_id, - parent.request.vshard_id, - // The sub-plan is part of the parent statement, so it runs on what - // is left of the parent's budget rather than a fresh one. - parent.request.deadline, - ); + // The sub-plan is part of the parent statement, so it runs on what is + // left of the parent's budget rather than a fresh one. + let task = Self::build_dummy_task_at(tid, &parent.request); self.execute_tx_sub_plan_with_task(&task, tid, plan, undo_log, crdt_deltas, user_roles) } @@ -103,7 +98,7 @@ impl CoreLoop { | PhysicalPlan::Array(_) | PhysicalPlan::ClusterArray(_) | PhysicalPlan::ClusterEvent(_) => { - self.exec_tx_passthrough(tid, plan, dummy_task.request.deadline) + self.exec_tx_passthrough(tid, plan, &dummy_task.request) } } } @@ -115,24 +110,40 @@ impl CoreLoop { /// only carries request metadata for response building. #[cfg(test)] pub(super) fn build_dummy_task(tid: u64) -> ExecutionTask { - Self::build_dummy_task_at( + Self::build_dummy_task_with( tid, DatabaseId::DEFAULT, crate::types::VShardId::new(0), // no-determinism: test-only dummy deadline, never written to Calvin state std::time::Instant::now() + std::time::Duration::from_secs(60), + Admission::Exempt(crate::bridge::envelope::ExemptReason::Read), + ) + } + + /// Copies `parent`'s database, vShard, `deadline`, and `admission`. The + /// dummy task's + /// [`execution_deadline`](crate::bridge::envelope::Request::execution_deadline) + /// then equals the parent's. Every sub-plan this task carries stops when + /// the statement does. An already-ordered parent, such as a Calvin apply, + /// has no execution deadline, and its sub-plans run to completion. + fn build_dummy_task_at(tid: u64, parent: &Request) -> ExecutionTask { + Self::build_dummy_task_with( + tid, + parent.database_id, + parent.vshard_id, + parent.deadline, + parent.admission, ) } - /// `deadline` is the enclosing statement's, copied from the parent task, so - /// every sub-plan this task carries stops when the statement does. - fn build_dummy_task_at( + fn build_dummy_task_with( tid: u64, database_id: DatabaseId, vshard_id: crate::types::VShardId, deadline: std::time::Instant, + admission: Admission, ) -> ExecutionTask { - ExecutionTask::new(crate::bridge::envelope::Request { + ExecutionTask::new(Request { request_id: crate::types::RequestId::new(0), tenant_id: TenantId::new(tid), database_id, @@ -153,9 +164,7 @@ impl CoreLoop { txn_id: None, wal_lsn: None, resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::Read, - ), + admission, }) } @@ -246,7 +255,7 @@ impl CoreLoop { undo_log, ), - _ => self.exec_tx_passthrough(tid, plan, dummy_task.request.deadline), + _ => self.exec_tx_passthrough(tid, plan, &dummy_task.request), } } @@ -293,7 +302,7 @@ impl CoreLoop { undo_log, )), - _ => self.exec_tx_passthrough(tid, plan, dummy_task.request.deadline), + _ => self.exec_tx_passthrough(tid, plan, &dummy_task.request), } } @@ -394,7 +403,7 @@ impl CoreLoop { Ok(response) } - _ => self.exec_tx_passthrough(tid, plan, dummy_task.request.deadline), + _ => self.exec_tx_passthrough(tid, plan, &dummy_task.request), } } @@ -416,7 +425,7 @@ impl CoreLoop { detail: "CRDT Apply is not supported inside transaction batches".into(), }) } - _ => self.exec_tx_passthrough(tid, plan, dummy_task.request.deadline), + _ => self.exec_tx_passthrough(tid, plan, &dummy_task.request), } } @@ -475,7 +484,7 @@ impl CoreLoop { } TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_) => { - self.exec_tx_passthrough(tid, plan, dummy_task.request.deadline) + self.exec_tx_passthrough(tid, plan, &dummy_task.request) } } } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_columnar.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_columnar.rs index 0d26e3908..a3380b74e 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_columnar.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_columnar.rs @@ -137,7 +137,7 @@ impl CoreLoop { ColumnarOp::Scan { .. } | ColumnarOp::MaterializeScan { .. } | ColumnarOp::ResolveDml { .. } => { - self.exec_tx_passthrough(tid, plan, dummy_task.request.deadline) + self.exec_tx_passthrough(tid, plan, &dummy_task.request) } } } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ops.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ops.rs index a5d41f04c..f3b02b8c5 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ops.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ops.rs @@ -161,7 +161,7 @@ impl CoreLoop { // COMMIT, the same passthrough Document `BulkUpdate`/`BulkDelete` // take in `exec_tx_document`. KvOp::PredicateUpdate { .. } | KvOp::PredicateDelete { .. } => { - self.exec_tx_passthrough(tid, plan, task.request.deadline) + self.exec_tx_passthrough(tid, plan, &task.request) } } } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs index 086c11f89..4ccbd8a71 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs @@ -333,16 +333,18 @@ impl CoreLoop { /// /// None of these variants mutate engine state, so no undo entry is needed. /// - /// `deadline` is the enclosing statement's, copied from the parent task. A - /// sub-plan is part of the statement that spawned it, so it inherits that - /// statement's remaining budget; minting a fresh one here would let a - /// transaction outlive the `statement_timeout` its client set by one - /// sub-plan's worth of work per sub-plan. + /// The sub-plan task copies `parent`'s `deadline` and `admission`, so its + /// [`execution_deadline`](crate::bridge::envelope::Request::execution_deadline) + /// equals the parent's. A sub-plan is part of the statement that spawned + /// it, so it runs on that statement's remaining budget. A fresh budget per + /// sub-plan lets a transaction outlive its client's `statement_timeout`. + /// An already-ordered parent has no execution deadline, and neither does + /// its sub-plan. pub(super) fn exec_tx_passthrough( &mut self, tid: u64, plan: &PhysicalPlan, - deadline: std::time::Instant, + parent: &crate::bridge::envelope::Request, ) -> Result { let resp = self.execute(&ExecutionTask::new(crate::bridge::envelope::Request { request_id: crate::types::RequestId::new(0), @@ -351,7 +353,7 @@ impl CoreLoop { vshard_id: crate::types::VShardId::new(0), plan: plan.clone(), // no-determinism: sub-plan deadline is ephemeral, not written to WAL - deadline, + deadline: parent.deadline, priority: crate::bridge::envelope::Priority::Normal, trace_id: TraceId::ZERO, consistency: crate::types::ReadConsistency::Strong, @@ -363,9 +365,7 @@ impl CoreLoop { txn_id: None, wal_lsn: None, resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::AlreadyOrdered, - ), + admission: parent.admission, })); if resp.status == Status::Error { return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { diff --git a/nodedb/src/data/executor/task.rs b/nodedb/src/data/executor/task.rs index faa4c676e..59ed29e4e 100644 --- a/nodedb/src/data/executor/task.rs +++ b/nodedb/src/data/executor/task.rs @@ -103,8 +103,12 @@ impl ExecutionTask { &self.request.plan } + /// Whether the task's execution deadline has passed. Already-ordered work + /// has no execution deadline and never expires. pub fn is_expired(&self) -> bool { - std::time::Instant::now() > self.request.deadline + self.request + .execution_deadline() + .is_some_and(|deadline| std::time::Instant::now() > deadline) } } @@ -153,6 +157,32 @@ mod tests { assert_eq!(task.wal_lsn(), Some(lsn)); } + fn past_deadline_task(admission: crate::bridge::envelope::Admission) -> ExecutionTask { + ExecutionTask::new(Request { + deadline: Instant::now() - Duration::from_secs(1), + admission, + ..request_with_wal_lsn(None) + }) + } + + #[test] + fn already_ordered_task_past_its_deadline_is_not_expired() { + let task = past_deadline_task(crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::AlreadyOrdered, + )); + assert!(!task.is_expired()); + } + + #[test] + fn admitted_or_read_task_past_its_deadline_is_expired() { + let admitted = past_deadline_task(crate::bridge::envelope::Admission::Admitted); + assert!(admitted.is_expired()); + let read = past_deadline_task(crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::Read, + )); + assert!(read.is_expired()); + } + #[test] fn new_leaves_wal_lsn_none_for_reads() { let task = ExecutionTask::new(request_with_wal_lsn(None)); diff --git a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs index 2167a0343..28f040b40 100644 --- a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs +++ b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs @@ -711,3 +711,70 @@ fn absent_document_read_without_matching_insert_still_commits() { collection" ); } + +/// Push one prebuilt request through the ring and return its response. +fn send_request( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + request: Request, +) -> Response { + tx.try_push(BridgeRequest { inner: request }).unwrap(); + core.tick(); + rx.try_pop().unwrap().inner +} + +/// Calvin sub-operations are already ordered: every replica must run them. +/// A stage and a flush whose envelope deadline has already passed still +/// execute, and the write becomes visible. They never answer +/// `DeadlineExceeded`. +#[test] +fn already_ordered_stage_and_flush_run_past_their_deadline() { + let (mut core, mut tx, mut rx, _dir) = make_core(); + let already_ordered = |plan: PhysicalPlan| Request { + deadline: Instant::now() - Duration::from_secs(1), + admission: nodedb::bridge::envelope::Admission::Exempt( + nodedb::bridge::envelope::ExemptReason::AlreadyOrdered, + ), + ..make_request(plan, 0, None) + }; + + let staged = send_request( + &mut core, + &mut tx, + &mut rx, + already_ordered(stage_static( + 9, + 0, + vec![kv_put("latecoll", b"lk", b"lv")], + Vec::new(), + )), + ); + assert_eq!(staged.status, Status::Ok, "late stage must run: {staged:?}"); + assert_eq!(staged.read_set_valid, Some(true)); + + let flush = send_request( + &mut core, + &mut tx, + &mut rx, + already_ordered(PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch: 9, + position: 0, + })), + ); + assert_eq!(flush.status, Status::Ok, "late flush must run: {flush:?}"); + + let after = send( + &mut core, + &mut tx, + &mut rx, + kv_get("latecoll", b"lk"), + 0, + None, + ); + assert_eq!(after.status, Status::Ok, "read after flush: {after:?}"); + assert!( + !after.payload.is_empty(), + "the late flush must make the staged write visible" + ); +} From 374114238e96f6608ec90b6260a82d2567aaf83c Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 14:29:36 +0800 Subject: [PATCH 06/64] fix(executor): run transaction sub-plans in the session's database exec_tx_passthrough hardcoded DatabaseId::DEFAULT and vShard 0 for every sub-plan request, so a predicate UPDATE/DELETE issued inside a transaction against a non-default session database silently applied to the default database's collection instead. Extract the sub-plan request-building shared by exec_tx_passthrough and build_dummy_task_at into a SubRequestScope that copies the parent's database, vShard, deadline, and admission, and use it from both call sites. --- .../data/executor/handlers/transaction/mod.rs | 1 + .../executor/handlers/transaction/sub_plan.rs | 71 +++-------- .../handlers/transaction/sub_plan_write.rs | 40 ++---- .../handlers/transaction/sub_request.rs | 114 ++++++++++++++++++ .../sql_transactions_bulk_dml_overlay.rs | 57 +++++++++ .../sql_transactions_kv_predicate_overlay.rs | 35 ++++++ 6 files changed, 235 insertions(+), 83 deletions(-) create mode 100644 nodedb/src/data/executor/handlers/transaction/sub_request.rs diff --git a/nodedb/src/data/executor/handlers/transaction/mod.rs b/nodedb/src/data/executor/handlers/transaction/mod.rs index b9cd821b2..6666aa559 100644 --- a/nodedb/src/data/executor/handlers/transaction/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/mod.rs @@ -17,6 +17,7 @@ mod sub_plan_kv_ops; mod sub_plan_kv_ttl_sorted; mod sub_plan_kv_writes; mod sub_plan_write; +mod sub_request; pub(in crate::data::executor::handlers) mod undo; mod write_version; mod write_version_kv; diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan.rs index 03c544a6b..79c0b1691 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan.rs @@ -6,14 +6,15 @@ //! and record undo entries) live in `sub_plan_write.rs`; this file only //! routes each `PhysicalPlan` variant to its engine-specific handler. -use crate::bridge::envelope::{Admission, ErrorCode, PhysicalPlan, Request, Response, Status}; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Request, Response, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -use crate::types::{DatabaseId, TenantId, TraceId}; +use crate::types::{RequestId, TenantId}; use nodedb_physical::physical_plan::{CrdtOp, DocumentOp, GraphOp, MetaOp, TimeseriesOp, VectorOp}; use super::sub_plan_doc::{TxPointDelete, TxPointPut}; use super::sub_plan_write::{TxEdgeDeleteParams, TxEdgePutParams, TxVectorInsertParams}; +use super::sub_request::SubRequestScope; use super::undo::UndoEntry; impl CoreLoop { @@ -110,61 +111,29 @@ impl CoreLoop { /// only carries request metadata for response building. #[cfg(test)] pub(super) fn build_dummy_task(tid: u64) -> ExecutionTask { - Self::build_dummy_task_with( - tid, - DatabaseId::DEFAULT, - crate::types::VShardId::new(0), + use crate::bridge::envelope::{Admission, ExemptReason}; + use crate::types::{DatabaseId, VShardId}; + + let scope = SubRequestScope { + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), // no-determinism: test-only dummy deadline, never written to Calvin state - std::time::Instant::now() + std::time::Duration::from_secs(60), - Admission::Exempt(crate::bridge::envelope::ExemptReason::Read), - ) + deadline: std::time::Instant::now() + std::time::Duration::from_secs(60), + admission: Admission::Exempt(ExemptReason::Read), + }; + ExecutionTask::new(scope.request(tid, Self::dummy_plan())) } - /// Copies `parent`'s database, vShard, `deadline`, and `admission`. The - /// dummy task's - /// [`execution_deadline`](crate::bridge::envelope::Request::execution_deadline) - /// then equals the parent's. Every sub-plan this task carries stops when - /// the statement does. An already-ordered parent, such as a Calvin apply, - /// has no execution deadline, and its sub-plans run to completion. + /// The dummy task runs under [`SubRequestScope::of`] `parent`: the + /// parent's database, vShard, deadline, and admission. Every sub-plan + /// this task carries stops when the statement does. fn build_dummy_task_at(tid: u64, parent: &Request) -> ExecutionTask { - Self::build_dummy_task_with( - tid, - parent.database_id, - parent.vshard_id, - parent.deadline, - parent.admission, - ) + ExecutionTask::new(SubRequestScope::of(parent).request(tid, Self::dummy_plan())) } - fn build_dummy_task_with( - tid: u64, - database_id: DatabaseId, - vshard_id: crate::types::VShardId, - deadline: std::time::Instant, - admission: Admission, - ) -> ExecutionTask { - ExecutionTask::new(Request { - request_id: crate::types::RequestId::new(0), - tenant_id: TenantId::new(tid), - database_id, - vshard_id, - plan: PhysicalPlan::Meta(MetaOp::Cancel { - target_request_id: crate::types::RequestId::new(0), - }), - // no-determinism: ephemeral deadline is not written to Calvin state. - deadline, - priority: crate::bridge::envelope::Priority::Normal, - trace_id: TraceId::ZERO, - consistency: crate::types::ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission, + fn dummy_plan() -> PhysicalPlan { + PhysicalPlan::Meta(MetaOp::Cancel { + target_request_id: RequestId::new(0), }) } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs index 4ccbd8a71..34cffb8f6 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs @@ -10,8 +10,8 @@ use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -use crate::types::{DatabaseId, TenantId, TraceId}; +use super::sub_request::SubRequestScope; use super::undo::UndoEntry; /// Fields for a transactional primary-vector insert (see `VectorOp::Insert`). @@ -329,44 +329,20 @@ impl CoreLoop { Ok(resp) } - /// Execute a read-only / DDL sub-plan via the standard dispatch path. + /// Execute a sub-plan with no undo-tracked arm via the standard dispatch + /// path. No undo entry is recorded. /// - /// None of these variants mutate engine state, so no undo entry is needed. - /// - /// The sub-plan task copies `parent`'s `deadline` and `admission`, so its - /// [`execution_deadline`](crate::bridge::envelope::Request::execution_deadline) - /// equals the parent's. A sub-plan is part of the statement that spawned - /// it, so it runs on that statement's remaining budget. A fresh budget per - /// sub-plan lets a transaction outlive its client's `statement_timeout`. - /// An already-ordered parent has no execution deadline, and neither does - /// its sub-plan. + /// The sub-plan runs under [`SubRequestScope::of`] `parent`: the parent's + /// database, vShard, deadline, and admission. A fresh budget per sub-plan + /// lets a transaction outlive its client's `statement_timeout`. pub(super) fn exec_tx_passthrough( &mut self, tid: u64, plan: &PhysicalPlan, parent: &crate::bridge::envelope::Request, ) -> Result { - let resp = self.execute(&ExecutionTask::new(crate::bridge::envelope::Request { - request_id: crate::types::RequestId::new(0), - tenant_id: TenantId::new(tid), - database_id: DatabaseId::DEFAULT, - vshard_id: crate::types::VShardId::new(0), - plan: plan.clone(), - // no-determinism: sub-plan deadline is ephemeral, not written to WAL - deadline: parent.deadline, - priority: crate::bridge::envelope::Priority::Normal, - trace_id: TraceId::ZERO, - consistency: crate::types::ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: parent.admission, - })); + let request = SubRequestScope::of(parent).request(tid, plan.clone()); + let resp = self.execute(&ExecutionTask::new(request)); if resp.status == Status::Error { return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { detail: "sub-plan execution failed".into(), diff --git a/nodedb/src/data/executor/handlers/transaction/sub_request.rs b/nodedb/src/data/executor/handlers/transaction/sub_request.rs new file mode 100644 index 000000000..3240f092c --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/sub_request.rs @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The request envelope a transaction sub-plan runs under. + +use crate::bridge::envelope::{Admission, PhysicalPlan, Priority, Request}; +use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; + +/// Request fields a sub-plan inherits from the batch that carries it. +pub(super) struct SubRequestScope { + /// Data Plane handlers key storage and collection metadata by database. + pub database_id: DatabaseId, + /// Handlers decide shard-owner side effects by vShard. + pub vshard_id: VShardId, + pub deadline: std::time::Instant, + pub admission: Admission, +} + +impl SubRequestScope { + /// Copy `parent`'s database, vShard, `deadline`, and `admission`. + /// + /// A sub-request's + /// [`execution_deadline`](Request::execution_deadline) then equals the + /// parent's. A sub-plan is part of the statement that spawned it, so it + /// runs on that statement's remaining budget. An already-ordered parent, + /// such as a Calvin apply, has no execution deadline, and neither does + /// its sub-plan. + pub(super) fn of(parent: &Request) -> Self { + Self { + database_id: parent.database_id, + vshard_id: parent.vshard_id, + // no-determinism: sub-plan deadline is ephemeral, not written to WAL + deadline: parent.deadline, + admission: parent.admission, + } + } + + /// Build the request that runs `plan` for tenant `tid` in this scope. + pub(super) fn request(&self, tid: u64, plan: PhysicalPlan) -> Request { + Request { + request_id: RequestId::new(0), + tenant_id: TenantId::new(tid), + database_id: self.database_id, + vshard_id: self.vshard_id, + plan, + // no-determinism: sub-plan deadline is ephemeral, not written to WAL + deadline: self.deadline, + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: self.admission, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::ExemptReason; + use nodedb_physical::physical_plan::MetaOp; + + fn cancel_plan() -> PhysicalPlan { + PhysicalPlan::Meta(MetaOp::Cancel { + target_request_id: RequestId::new(0), + }) + } + + #[test] + fn sub_request_runs_in_parent_database_and_vshard() { + let parent = SubRequestScope { + database_id: DatabaseId::new(7), + vshard_id: VShardId::new(3), + deadline: std::time::Instant::now() + std::time::Duration::from_secs(5), + admission: Admission::Admitted, + } + .request(9, cancel_plan()); + + let sub = SubRequestScope::of(&parent).request(9, PhysicalPlan::Meta(MetaOp::Compact)); + + assert_eq!(sub.database_id, DatabaseId::new(7)); + assert_eq!(sub.vshard_id, VShardId::new(3)); + assert_eq!(sub.tenant_id, TenantId::new(9)); + assert!(matches!(sub.plan, PhysicalPlan::Meta(MetaOp::Compact))); + } + + #[test] + fn sub_request_shares_parent_execution_deadline() { + for admission in [ + Admission::Admitted, + Admission::Exempt(ExemptReason::Read), + Admission::Exempt(ExemptReason::AlreadyOrdered), + ] { + let parent = SubRequestScope { + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + deadline: std::time::Instant::now() + std::time::Duration::from_secs(5), + admission, + } + .request(1, cancel_plan()); + + let sub = SubRequestScope::of(&parent).request(1, cancel_plan()); + + assert_eq!(sub.admission, parent.admission); + assert_eq!(sub.execution_deadline(), parent.execution_deadline()); + } + } +} diff --git a/nodedb/tests/wire/cases/sql_transactions_bulk_dml_overlay.rs b/nodedb/tests/wire/cases/sql_transactions_bulk_dml_overlay.rs index ae347529d..d47a34134 100644 --- a/nodedb/tests/wire/cases/sql_transactions_bulk_dml_overlay.rs +++ b/nodedb/tests/wire/cases/sql_transactions_bulk_dml_overlay.rs @@ -430,3 +430,60 @@ async fn bulk_update_returning_in_txn_is_refused(engine: &str, coll: &str) { async fn bulk_update_returning_in_txn_is_refused_case() { bulk_update_returning_in_txn_is_refused("document_schemaless", "bu_ret_ov_upd").await; } + +/// Sorted `id=n` pairs of the whole collection in the session's database. +async fn id_n_pairs(server: &TestServer, coll: &str) -> Vec { + let mut v: Vec = server + .query_rows(&format!("SELECT id, n FROM {coll}")) + .await + .unwrap_or_else(|e| panic!("read {coll}: {e}")) + .into_iter() + .map(|row| row.join("=")) + .collect(); + v.sort(); + v +} + +/// Predicate DML committed in a named database applies to that database's +/// collection. The same-named collection in `default` keeps its rows. +async fn bulk_dml_commit_applies_in_session_database(engine: &str, coll: &str) { + let server = TestServer::start().await; + setup(&server, coll, engine).await; + server.exec("CREATE DATABASE bulk_dml_named").await.unwrap(); + server.exec("USE DATABASE bulk_dml_named").await.unwrap(); + setup(&server, coll, engine).await; + + server.exec("BEGIN").await.unwrap(); + for sql in [ + format!("UPDATE {coll} SET n = 7 WHERE n = 1"), + format!("DELETE FROM {coll} WHERE n = 2"), + ] { + server + .exec(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + server.exec("COMMIT").await.unwrap(); + + assert_eq!( + id_n_pairs(&server, coll).await, + vec!["a=7", "b=7", "unrelated=100"], + "{engine}: COMMIT applies the predicate DML in the session's database" + ); + server.exec("USE DATABASE default").await.unwrap(); + assert_eq!( + id_n_pairs(&server, coll).await, + vec!["a=1", "b=1", "c=2", "unrelated=100"], + "{engine}: a transaction in another database leaves `default` untouched" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn schemaless_bulk_dml_commit_applies_in_session_database() { + bulk_dml_commit_applies_in_session_database("document_schemaless", "bu_db_s").await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn strict_bulk_dml_commit_applies_in_session_database() { + bulk_dml_commit_applies_in_session_database("document_strict", "bu_db_t").await; +} diff --git a/nodedb/tests/wire/cases/sql_transactions_kv_predicate_overlay.rs b/nodedb/tests/wire/cases/sql_transactions_kv_predicate_overlay.rs index 2d2ec0c4b..9e86d8097 100644 --- a/nodedb/tests/wire/cases/sql_transactions_kv_predicate_overlay.rs +++ b/nodedb/tests/wire/cases/sql_transactions_kv_predicate_overlay.rs @@ -396,3 +396,38 @@ async fn kv_predicate_update_after_truncate_in_transaction_matches_nothing() { "COMMIT replays the truncate and the no-op predicate writes" ); } + +/// Predicate DML committed in a named database applies to that database's +/// collection. The same-named collection in `default` keeps its rows. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn predicate_dml_commit_applies_in_session_database() { + let server = TestServer::start().await; + setup(&server, "kvp_db").await; + server.exec("CREATE DATABASE kvp_named").await.unwrap(); + server.exec("USE DATABASE kvp_named").await.unwrap(); + setup(&server, "kvp_db").await; + + server.exec("BEGIN").await.unwrap(); + for sql in [ + "UPDATE kvp_db SET n = 7 WHERE n = 1", + "DELETE FROM kvp_db WHERE n = 2", + ] { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + server.exec("COMMIT").await.unwrap(); + + assert_eq!( + rows(&server, "kvp_db").await, + pairs(&[("a", "7"), ("b", "7"), ("unrelated", "100")]), + "COMMIT applies the predicate DML in the session's database" + ); + server.exec("USE DATABASE default").await.unwrap(); + assert_eq!( + rows(&server, "kvp_db").await, + pairs(&[("a", "1"), ("b", "1"), ("c", "2"), ("unrelated", "100")]), + "a transaction in another database leaves `default` untouched" + ); +} From a17406cd8f97aaf809a34eeb830ced0688032b19 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 15:09:36 +0800 Subject: [PATCH 07/64] feat(scheduler): halt a vShard's Calvin scheduler on apply failure Every replica must mark a sequenced txn applied only after it applied on this replica, or after an abort every replica reaches identically. A replica-local infrastructure error (dispatch refusal, a disconnected executor response, a failed resolve/flush/stage, a failed identity bind, or a failed WAL append) was neither, but the scheduler had no way to stop rather than silently diverge from its peers. Add a HaltLatch to each Scheduler that records the first such failure, holds the stuck txn's locks and pending entry in place, closes intake via IntakeClosure::ApplyHalted, and stops the deferred re-send and catch-up drain, while continuing to route responses, verdicts, and promotions for every other in-flight txn. The first cause wins and is recorded as a CalvinApplyHalt on the node-wide CalvinApplyHaltMarker (exposed through SequencerHaltMarker::apply_halt), which /healthz and the native status report read the same way they already read a halted sequencer or a wedged metadata applier. A halt during node shutdown is attributed to draining, logs at info instead of error, and sets no node marker. Expose the new gauge reason and step labels as scheduler metrics. --- .../scheduler/driver/core/commit_redo.rs | 116 ++++--- .../driver/core/commit_resolve/apply_tail.rs | 150 +++++++--- .../driver/core/commit_resolve/verdict.rs | 108 ++++++- .../driver/core/commit_resolve/vote.rs | 8 + .../scheduler/driver/core/completion_route.rs | 147 ++++++--- .../calvin/scheduler/driver/core/deferred.rs | 143 +++++++-- .../driver/core/dispatch/active_dispatch.rs | 9 +- .../driver/core/dispatch/bind_identities.rs | 28 +- .../driver/core/dispatch/static_dispatch.rs | 9 +- .../calvin/scheduler/driver/core/halt.rs | 283 ++++++++++++++++++ .../calvin/scheduler/driver/core/intake.rs | 49 ++- .../calvin/scheduler/driver/core/mod.rs | 56 +--- .../calvin/scheduler/driver/core/scheduler.rs | 8 +- .../scheduler/driver/core/test_support.rs | 32 +- .../cluster/calvin/scheduler/driver/types.rs | 6 + .../cluster/calvin/scheduler/metrics.rs | 60 +++- nodedb/src/control/cluster/mod.rs | 21 +- nodedb/src/control/cluster/sequencer_halt.rs | 77 ++++- .../src/control/server/http/routes/health.rs | 96 ++++++ .../control/server/native/session/request.rs | 4 +- nodedb/src/control/state/fields.rs | 8 +- nodedb/src/diag/context/data_plane.rs | 45 +++ nodedb/src/diag/context/mod.rs | 4 +- nodedb/src/diag/mod.rs | 18 +- nodedb/src/diag/recording/data_plane.rs | 28 ++ nodedb/src/diag/recording/mod.rs | 3 +- 26 files changed, 1246 insertions(+), 270 deletions(-) create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index 4b97d6d3c..6a673422d 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -13,10 +13,10 @@ use super::super::types::CommitState; use super::deferred::{DispatchOutcome, DispatchStep}; +use super::halt::{HaltReason, HaltStep, error_response_text}; use super::scheduler::Scheduler; use crate::bridge::envelope::{Response, Status}; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; -use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; use crate::types::VShardId; use crate::wal::{CalvinStamp, RedoRecord}; use nodedb_physical::physical_plan::PhysicalPlan; @@ -27,35 +27,34 @@ impl Scheduler { /// `RedoRecord`, WAL-append it (unless its op set is empty), then dispatch /// the flush stamped with that record's LSN. /// - /// A non-`Ok` response, a decode failure, or a WAL-append failure is a - /// loud infra abort — never a silent fall-through to a non-durable flush. + /// The verdict is already COMMIT, so a skipped resolve would tear the + /// committed txn on this replica. A non-`Ok` response, a decode failure, + /// or a WAL-append failure halts the scheduler: the txn keeps its + /// `pending` entry and locks, and its position stays unapplied. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn finish_redo_resolve( &mut self, txn_id: TxnId, response: Response, ) { if response.status != Status::Ok { - tracing::warn!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - "calvin: CalvinResolve response was not Ok; locks NOT released (shard degraded)" + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + error_response_text("CalvinResolve", &response), ); - self.complete_infra_abort(txn_id); return; } let mut redo = match RedoRecord::from_bytes(response.payload.as_bytes()) { Ok(r) => r, Err(e) => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: CalvinResolve redo record decode failed" + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + format!("CalvinResolve redo record decode failed: {e}"), ); - self.complete_infra_abort(txn_id); return; } }; @@ -86,14 +85,12 @@ impl Scheduler { ) { Ok(lsn) => Some(lsn), Err(e) => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: TransactionRedo WAL append failed" + self.halt_apply( + txn_id, + HaltReason::WalAppendFailed, + HaltStep::RedoAppend, + format!("TransactionRedo WAL append failed: {e}"), ); - self.complete_infra_abort(txn_id); return; } } @@ -116,21 +113,6 @@ impl Scheduler { } } - /// Complete `txn_id` as an infra error: releases its locks so the epoch - /// advances rather than stalling. Shared by every `finish_redo_resolve` - /// failure branch and every terminal resolve, flush, or drop dispatch - /// refusal. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn complete_infra_abort( - &mut self, - txn_id: TxnId, - ) { - self.metrics.record_executor_error(); - self.metrics - .record_infra_abort(infra_abort_reason::IO_ERROR); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } - /// Dispatch `MetaOp::CalvinResolve` to this vShard's core, registering a /// response bridge so the resolve response re-enters the completion loop /// under `CommitState::AwaitingRedoResolve`. @@ -177,3 +159,61 @@ pub(in crate::control::cluster::calvin::scheduler::driver::core) fn missing_pend ), } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + error_response, scheduler_with_pending, staged_response, + }; + + /// A resolve that returns an error under a COMMIT verdict holds the txn + /// unapplied and halts: skipping it would tear the committed txn. + #[tokio::test] + async fn resolve_error_response_holds_committed_txn_unapplied() { + let txn_id = TxnId::new(8, 0); + let (mut scheduler, _dir) = + scheduler_with_pending(txn_id, CommitState::AwaitingRedoResolve); + + scheduler.finish_redo_resolve( + txn_id, + error_response(ErrorCode::Internal { + detail: "resolve failed".to_string(), + }), + ); + + assert!(!scheduler.applied.is_applied(8, 0)); + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::ResolveFailed) + ); + assert!(scheduler.shared.sequencer_halt.apply_halt().is_halted()); + } + + /// A resolve whose redo record does not decode holds the txn unapplied + /// and halts. + #[tokio::test] + async fn undecodable_resolve_payload_holds_committed_txn_unapplied() { + let txn_id = TxnId::new(8, 0); + let (mut scheduler, _dir) = + scheduler_with_pending(txn_id, CommitState::AwaitingRedoResolve); + let mut response = staged_response(Status::Ok, None); + response.payload = Payload::from_vec(vec![0xff, 0x00, 0x13]); + + scheduler.finish_redo_resolve(txn_id, response); + + assert!(!scheduler.applied.is_applied(8, 0)); + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::ResolveFailed) + ); + assert_eq!( + scheduler.pending.get(&txn_id).and_then(|p| p.commit_state), + Some(CommitState::AwaitingRedoResolve), + "no flush is dispatched" + ); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs index d9cd61074..59d3b3c75 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -7,6 +7,9 @@ use nodedb_cluster::calvin::SequencerEntry; use crate::bridge::envelope::{Response, Status}; +use crate::control::cluster::calvin::scheduler::driver::core::halt::{ + HaltReason, HaltStep, error_response_text, +}; use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; @@ -19,7 +22,12 @@ impl Scheduler { /// successful drop only the `CompletionAck` is proposed — the coordinator's /// completion waiter still fires and the epoch advances, but nothing was /// written so there is no result to deposit, no apply LSN, and no versions - /// to record. A non-`Ok` resolve response is treated as an executor error. + /// to record. + /// + /// A non-`Ok` flush halts the scheduler. The flush handler removes the + /// staged buffer before it applies, so a second flush applies nothing, and + /// a skipped flush tears the committed txn on this replica. A non-`Ok` drop + /// completes the txn: under an abort verdict no replica writes anything. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn finish_resolved_commit( &mut self, txn_id: TxnId, @@ -27,52 +35,50 @@ impl Scheduler { committed: bool, redo_lsn: Option, ) { - let completed = if response.status == Status::Ok { + if response.status != Status::Ok { if committed { - self.commit_apply_tail(txn_id, response, redo_lsn) - } else { - self.propose_sequencer_entry( - SequencerEntry::CompletionAck { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }, + self.halt_apply( txn_id, - "completion ack (dropped)", + HaltReason::FlushFailed, + HaltStep::Flush, + error_response_text("CalvinFlush", &response), ); - true + return; } - } else { tracing::error!( vshard_id = self.vshard_id, epoch = txn_id.epoch, position = txn_id.position, - committed, - "calvin: flush/drop response was not Ok while applying an already-committed \ - verdict; forcing infra-abort completion so locks release and the epoch advances" + "calvin: drop response was not Ok under an abort verdict; completing the \ + aborted txn, since no replica writes it" ); - false - }; - - if completed { - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } else { - // The cross-shard verdict is already globally durable, and a commit's - // resolved redo was WAL-appended before this flush — so recovery - // re-applies the write. A local flush/apply or WAL-marker failure is - // therefore an infrastructure event, NOT an outcome change. It must - // never leave the txn parked: holding its locks forever wedges every - // txn queued behind those keys and freezes this vShard's epoch - // watermark (which anchors cross-shard BEGIN snapshots), and nothing - // re-drives a non-`AwaitingVerdict` pending entry. Surface the infra - // abort and force completion — the same forward-progress contract the - // resolve/drop dispatch-failure path in `resume_on_verdict` follows. self.metrics.record_executor_error(); self.metrics .record_infra_abort(infra_abort_reason::IO_ERROR); self.metrics.record_completed(); self.on_txn_complete(txn_id); + return; + } + + let completed = if committed { + self.commit_apply_tail(txn_id, response, redo_lsn) + } else { + self.propose_sequencer_entry( + SequencerEntry::CompletionAck { + epoch: txn_id.epoch, + position: txn_id.position, + vshard_id: self.vshard_id, + }, + txn_id, + "completion ack (dropped)", + ); + true + }; + // `false` means the commit tail halted the scheduler: the txn stays + // pending and unapplied. + if completed { + self.metrics.record_completed(); + self.on_txn_complete(txn_id); } } @@ -82,6 +88,10 @@ impl Scheduler { /// Shared by the flush-completion path and the direct-apply (dependent / /// active) apply path. /// + /// Returns `false` once a failed `CalvinApplied` WAL append halted the + /// scheduler: the position must not be marked applied without its marker, + /// so the caller leaves the txn pending. + /// /// `redo_lsn` is `Some(lsn)` when a `TransactionRedo` record was already /// WAL-appended for this commit's non-empty write set (`finish_redo_resolve`) /// — that record already IS the durable applied marker, so only write @@ -202,12 +212,11 @@ impl Scheduler { Some(applied_lsn) } Err(e) => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: failed to write CalvinApplied WAL record" + self.halt_apply( + txn_id, + HaltReason::WalAppendFailed, + HaltStep::AppliedMarker, + format!("CalvinApplied WAL append failed: {e}"), ); None } @@ -216,7 +225,8 @@ impl Scheduler { let Some(lsn) = applied_lsn else { // The apply cannot be acknowledged without a durable participant // LSN: CDC and write-version consumers would otherwise observe a - // successful commit with no authoritative ordering point. + // successful commit with no authoritative ordering point. The + // scheduler halted above. return false; }; // Control change-stream events are distinct from Data-Plane @@ -249,3 +259,63 @@ impl Scheduler { true } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::ErrorCode; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + error_response, scheduler_with_pending, + }; + use crate::control::cluster::calvin::scheduler::driver::types::CommitState; + + fn internal_error() -> Response { + error_response(ErrorCode::Internal { + detail: "core failed".to_string(), + }) + } + + /// A flush that returns an error under a COMMIT verdict holds the txn + /// unapplied and halts: a second flush would apply nothing. + #[tokio::test] + async fn flush_error_response_holds_committed_txn_unapplied() { + let txn_id = TxnId::new(9, 2); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }, + ); + + scheduler.finish_resolved_commit(txn_id, internal_error(), true, None); + + assert!(!scheduler.applied.is_applied(9, 2)); + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::FlushFailed) + ); + assert!(scheduler.shared.sequencer_halt.apply_halt().is_halted()); + } + + /// A drop that returns an error under an abort verdict still completes + /// the txn: no replica writes an aborted txn. + #[tokio::test] + async fn drop_error_response_under_abort_completes_txn() { + let txn_id = TxnId::new(9, 2); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: false, + redo_lsn: None, + }, + ); + + scheduler.finish_resolved_commit(txn_id, internal_error(), false, None); + + assert!(scheduler.applied.is_applied(9, 2)); + assert!(!scheduler.pending.contains_key(&txn_id)); + assert!(!scheduler.is_apply_halted()); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs index cee99eae4..f9807529c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs @@ -11,6 +11,7 @@ use nodedb_cluster::calvin::VerdictSignal; use crate::control::cluster::calvin::scheduler::driver::core::deferred::{ DispatchOutcome, DispatchStep, }; +use crate::control::cluster::calvin::scheduler::driver::core::halt::{HaltReason, HaltStep}; use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::driver::types::CommitState; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; @@ -33,6 +34,10 @@ impl Scheduler { /// txn is still `Some(AwaitingVerdict)` — if it already transitioned out /// (resolve/drop dispatched, or completed), this is a no-op. This guarantees /// the flush/drop is dispatched exactly once. + /// + /// A COMMIT verdict for a txn this replica failed to stage halts the + /// scheduler: the txn stays parked with its locks, its stall deadline + /// cleared, and its position unapplied. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resume_on_verdict( &mut self, txn_id: TxnId, @@ -47,6 +52,20 @@ impl Scheduler { return; } + if committed + && let Some(pending) = self.pending.get_mut(&txn_id) + && let Some(stage_error) = pending.stage_error.clone() + { + pending.verdict_deadline = None; + self.halt_apply( + txn_id, + HaltReason::LocalStageFailed, + HaltStep::Stage, + format!("COMMIT verdict for a txn this replica did not stage: {stage_error}"), + ); + return; + } + let (outcome, step) = if committed { // Resolve the staged post-images into a replayable `RedoRecord` // first; the redo is WAL-appended (in `finish_redo_resolve`) before @@ -60,9 +79,12 @@ impl Scheduler { ) }; if let DispatchOutcome::Failed(error) = outcome { - // Terminal resolve/drop refusal: complete the txn as an infra error - // so its locks release and the epoch advances rather than stalling. - // The staged buffer is reclaimed by a later drop or on core teardown. + // Terminal resolve/drop refusal: the scheduler halts and holds + // the txn parked with its locks and staged buffer. The cleared + // deadline keeps the stall sweep from re-sending it. + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.verdict_deadline = None; + } self.fail_dispatch_step(txn_id, step, error); return; } @@ -173,10 +195,13 @@ mod tests { use super::*; use crate::bridge::dispatch::CoreChannelDataSide; + use crate::bridge::envelope::ErrorCode; use crate::bridge::envelope::{Payload, Status}; + use crate::control::cluster::calvin::scheduler::driver::core::halt::HaltReason; use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ - await_data_plane_request, build_test_scheduler_with_data_side, fill_tenant_inflight, - make_sequenced_txn, release_filler, spawn_scheduler_loop, staged_pending, staged_response, + await_data_plane_request, build_test_scheduler_with_data_side, error_response, + fill_tenant_inflight, make_sequenced_txn, release_filler, spawn_scheduler_loop, + staged_pending, staged_response, }; use crate::control::state::SharedState; use crate::types::RequestId; @@ -453,4 +478,77 @@ mod tests { assert!(data_side.request_rx.try_pop().is_err()); } } + + /// A follower scheduler whose stage failed, parked on the verdict barrier. + fn follower_with_failed_stage( + txn_id: TxnId, + ) -> (Scheduler, tempfile::TempDir, CoreChannelDataSide) { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, dir, data_side) = build_test_scheduler_with_data_side(7, registry); + assert!( + !scheduler.is_group_leader(), + "the fixture hosts no data group" + ); + scheduler.pending.insert( + txn_id, + staged_pending(make_sequenced_txn(txn_id.epoch, txn_id.position), txn_id), + ); + scheduler.resolve_staged_commit( + txn_id, + &error_response(ErrorCode::Internal { + detail: "stage failed".to_string(), + }), + ); + (scheduler, dir, data_side) + } + + /// A COMMIT verdict for a txn this follower failed to stage halts the + /// scheduler: the leader staged and voted commit, so this replica cannot + /// apply it. No resolve is dispatched, and the stall sweep stops. + #[tokio::test] + async fn commit_verdict_after_local_stage_error_halts_unapplied() { + let txn_id = TxnId::new(14, 2); + let (mut scheduler, _dir, mut data_side) = follower_with_failed_stage(txn_id); + + scheduler.resume_on_verdict(txn_id, true); + + assert!(!scheduler.applied.is_applied(14, 2)); + let pending = scheduler + .pending + .get(&txn_id) + .expect("the txn stays pending"); + assert_eq!(pending.commit_state, Some(CommitState::AwaitingVerdict)); + assert_eq!(pending.verdict_deadline, None); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::LocalStageFailed) + ); + assert!( + data_side.request_rx.try_pop().is_err(), + "no resolve reaches the Data Plane" + ); + } + + /// An abort verdict for a txn this replica failed to stage drops it as + /// usual: every replica reaches the same abort. + #[tokio::test] + async fn abort_verdict_after_local_stage_error_drops_without_halting() { + let txn_id = TxnId::new(14, 2); + let (mut scheduler, _dir, mut data_side) = follower_with_failed_stage(txn_id); + + scheduler.resume_on_verdict(txn_id, false); + + assert!(!scheduler.is_apply_halted()); + let request = data_side + .request_rx + .try_pop() + .expect("the abort dispatches a drop"); + assert!(matches!( + request.inner.plan, + PhysicalPlan::Meta(MetaOp::CalvinDrop { + epoch: 14, + position: 2 + }) + )); + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs index 62564932a..fefb9e9c9 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs @@ -8,6 +8,7 @@ use std::time::Instant; use nodedb_cluster::calvin::SequencerEntry; use crate::bridge::envelope::Response; +use crate::control::cluster::calvin::scheduler::driver::core::halt::error_response_text; use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::driver::core::staged_vote::{ StagedVote, staged_commit_vote, @@ -86,9 +87,16 @@ impl Scheduler { // deadline. Do NOT dispatch resolve/drop here — the GLOBAL verdict, not // this local vote, decides. If the txn already vanished (torn down // elsewhere), there is nothing to park. + // + // A stage error parks too. A deterministic error fails on every + // replica, the leader votes abort, and every replica drops. A local + // error on a follower leaves the leader's commit vote standing: + // `resume_on_verdict` halts on that COMMIT verdict. match self.pending.get_mut(&txn_id) { Some(pending) => { pending.commit_state = Some(CommitState::AwaitingVerdict); + pending.stage_error = (vote == StagedVote::ParticipantError) + .then(|| error_response_text("stage", staged_response)); // no-determinism: local stall-warning deadline only; the global replicated verdict, not this wall-clock, decides commit/abort. pending.verdict_deadline = Some(Instant::now() + self.config.verdict_stall_warn()); } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs index eb0d188d9..d0541b521 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs @@ -13,6 +13,7 @@ use nodedb_cluster::calvin::SequencerEntry; use super::super::types::CommitState; +use super::halt::{HaltReason, HaltStep}; use super::scheduler::Scheduler; use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; @@ -31,20 +32,20 @@ impl Scheduler { let response = match resp_opt { Some(r) => r, None => { - // Bridge task observed a closed channel before any response. - tracing::warn!( - vshard_id = self.vshard_id, - request_id = request_id.as_u64(), - epoch = txn_id.epoch, - position = txn_id.position, - "calvin: executor response channel disconnected" - ); + // The bridge task saw the channel close before any response, + // so the request's outcome on this replica is unknown. Hold + // the txn unapplied and halt. self.metrics.record_executor_error(); - self.metrics.record_infra_abort( - crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason::IO_ERROR, + let state = self.pending.get(&txn_id).and_then(|p| p.commit_state); + self.halt_apply( + txn_id, + HaltReason::ResponseDisconnected, + HaltStep::awaited_by(state), + format!( + "executor response channel for request {} closed before a response", + request_id.as_u64() + ), ); - self.metrics.record_completed(); - self.on_txn_complete(txn_id); return; } }; @@ -157,20 +158,11 @@ impl Scheduler { None => {} } - let completed = if response.status == crate::bridge::envelope::Status::Ok { - // Observe whether the applying participant reported its slice of the - // transaction's reads as no longer current against the local write - // versions. Direct-apply (dependent/active) observation only: the - // staged path folds this into its commit vote instead. `None` means - // no read-set was checked. - if response.read_set_valid == Some(false) { - self.shared - .calvin_counters - .read_set_validation_failures - .fetch_add(1, std::sync::atomic::Ordering::Relaxed); - } - self.commit_apply_tail(txn_id, response, None) - } else { + if response.status != crate::bridge::envelope::Status::Ok { + // A failed direct apply must not leave the txn parked with its locks + // held: that wedges every txn queued behind those keys and freezes + // this vShard's epoch watermark, and no sweep re-drives the entry. + // Surface the infra abort and force completion. tracing::error!( vshard_id = self.vshard_id, epoch = txn_id.epoch, @@ -178,25 +170,102 @@ impl Scheduler { "calvin: executor response was not Ok; forcing infra-abort completion so locks \ release and the epoch advances" ); - false - }; - - if completed { - self.metrics.record_completed(); - self.on_txn_complete(txn_id); - } else { - // A failed direct apply must not leave the txn parked with its locks - // held: that wedges every txn queued behind those keys and freezes - // this vShard's epoch watermark, and no sweep re-drives the entry. - // Surface the infra abort and force completion — the same - // forward-progress contract the disconnected-channel path above - // follows. self.metrics.record_executor_error(); self.metrics.record_infra_abort( crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason::IO_ERROR, ); self.metrics.record_completed(); self.on_txn_complete(txn_id); + return; + } + + // Observe whether the applying participant reported its slice of the + // transaction's reads as no longer current against the local write + // versions. Direct-apply (dependent/active) observation only: the + // staged path folds this into its commit vote instead. `None` means + // no read-set was checked. + if response.read_set_valid == Some(false) { + self.shared + .calvin_counters + .read_set_validation_failures + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); } + // `false` means the commit tail halted the scheduler: the txn stays + // pending and unapplied. + if self.commit_apply_tail(txn_id, response, None) { + self.metrics.record_completed(); + self.on_txn_complete(txn_id); + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::Ordering; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::halt::HaltReason; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + make_sequenced_txn, scheduler_with_pending, staged_pending, + }; + use crate::control::cluster::calvin::scheduler::metrics::apply_halt_reason; + + /// A response channel that closes before a response leaves the txn's + /// outcome unknown: the txn stays pending and unapplied, the scheduler + /// halts, and the node marker names the txn. + #[tokio::test] + async fn disconnected_response_holds_txn_unapplied_and_sets_node_marker() { + let txn_id = TxnId::new(5, 1); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }, + ); + + scheduler.handle_completion(txn_id, RequestId::new(9), None); + + assert!( + !scheduler.applied.is_applied(5, 1), + "an unknown outcome must not mark the position applied" + ); + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::ResponseDisconnected) + ); + let marker = scheduler.shared.sequencer_halt.apply_halt().report(); + assert_eq!( + marker.map(|h| (h.vshard_id, h.epoch, h.position, h.step)), + Some((7, 5, 1, "flush")) + ); + assert_eq!(scheduler.metrics.apply_halted.load(Ordering::Relaxed), 1); + assert_eq!( + scheduler.metrics.apply_halt_reason.load(Ordering::Relaxed), + apply_halt_reason::RESPONSE_DISCONNECTED as u64 + ); + } + + /// A second disconnect keeps the first cause and holds its txn too. + #[tokio::test] + async fn second_halt_keeps_the_first_cause() { + let first = TxnId::new(5, 1); + let second = TxnId::new(6, 0); + let (mut scheduler, _dir) = scheduler_with_pending(first, CommitState::Staged); + let mut pending = staged_pending(make_sequenced_txn(6, 0), second); + pending.commit_state = Some(CommitState::AwaitingRedoResolve); + scheduler.pending.insert(second, pending); + + scheduler.handle_completion(first, RequestId::new(9), None); + scheduler.handle_completion(second, RequestId::new(10), None); + + assert!(!scheduler.applied.is_applied(6, 0)); + assert!(scheduler.pending.contains_key(&second)); + let marker = scheduler.shared.sequencer_halt.apply_halt().report(); + assert_eq!( + marker.map(|h| (h.epoch, h.position, h.step)), + Some((5, 1, "stage")) + ); } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs index 3fc7aea53..37d5faefa 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs @@ -6,7 +6,8 @@ //! every scheduler dispatch goes through [`Scheduler::dispatch_sequenced`]. //! A capacity refusal parks the request in a FIFO. The txn keeps its locks //! and its `pending` entry, and never reaches `on_txn_complete`. The run loop -//! re-sends parked requests once a routed response frees capacity. +//! re-sends parked requests once a routed response frees capacity. A terminal +//! refusal halts the scheduler (see [`super::halt`]). use std::collections::VecDeque; use std::sync::atomic::Ordering; @@ -16,6 +17,7 @@ use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; use tokio::sync::mpsc; +use super::halt::{HaltReason, HaltStep}; use super::scheduler::Scheduler; use crate::bridge::dispatch::DispatchRefusal; use crate::bridge::envelope::{Request, Response}; @@ -103,6 +105,14 @@ impl Scheduler { !self.deferred.is_empty() } + /// Whether the run loop re-sends parked requests: some wait, and the + /// scheduler has not halted. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resends_deferred( + &self, + ) -> bool { + self.has_deferred_dispatch() && !self.is_apply_halted() + } + /// Number of refused requests waiting for capacity. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn deferred_dispatch_len( &self, @@ -113,7 +123,8 @@ impl Scheduler { /// Re-send parked requests in FIFO order. /// /// Stops at the first capacity refusal, which goes back to the FIFO - /// front. A terminal refusal runs the step's terminal handling. + /// front. A terminal refusal runs the step's terminal handling, and stops + /// the pass once the scheduler halted. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn redispatch_deferred( &mut self, ) { @@ -140,7 +151,12 @@ impl Scheduler { self.deferred.push_front(*parked); break; } - Attempt::Failed(error) => self.fail_dispatch_step(txn_id, step, error), + Attempt::Failed(error) => { + self.fail_dispatch_step(txn_id, step, error); + if self.is_apply_halted() { + break; + } + } } } self.metrics @@ -150,38 +166,22 @@ impl Scheduler { /// Run the terminal handling of a dispatch the dispatcher refused for a /// reason other than capacity. /// - /// Stage steps release the txn's locks. Resolve, flush, and drop steps - /// complete the txn as an infra abort. A write-version record is dropped - /// with a warning, because the commit does not depend on it. + /// A stage, resolve, flush, or drop refusal halts the scheduler: the txn + /// keeps its `pending` entry and locks, and its position stays unapplied. + /// A refusal during shutdown holds the txn the same way. A write-version + /// record is dropped with a warning, because the commit does not depend on + /// it. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn fail_dispatch_step( &mut self, txn_id: TxnId, step: DispatchStep, error: crate::Error, ) { - match step { - DispatchStep::StageStatic | DispatchStep::StageActive => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - ?step, - %error, - "calvin scheduler: dispatch failed; releasing locks" - ); - self.on_txn_complete(txn_id); - } - DispatchStep::Resolve | DispatchStep::Flush | DispatchStep::Drop => { - tracing::error!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - ?step, - %error, - "calvin: commit resolution dispatch failed" - ); - self.complete_infra_abort(txn_id); - } + let halt_step = match step { + DispatchStep::StageStatic | DispatchStep::StageActive => HaltStep::Stage, + DispatchStep::Resolve => HaltStep::Resolve, + DispatchStep::Flush => HaltStep::Flush, + DispatchStep::Drop => HaltStep::Drop, DispatchStep::WriteVersionRecord => { tracing::warn!( vshard_id = self.vshard_id, @@ -190,8 +190,15 @@ impl Scheduler { %error, "calvin: write-version record dispatch failed" ); + return; } - } + }; + self.halt_apply( + txn_id, + HaltReason::DispatchRefused, + halt_step, + error.to_string(), + ); } /// One send attempt: register, dispatch, and cancel on refusal. @@ -289,3 +296,79 @@ impl Scheduler { } } } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_cluster::calvin::types::SchedulerInput; + use nodedb_types::TenantId; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::halt::HaltReason; + use crate::control::cluster::calvin::scheduler::driver::core::intake::IntakeClosure; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + begin_data_plane_drain, build_test_scheduler_with_data_side, fill_tenant_inflight, + make_validate_only_txn, test_coll_vshard, + }; + + /// A stage refused because the Data Plane drains stays pending and + /// unapplied, keeps its locks, and closes intake. Shutdown sets no node + /// marker. + #[tokio::test] + async fn stage_refused_while_draining_is_held_unapplied_without_node_marker() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, _data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + begin_data_plane_drain(&scheduler.shared); + let txn_id = TxnId::new(3, 0); + + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(4, 0))); + + assert!( + !scheduler.applied.is_applied(3, 0), + "a terminal refusal must not mark the position applied" + ); + assert!( + scheduler.pending.contains_key(&txn_id), + "the refused txn keeps its pending entry" + ); + assert!( + scheduler.blocked.contains_key(&TxnId::new(4, 0)), + "the refused txn keeps its key locks" + ); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::Draining) + ); + assert_eq!(scheduler.intake_closure(), Some(IntakeClosure::ApplyHalted)); + assert!( + !scheduler.shared.sequencer_halt.apply_halt().is_halted(), + "a shutdown halt sets no node marker" + ); + } + + /// A parked stage whose re-send is refused terminally stays pending and + /// unapplied, and the scheduler stops re-sending. + #[tokio::test] + async fn parked_stage_refused_terminally_on_resend_is_held_unapplied() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let shared = Arc::clone(&scheduler.shared); + fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); + let txn_id = TxnId::new(3, 0); + scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + assert!(scheduler.has_deferred_dispatch(), "the stage parks"); + + begin_data_plane_drain(&shared); + scheduler.redispatch_deferred(); + + assert!(!scheduler.applied.is_applied(3, 0)); + assert!(scheduler.pending.contains_key(&txn_id)); + assert!(scheduler.is_apply_halted()); + assert!(!scheduler.resends_deferred()); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs index 13a93c247..59e7e809c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs @@ -92,13 +92,7 @@ impl Scheduler { return; } }; - if !self.bind_local_identities( - &mut plans, - txn.tx_class.database_id, - tenant_id, - txn_id, - lock_owner, - ) { + if !self.bind_local_identities(&mut plans, txn.tx_class.database_id, tenant_id, txn_id) { return; } let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); @@ -140,6 +134,7 @@ impl Scheduler { commit_state: Some(super::super::super::types::CommitState::Staged), // Set only once the txn parks in `AwaitingVerdict`. verdict_deadline: None, + stage_error: None, }, ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs index e00a6202d..9481dc9f2 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/bind_identities.rs @@ -8,8 +8,8 @@ //! or a later point read by primary key resolves nothing on this node. use nodedb_physical::physical_plan::PhysicalPlan; -use tracing::error; +use super::super::halt::{HaltReason, HaltStep}; use super::super::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::control::surrogate::bind_plan_identities; @@ -18,19 +18,18 @@ use crate::types::{DatabaseId, TenantId}; impl Scheduler { /// Bind every identity in `plans` first-wins and rewrite each surrogate /// slot with the authoritative value, the same walk the replicated-write - /// decoder runs. On a catalog error the txn is terminated as a routing - /// failure: applying rows nobody can resolve by key is worse than aborting. + /// decoder runs. /// - /// Returns `false` after terminating the txn; the caller returns at once. - /// The txn is not yet in `pending`, so its locks release under - /// `lock_owner`. + /// A catalog error is local to this replica, and its peers apply the + /// slice. So the scheduler halts: the txn is not yet in `pending`, its + /// locks stay held under its lock owner, and its position stays + /// unapplied. Returns `false` after the halt; the caller returns at once. pub(super) fn bind_local_identities( &mut self, plans: &mut [PhysicalPlan], database_id: DatabaseId, tenant_id: TenantId, txn_id: TxnId, - lock_owner: TxnId, ) -> bool { let assigner = &self.shared.surrogate_assigner; let bound = plans @@ -39,17 +38,12 @@ impl Scheduler { match bound { Ok(()) => true, Err(e) => { - let epoch = txn_id.epoch; - let position = txn_id.position; - error!( - vshard_id = self.vshard_id, - epoch, - position, - error = %e, - "calvin scheduler: surrogate binding failed; releasing locks" + self.halt_apply( + txn_id, + HaltReason::IdentityBindFailed, + HaltStep::IdentityBind, + format!("surrogate binding failed: {e}"), ); - self.propose_routing_failure(epoch, position, txn_id, &e); - self.on_unpending_txn_complete(txn_id, lock_owner); false } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs index df973a1a9..306cd9614 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs @@ -170,13 +170,7 @@ impl Scheduler { return; } }; - if !self.bind_local_identities( - &mut local, - txn.tx_class.database_id, - tenant_id, - txn_id, - lock_owner, - ) { + if !self.bind_local_identities(&mut local, txn.tx_class.database_id, tenant_id, txn_id) { return; } @@ -291,6 +285,7 @@ impl Scheduler { commit_state: Some(super::super::super::types::CommitState::Staged), // Set only once the txn parks in `AwaitingVerdict`. verdict_deadline: None, + stage_error: None, }, ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs new file mode 100644 index 000000000..d37b8e6bb --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs @@ -0,0 +1,283 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply halt for one Calvin scheduler. +//! +//! Every replica applies every sequenced txn. A position is marked applied +//! only after it applied on this replica, or after an abort that every +//! replica reaches identically. A replica-local infrastructure error is +//! neither: the txn's effect on this replica is unknown or missing. So the +//! scheduler halts and never marks that position applied. +//! +//! A halted scheduler: +//! - keeps the stuck txn unapplied, with its locks and its `pending` entry; +//! - closes intake with [`IntakeClosure::ApplyHalted`]. The fan-out then drops +//! new input and arms catch-up, which keeps the sequencer log retained; +//! - stops the deferred re-send and the catch-up drain; +//! - keeps routing responses, verdicts, and promotions for other in-flight +//! txns. Each one holds locks disjoint from the stuck txn, so its outcome +//! does not depend on it. Peer vShards wait on its votes, and the per-position +//! applied gate keeps restart replay exact. A later infrastructure error on +//! another txn holds that txn the same way. +//! +//! The first cause wins. It logs one `error!` line, sets the +//! `nodedb_calvin_apply_halted` gauge, and records the node-wide +//! [`CalvinApplyHaltMarker`] that `/healthz` and the native status report. +//! While the node shuts down (the Data Plane drains), the scheduler holds the +//! same way, logs at info, and sets no node marker: the process is exiting. +//! +//! [`IntakeClosure::ApplyHalted`]: super::intake::IntakeClosure::ApplyHalted +//! [`CalvinApplyHaltMarker`]: crate::control::cluster::CalvinApplyHaltMarker + +use tracing::{debug, error, info, warn}; + +use super::super::types::CommitState; +use super::scheduler::Scheduler; +use crate::bridge::envelope::Response; +use crate::control::cluster::CalvinApplyHalt; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::cluster::calvin::scheduler::metrics::apply_halt_reason; + +/// Why a scheduler halted. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum HaltReason { + /// The node shuts down and its Data Plane drains. + Draining, + /// The dispatcher refused a request for a reason other than capacity. + DispatchRefused, + /// The executor response channel closed before a response arrived. + ResponseDisconnected, + /// The resolve of a committed txn failed or returned an undecodable record. + ResolveFailed, + /// The flush of a committed txn returned an error. + FlushFailed, + /// This replica failed to stage a txn whose verdict is COMMIT. + LocalStageFailed, + /// The surrogate catalog refused a coordinator-assigned identity. + IdentityBindFailed, + /// A redo or `CalvinApplied` WAL append failed. + WalAppendFailed, +} + +impl HaltReason { + /// The `nodedb_calvin_apply_halted` reason index. + fn metric_reason(self) -> usize { + match self { + Self::Draining => apply_halt_reason::DRAINING, + Self::DispatchRefused => apply_halt_reason::DISPATCH_REFUSED, + Self::ResponseDisconnected => apply_halt_reason::RESPONSE_DISCONNECTED, + Self::ResolveFailed => apply_halt_reason::RESOLVE_FAILED, + Self::FlushFailed => apply_halt_reason::FLUSH_FAILED, + Self::LocalStageFailed => apply_halt_reason::LOCAL_STAGE_FAILED, + Self::IdentityBindFailed => apply_halt_reason::IDENTITY_BIND_FAILED, + Self::WalAppendFailed => apply_halt_reason::WAL_APPEND_FAILED, + } + } + + /// The reason label on the gauge and the node marker. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn label( + self, + ) -> &'static str { + apply_halt_reason::LABELS[self.metric_reason()] + } +} + +/// The sub-operation of the stuck txn that failed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum HaltStep { + /// The stage request, or the stage result under a COMMIT verdict. + Stage, + /// The resolve request of a committed txn. + Resolve, + /// The flush request of a committed txn. + Flush, + /// The drop request of an aborted txn. + Drop, + /// The direct apply of a txn with no commit state. + Apply, + /// The `TransactionRedo` WAL append. + RedoAppend, + /// The `CalvinApplied` WAL append. + AppliedMarker, + /// The surrogate identity binding before the stage dispatch. + IdentityBind, +} + +impl HaltStep { + /// The step label on the node marker and the log line. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn label( + self, + ) -> &'static str { + match self { + Self::Stage => "stage", + Self::Resolve => "resolve", + Self::Flush => "flush", + Self::Drop => "drop", + Self::Apply => "apply", + Self::RedoAppend => "redo_append", + Self::AppliedMarker => "applied_marker", + Self::IdentityBind => "identity_bind", + } + } + + /// The request a txn in `state` awaits a response for. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn awaited_by( + state: Option, + ) -> Self { + match state { + Some(CommitState::Staged | CommitState::AwaitingVerdict) => Self::Stage, + Some(CommitState::AwaitingRedoResolve) => Self::Resolve, + Some(CommitState::AwaitingResolve { + committed: true, .. + }) => Self::Flush, + Some(CommitState::AwaitingResolve { + committed: false, .. + }) => Self::Drop, + None => Self::Apply, + } + } +} + +/// The first halt of one scheduler. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct ApplyHalt { + pub reason: HaltReason, + /// The stuck txn, step, and error text, in the node marker's shape. + pub report: CalvinApplyHalt, +} + +/// First-cause-wins halt latch of one scheduler. Never clears. +#[derive(Debug, Default)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct HaltLatch { + first: Option, +} + +/// Error text for an executor response that was not `Ok`. +pub(in crate::control::cluster::calvin::scheduler::driver::core) fn error_response_text( + request: &str, + response: &Response, +) -> String { + format!( + "{request} returned {:?} with error code {:?}", + response.status, + response.error_code.as_deref() + ) +} + +impl Scheduler { + /// Whether this scheduler halted. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn is_apply_halted( + &self, + ) -> bool { + self.halt.first.is_some() + } + + /// The first halt, if this scheduler halted. + #[cfg(test)] + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn apply_halt( + &self, + ) -> Option<&ApplyHalt> { + self.halt.first.as_ref() + } + + /// Hold `txn_id` unapplied and halt this scheduler. + /// + /// The caller leaves the txn's `pending` entry and locks in place and + /// never calls `on_txn_complete` for it. A halt while the node shuts down + /// records [`HaltReason::Draining`] whatever `reason` says. A second halt + /// only logs: the first cause stays. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn halt_apply( + &mut self, + txn_id: TxnId, + reason: HaltReason, + step: HaltStep, + error: String, + ) { + let reason = if self.node_shutting_down() { + HaltReason::Draining + } else { + reason + }; + if let Some(first) = &self.halt.first { + if first.reason == HaltReason::Draining || reason == HaltReason::Draining { + debug!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + step = step.label(), + %error, + "calvin scheduler: shutting down; holding txn unapplied" + ); + } else { + warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + reason = reason.label(), + step = step.label(), + %error, + halted_epoch = first.report.epoch, + halted_position = first.report.position, + "calvin scheduler: apply halted; holding another txn unapplied" + ); + } + return; + } + + let report = CalvinApplyHalt { + vshard_id: self.vshard_id, + epoch: txn_id.epoch, + position: txn_id.position, + reason: reason.label(), + step: step.label(), + error, + }; + self.metrics.set_apply_halted(reason.metric_reason()); + if reason == HaltReason::Draining { + info!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + step = report.step, + error = %report.error, + "calvin scheduler: shutting down; holding txn unapplied and closing intake" + ); + } else { + error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + reason = report.reason, + step = report.step, + error = %report.error, + "calvin scheduler: apply halted; txn held unapplied with its locks, \ + intake closed until restart" + ); + crate::diag::calvin_apply_halted( + report.vshard_id, + report.epoch, + report.position, + report.reason, + report.step, + &report.error, + ); + self.shared + .sequencer_halt + .apply_halt() + .record(report.clone()); + } + self.halt.first = Some(ApplyHalt { reason, report }); + } + + /// Whether the node shuts down: the shutdown watch fired, or the + /// dispatcher closed its Data Plane enqueue gate. + fn node_shutting_down(&self) -> bool { + if self.shared.shutdown.is_shutdown() { + return true; + } + self.shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_data_plane_draining() + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs index a5c1c1c60..22dcbee8c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs @@ -3,8 +3,9 @@ //! Intake gate for the Calvin scheduler. //! //! The scheduler takes new sequenced input only while it can make progress. -//! The gate closes while a capacity-refused dispatch waits for re-send, or -//! while the in-flight backlog sits at [`SchedulerConfig::max_inflight_backlog`]. +//! The gate closes for good once the scheduler halts (see [`super::halt`]). +//! It closes while a capacity-refused dispatch waits for re-send, or while +//! the in-flight backlog sits at [`SchedulerConfig::max_inflight_backlog`]. //! A closed gate disables the run loop's receiver arm and skips the catch-up //! drain. Completions, verdicts, read results, promotions, and capacity //! wakeups stay active, because they drain the backlog. @@ -26,6 +27,8 @@ use crate::control::cluster::calvin::scheduler::metrics::intake_closure_reason; /// Why the intake gate is closed. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(in crate::control::cluster::calvin::scheduler::driver::core) enum IntakeClosure { + /// The scheduler halted on a txn it cannot mark applied. + ApplyHalted, /// A capacity-refused dispatch waits in the deferred FIFO. DeferredDispatch, /// The in-flight backlog is at its bound, and some of it progresses @@ -37,6 +40,7 @@ impl IntakeClosure { /// The `nodedb_calvin_intake_gate_closed_total` reason index. fn metric_reason(self) -> usize { match self { + Self::ApplyHalted => intake_closure_reason::APPLY_HALTED, Self::DeferredDispatch => intake_closure_reason::DEFERRED_DISPATCH, Self::BacklogFull => intake_closure_reason::BACKLOG_FULL, } @@ -66,6 +70,9 @@ impl Scheduler { pub(in crate::control::cluster::calvin::scheduler::driver::core) fn intake_closure( &self, ) -> Option { + if self.is_apply_halted() { + return Some(IntakeClosure::ApplyHalted); + } if self.has_deferred_dispatch() { return Some(IntakeClosure::DeferredDispatch); } @@ -137,6 +144,7 @@ mod tests { }; use crate::control::cluster::calvin::scheduler::driver::types::BlockedTxn; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + use crate::types::RequestId; /// How long a held input must stay unread. Several liveness ticks of the /// spawned loop fit in it. @@ -282,4 +290,41 @@ mod tests { 1 ); } + + /// A halted scheduler closes intake with `ApplyHalted`, and the run loop + /// leaves new input unread. + #[tokio::test] + async fn halted_scheduler_closes_intake_and_reads_no_input() { + let registry = CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, _data_side) = + build_test_scheduler_with_data_side(test_coll_vshard(), registry); + let held = TxnId::new(3, 0); + scheduler + .pending + .insert(held, staged_pending(make_validate_only_txn(3, 0), held)); + scheduler.handle_completion(held, RequestId::new(9), None); + assert_eq!(scheduler.intake_closure(), Some(IntakeClosure::ApplyHalted)); + let metrics = Arc::clone(&scheduler.metrics); + + let running = spawn_scheduler_loop(scheduler); + running + .input_tx() + .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .await + .expect("the loop's input channel is open"); + + let read_while_halted = inputs_consumed_within(running.input_tx(), HOLD_WAIT).await; + running.stop().await; + + assert!( + !read_while_halted, + "the loop must not read input once the scheduler halted" + ); + assert_eq!(metrics.intake_gate_closed.load(Ordering::Relaxed), 1); + assert_eq!( + metrics.intake_gate_closed_counts[intake_closure_reason::APPLY_HALTED] + .load(Ordering::Relaxed), + 1 + ); + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index 140908580..5cbe3f58c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -3,51 +3,16 @@ //! Calvin scheduler driver core. //! //! One [`Scheduler`] task runs per vshard hosted on this node. It receives -//! [`SequencedTxn`]s from the sequencer, acquires deterministic locks, -//! dispatches static / dependent-read transactions to the Data Plane, -//! waits for executor responses, and writes `CalvinApplied` WAL records. -//! -//! Sub-modules (one concern per file): -//! -//! - [`scheduler`] — `Scheduler` struct, ctor, run loop. -//! - [`completion_route`] — routes each executor response (disconnect, OLLP -//! mismatch, staged commit-resolution state, or direct apply) to its handler. -//! - [`process`] — new-txn processing, dependent-read barrier setup, -//! txn-completion bookkeeping. -//! - [`catch_up`] — sequencer-fan-out catch-up drain: replays inputs dropped on -//! this replica (channel Full/Closed) from the committed sequencer Raft log. -//! - [`intake`] — intake gate: stops reading new sequenced input while a -//! dispatch is deferred or the in-flight backlog is at its bound. -//! - [`dispatch`] — static / active dispatch to the Data Plane executor. -//! - [`deferred`] — capacity-safe dispatch: parks a request the bridge refuses -//! at capacity and re-sends it once capacity frees. -//! - [`routing`] — exhaustive `PhysicalPlan` → vshard routing oracle used by -//! `dispatch`'s local-plan filtering. -//! - [`commit_resolve`] — verdict-driven flush-or-drop of a staged static -//! transaction, plus the shared commit tail. -//! - [`staged_vote`] — derives a participant's local commit vote from its -//! staged executor response, keeping the two abort causes apart. -//! - [`commit_redo`] — resolves a committed staged transaction's post-images -//! into a replayable `TransactionRedo` WAL record ahead of the flush. -//! - [`read_result`] — `CalvinReadResult` handling and barrier timeouts. -//! - [`propose`] — propose `CalvinReadResult` Raft entries. -//! - [`request`] — shared `Request` construction for already-sequenced Calvin -//! sub-operations. -//! - [`write_version_record`] — post-apply write-version recording for -//! committed Calvin transactions (at the CalvinApplied WAL LSN). -//! -//! # Determinism -//! -//! All bookkeeping uses `BTreeMap`/`BTreeSet` — never `HashMap`/`HashSet`. -//! Dispatch order is `(epoch, position)` order. -//! -//! # Timing / `Instant::now()` -//! -//! `Instant::now()` is used for: -//! - Lock-wait latency metrics (observability only). -//! - Dependent-read barrier `timeout_at` (off-WAL path only). -//! -//! Never used for WAL-influencing values. +//! sequenced txns from the sequencer, acquires deterministic locks, dispatches +//! static / dependent-read transactions to the Data Plane, waits for executor +//! responses, and writes `CalvinApplied` WAL records. Each sub-module owns one +//! concern; see that file's own doc comment for what it does. +//! +//! All bookkeeping uses `BTreeMap`/`BTreeSet` — never `HashMap`/`HashSet` — +//! and dispatch order is `(epoch, position)` order. `Instant::now()` is used +//! only for lock-wait latency metrics and the dependent-read barrier +//! `timeout_at`, both off the WAL-influencing path; every call site carries a +//! `// no-determinism:` marker. pub mod catch_up; pub mod commit_redo; @@ -56,6 +21,7 @@ pub mod commit_resolve; pub mod completion_route; pub mod deferred; pub mod dispatch; +pub mod halt; pub mod intake; pub mod process; pub mod propose; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index 5bf93087f..6b869f713 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -20,6 +20,7 @@ use super::super::config::SchedulerConfig; use super::super::types::{BlockedTxn, PendingTxn}; use super::catch_up::CatchUpDrain; use super::deferred::DeferredQueue; +use super::halt::HaltLatch; use super::intake::IntakeGate; use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; @@ -157,6 +158,8 @@ pub struct Scheduler { pub(in crate::control::cluster::calvin::scheduler::driver::core) capacity_freed: Arc, /// Last observed intake gate state. See [`super::intake`]. pub(in crate::control::cluster::calvin::scheduler::driver::core) intake: IntakeGate, + /// First halt cause, once set. See [`super::halt`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) halt: HaltLatch, } /// Parameters for [`Scheduler::new`]. @@ -249,6 +252,7 @@ impl Scheduler { deferred: DeferredQueue::new(), capacity_freed, intake: IntakeGate::default(), + halt: HaltLatch::default(), } } @@ -350,7 +354,7 @@ impl Scheduler { let capacity_notified = capacity_freed.notified(); tokio::pin!(capacity_notified); capacity_notified.as_mut().enable(); - if self.has_deferred_dispatch() { + if self.resends_deferred() { self.redispatch_deferred(); } @@ -402,7 +406,7 @@ impl Scheduler { } } - _ = &mut capacity_notified, if self.has_deferred_dispatch() => { + _ = &mut capacity_notified, if self.resends_deferred() => { // Capacity freed: the next loop pass re-sends deferred // requests in FIFO order. } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs index 123330f76..0c46a9ca4 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -20,7 +20,7 @@ use tokio::sync::mpsc; use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; use crate::bridge::envelope::{ - Admission, ExemptReason, Payload, Priority, Request, Response, Status, + Admission, ErrorCode, ExemptReason, Payload, Priority, Request, Response, Status, }; use crate::control::cluster::calvin::scheduler::driver::barrier::ReadResultEvent; use crate::control::cluster::calvin::scheduler::driver::core::scheduler::{ @@ -424,6 +424,7 @@ pub(super) fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { change_sets: Vec::new(), commit_state: Some(CommitState::Staged), verdict_deadline: None, + stage_error: None, } } @@ -442,3 +443,32 @@ pub(super) fn staged_response(status: Status, read_set_valid: Option) -> R write_set: Vec::new(), } } + +/// An executor `Response` with `Status::Error` carrying `code`. +pub(super) fn error_response(code: ErrorCode) -> Response { + let mut response = staged_response(Status::Error, None); + response.error_code = Some(Box::new(code)); + response +} + +/// Close the dispatcher's Data Plane enqueue gate, as a node shutdown does. +/// Every later dispatch is refused terminally. +pub(super) fn begin_data_plane_drain(shared: &SharedState) { + shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .begin_data_plane_drain(); +} + +/// A scheduler on vShard 7 with `txn_id` pending in `state`. +pub(super) fn scheduler_with_pending( + txn_id: TxnId, + state: CommitState, +) -> (Scheduler, tempfile::TempDir) { + let (mut scheduler, dir) = build_test_scheduler(7); + let mut pending = staged_pending(make_sequenced_txn(txn_id.epoch, txn_id.position), txn_id); + pending.commit_state = Some(state); + scheduler.pending.insert(txn_id, pending); + (scheduler, dir) +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs index 64c65d63d..af7e39652 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs @@ -64,6 +64,12 @@ pub(super) struct PendingTxn { /// `Instant::now()` is used for this deadline (observability / liveness /// only; never influences WAL bytes). pub verdict_deadline: Option, + /// Error text of a stage response that was not `Ok` on this replica. + /// + /// `Some` when this replica never staged the txn. An abort verdict drops it + /// as usual. A COMMIT verdict halts the scheduler, because the txn cannot + /// apply here while its peers apply it. + pub stage_error: Option, } /// Commit-resolution state of a staged static Calvin transaction. diff --git a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs index 1bc1e68dd..7a70d101b 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs @@ -70,7 +70,12 @@ pub struct SchedulerMetrics { pub intake_backlog: AtomicU64, /// Intake gate closures by reason. Indexes are the constants in /// [`intake_closure_reason`]. - pub intake_gate_closed_counts: [AtomicU64; 2], + pub intake_gate_closed_counts: [AtomicU64; 3], + /// Apply halt state: 1 once the scheduler halted, 0 while it applies. + pub apply_halted: AtomicU64, + /// Reason of the halt. An index into [`apply_halt_reason`], read only + /// while `apply_halted` is 1. + pub apply_halt_reason: AtomicU64, } /// Reason codes for `nodedb_calvin_infra_abort_total`. @@ -96,8 +101,32 @@ pub mod infra_abort_reason { pub mod intake_closure_reason { pub const DEFERRED_DISPATCH: usize = 0; pub const BACKLOG_FULL: usize = 1; + pub const APPLY_HALTED: usize = 2; - pub const LABELS: &[&str] = &["deferred_dispatch", "backlog_full"]; + pub const LABELS: &[&str] = &["deferred_dispatch", "backlog_full", "apply_halted"]; +} + +/// Reason codes for `nodedb_calvin_apply_halted`. +pub mod apply_halt_reason { + pub const DRAINING: usize = 0; + pub const DISPATCH_REFUSED: usize = 1; + pub const RESPONSE_DISCONNECTED: usize = 2; + pub const RESOLVE_FAILED: usize = 3; + pub const FLUSH_FAILED: usize = 4; + pub const LOCAL_STAGE_FAILED: usize = 5; + pub const IDENTITY_BIND_FAILED: usize = 6; + pub const WAL_APPEND_FAILED: usize = 7; + + pub const LABELS: &[&str] = &[ + "draining", + "dispatch_refused", + "response_disconnected", + "resolve_failed", + "flush_failed", + "local_stage_failed", + "identity_bind_failed", + "wal_append_failed", + ]; } impl SchedulerMetrics { @@ -182,6 +211,15 @@ impl SchedulerMetrics { .store(u64::from(closed), Ordering::Relaxed); } + /// Set the apply halt gauge for `reason`. + /// + /// `reason` must be one of the constants in [`apply_halt_reason`]. + pub fn set_apply_halted(&self, reason: usize) { + self.apply_halt_reason + .store(reason as u64, Ordering::Relaxed); + self.apply_halted.store(1, Ordering::Relaxed); + } + /// Set the in-flight backlog gauge. pub fn set_intake_backlog(&self, backlog: usize) { self.intake_backlog.store(backlog as u64, Ordering::Relaxed); @@ -404,6 +442,22 @@ impl SchedulerMetrics { ); } + let _ = writeln!( + out, + "# HELP nodedb_calvin_apply_halted \ + 1 once the scheduler halted on an apply error it cannot mark applied, by reason." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_apply_halted gauge"); + let halted = self.apply_halted.load(Ordering::Relaxed) == 1; + let halt_reason = self.apply_halt_reason.load(Ordering::Relaxed); + for (i, &reason_label) in apply_halt_reason::LABELS.iter().enumerate() { + let value = u64::from(halted && halt_reason == i as u64); + let _ = writeln!( + out, + "nodedb_calvin_apply_halted{{{label},reason=\"{reason_label}\"}} {value}" + ); + } + out } } @@ -428,6 +482,8 @@ impl Default for SchedulerMetrics { intake_gate_closed: AtomicU64::new(0), intake_backlog: AtomicU64::new(0), intake_gate_closed_counts: std::array::from_fn(|_| AtomicU64::new(0)), + apply_halted: AtomicU64::new(0), + apply_halt_reason: AtomicU64::new(0), } } } diff --git a/nodedb/src/control/cluster/mod.rs b/nodedb/src/control/cluster/mod.rs index c704618ff..547f495a5 100644 --- a/nodedb/src/control/cluster/mod.rs +++ b/nodedb/src/control/cluster/mod.rs @@ -2,23 +2,8 @@ //! Cluster mode startup and integration. //! -//! Bridges `nodedb-cluster` (Raft, transport, routing, metadata group) -//! into the main server. Split into one concern per file: -//! -//! - [`init`] — cluster startup (transport, catalog, bootstrap/join/restart). -//! - [`start_raft`] — Raft event loop + RPC server + applier wiring. -//! - [`handle`] — the `ClusterHandle` passed between init and start_raft. -//! - [`core_stall`] — samples every Data Plane core's event-loop -//! liveness counter and marks the cores that stopped advancing. -//! - [`decommission_bridge`] — drives `nodedb-cluster`'s per-node -//! decommission signal into this process's `ShutdownWatch`. -//! - [`spsc_applier`] — committed data-group entries → SPSC bridge. -//! - [`metadata_applier`] — committed metadata-group entries → -//! `MetadataCache` + optional redb writeback. The per-Raft-group -//! apply watermark watchers themselves now live in -//! [`nodedb_cluster::GroupAppliedWatchers`] and are bumped from -//! the Raft tick loop so every group (metadata + data) shares one -//! primitive. +//! Bridges `nodedb-cluster` (Raft, transport, routing, metadata group) into +//! the main server. Each sub-module documents its own concern. pub mod array_cluster_exec; pub mod array_cluster_helpers; @@ -54,7 +39,7 @@ pub use init::{init_cluster, init_cluster_with_transport, init_single_node_calvi pub use metadata_applier::MetadataCommitApplier; pub use read_index::{MultiRaftReadGate, RaftReadGate, ReadIndexRefusal}; pub use recovery_check::{VerifyReport, verify_and_repair}; -pub use sequencer_halt::SequencerHaltMarker; +pub use sequencer_halt::{CalvinApplyHalt, CalvinApplyHaltMarker, SequencerHaltMarker}; pub use spsc_applier::SpscCommitApplier; pub use start_raft::start_raft; pub use tls::resolve_credentials; diff --git a/nodedb/src/control/cluster/sequencer_halt.rs b/nodedb/src/control/cluster/sequencer_halt.rs index 279c5d2a2..ff4d9e466 100644 --- a/nodedb/src/control/cluster/sequencer_halt.rs +++ b/nodedb/src/control/cluster/sequencer_halt.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Node-wide marker for a halted Calvin sequencer state machine. +//! Node-wide markers for a halted Calvin sequencer state machine and a halted +//! Calvin scheduler. //! //! The sequencer stops applying epoch batches when a NEW committed entry //! re-mints an epoch this replica already consumed — the one divergence that @@ -16,6 +17,11 @@ //! fast instead of hanging, and this marker makes the degradation visible on the //! same surfaces a wedged metadata applier uses, so it can never be mistaken for //! a healthy node. +//! +//! A Calvin scheduler halts one vShard when a replica-local error leaves a +//! sequenced txn neither applied nor identically aborted on this replica. The +//! same scoping holds: the node keeps serving, and [`CalvinApplyHaltMarker`] +//! makes the lost vShard visible on the same surfaces. use std::sync::OnceLock; @@ -29,6 +35,7 @@ use nodedb_cluster::calvin::SequencerHalt; #[derive(Debug, Default)] pub struct SequencerHaltMarker { halt: OnceLock, + apply: CalvinApplyHaltMarker, } impl SequencerHaltMarker { @@ -45,6 +52,49 @@ impl SequencerHaltMarker { pub fn is_halted(&self) -> bool { self.halt.get().is_some() } + + /// The node-wide marker for halted Calvin schedulers. + pub fn apply_halt(&self) -> &CalvinApplyHaltMarker { + &self.apply + } +} + +/// Why one vShard's Calvin scheduler stopped applying sequenced txns. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CalvinApplyHalt { + pub vshard_id: u32, + /// Epoch of the txn held unapplied. + pub epoch: u64, + /// Position of the txn held unapplied. + pub position: u32, + /// Halt reason label, as on `nodedb_calvin_apply_halted`. + pub reason: &'static str, + /// Sub-operation of the txn that failed. + pub step: &'static str, + pub error: String, +} + +/// First-writer-wins record of the first Calvin scheduler that halted on this +/// node. A halted scheduler stays halted until restart, so nothing clears it. +#[derive(Debug, Default)] +pub struct CalvinApplyHaltMarker { + halt: OnceLock, +} + +impl CalvinApplyHaltMarker { + /// Record the first halt. Later calls are ignored. + pub fn record(&self, halt: CalvinApplyHalt) { + let _ = self.halt.set(halt); + } + + /// The recorded halt, if a scheduler on this node halted. + pub fn report(&self) -> Option<&CalvinApplyHalt> { + self.halt.get() + } + + pub fn is_halted(&self) -> bool { + self.halt.get().is_some() + } } #[cfg(test)] @@ -67,6 +117,31 @@ mod tests { assert!(marker.report().is_none()); } + fn apply_halt(vshard_id: u32) -> CalvinApplyHalt { + CalvinApplyHalt { + vshard_id, + epoch: 9, + position: 1, + reason: "flush_failed", + step: "flush", + error: "flush returned Error".to_string(), + } + } + + #[test] + fn apply_marker_keeps_the_first_recorded_halt() { + let marker = SequencerHaltMarker::default(); + assert!(!marker.apply_halt().is_halted()); + marker.apply_halt().record(apply_halt(3)); + marker.apply_halt().record(apply_halt(4)); + assert!(marker.apply_halt().is_halted()); + assert_eq!(marker.apply_halt().report().map(|h| h.vshard_id), Some(3)); + assert!( + !marker.is_halted(), + "a scheduler halt leaves the sequencer halt clear" + ); + } + #[test] fn marker_keeps_the_first_recorded_halt() { let marker = SequencerHaltMarker::default(); diff --git a/nodedb/src/control/server/http/routes/health.rs b/nodedb/src/control/server/http/routes/health.rs index 2dd592efa..e747def5b 100644 --- a/nodedb/src/control/server/http/routes/health.rs +++ b/nodedb/src/control/server/http/routes/health.rs @@ -109,6 +109,24 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); } + // A halted Calvin scheduler holds one vShard's sequenced txns unapplied. + // The node serves everything else, so it reports degraded, like a halted + // sequencer. + if let Some(halt) = state.shared.sequencer_halt.apply_halt().report() { + let body = json!({ + "status": "degraded", + "reason": "calvin_apply_halted", + "node_id": state.shared.node_id, + "vshard_id": halt.vshard_id, + "epoch": halt.epoch, + "position": halt.position, + "halt_reason": halt.reason, + "step": halt.step, + "error": halt.error, + }); + return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); + } + // A core that stops completing event-loop iterations panics nothing, so // the per-core panic watchdog stays quiet and every other check above // still passes. Fail readiness and name the cores: work routed to a @@ -252,3 +270,81 @@ pub async fn drain( })), )) } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::config::auth::AuthMode; + use crate::control::cluster::CalvinApplyHalt; + use crate::control::state::SharedState; + use crate::wal::WalManager; + + fn app_state(dir: &tempfile::TempDir) -> AppState { + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("health.wal")).expect("open WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let shared = SharedState::new(dispatcher, wal).expect("shared state"); + AppState { + shutdown_bus: crate::control::shutdown::ShutdownBus::new(Arc::clone(&shared.shutdown)) + .0, + query_ctx: Arc::new(crate::control::planner::context::QueryContext::for_state( + &shared, + )), + shared, + auth_mode: AuthMode::Trust, + } + } + + async fn healthz_body(state: AppState) -> (StatusCode, serde_json::Value) { + let response = healthz(State(state)).await.into_response(); + let status = response.status(); + let bytes = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .expect("read healthz body"); + let body = sonic_rs::from_slice::(&bytes).expect("healthz body is JSON"); + (status, body) + } + + #[tokio::test] + async fn healthz_reports_a_halted_calvin_scheduler_as_degraded() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + state + .shared + .sequencer_halt + .apply_halt() + .record(CalvinApplyHalt { + vshard_id: 12, + epoch: 40, + position: 3, + reason: "flush_failed", + step: "flush", + error: "CalvinFlush returned Error".to_string(), + }); + + let (status, body) = healthz_body(state).await; + + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(body["status"], "degraded"); + assert_eq!(body["reason"], "calvin_apply_halted"); + assert_eq!(body["vshard_id"], 12); + assert_eq!(body["epoch"], 40); + assert_eq!(body["position"], 3); + assert_eq!(body["halt_reason"], "flush_failed"); + assert_eq!(body["step"], "flush"); + } + + #[tokio::test] + async fn healthz_without_a_calvin_halt_names_no_calvin_halt() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + + let (_status, body) = healthz_body(state).await; + + assert_ne!(body["reason"], "calvin_apply_halted"); + } +} diff --git a/nodedb/src/control/server/native/session/request.rs b/nodedb/src/control/server/native/session/request.rs index 8a92723fd..a81b253d2 100644 --- a/nodedb/src/control/server/native/session/request.rs +++ b/nodedb/src/control/server/native/session/request.rs @@ -52,12 +52,14 @@ impl NativeSession { // status surfaces must agree that this node cannot serve. // A halted sequencer is the same shape of after-boot degradation as // a wedged applier: the node still serves, but a whole class of - // writes no longer completes. Both surfaces must say so. + // writes no longer completes. Both surfaces must say so. A halted + // Calvin scheduler is the same shape, scoped to one vShard. // A stalled Data Plane core is a third after-boot degradation with // the same consequence: the gate reads Ok while work sent to that // core never completes. One atomic load, so it stays on this path. let native_status = if self.state.metadata_apply_wedge.is_wedged() || self.state.sequencer_halt.is_halted() + || self.state.sequencer_halt.apply_halt().is_halted() || self.state.core_stall.is_stalled() { crate::control::startup::health::NativeStatus::Failed diff --git a/nodedb/src/control/state/fields.rs b/nodedb/src/control/state/fields.rs index 4d3236139..d66729a9d 100644 --- a/nodedb/src/control/state/fields.rs +++ b/nodedb/src/control/state/fields.rs @@ -131,10 +131,10 @@ pub struct SharedState { /// cannot clear. The readiness probe reads it so a wedged node stops /// reporting itself healthy while every query dies on a lease timeout. pub metadata_apply_wedge: Arc, - /// Set once when the Calvin sequencer state machine halts on an epoch - /// regression. Every non-Calvin path keeps serving, so the node stays up — - /// the health surfaces read this to make the lost capability visible rather - /// than letting it look like an ordinary node. + /// Set once when the Calvin sequencer halts on an epoch regression, and + /// (`apply_halt()`) once when a Calvin scheduler halts a vShard. The node + /// keeps serving every other path; the health surfaces read both markers + /// so the lost capability never looks like an ordinary node. pub sequencer_halt: Arc, /// Which Data Plane cores have stopped completing event-loop iterations, /// as of the last sampling window. Replaced every window rather than diff --git a/nodedb/src/diag/context/data_plane.rs b/nodedb/src/diag/context/data_plane.rs index dc1955ff1..24c2e42bb 100644 --- a/nodedb/src/diag/context/data_plane.rs +++ b/nodedb/src/diag/context/data_plane.rs @@ -101,3 +101,48 @@ impl DomainContext for CalvinCompletionTimeout { }) } } + +/// A Calvin scheduler held a sequenced txn unapplied and halted its vShard, +/// because a replica-local error left the txn's effect on this replica +/// unknown or missing. +pub(in crate::diag) struct CalvinApplyHalted<'a> { + pub vshard_id: u32, + pub epoch: u64, + pub position: u32, + /// Halt reason label (`dispatch_refused`, `flush_failed`, ...). + pub reason: &'a str, + /// Sub-operation of the txn that failed (`stage`, `flush`, ...). + pub step: &'a str, + pub error: &'a str, +} + +impl DomainContext for CalvinApplyHalted<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb.calvin_apply_halted" + } + + fn grouping_key(&self) -> String { + // Reason + step name the bug; the vShard and txn are the occurrence. + format!("reason={};step={}", self.reason, self.step) + } + + fn to_json(&self) -> Value { + json!({ + "vshard_id": self.vshard_id, + "epoch": self.epoch, + "position": self.position, + "reason": self.reason, + "step": self.step, + "error": self.error, + "why_fatal": "every replica must apply every sequenced txn. Marking this \ + position applied would skip it on this replica alone, so the \ + scheduler holds it unapplied with its locks and reads no new \ + sequenced input for this vShard. Calvin writes that touch the \ + vShard stop completing on this node", + "operator_action": "fix the named cause (Data Plane core, WAL disk, surrogate \ + catalog), then restart the node. The sequencer log stays \ + retained, and restart replay applies the held position \ + and every later one", + }) + } +} diff --git a/nodedb/src/diag/context/mod.rs b/nodedb/src/diag/context/mod.rs index 075ef8d17..73b5d60fa 100644 --- a/nodedb/src/diag/context/mod.rs +++ b/nodedb/src/diag/context/mod.rs @@ -22,7 +22,9 @@ pub(in crate::diag) use catalog::{ }; pub(in crate::diag) use crdt::HistoryCompactionNotApplied; pub use data_plane::LostResponseWrite; -pub(in crate::diag) use data_plane::{CalvinCompletionTimeout, DataPlaneResponseLost}; +pub(in crate::diag) use data_plane::{ + CalvinApplyHalted, CalvinCompletionTimeout, DataPlaneResponseLost, +}; pub(in crate::diag) use ingest::IlpAcceptedLinesDropped; pub use ingest::IlpFlushOutcome; pub use quota::{DATABASE_SCOPE, TENANT_SCOPE}; diff --git a/nodedb/src/diag/mod.rs b/nodedb/src/diag/mod.rs index fded5b163..20c0239c1 100644 --- a/nodedb/src/diag/mod.rs +++ b/nodedb/src/diag/mod.rs @@ -10,13 +10,13 @@ mod recording; pub use context::{DATABASE_SCOPE, IlpFlushOutcome, LostResponseWrite, TENANT_SCOPE}; pub use recording::{ - batch_insert_without_surrogates, calvin_completion_timeout, catalog_apply_orphan_row, - collection_purge_row_missing, consumer_group_offsets_retained, data_plane_response_lost, - data_plane_responses_lost, entry_kind, fts_index_update_failed, history_compaction_not_applied, - ilp_invalid_utf8_drop, ilp_line_read_drop, metadata_apply_wedged, - orphaned_index_entry_after_delete, quota_row_invalid, quota_row_undecodable, - quota_row_write_failed, quota_scope_purge_incomplete, quota_scope_replay_aborted, - replay_record_unapplied, retention_autowire_orphaned, scope_quota_not_installed, - strict_row_undecodable, synonym_group_not_applied, vector_index_not_applied, - wal_archival_failed_truncation_held, write_acked_without_durability, + batch_insert_without_surrogates, calvin_apply_halted, calvin_completion_timeout, + catalog_apply_orphan_row, collection_purge_row_missing, consumer_group_offsets_retained, + data_plane_response_lost, data_plane_responses_lost, entry_kind, fts_index_update_failed, + history_compaction_not_applied, ilp_invalid_utf8_drop, ilp_line_read_drop, + metadata_apply_wedged, orphaned_index_entry_after_delete, quota_row_invalid, + quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, + quota_scope_replay_aborted, replay_record_unapplied, retention_autowire_orphaned, + scope_quota_not_installed, strict_row_undecodable, synonym_group_not_applied, + vector_index_not_applied, wal_archival_failed_truncation_held, write_acked_without_durability, }; diff --git a/nodedb/src/diag/recording/data_plane.rs b/nodedb/src/diag/recording/data_plane.rs index d0447987c..12b0d650c 100644 --- a/nodedb/src/diag/recording/data_plane.rs +++ b/nodedb/src/diag/recording/data_plane.rs @@ -60,3 +60,31 @@ pub fn calvin_completion_timeout( .with_backtrace() .emit(); } + +/// Report a Calvin scheduler that held a sequenced txn unapplied and halted +/// its vShard. Called once per scheduler, from its halt latch on the first +/// cause. +pub fn calvin_apply_halted( + vshard_id: u32, + epoch: u64, + position: u32, + reason: &str, + step: &str, + error: &str, +) { + let ctx = context::CalvinApplyHalted { + vshard_id, + epoch, + position, + reason, + step, + error, + }; + let _ = Capture::new( + EventKind::InvariantViolation, + "Calvin scheduler halted: a sequenced txn could not be applied on this replica", + ) + .domain(&ctx) + .with_backtrace() + .emit(); +} diff --git a/nodedb/src/diag/recording/mod.rs b/nodedb/src/diag/recording/mod.rs index 3e39b978c..b6a6db5d0 100644 --- a/nodedb/src/diag/recording/mod.rs +++ b/nodedb/src/diag/recording/mod.rs @@ -24,7 +24,8 @@ pub use catalog::{ }; pub use crdt::history_compaction_not_applied; pub use data_plane::{ - calvin_completion_timeout, data_plane_response_lost, data_plane_responses_lost, + calvin_apply_halted, calvin_completion_timeout, data_plane_response_lost, + data_plane_responses_lost, }; pub use ingest::{ilp_invalid_utf8_drop, ilp_line_read_drop}; pub use quota::{ From e24f5ba649221d8d536abeb1e70d2fe69389acb4 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 17:03:25 +0800 Subject: [PATCH 08/64] feat(scheduler): retry a Calvin scheduler's sequencer entries until applied propose_sequencer_entry sent a vote, completion ack, OLLP mismatch, or routing-failure signal to the sequencer group once and dropped it on any failure. A refused proposal or a leader change that dropped an appended entry then left every participant waiting for a verdict or ack forever, since nothing proposed it again. Add an owed-entries table that keeps each proposal until this node's completion registry shows its effect, and a stall-tick sweep that proposes every remaining owed entry again. Read progress from a new CalvinCompletionRegistry::participant_progress so the sweep can tell an applied entry from an outstanding one. Route every proposal through a new SequencerProposer seam instead of locking MultiRaft directly: RaftSequencerProposer appends locally on the sequencer leader and forwards to it otherwise, reusing the data-group forward RPC. Extend DataProposeRequest's target from a bare vshard_id to a ProposeTarget enum (VShard or Sequencer) so the same RPC and raft-loop handler carry both proposal kinds, and give each node one shared proposer so its forward concurrency limit bounds the whole node rather than one scheduler. Expose the retry counts as a per-kind Prometheus counter, and split the scheduler's flow-metric rendering into its own module alongside it. --- nodedb-cluster/src/calvin/completion.rs | 78 ++++ nodedb-cluster/src/calvin/mod.rs | 3 +- .../src/raft_loop/handle_rpc/plan_dispatch.rs | 102 +++++- nodedb-cluster/src/raft_loop/proposals.rs | 2 +- nodedb-cluster/src/rpc_codec/data_propose.rs | 59 +++- nodedb-cluster/src/rpc_codec/mod.rs | 2 +- nodedb/src/control/cluster/calvin/mod.rs | 4 +- .../driver/core/commit_resolve/apply_tail.rs | 23 +- .../driver/core/commit_resolve/vote.rs | 26 +- .../scheduler/driver/core/completion_route.rs | 12 +- .../driver/core/dispatch/active_dispatch.rs | 4 +- .../driver/core/dispatch/static_dispatch.rs | 20 +- .../calvin/scheduler/driver/core/mod.rs | 5 + .../scheduler/driver/core/owed/entry.rs | 243 +++++++++++++ .../calvin/scheduler/driver/core/owed/mod.rs | 8 + .../scheduler/driver/core/owed/retry.rs | 333 ++++++++++++++++++ .../calvin/scheduler/driver/core/scheduler.rs | 66 ++-- .../driver/core/sequencer_proposer/mod.rs | 10 + .../driver/core/sequencer_proposer/raft.rs | 291 +++++++++++++++ .../driver/core/sequencer_proposer/seam.rs | 52 +++ .../scheduler/driver/core/test_proposer.rs | 102 ++++++ .../scheduler/driver/core/test_support.rs | 3 + .../cluster/calvin/scheduler/driver/mod.rs | 5 +- .../cluster/calvin/scheduler/metrics.rs | 102 ++---- .../cluster/calvin/scheduler/metrics_flow.rs | 135 +++++++ .../control/cluster/calvin/scheduler/mod.rs | 5 +- .../src/control/cluster/start_raft_helpers.rs | 14 +- 27 files changed, 1492 insertions(+), 217 deletions(-) create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/mod.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/mod.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/test_proposer.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/metrics_flow.rs diff --git a/nodedb-cluster/src/calvin/completion.rs b/nodedb-cluster/src/calvin/completion.rs index 05d42629a..f2c3fb1e7 100644 --- a/nodedb-cluster/src/calvin/completion.rs +++ b/nodedb-cluster/src/calvin/completion.rs @@ -73,6 +73,24 @@ impl VerdictOutcome { } } +/// What this node's registry has applied for one participant vShard of a txn. +/// +/// A scheduler reads it to learn whether a sequencer entry it proposed has +/// been applied here. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ParticipantProgress { + /// A `Vote` or `AbortVote` from this vShard is in the tally. + pub voted: bool, + /// A `CompletionAck` from this vShard is recorded. + pub acked: bool, + /// The txn's global verdict is stored. + pub has_verdict: bool, + /// An `OllpMismatch` for the txn is recorded. + pub mismatched: bool, + /// A `TxnRoutingFailed` for the txn is recorded and not yet delivered. + pub routing_failed: bool, +} + pub(crate) struct PendingCompletion { /// `pub(crate)`: also read/written by the vote/verdict-tally methods in /// `completion_verdict.rs` (a sibling module in the same crate). @@ -384,6 +402,25 @@ impl CalvinCompletionRegistry { } } + /// What this registry holds for participant `vshard` of `txn`. + /// + /// `None` means no entry exists for `txn`. That is either a txn this node + /// never seeded, or one whose outcome already fired and evicted its entry. + pub fn participant_progress(&self, txn: TxnId, vshard: u32) -> Option { + self.inner + .lock() + .unwrap_or_else(|p| p.into_inner()) + .completions + .get(&txn) + .map(|entry| ParticipantProgress { + voted: entry.votes.contains_key(&vshard), + acked: entry.acked_vshards.contains(&vshard), + has_verdict: entry.verdict.is_some(), + mismatched: entry.mismatched, + routing_failed: entry.routing_failed.is_some(), + }) + } + /// Test-only: returns the number of pending completion entries. /// Used to verify entries are removed once all acks arrive (no leak). #[cfg(test)] @@ -711,4 +748,45 @@ mod tests { "entry must be evicted once mismatch is signalled" ); } + + #[tokio::test] + async fn participant_progress_is_none_before_any_entry_exists() { + let reg = CalvinCompletionRegistry::new_detached(); + assert_eq!(reg.participant_progress(TxnId::new(40, 0), 1), None); + } + + #[tokio::test] + async fn participant_progress_reports_each_applied_signal_for_its_vshard() { + let reg = CalvinCompletionRegistry::new_detached(); + let txn = TxnId::new(40, 1); + reg.seed_expected(txn, 2); + let empty = reg.participant_progress(txn, 1).expect("seeded entry"); + assert!(!empty.voted && !empty.acked && !empty.has_verdict); + assert!(!empty.mismatched && !empty.routing_failed); + + reg.note_vote(txn, 1, ParticipantVote::Commit); + reg.note_completion_ack(txn, 1); + let own = reg.participant_progress(txn, 1).expect("entry"); + assert!(own.voted && own.acked); + let peer = reg.participant_progress(txn, 2).expect("entry"); + assert!( + !peer.voted && !peer.acked, + "another vShard's signals do not count" + ); + + reg.note_verdict(txn, VerdictOutcome::Commit); + reg.note_ollp_mismatch(txn); + reg.note_routing_failed(txn, "unroutable".to_string()); + let txn_wide = reg.participant_progress(txn, 2).expect("entry"); + assert!(txn_wide.has_verdict && txn_wide.mismatched && txn_wide.routing_failed); + } + + #[tokio::test] + async fn participant_progress_is_none_once_the_outcome_fired() { + let reg = CalvinCompletionRegistry::new_detached(); + let txn = TxnId::new(40, 2); + let _rx = reg.register_completion(txn, 1); + reg.note_completion_ack(txn, 1); + assert_eq!(reg.participant_progress(txn, 1), None); + } } diff --git a/nodedb-cluster/src/calvin/mod.rs b/nodedb-cluster/src/calvin/mod.rs index 49ae97404..3b24f07ad 100644 --- a/nodedb-cluster/src/calvin/mod.rs +++ b/nodedb-cluster/src/calvin/mod.rs @@ -6,7 +6,8 @@ pub mod sequencer; pub mod types; pub use completion::{ - AttemptOutcome, CalvinCompletionRegistry, ParticipantVote, TxnId, VerdictOutcome, + AttemptOutcome, CalvinCompletionRegistry, ParticipantProgress, ParticipantVote, TxnId, + VerdictOutcome, }; pub use completion_verdict::VerdictSignal; pub use sequencer::{ diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs index 83129fbe5..765e3354f 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs @@ -3,10 +3,13 @@ //! Physical-plan execution (C-β), metadata/data propose forwarding, and //! VShardEnvelope routing RPC bodies. +use crate::calvin::SEQUENCER_GROUP_ID; use crate::error::{ClusterError, Result}; use crate::forward::{ChunkSink, PlanExecutor}; +use crate::multi_raft::MultiRaft; use crate::rpc_codec::{ - DataProposeRequest, ExecuteRequest, MetadataProposeRequest, RaftRpc, TypedClusterError, + DataProposeRequest, DataProposeResponse, ExecuteRequest, MetadataProposeRequest, ProposeTarget, + RaftRpc, TypedClusterError, }; use super::super::loop_core::{CommitApplier, RaftLoop}; @@ -37,21 +40,13 @@ impl RaftLoop { Ok(RaftRpc::MetadataProposeResponse(resp)) } - // Data-group proposal forwarding — apply locally if we are the - // data-group leader for the given vshard, otherwise return - // NotLeader with a hint so the forwarder can chase the redirect. + // Data-group and sequencer-group proposal forwarding — apply locally if + // we lead the target group, otherwise return NotLeader with a hint so the + // forwarder can chase the redirect. pub(super) fn handle_data_propose_rpc(&self, req: DataProposeRequest) -> Result { let resp = { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - match mr.propose(req.vshard_id, req.bytes) { - Ok((group_id, log_index)) => { - crate::rpc_codec::DataProposeResponse::ok(group_id, log_index) - } - Err(crate::error::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { - leader_hint, - })) => crate::rpc_codec::DataProposeResponse::err("not leader", leader_hint), - Err(e) => crate::rpc_codec::DataProposeResponse::err(e.to_string(), None), - } + propose_forwarded(&mut mr, req) }; Ok(RaftRpc::DataProposeResponse(resp)) } @@ -79,3 +74,84 @@ impl RaftLoop { self.plan_executor.execute_plan_streaming(req, sink).await } } + +/// Propose a forwarded entry to its target group on this node. +/// +/// Answers `not leader` with the known leader as a hint when this node does +/// not lead the target group. +fn propose_forwarded(mr: &mut MultiRaft, req: DataProposeRequest) -> DataProposeResponse { + let proposed = match req.target { + ProposeTarget::VShard(vshard_id) => mr.propose(vshard_id, req.bytes), + ProposeTarget::Sequencer => mr + .propose_to_group(SEQUENCER_GROUP_ID, req.bytes) + .map(|log_index| (SEQUENCER_GROUP_ID, log_index)), + }; + match proposed { + Ok((group_id, log_index)) => DataProposeResponse::ok(group_id, log_index), + Err(ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint })) => { + DataProposeResponse::err("not leader", leader_hint) + } + Err(e) => DataProposeResponse::err(e.to_string(), None), + } +} + +#[cfg(test)] +mod tests { + use std::time::{Duration, Instant}; + + use super::*; + use crate::routing::RoutingTable; + + fn multi_raft_with_sequencer(dir: &std::path::Path) -> MultiRaft { + let mut mr = MultiRaft::new(1, RoutingTable::uniform(1, &[1], 1), dir.to_path_buf()); + mr.add_group(SEQUENCER_GROUP_ID, vec![]) + .expect("add sequencer group"); + mr + } + + fn elect_sequencer_leader(mr: &mut MultiRaft) { + if let Some(node) = mr.groups_mut().get_mut(&SEQUENCER_GROUP_ID) { + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + for _ in 0..20 { + mr.tick().expect("tick"); + if mr.is_group_leader(SEQUENCER_GROUP_ID) { + return; + } + } + panic!("sequencer group did not elect this single node"); + } + + fn sequencer_request() -> DataProposeRequest { + DataProposeRequest { + target: ProposeTarget::Sequencer, + bytes: vec![7, 7, 7], + } + } + + #[test] + fn sequencer_target_is_proposed_to_the_sequencer_group_on_its_leader() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut mr = multi_raft_with_sequencer(dir.path()); + elect_sequencer_leader(&mut mr); + let before = mr.last_log_index(SEQUENCER_GROUP_ID).unwrap_or(0); + + let resp = propose_forwarded(&mut mr, sequencer_request()); + + assert!(resp.success, "{}", resp.error_message); + assert_eq!(resp.group_id, SEQUENCER_GROUP_ID); + assert!(resp.log_index > before); + assert_eq!(mr.last_log_index(SEQUENCER_GROUP_ID), Some(resp.log_index)); + } + + #[test] + fn sequencer_target_on_a_non_leader_answers_not_leader() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut mr = multi_raft_with_sequencer(dir.path()); + + let resp = propose_forwarded(&mut mr, sequencer_request()); + + assert!(!resp.success); + assert_eq!(resp.error_message, "not leader"); + } +} diff --git a/nodedb-cluster/src/raft_loop/proposals.rs b/nodedb-cluster/src/raft_loop/proposals.rs index 8417a779e..f3acd8f22 100644 --- a/nodedb-cluster/src/raft_loop/proposals.rs +++ b/nodedb-cluster/src/raft_loop/proposals.rs @@ -244,7 +244,7 @@ impl RaftLoop { let req = crate::rpc_codec::RaftRpc::DataProposeRequest(crate::rpc_codec::DataProposeRequest { - vshard_id, + target: crate::rpc_codec::ProposeTarget::VShard(vshard_id), bytes: data, }); let resp = self.transport.send_rpc(leader_id, req).await?; diff --git a/nodedb-cluster/src/rpc_codec/data_propose.rs b/nodedb-cluster/src/rpc_codec/data_propose.rs index 25b61c18d..c1f459c0f 100644 --- a/nodedb-cluster/src/rpc_codec/data_propose.rs +++ b/nodedb-cluster/src/rpc_codec/data_propose.rs @@ -2,23 +2,31 @@ //! DataProposeRequest / DataProposeResponse wire types and codecs. //! -//! Used to forward data-group (non-metadata) Raft proposals from a follower -//! node to the group leader. The leader applies the proposal locally and -//! returns `(group_id, log_index)` so the forwarder can register a -//! `ProposeTracker` waiter and await commit. +//! Used to forward a non-metadata Raft proposal from a node that does not +//! lead the target group to the group leader. The target is a vShard's data +//! group or the Calvin sequencer group. The leader applies the proposal +//! locally and returns `(group_id, log_index)`. use super::discriminants::*; use super::header::write_frame; use super::raft_rpc::RaftRpc; use crate::error::{ClusterError, Result}; -/// Forward an opaque data-group proposal payload to the data-group leader. -/// -/// `vshard_id` identifies the vShard (and thus the Raft group) the entry -/// belongs to. `bytes` is the serialized `ReplicatedEntry`. +/// The Raft group a forwarded proposal is for. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum ProposeTarget { + /// The data group that owns this vShard. The bytes are a serialized + /// `ReplicatedEntry`. + VShard(u32), + /// The Calvin sequencer group. The bytes are a msgpack-encoded + /// `SequencerEntry`. + Sequencer, +} + +/// Forward an opaque proposal payload to the leader of its target group. #[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub struct DataProposeRequest { - pub vshard_id: u32, + pub target: ProposeTarget, pub bytes: Vec, } @@ -95,3 +103,36 @@ pub(super) fn decode_data_propose_resp(payload: &[u8]) -> Result { "DataProposeResponse" )?)) } + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + fn roundtrip(target: ProposeTarget) -> DataProposeRequest { + let rpc = RaftRpc::DataProposeRequest(DataProposeRequest { + target, + bytes: vec![1, 2, 3], + }); + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + match decode(&encoded, &epoch).expect("decode") { + RaftRpc::DataProposeRequest(req) => req, + other => panic!("decoded the wrong variant: {other:?}"), + } + } + + #[test] + fn sequencer_target_survives_the_wire() { + let req = roundtrip(ProposeTarget::Sequencer); + assert_eq!(req.target, ProposeTarget::Sequencer); + assert_eq!(req.bytes, vec![1, 2, 3]); + } + + #[test] + fn vshard_target_survives_the_wire() { + let req = roundtrip(ProposeTarget::VShard(42)); + assert_eq!(req.target, ProposeTarget::VShard(42)); + } +} diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index a962dae83..507561f44 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -38,7 +38,7 @@ pub use cluster_mgmt::{ PongResponse, TopologyAck, TopologyUpdate, }; pub use data_plane_error::DataPlaneErrorCode; -pub use data_propose::{DataProposeRequest, DataProposeResponse}; +pub use data_propose::{DataProposeRequest, DataProposeResponse, ProposeTarget}; pub use execute::{ DescriptorVersionEntry, ExecuteRequest, ExecuteResponse, ExecuteStreamChunk, ExecuteStreamEnd, PLAN_DECODE_FAILED, TypedClusterError, diff --git a/nodedb/src/control/cluster/calvin/mod.rs b/nodedb/src/control/cluster/calvin/mod.rs index 51680f5bf..13c67536a 100644 --- a/nodedb/src/control/cluster/calvin/mod.rs +++ b/nodedb/src/control/cluster/calvin/mod.rs @@ -5,6 +5,6 @@ pub mod scheduler; pub use executor::{OllpConfig, OllpError, OllpOrchestrator}; pub use scheduler::{ - CalvinReadResultProposal, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, - propose_calvin_read_result, + CalvinReadResultProposal, RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, + SchedulerParams, SequencerProposer, propose_calvin_read_result, }; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs index 59d3b3c75..7ccb67e49 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -4,12 +4,11 @@ //! applied result, marking the apply durable, recording write versions, and //! proposing the `CompletionAck`. -use nodedb_cluster::calvin::SequencerEntry; - use crate::bridge::envelope::{Response, Status}; use crate::control::cluster::calvin::scheduler::driver::core::halt::{ HaltReason, HaltStep, error_response_text, }; +use crate::control::cluster::calvin::scheduler::driver::core::owed::SchedulerProposal; use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; @@ -63,15 +62,7 @@ impl Scheduler { let completed = if committed { self.commit_apply_tail(txn_id, response, redo_lsn) } else { - self.propose_sequencer_entry( - SequencerEntry::CompletionAck { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }, - txn_id, - "completion ack (dropped)", - ); + self.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); true }; // `false` means the commit tail halted the scheduler: the txn stays @@ -247,15 +238,7 @@ impl Scheduler { ); } } - self.propose_sequencer_entry( - SequencerEntry::CompletionAck { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }, - txn_id, - "completion ack", - ); + self.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); true } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs index fefb9e9c9..74edb77c3 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs @@ -5,10 +5,9 @@ use std::sync::atomic::Ordering; use std::time::Instant; -use nodedb_cluster::calvin::SequencerEntry; - use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::driver::core::halt::error_response_text; +use crate::control::cluster::calvin::scheduler::driver::core::owed::SchedulerProposal; use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::driver::core::staged_vote::{ StagedVote, staged_commit_vote, @@ -53,23 +52,16 @@ impl Scheduler { // leader ran read-set validation, so only a leader's vote is // authoritative. The sequencer aggregates every participant's vote into // the single global verdict this txn parks on below. An abort travels as - // `AbortVote` so its cause survives to the coordinator. + // `AbortVote` so its cause survives to the coordinator. The vote stays + // owed until the tally holds it, so a refused or dropped proposal is + // proposed again rather than lost. if self.is_group_leader() { - let entry = match vote.abort_reason() { - Some(reason) => SequencerEntry::AbortVote { - epoch: txn_id.epoch, - position: txn_id.position, - vshard: self.vshard_id, - reason, - }, - None => SequencerEntry::Vote { - epoch: txn_id.epoch, - position: txn_id.position, - vshard: self.vshard_id, - commit: true, + self.propose_sequencer_entry( + txn_id, + SchedulerProposal::Vote { + abort: vote.abort_reason(), }, - }; - self.propose_sequencer_entry(entry, txn_id, "commit vote"); + ); } if vote == StagedVote::SerializationConflict { diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs index d0541b521..90ba3f6dc 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs @@ -10,10 +10,9 @@ //! [`CommitState::AwaitingVerdict`] has no outstanding bridge, so a completion //! for it is a no-op that keeps it parked. -use nodedb_cluster::calvin::SequencerEntry; - use super::super::types::CommitState; use super::halt::{HaltReason, HaltStep}; +use super::owed::SchedulerProposal; use super::scheduler::Scheduler; use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; @@ -102,14 +101,7 @@ impl Scheduler { // OllpMismatch calls note_ollp_mismatch, waking the coordinator's // retry-loop waiter wherever it is — mirrors how CompletionAck is // delivered to remote coordinators. - self.propose_sequencer_entry( - SequencerEntry::OllpMismatch { - epoch: txn_id.epoch, - position: txn_id.position, - }, - txn_id, - "OLLP mismatch signal", - ); + self.propose_sequencer_entry(txn_id, SchedulerProposal::OllpMismatch); return; } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs index 59e7e809c..d5adbc5d0 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs @@ -75,7 +75,7 @@ impl Scheduler { error = %e, "calvin scheduler: active txn homes no local writes; releasing locks" ); - self.propose_routing_failure(epoch, position, txn_id, &e); + self.propose_routing_failure(txn_id, &e); self.on_unpending_txn_complete(txn_id, lock_owner); return; } @@ -87,7 +87,7 @@ impl Scheduler { error = %e, "calvin scheduler: active txn routing failed; releasing locks" ); - self.propose_routing_failure(epoch, position, txn_id, &e); + self.propose_routing_failure(txn_id, &e); self.on_unpending_txn_complete(txn_id, lock_owner); return; } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs index 306cd9614..37a761a0d 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs @@ -12,6 +12,7 @@ use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; use super::super::deferred::{DispatchOutcome, DispatchStep}; +use super::super::owed::SchedulerProposal; use super::super::routing::PlanRouting; use super::super::scheduler::Scheduler; use super::primary_write::{ @@ -51,21 +52,12 @@ impl Scheduler { /// full deadline and report a generic timeout. Mirrors the OllpMismatch /// broadcast in `handle_executor_response`. Shared by `dispatch_txn` and /// `dispatch_active_txn`. - pub(super) fn propose_routing_failure( - &self, - epoch: u64, - position: u32, - txn_id: TxnId, - err: &crate::Error, - ) { + pub(super) fn propose_routing_failure(&mut self, txn_id: TxnId, err: &crate::Error) { self.propose_sequencer_entry( - nodedb_cluster::calvin::SequencerEntry::TxnRoutingFailed { - epoch, - position, + txn_id, + SchedulerProposal::RoutingFailed { detail: err.to_string(), }, - txn_id, - "txn routing-failure signal", ); } @@ -165,7 +157,7 @@ impl Scheduler { error = %e, "calvin scheduler: static txn routing failed; releasing locks" ); - self.propose_routing_failure(epoch, position, txn_id, &e); + self.propose_routing_failure(txn_id, &e); self.on_unpending_txn_complete(txn_id, lock_owner); return; } @@ -200,7 +192,7 @@ impl Scheduler { error = %e, "calvin scheduler: static txn homes no local work; releasing locks" ); - self.propose_routing_failure(epoch, position, txn_id, &e); + self.propose_routing_failure(txn_id, &e); self.on_unpending_txn_complete(txn_id, lock_owner); return; } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index 5cbe3f58c..2c5dce1dd 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -23,16 +23,21 @@ pub mod deferred; pub mod dispatch; pub mod halt; pub mod intake; +pub mod owed; pub mod process; pub mod propose; pub mod read_result; pub mod request; pub mod routing; pub mod scheduler; +pub mod sequencer_proposer; pub mod staged_vote; #[cfg(test)] +mod test_proposer; +#[cfg(test)] mod test_support; pub mod write_version_record; pub use propose::{CalvinReadResultProposal, propose_calvin_read_result}; pub use scheduler::{Scheduler, SchedulerParams}; +pub use sequencer_proposer::{RaftSequencerProposer, SequencerProposer}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs new file mode 100644 index 000000000..31453d838 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs @@ -0,0 +1,243 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Owed sequencer entries: what the scheduler proposes, and how it tells +//! that an entry is applied. +//! +//! A proposal can be lost: a node that does not lead the sequencer group +//! refuses it, and a leader change can drop an appended entry. A lost vote +//! leaves every participant waiting for a verdict forever. So the scheduler +//! keeps each proposal in [`OwedEntries`] and proposes it again on the stall +//! tick until this node's completion registry shows its effect. + +use std::collections::BTreeMap; + +use nodedb_cluster::calvin::{AbortReason, ParticipantProgress, SequencerEntry}; + +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::cluster::calvin::scheduler::metrics::sequencer_propose_kind; + +/// A sequencer entry the scheduler proposes for one txn on its vShard. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum SchedulerProposal { + /// This vShard's commit vote. `Some(reason)` is an abort vote. + Vote { abort: Option }, + /// This vShard applied or dropped the txn. + CompletionAck, + /// The active executor saw its OLLP prediction drift. + OllpMismatch, + /// The txn's local plan routing failed for good. + RoutingFailed { detail: String }, +} + +impl SchedulerProposal { + /// The owed-entry kind this proposal fills. + pub fn kind(&self) -> OwedKind { + match self { + Self::Vote { .. } => OwedKind::Vote, + Self::CompletionAck => OwedKind::CompletionAck, + Self::OllpMismatch => OwedKind::OllpMismatch, + Self::RoutingFailed { .. } => OwedKind::RoutingFailed, + } + } + + /// The sequencer entry for `txn_id` proposed by `vshard`. + pub fn entry(&self, txn_id: TxnId, vshard: u32) -> SequencerEntry { + let (epoch, position) = (txn_id.epoch, txn_id.position); + match self { + Self::Vote { abort: None } => SequencerEntry::Vote { + epoch, + position, + vshard, + commit: true, + }, + Self::Vote { + abort: Some(reason), + } => SequencerEntry::AbortVote { + epoch, + position, + vshard, + reason: *reason, + }, + Self::CompletionAck => SequencerEntry::CompletionAck { + epoch, + position, + vshard_id: vshard, + }, + Self::OllpMismatch => SequencerEntry::OllpMismatch { epoch, position }, + Self::RoutingFailed { detail } => SequencerEntry::TxnRoutingFailed { + epoch, + position, + detail: detail.clone(), + }, + } + } +} + +/// The kind of an owed entry. A txn owes at most one entry of each kind. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum OwedKind { + Vote, + CompletionAck, + OllpMismatch, + RoutingFailed, +} + +impl OwedKind { + /// Short name for logs. + pub fn label(self) -> &'static str { + sequencer_propose_kind::LABELS[self.metric_index()] + } + + /// Index into the `sequencer_propose_kind` metric labels. + pub fn metric_index(self) -> usize { + match self { + Self::Vote => sequencer_propose_kind::VOTE, + Self::CompletionAck => sequencer_propose_kind::COMPLETION_ACK, + Self::OllpMismatch => sequencer_propose_kind::OLLP_MISMATCH, + Self::RoutingFailed => sequencer_propose_kind::ROUTING_FAILED, + } + } + + /// Whether this node applied the entry of this kind. + /// + /// `progress` is the registry's view of the txn for the scheduler's + /// vShard. `entry_seen` is whether the registry held an entry for the + /// txn at an earlier check. The registry removes an entry only once the + /// txn's outcome fired, and a txn with a fired outcome needs no entry of + /// any kind. A missing entry that was never seen means this node has not + /// seeded the txn, so the entry is still owed. + /// + /// A stored verdict also settles a vote: the verdict forms only once + /// every participant's vote is in the tally. + pub fn is_applied(self, progress: Option, entry_seen: bool) -> bool { + let Some(p) = progress else { + return entry_seen; + }; + match self { + Self::Vote => p.voted || p.has_verdict, + Self::CompletionAck => p.acked, + Self::OllpMismatch => p.mismatched, + Self::RoutingFailed => p.routing_failed, + } + } +} + +/// One owed sequencer entry. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct OwedEntry { + /// The msgpack-encoded `SequencerEntry`. + pub bytes: Vec, + /// The registry held an entry for the txn at some check. + pub entry_seen: bool, + /// The last proposal left this node after the previous sweep. The next + /// sweep skips the entry once, which gives it one tick to apply. + pub in_flight: bool, +} + +/// Owed entries keyed by txn and kind, in deterministic order. +/// +/// Bounded: it holds at most one entry per (txn, kind), and a txn owes at +/// most two kinds (a vote and a completion ack, or a single terminal +/// signal). An entry leaves once this node applies it. Entries pile up only +/// while no sequencer leader is reachable, and then the sequencer admits no +/// new txns either. +pub type OwedEntries = BTreeMap<(TxnId, OwedKind), OwedEntry>; + +#[cfg(test)] +mod tests { + use super::*; + + fn progress() -> ParticipantProgress { + ParticipantProgress { + voted: false, + acked: false, + has_verdict: false, + mismatched: false, + routing_failed: false, + } + } + + const ALL_KINDS: [OwedKind; 4] = [ + OwedKind::Vote, + OwedKind::CompletionAck, + OwedKind::OllpMismatch, + OwedKind::RoutingFailed, + ]; + + #[test] + fn nothing_is_applied_while_the_registry_shows_no_effect() { + for kind in ALL_KINDS { + assert!(!kind.is_applied(Some(progress()), true), "{kind:?}"); + } + } + + #[test] + fn each_kind_is_applied_once_its_own_effect_shows() { + let voted = ParticipantProgress { + voted: true, + ..progress() + }; + let acked = ParticipantProgress { + acked: true, + ..progress() + }; + let mismatched = ParticipantProgress { + mismatched: true, + ..progress() + }; + let routing_failed = ParticipantProgress { + routing_failed: true, + ..progress() + }; + assert!(OwedKind::Vote.is_applied(Some(voted), false)); + assert!(!OwedKind::CompletionAck.is_applied(Some(voted), false)); + assert!(OwedKind::CompletionAck.is_applied(Some(acked), false)); + assert!(!OwedKind::Vote.is_applied(Some(acked), false)); + assert!(OwedKind::OllpMismatch.is_applied(Some(mismatched), false)); + assert!(OwedKind::RoutingFailed.is_applied(Some(routing_failed), false)); + } + + #[test] + fn a_stored_verdict_settles_the_vote_only() { + let decided = ParticipantProgress { + has_verdict: true, + ..progress() + }; + assert!(OwedKind::Vote.is_applied(Some(decided), false)); + assert!(!OwedKind::CompletionAck.is_applied(Some(decided), false)); + } + + #[test] + fn a_removed_entry_settles_every_kind_only_after_it_was_seen() { + for kind in ALL_KINDS { + assert!(kind.is_applied(None, true), "{kind:?}"); + assert!(!kind.is_applied(None, false), "{kind:?}"); + } + } + + #[test] + fn a_vote_proposal_carries_its_abort_reason() { + let txn = TxnId::new(3, 4); + assert!(matches!( + SchedulerProposal::Vote { abort: None }.entry(txn, 9), + SequencerEntry::Vote { + epoch: 3, + position: 4, + vshard: 9, + commit: true + } + )); + assert!(matches!( + SchedulerProposal::Vote { + abort: Some(AbortReason::SerializationConflict) + } + .entry(txn, 9), + SequencerEntry::AbortVote { + epoch: 3, + position: 4, + vshard: 9, + reason: AbortReason::SerializationConflict + } + )); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/mod.rs new file mode 100644 index 000000000..60e264f97 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/mod.rs @@ -0,0 +1,8 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sequencer entries the scheduler owes until this node applies them. + +pub mod entry; +pub mod retry; + +pub use entry::{OwedEntries, OwedEntry, OwedKind, SchedulerProposal}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs new file mode 100644 index 000000000..7b8fcc395 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs @@ -0,0 +1,333 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Owing, proposing, and re-proposing the scheduler's sequencer entries. +//! +//! Re-proposing is safe because every entry kind applies idempotently in the +//! completion registry. A vote is stored per vShard, so a repeat overwrites +//! it with the same value, and the verdict it completes is emitted once. An +//! ack is a set insert per vShard, and the completion fires once. The +//! mismatch and routing-failure signals set a flag, and each fires its +//! waiter once. + +use tracing::{debug, error}; + +use nodedb_cluster::calvin::ParticipantProgress; + +use super::entry::{OwedEntry, OwedKind, SchedulerProposal}; +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::driver::core::sequencer_proposer::SequencerProposer; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Owe `proposal` for `txn_id` and propose it once. + /// + /// The entry stays owed until this node's completion registry shows it + /// applied. [`Self::retry_owed_sequencer_entries`] proposes it again on + /// the stall tick until then. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn propose_sequencer_entry( + &mut self, + txn_id: TxnId, + proposal: SchedulerProposal, + ) { + let kind = proposal.kind(); + let bytes = match zerompk::to_msgpack_vec(&proposal.entry(txn_id, self.vshard_id)) { + Ok(bytes) => bytes, + Err(e) => { + error!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + kind = kind.label(), + error = %e, + "calvin: failed to encode a sequencer entry; it cannot be proposed", + ); + return; + } + }; + let entry_seen = self.participant_progress(txn_id).is_some(); + let in_flight = propose_owed( + self.sequencer_proposer.as_ref(), + self.vshard_id, + txn_id, + kind, + bytes.clone(), + ); + self.owed.insert( + (txn_id, kind), + OwedEntry { + bytes, + entry_seen, + in_flight, + }, + ); + } + + /// Drop every owed entry this node applied, and propose the rest again. + /// + /// Runs on the stall tick. An entry proposed since the previous sweep is + /// skipped once, so an entry on its way through Raft is not proposed a + /// second time before it can apply. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn retry_owed_sequencer_entries( + &mut self, + ) { + let registry = &self.registry; + let vshard_id = self.vshard_id; + self.owed.retain(|(txn_id, kind), owed| { + let progress = registry.participant_progress(cluster_txn_id(*txn_id), vshard_id); + owed.entry_seen |= progress.is_some(); + !kind.is_applied(progress, owed.entry_seen) + }); + + for ((txn_id, kind), owed) in self.owed.iter_mut() { + if owed.in_flight { + owed.in_flight = false; + continue; + } + self.metrics + .record_sequencer_propose_retry(kind.metric_index()); + owed.in_flight = propose_owed( + self.sequencer_proposer.as_ref(), + vshard_id, + *txn_id, + *kind, + owed.bytes.clone(), + ); + } + } + + /// The registry's view of `txn_id` for this scheduler's vShard. + fn participant_progress(&self, txn_id: TxnId) -> Option { + self.registry + .participant_progress(cluster_txn_id(txn_id), self.vshard_id) + } +} + +/// The completion registry's key for `txn_id`. +fn cluster_txn_id(txn_id: TxnId) -> nodedb_cluster::calvin::TxnId { + nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position) +} + +/// Propose `bytes`. Returns `true` when the entry left this node. +fn propose_owed( + proposer: &dyn SequencerProposer, + vshard_id: u32, + txn_id: TxnId, + kind: OwedKind, + bytes: Vec, +) -> bool { + match proposer.propose(bytes) { + Ok(_) => true, + Err(e) => { + debug!( + vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + kind = kind.label(), + error = %e, + "calvin: sequencer entry not proposed; the stall tick proposes it again", + ); + false + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::Ordering; + use std::time::Duration; + + use nodedb_cluster::calvin::{ParticipantVote, SequencerEntry}; + + use super::*; + use crate::bridge::envelope::Status; + use crate::control::cluster::calvin::scheduler::driver::core::test_proposer::{ + CapturingProposer, elect_data_group_leader, + }; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler, make_sequenced_txn, scheduler_with_pending, spawn_scheduler_loop, + staged_pending, staged_response, + }; + use crate::control::cluster::calvin::scheduler::driver::types::CommitState; + use crate::control::cluster::calvin::scheduler::metrics::sequencer_propose_kind; + + const VSHARD: u32 = 7; + + fn retries(scheduler: &Scheduler, kind: usize) -> u64 { + scheduler.metrics.sequencer_propose_retry_counts[kind].load(Ordering::Relaxed) + } + + /// The leader's vote survives two refused proposals, then stops being + /// proposed once the tally holds it. + #[tokio::test] + async fn staged_leader_vote_is_reproposed_until_the_tally_shows_it() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::failing_first(2); + scheduler.sequencer_proposer = proposer.clone(); + elect_data_group_leader(&scheduler); + let txn_id = TxnId::new(20, 1); + scheduler.registry.seed_expected(cluster_txn_id(txn_id), 2); + scheduler + .pending + .insert(txn_id, staged_pending(make_sequenced_txn(20, 1), txn_id)); + + scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); + assert_eq!(proposer.attempt_count(), 1, "the first proposal is refused"); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 2, "the second is refused too"); + assert!(proposer.accepted().is_empty()); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.accepted(), + vec![SequencerEntry::Vote { + epoch: 20, + position: 1, + vshard: VSHARD, + commit: true, + }] + ); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.attempt_count(), + 3, + "an accepted entry gets one tick to apply" + ); + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.attempt_count(), + 4, + "an accepted entry that did not apply is proposed again" + ); + + scheduler + .registry + .note_vote(cluster_txn_id(txn_id), VSHARD, ParticipantVote::Commit); + scheduler.retry_owed_sequencer_entries(); + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.attempt_count(), + 4, + "an applied vote is not proposed" + ); + assert!(scheduler.owed.is_empty()); + assert_eq!(retries(&scheduler, sequencer_propose_kind::VOTE), 3); + } + + /// The completion ack stays owed after the txn leaves `pending`, and + /// stops being proposed once the registry records it. + #[tokio::test] + async fn completion_ack_is_reproposed_until_the_registry_records_it() { + let txn_id = TxnId::new(21, 0); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: false, + redo_lsn: None, + }, + ); + let proposer = CapturingProposer::failing_first(1); + scheduler.sequencer_proposer = proposer.clone(); + scheduler.registry.seed_expected(cluster_txn_id(txn_id), 2); + + scheduler.finish_resolved_commit(txn_id, staged_response(Status::Ok, None), false, None); + assert!(!scheduler.pending.contains_key(&txn_id)); + assert_eq!(proposer.attempt_count(), 1, "the first proposal is refused"); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.accepted(), + vec![SequencerEntry::CompletionAck { + epoch: 21, + position: 0, + vshard_id: VSHARD, + }] + ); + scheduler.retry_owed_sequencer_entries(); + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 3); + + scheduler + .registry + .note_completion_ack(cluster_txn_id(txn_id), VSHARD); + scheduler.retry_owed_sequencer_entries(); + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.attempt_count(), + 3, + "an applied ack is not proposed" + ); + assert!(scheduler.owed.is_empty()); + assert_eq!( + retries(&scheduler, sequencer_propose_kind::COMPLETION_ACK), + 2 + ); + } + + /// The registry removes a txn's entry once its outcome fired. An owed + /// entry for such a txn is settled, not proposed again. + #[tokio::test] + async fn entry_for_a_txn_whose_outcome_fired_is_settled() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::failing_first(1); + scheduler.sequencer_proposer = proposer.clone(); + let txn_id = TxnId::new(22, 0); + let _outcome = scheduler + .registry + .register_completion(cluster_txn_id(txn_id), 1); + + scheduler.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); + scheduler + .registry + .note_completion_ack(cluster_txn_id(txn_id), VSHARD); + assert_eq!(scheduler.participant_progress(txn_id), None); + + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 1); + assert!(scheduler.owed.is_empty()); + } + + /// A txn this node never seeded keeps its entry owed: a missing registry + /// entry that was never seen does not settle it. + #[tokio::test] + async fn entry_for_an_unseeded_txn_stays_owed() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::accepting(); + scheduler.sequencer_proposer = proposer.clone(); + let txn_id = TxnId::new(23, 0); + + scheduler.propose_sequencer_entry(txn_id, SchedulerProposal::OllpMismatch); + scheduler.retry_owed_sequencer_entries(); + scheduler.retry_owed_sequencer_entries(); + + assert_eq!(proposer.attempt_count(), 2); + assert_eq!(scheduler.owed.len(), 1); + } + + /// The run loop's stall tick drives the retry. + #[tokio::test] + async fn run_loop_reproposes_a_refused_entry_on_the_stall_tick() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::failing_first(1); + scheduler.sequencer_proposer = proposer.clone(); + scheduler.propose_sequencer_entry( + TxnId::new(24, 0), + SchedulerProposal::RoutingFailed { + detail: "unroutable".to_string(), + }, + ); + assert!(proposer.accepted().is_empty()); + + let running = spawn_scheduler_loop(scheduler); + let accepted = tokio::time::timeout(Duration::from_secs(5), async { + while proposer.accepted().is_empty() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await; + running.stop().await; + + assert!(accepted.is_ok(), "the stall tick proposes the entry again"); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index 6b869f713..a42f4ad24 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -10,10 +10,7 @@ use tracing::info; use nodedb_cluster::MultiRaft; use nodedb_cluster::calvin::types::SchedulerInput; -use nodedb_cluster::calvin::{ - CalvinCompletionRegistry, SEQUENCER_GROUP_ID, SequencerEntry, SequencerStateMachine, - VerdictSignal, -}; +use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerStateMachine, VerdictSignal}; use super::super::barrier::{PendingDependentBarrier, ReadResultEvent}; use super::super::config::SchedulerConfig; @@ -22,6 +19,8 @@ use super::catch_up::CatchUpDrain; use super::deferred::DeferredQueue; use super::halt::HaltLatch; use super::intake::IntakeGate; +use super::owed::OwedEntries; +use super::sequencer_proposer::SequencerProposer; use crate::bridge::envelope::Response; use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; @@ -53,10 +52,17 @@ pub struct Scheduler { /// Shared control-plane state used for dispatch, response tracking, WAL, /// and request-id allocation. pub(in crate::control::cluster::calvin::scheduler::driver::core) shared: Arc, - /// Handle to MultiRaft so completion acknowledgements can be proposed to - /// the sequencer group. + /// Handle to MultiRaft for the data-group leader check and the catch-up + /// read of the sequencer log. pub(in crate::control::cluster::calvin::scheduler::driver::core) multi_raft: Arc>, + /// Hands sequencer entries to the sequencer group, locally on its leader + /// and by forward from any other node. + pub(in crate::control::cluster::calvin::scheduler::driver::core) sequencer_proposer: + Arc, + /// Sequencer entries proposed and not yet seen applied. See + /// [`super::owed`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) owed: OwedEntries, /// Shared handle to the sequencer state machine. The state machine records, /// per vShard, the earliest Raft index whose fan-out `try_send` was DROPPED /// (channel Full/Closed) so a dropped `SchedulerInput` never permanently @@ -168,6 +174,9 @@ pub struct SchedulerParams { pub receiver: mpsc::Receiver, pub shared: Arc, pub multi_raft: Arc>, + /// The node's sequencer proposer. Production passes one + /// `RaftSequencerProposer` shared by every scheduler on the node. + pub sequencer_proposer: Arc, /// Shared sequencer state machine, source of the per-vShard catch-up index /// the drain replays from. Same `Arc` the Raft apply loop drives. pub sequencer_state_machine: Arc>, @@ -204,6 +213,7 @@ impl Scheduler { receiver, shared, multi_raft, + sequencer_proposer, sequencer_state_machine, fully_applied_epoch, applied_tail, @@ -233,6 +243,8 @@ impl Scheduler { receiver, shared, multi_raft, + sequencer_proposer, + owed: OwedEntries::new(), sequencer_state_machine, lock_manager, pending: BTreeMap::new(), @@ -432,6 +444,8 @@ impl Scheduler { // A closed intake gate skips the drain until it opens. catch_up_resume = !intake_open || self.drain_catch_up() == CatchUpDrain::Remaining; + // Propose again every owed sequencer entry not yet applied. + self.retry_owed_sequencer_entries(); // The top-of-loop check_awaiting_verdict_stalls / // check_dependent_barrier_timeouts and the deferred re-send // pass run on every wake; this arm guarantees the loop wakes @@ -441,46 +455,6 @@ impl Scheduler { } } - /// Encode `entry` as MessagePack and propose it to the sequencer Raft group. - /// - /// Logs a warning on encode failure or propose failure; never panics. - /// `op_name` is a short human-readable label used in warning messages - /// (e.g. `"completion ack"`, `"OLLP mismatch signal"`). - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn propose_sequencer_entry( - &self, - entry: SequencerEntry, - txn_id: TxnId, - op_name: &str, - ) { - match zerompk::to_msgpack_vec(&entry) { - Ok(bytes) => { - if let Err(e) = self - .multi_raft - .lock() - .unwrap_or_else(|p| p.into_inner()) - .propose_to_group(SEQUENCER_GROUP_ID, bytes) - { - tracing::warn!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: failed to propose {op_name}", - ); - } - } - Err(e) => { - tracing::warn!( - vshard_id = self.vshard_id, - epoch = txn_id.epoch, - position = txn_id.position, - error = %e, - "calvin: failed to encode {op_name}", - ); - } - } - } - /// Allocate a fresh request ID for a dispatch. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn next_request_id( &self, diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/mod.rs new file mode 100644 index 000000000..392949c8e --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/mod.rs @@ -0,0 +1,10 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! How the Calvin scheduler hands its sequencer entries to the sequencer +//! Raft group. + +pub mod raft; +pub mod seam; + +pub use raft::{MAX_INFLIGHT_SEQUENCER_FORWARDS, RaftSequencerProposer}; +pub use seam::{ProposeDispatch, SequencerProposeError, SequencerProposer}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs new file mode 100644 index 000000000..68bbdb80f --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs @@ -0,0 +1,291 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! [`RaftSequencerProposer`]: the production [`SequencerProposer`]. +//! +//! On the sequencer leader it appends the entry to the local sequencer +//! group. On any other node it forwards the entry to the leader over the +//! cluster transport as a `DataProposeRequest` with the `Sequencer` target, +//! the same RPC that forwards data-group proposals. The leader's handler +//! proposes it to its sequencer group. + +use std::collections::BTreeSet; +use std::sync::{Arc, Mutex}; + +use tokio::sync::Semaphore; +use tracing::debug; + +use nodedb_cluster::MultiRaft; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::rpc_codec::{DataProposeRequest, ProposeTarget, RaftRpc}; + +use super::seam::{ProposeDispatch, SequencerProposeError, SequencerProposer}; +use crate::control::server::exchange::resolve::register_peers_from_topology; +use crate::control::state::SharedState; + +/// Most forward RPCs one proposer keeps in flight. A proposal past this +/// limit fails with [`SequencerProposeError::ForwardBusy`], and the +/// scheduler proposes it again on a later stall tick. +pub const MAX_INFLIGHT_SEQUENCER_FORWARDS: usize = 64; + +/// How a proposal reaches the sequencer group from this node. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum SequencerRoute { + /// This node leads the sequencer group. + Local, + /// Another node leads the sequencer group. + Forward { leader: u64 }, +} + +/// Pick the route from this node's view of the sequencer group. +/// +/// `leader` is the leader this node observes, `0` while unknown. A node +/// that is not leader but observes itself as leader has a stale view, so it +/// has no leader to send to. +pub(super) fn sequencer_route( + is_leader: bool, + leader: u64, + local_node: u64, +) -> Result { + if is_leader { + return Ok(SequencerRoute::Local); + } + if leader == 0 || leader == local_node { + return Err(SequencerProposeError::NoLeader); + } + Ok(SequencerRoute::Forward { leader }) +} + +/// Proposes sequencer entries locally on the sequencer leader and forwards +/// them to the leader from every other node. +pub struct RaftSequencerProposer { + node_id: u64, + multi_raft: Arc>, + /// Source of the cluster transport and topology for forwards. + shared: Arc, + /// One permit per forward RPC in flight. + forwards: Arc, +} + +impl RaftSequencerProposer { + pub fn new(node_id: u64, multi_raft: Arc>, shared: Arc) -> Self { + Self { + node_id, + multi_raft, + shared, + forwards: Arc::new(Semaphore::new(MAX_INFLIGHT_SEQUENCER_FORWARDS)), + } + } + + /// Send `bytes` to `leader` on a spawned task. + /// + /// The task logs a refused or failed forward at `debug`. The scheduler + /// does not need the RPC result: it proposes the entry again until the + /// completion registry shows it applied. + fn forward( + &self, + leader: u64, + bytes: Vec, + ) -> Result { + let Some(transport) = self.shared.cluster_transport.as_ref() else { + return Err(SequencerProposeError::NoTransport { leader }); + }; + let permit = Arc::clone(&self.forwards) + .try_acquire_owned() + .map_err(|_| SequencerProposeError::ForwardBusy { + leader, + limit: MAX_INFLIGHT_SEQUENCER_FORWARDS, + })?; + register_peers_from_topology(&self.shared, transport, &BTreeSet::from([leader])); + let transport = Arc::clone(transport); + tokio::spawn(async move { + let _permit = permit; + let rpc = RaftRpc::DataProposeRequest(DataProposeRequest { + target: ProposeTarget::Sequencer, + bytes, + }); + match transport.send_rpc(leader, rpc).await { + Ok(RaftRpc::DataProposeResponse(resp)) if resp.success => {} + Ok(RaftRpc::DataProposeResponse(resp)) => debug!( + leader, + leader_hint = ?resp.leader_hint, + error = %resp.error_message, + "calvin: sequencer leader refused a forwarded entry", + ), + Ok(other) => debug!( + leader, + response = ?other, + "calvin: unexpected reply to a forwarded sequencer entry", + ), + Err(e) => debug!( + leader, + error = %e, + "calvin: forward of a sequencer entry failed", + ), + } + }); + Ok(ProposeDispatch::Forwarded { leader }) + } +} + +impl SequencerProposer for RaftSequencerProposer { + fn propose(&self, bytes: Vec) -> Result { + let leader = { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let route = sequencer_route( + mr.is_group_leader(SEQUENCER_GROUP_ID), + mr.group_leader(SEQUENCER_GROUP_ID), + self.node_id, + )?; + match route { + SequencerRoute::Local => { + mr.propose_to_group(SEQUENCER_GROUP_ID, bytes)?; + return Ok(ProposeDispatch::Local); + } + SequencerRoute::Forward { leader } => leader, + } + }; + self.forward(leader, bytes) + } +} + +#[cfg(test)] +mod tests { + use std::time::{Duration, Instant}; + + use nodedb_cluster::RoutingTable; + use nodedb_raft::message::AppendEntriesRequest; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + const LOCAL_NODE: u64 = 1; + const REMOTE_LEADER: u64 = 2; + + fn shared_state(dir: &std::path::Path) -> Arc { + let wal = Arc::new(WalManager::open_for_testing(&dir.join("test.wal")).expect("wal")); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + SharedState::new(dispatcher, wal).expect("shared state") + } + + /// A `MultiRaft` on `LOCAL_NODE` with a sequencer group of `peers`. + fn multi_raft(dir: &std::path::Path, peers: Vec) -> Arc> { + let rt = RoutingTable::uniform(1, &[LOCAL_NODE], 1); + let mut mr = MultiRaft::new(LOCAL_NODE, rt, dir.to_path_buf()); + mr.add_group(SEQUENCER_GROUP_ID, peers) + .expect("add sequencer group"); + Arc::new(Mutex::new(mr)) + } + + #[test] + fn route_is_local_on_the_leader() { + assert_eq!( + sequencer_route(true, LOCAL_NODE, LOCAL_NODE).expect("route"), + SequencerRoute::Local + ); + } + + #[test] + fn route_forwards_to_a_remote_leader() { + assert_eq!( + sequencer_route(false, REMOTE_LEADER, LOCAL_NODE).expect("route"), + SequencerRoute::Forward { + leader: REMOTE_LEADER + } + ); + } + + #[test] + fn route_has_no_target_without_a_known_leader() { + assert!(matches!( + sequencer_route(false, 0, LOCAL_NODE), + Err(SequencerProposeError::NoLeader) + )); + assert!(matches!( + sequencer_route(false, LOCAL_NODE, LOCAL_NODE), + Err(SequencerProposeError::NoLeader) + )); + } + + #[tokio::test] + async fn sequencer_leader_appends_the_entry_locally() { + let dir = tempfile::tempdir().expect("tempdir"); + let mr = multi_raft(dir.path(), vec![]); + { + let mut guard = mr.lock().unwrap_or_else(|p| p.into_inner()); + if let Some(node) = guard.groups_mut().get_mut(&SEQUENCER_GROUP_ID) { + // no-determinism: test-only forced election deadline so the single voter campaigns immediately. + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + for _ in 0..20 { + guard.tick().expect("tick"); + if guard.is_group_leader(SEQUENCER_GROUP_ID) { + break; + } + } + assert!(guard.is_group_leader(SEQUENCER_GROUP_ID)); + } + let before = mr + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_log_index(SEQUENCER_GROUP_ID) + .unwrap_or(0); + let proposer = + RaftSequencerProposer::new(LOCAL_NODE, Arc::clone(&mr), shared_state(dir.path())); + + let dispatch = proposer.propose(vec![9, 9]).expect("local propose"); + + assert_eq!(dispatch, ProposeDispatch::Local); + let after = mr + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_log_index(SEQUENCER_GROUP_ID) + .unwrap_or(0); + assert_eq!(after, before + 1); + } + + /// A follower that knows the remote leader takes the forward path. The + /// fixture has no cluster transport, so the forward stops there with + /// the leader named, and nothing is appended to the local log. + #[tokio::test] + async fn follower_forwards_to_the_sequencer_leader() { + let dir = tempfile::tempdir().expect("tempdir"); + let mr = multi_raft(dir.path(), vec![REMOTE_LEADER]); + let before = { + let mut guard = mr.lock().unwrap_or_else(|p| p.into_inner()); + guard + .handle_append_entries(&AppendEntriesRequest { + term: 1, + leader_id: REMOTE_LEADER, + prev_log_index: 0, + prev_log_term: 0, + entries: Vec::new(), + leader_commit: 0, + group_id: SEQUENCER_GROUP_ID, + }) + .expect("heartbeat"); + assert_eq!(guard.group_leader(SEQUENCER_GROUP_ID), REMOTE_LEADER); + guard.last_log_index(SEQUENCER_GROUP_ID).unwrap_or(0) + }; + let proposer = + RaftSequencerProposer::new(LOCAL_NODE, Arc::clone(&mr), shared_state(dir.path())); + + let result = proposer.propose(vec![9, 9]); + + assert!( + matches!( + result, + Err(SequencerProposeError::NoTransport { + leader: REMOTE_LEADER + }) + ), + "{result:?}" + ); + let after = mr + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_log_index(SEQUENCER_GROUP_ID) + .unwrap_or(0); + assert_eq!(after, before, "a follower appends nothing locally"); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs new file mode 100644 index 000000000..a39513bba --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs @@ -0,0 +1,52 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The [`SequencerProposer`] seam and its result types. + +use nodedb_cluster::error::ClusterError; + +/// Where a sequencer proposal went. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ProposeDispatch { + /// This node leads the sequencer group and appended the entry. + Local, + /// A forward task carries the entry to the sequencer leader. + Forwarded { leader: u64 }, +} + +/// Why a sequencer proposal did not leave this node. +#[derive(Debug, thiserror::Error)] +pub enum SequencerProposeError { + /// This node sees no sequencer leader, or sees itself as leader after it + /// stepped down. + #[error("no sequencer leader is known")] + NoLeader, + /// This node does not lead the sequencer group and has no cluster + /// transport to reach the leader. + #[error("sequencer leader is node {leader}, and this node has no cluster transport")] + NoTransport { leader: u64 }, + /// The forward limit is reached. The owed-entry sweep proposes the entry + /// again on a later tick. + #[error("{limit} sequencer forwards are in flight; the entry for node {leader} waits")] + ForwardBusy { leader: u64, limit: usize }, + /// The local sequencer group refused the proposal. + #[error("sequencer propose: {0}")] + Cluster(#[from] ClusterError), +} + +impl From for crate::Error { + fn from(e: SequencerProposeError) -> Self { + crate::Error::Dispatch { + detail: e.to_string(), + } + } +} + +/// Hands encoded sequencer entries to the sequencer Raft group. +/// +/// `Ok` means the entry left this node. It does not mean the entry is +/// applied: a leader change can drop it. The scheduler learns that an entry +/// is applied from the completion registry, and proposes it again until then. +pub trait SequencerProposer: Send + Sync { + /// Propose one msgpack-encoded `SequencerEntry`. + fn propose(&self, bytes: Vec) -> Result; +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_proposer.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_proposer.rs new file mode 100644 index 000000000..6602ec96e --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_proposer.rs @@ -0,0 +1,102 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Test fixtures for the scheduler's sequencer proposals: a capturing +//! [`SequencerProposer`] and a helper that makes the fixture node lead its +//! vShard's data group. + +use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; + +use nodedb_cluster::calvin::SequencerEntry; + +use super::scheduler::Scheduler; +use super::sequencer_proposer::{ProposeDispatch, SequencerProposeError, SequencerProposer}; + +/// Records every proposal it receives. The first `fail_first` proposals fail +/// the way a node that does not lead the sequencer group fails. Every later +/// one succeeds. +#[derive(Default)] +pub(super) struct CapturingProposer { + fail_first: usize, + /// Every proposal, decoded, with whether it succeeded. + attempts: Mutex>, +} + +impl CapturingProposer { + /// A proposer that accepts every proposal. + pub(super) fn accepting() -> Arc { + Arc::new(Self::default()) + } + + /// A proposer that refuses the first `n` proposals. + pub(super) fn failing_first(n: usize) -> Arc { + Arc::new(Self { + fail_first: n, + attempts: Mutex::new(Vec::new()), + }) + } + + /// Number of proposals received, failed ones included. + pub(super) fn attempt_count(&self) -> usize { + self.attempts + .lock() + .unwrap_or_else(|p| p.into_inner()) + .len() + } + + /// The proposals that succeeded, in order. + pub(super) fn accepted(&self) -> Vec { + self.attempts + .lock() + .unwrap_or_else(|p| p.into_inner()) + .iter() + .filter(|(_, ok)| *ok) + .map(|(entry, _)| entry.clone()) + .collect() + } +} + +impl SequencerProposer for CapturingProposer { + fn propose(&self, bytes: Vec) -> Result { + let entry: SequencerEntry = + zerompk::from_msgpack(&bytes).expect("scheduler proposes a valid SequencerEntry"); + let mut attempts = self.attempts.lock().unwrap_or_else(|p| p.into_inner()); + let ok = attempts.len() >= self.fail_first; + attempts.push((entry, ok)); + if ok { + Ok(ProposeDispatch::Local) + } else { + Err(SequencerProposeError::NoLeader) + } + } +} + +/// Make the fixture node the elected leader of the data group that owns +/// `scheduler`'s vShard, so leader-only proposals such as the commit vote +/// run. +pub(super) fn elect_data_group_leader(scheduler: &Scheduler) { + let mut mr = scheduler + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()); + let group_id = mr + .routing() + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(scheduler.vshard_id) + .expect("the fixture routing maps every vShard"); + if !mr.contains_group(group_id) { + mr.add_group(group_id, vec![]).expect("add data group"); + } + if let Some(node) = mr.groups_mut().get_mut(&group_id) { + // no-determinism: test-only forced election deadline so the single voter campaigns immediately. + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + for _ in 0..20 { + mr.tick().expect("tick"); + if mr.vshard_role_is_leader(scheduler.vshard_id) { + return; + } + } + panic!("data group did not elect the single fixture node"); +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs index 0c46a9ca4..d749ad559 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -26,6 +26,7 @@ use crate::control::cluster::calvin::scheduler::driver::barrier::ReadResultEvent use crate::control::cluster::calvin::scheduler::driver::core::scheduler::{ Scheduler, SchedulerParams, }; +use crate::control::cluster::calvin::scheduler::driver::core::test_proposer::CapturingProposer; use crate::control::cluster::calvin::scheduler::driver::types::{CommitState, PendingTxn}; use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; @@ -71,6 +72,7 @@ pub(super) fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::Temp receiver, shared, multi_raft, + sequencer_proposer: CapturingProposer::accepting(), sequencer_state_machine, // A freshly-built scheduler has applied nothing, so its watermark is the // not-yet-applied sentinel (matching `read_applied_recovery` for a clean @@ -129,6 +131,7 @@ pub(super) fn build_test_scheduler_with_data_side( receiver, shared, multi_raft, + sequencer_proposer: CapturingProposer::accepting(), sequencer_state_machine, fully_applied_epoch: NOT_YET_APPLIED_EPOCH, applied_tail: BTreeSet::new(), diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/mod.rs index 325fa7ea5..5baf6021c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/mod.rs @@ -8,4 +8,7 @@ pub mod types; pub use barrier::ReadResultEvent; pub use config::SchedulerConfig; -pub use core::{CalvinReadResultProposal, Scheduler, SchedulerParams, propose_calvin_read_result}; +pub use core::{ + CalvinReadResultProposal, RaftSequencerProposer, Scheduler, SchedulerParams, SequencerProposer, + propose_calvin_read_result, +}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs index 7a70d101b..9019eb688 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs @@ -76,6 +76,10 @@ pub struct SchedulerMetrics { /// Reason of the halt. An index into [`apply_halt_reason`], read only /// while `apply_halted` is 1. pub apply_halt_reason: AtomicU64, + /// Owed sequencer entries re-proposed because their effect was not yet + /// applied, by kind. Indexes are the constants in + /// [`sequencer_propose_kind`]. + pub sequencer_propose_retry_counts: [AtomicU64; 4], } /// Reason codes for `nodedb_calvin_infra_abort_total`. @@ -129,6 +133,16 @@ pub mod apply_halt_reason { ]; } +/// Kind codes for `nodedb_calvin_sequencer_propose_retry_total`. +pub mod sequencer_propose_kind { + pub const VOTE: usize = 0; + pub const COMPLETION_ACK: usize = 1; + pub const OLLP_MISMATCH: usize = 2; + pub const ROUTING_FAILED: usize = 3; + + pub const LABELS: &[&str] = &["vote", "completion_ack", "ollp_mismatch", "routing_failed"]; +} + impl SchedulerMetrics { pub fn new() -> Arc { Arc::new(Self::default()) @@ -220,6 +234,15 @@ impl SchedulerMetrics { self.apply_halted.store(1, Ordering::Relaxed); } + /// Record that an owed sequencer entry of `kind` was re-proposed. + /// + /// `kind` must be one of the constants in [`sequencer_propose_kind`]. + pub fn record_sequencer_propose_retry(&self, kind: usize) { + if let Some(counter) = self.sequencer_propose_retry_counts.get(kind) { + counter.fetch_add(1, Ordering::Relaxed); + } + } + /// Set the in-flight backlog gauge. pub fn set_intake_backlog(&self, backlog: usize) { self.intake_backlog.store(backlog as u64, Ordering::Relaxed); @@ -380,83 +403,7 @@ impl SchedulerMetrics { self.catch_up_log_compacted.load(Ordering::Relaxed) ); - let _ = writeln!( - out, - "# HELP nodedb_calvin_dispatch_deferred_total \ - Scheduler dispatches refused at dispatcher capacity and parked for re-send." - ); - let _ = writeln!(out, "# TYPE nodedb_calvin_dispatch_deferred_total counter"); - let _ = writeln!( - out, - "nodedb_calvin_dispatch_deferred_total{{{label}}} {}", - self.dispatch_deferred_count.load(Ordering::Relaxed) - ); - - let _ = writeln!( - out, - "# HELP nodedb_calvin_dispatch_deferred_depth \ - Scheduler requests parked for re-send, waiting for dispatcher capacity." - ); - let _ = writeln!(out, "# TYPE nodedb_calvin_dispatch_deferred_depth gauge"); - let _ = writeln!( - out, - "nodedb_calvin_dispatch_deferred_depth{{{label}}} {}", - self.dispatch_deferred_depth.load(Ordering::Relaxed) - ); - - let _ = writeln!( - out, - "# HELP nodedb_calvin_intake_gate_closed \ - 1 while the scheduler takes no new sequenced input." - ); - let _ = writeln!(out, "# TYPE nodedb_calvin_intake_gate_closed gauge"); - let _ = writeln!( - out, - "nodedb_calvin_intake_gate_closed{{{label}}} {}", - self.intake_gate_closed.load(Ordering::Relaxed) - ); - - let _ = writeln!( - out, - "# HELP nodedb_calvin_intake_backlog \ - Pending, blocked, and dependent-barrier txns in the scheduler." - ); - let _ = writeln!(out, "# TYPE nodedb_calvin_intake_backlog gauge"); - let _ = writeln!( - out, - "nodedb_calvin_intake_backlog{{{label}}} {}", - self.intake_backlog.load(Ordering::Relaxed) - ); - - let _ = writeln!( - out, - "# HELP nodedb_calvin_intake_gate_closed_total \ - Times the scheduler stopped taking new sequenced input, by reason." - ); - let _ = writeln!(out, "# TYPE nodedb_calvin_intake_gate_closed_total counter"); - for (i, &reason_label) in intake_closure_reason::LABELS.iter().enumerate() { - let _ = writeln!( - out, - "nodedb_calvin_intake_gate_closed_total{{{label},reason=\"{reason_label}\"}} {}", - self.intake_gate_closed_counts[i].load(Ordering::Relaxed) - ); - } - - let _ = writeln!( - out, - "# HELP nodedb_calvin_apply_halted \ - 1 once the scheduler halted on an apply error it cannot mark applied, by reason." - ); - let _ = writeln!(out, "# TYPE nodedb_calvin_apply_halted gauge"); - let halted = self.apply_halted.load(Ordering::Relaxed) == 1; - let halt_reason = self.apply_halt_reason.load(Ordering::Relaxed); - for (i, &reason_label) in apply_halt_reason::LABELS.iter().enumerate() { - let value = u64::from(halted && halt_reason == i as u64); - let _ = writeln!( - out, - "nodedb_calvin_apply_halted{{{label},reason=\"{reason_label}\"}} {value}" - ); - } + self.render_flow_prometheus(&mut out, &label); out } @@ -484,6 +431,7 @@ impl Default for SchedulerMetrics { intake_gate_closed_counts: std::array::from_fn(|_| AtomicU64::new(0)), apply_halted: AtomicU64::new(0), apply_halt_reason: AtomicU64::new(0), + sequencer_propose_retry_counts: std::array::from_fn(|_| AtomicU64::new(0)), } } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/metrics_flow.rs b/nodedb/src/control/cluster/calvin/scheduler/metrics_flow.rs new file mode 100644 index 000000000..019437c27 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/metrics_flow.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Prometheus rendering of the scheduler's flow metrics: deferred dispatch, +//! the intake gate, the apply halt, and sequencer propose retries. + +use std::fmt::Write as _; +use std::sync::atomic::Ordering; + +use super::metrics::{ + SchedulerMetrics, apply_halt_reason, intake_closure_reason, sequencer_propose_kind, +}; + +impl SchedulerMetrics { + /// Append the flow metrics for the vShard in `label` to `out`. + pub(super) fn render_flow_prometheus(&self, out: &mut String, label: &str) { + let _ = writeln!( + out, + "# HELP nodedb_calvin_dispatch_deferred_total \ + Scheduler dispatches refused at dispatcher capacity and parked for re-send." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_dispatch_deferred_total counter"); + let _ = writeln!( + out, + "nodedb_calvin_dispatch_deferred_total{{{label}}} {}", + self.dispatch_deferred_count.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_dispatch_deferred_depth \ + Scheduler requests parked for re-send, waiting for dispatcher capacity." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_dispatch_deferred_depth gauge"); + let _ = writeln!( + out, + "nodedb_calvin_dispatch_deferred_depth{{{label}}} {}", + self.dispatch_deferred_depth.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_gate_closed \ + 1 while the scheduler takes no new sequenced input." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_gate_closed gauge"); + let _ = writeln!( + out, + "nodedb_calvin_intake_gate_closed{{{label}}} {}", + self.intake_gate_closed.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_backlog \ + Pending, blocked, and dependent-barrier txns in the scheduler." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_backlog gauge"); + let _ = writeln!( + out, + "nodedb_calvin_intake_backlog{{{label}}} {}", + self.intake_backlog.load(Ordering::Relaxed) + ); + + let _ = writeln!( + out, + "# HELP nodedb_calvin_intake_gate_closed_total \ + Times the scheduler stopped taking new sequenced input, by reason." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_intake_gate_closed_total counter"); + for (i, &reason_label) in intake_closure_reason::LABELS.iter().enumerate() { + let _ = writeln!( + out, + "nodedb_calvin_intake_gate_closed_total{{{label},reason=\"{reason_label}\"}} {}", + self.intake_gate_closed_counts[i].load(Ordering::Relaxed) + ); + } + + let _ = writeln!( + out, + "# HELP nodedb_calvin_apply_halted \ + 1 once the scheduler halted on an apply error it cannot mark applied, by reason." + ); + let _ = writeln!(out, "# TYPE nodedb_calvin_apply_halted gauge"); + let halted = self.apply_halted.load(Ordering::Relaxed) == 1; + let halt_reason = self.apply_halt_reason.load(Ordering::Relaxed); + for (i, &reason_label) in apply_halt_reason::LABELS.iter().enumerate() { + let value = u64::from(halted && halt_reason == i as u64); + let _ = writeln!( + out, + "nodedb_calvin_apply_halted{{{label},reason=\"{reason_label}\"}} {value}" + ); + } + let _ = writeln!( + out, + "# HELP nodedb_calvin_sequencer_propose_retry_total \ + Owed sequencer entries re-proposed because their effect was not yet applied, by kind." + ); + let _ = writeln!( + out, + "# TYPE nodedb_calvin_sequencer_propose_retry_total counter" + ); + for (i, &kind_label) in sequencer_propose_kind::LABELS.iter().enumerate() { + let _ = writeln!( + out, + "nodedb_calvin_sequencer_propose_retry_total{{{label},kind=\"{kind_label}\"}} {}", + self.sequencer_propose_retry_counts[i].load(Ordering::Relaxed) + ); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn propose_retry_counter_renders_per_kind() { + let m = SchedulerMetrics::new(); + m.record_sequencer_propose_retry(sequencer_propose_kind::VOTE); + m.record_sequencer_propose_retry(sequencer_propose_kind::VOTE); + m.record_sequencer_propose_retry(sequencer_propose_kind::COMPLETION_ACK); + let out = m.render_prometheus(3); + assert!( + out.contains( + "nodedb_calvin_sequencer_propose_retry_total{vshard=\"3\",kind=\"vote\"} 2" + ) + ); + assert!(out.contains( + "nodedb_calvin_sequencer_propose_retry_total{vshard=\"3\",kind=\"completion_ack\"} 1" + )); + assert!(out.contains( + "nodedb_calvin_sequencer_propose_retry_total{vshard=\"3\",kind=\"routing_failed\"} 0" + )); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/mod.rs index 4e7249c27..ec86c0cd1 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/mod.rs @@ -4,12 +4,13 @@ pub mod applied_gate; pub mod driver; pub mod lock; pub mod metrics; +mod metrics_flow; pub mod recovery; pub use applied_gate::AppliedGate; pub use driver::{ - CalvinReadResultProposal, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, - propose_calvin_read_result, + CalvinReadResultProposal, RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, + SchedulerParams, SequencerProposer, propose_calvin_read_result, }; pub use lock::{AcquireOutcome, HotKeyTable, LockKey, LockManager, LockMode, TxnId}; // Existing call sites reference this module as `scheduler::lock_manager::…`; diff --git a/nodedb/src/control/cluster/start_raft_helpers.rs b/nodedb/src/control/cluster/start_raft_helpers.rs index 9626be8f8..bc2f2b3f3 100644 --- a/nodedb/src/control/cluster/start_raft_helpers.rs +++ b/nodedb/src/control/cluster/start_raft_helpers.rs @@ -11,7 +11,8 @@ use nodedb_cluster::wire::VShardEnvelope; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; use crate::control::cluster::calvin::scheduler::read_applied_recovery; use crate::control::cluster::calvin::{ - ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, + RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, + SequencerProposer, }; use crate::control::cluster::handle::ClusterHandle; use crate::control::state::SharedState; @@ -114,6 +115,8 @@ struct ReconcileSchedulersParams<'a> { routing: &'a Arc>, shared: &'a Arc, raft_loop_handle: &'a Arc>, + /// One proposer per node, so its forward limit bounds the whole node. + sequencer_proposer: &'a Arc, sequencer_state_machine: &'a Arc>, calvin_read_result_senders: &'a ReadResultSenders, calvin_completion_registry: &'a Arc, @@ -138,6 +141,7 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: routing, shared, raft_loop_handle, + sequencer_proposer, sequencer_state_machine, calvin_read_result_senders, calvin_completion_registry, @@ -237,6 +241,7 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: receiver: sequenced_rx, shared: Arc::clone(shared), multi_raft: raft_loop_handle.clone(), + sequencer_proposer: Arc::clone(sequencer_proposer), sequencer_state_machine: Arc::clone(sequencer_state_machine), fully_applied_epoch: recovery.fully_applied_epoch, applied_tail: recovery.applied_tail, @@ -321,6 +326,11 @@ pub(super) fn spawn_vshard_schedulers( let node_id = handle.node_id; let routing = Arc::clone(&handle.routing); + let sequencer_proposer: Arc = Arc::new(RaftSequencerProposer::new( + node_id, + Arc::clone(&raft_loop_handle), + Arc::clone(shared), + )); // Initial reconcile: schedulers for vShards this node already knows it hosts. reconcile_vshard_schedulers(ReconcileSchedulersParams { @@ -328,6 +338,7 @@ pub(super) fn spawn_vshard_schedulers( routing: &routing, shared, raft_loop_handle: &raft_loop_handle, + sequencer_proposer: &sequencer_proposer, sequencer_state_machine, calvin_read_result_senders, calvin_completion_registry, @@ -362,6 +373,7 @@ pub(super) fn spawn_vshard_schedulers( routing: &routing, shared: &shared_task, raft_loop_handle: &raft_loop_handle, + sequencer_proposer: &sequencer_proposer, sequencer_state_machine: &sm_task, calvin_read_result_senders: &rr_task, calvin_completion_registry: ®istry_task, From b32966d83816c5f6a8fce83fb7670292f1214cfa Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 18:52:05 +0800 Subject: [PATCH 09/64] refactor(scheduler): split Calvin lock manager into a directory module manager.rs held the lock table, acquire, wound-wait, release, try_acquire, and introspection logic in one file. It becomes a manager/ directory with types.rs, acquire.rs, wound_wait.rs, release.rs, try_acquire.rs, and introspection.rs, each keeping its existing logic. --- .../cluster/calvin/scheduler/lock/manager.rs | 1136 ----------------- .../calvin/scheduler/lock/manager/acquire.rs | 378 ++++++ .../scheduler/lock/manager/introspection.rs | 96 ++ .../calvin/scheduler/lock/manager/mod.rs | 38 + .../calvin/scheduler/lock/manager/release.rs | 334 +++++ .../scheduler/lock/manager/try_acquire.rs | 40 + .../calvin/scheduler/lock/manager/types.rs | 101 ++ .../scheduler/lock/manager/wound_wait.rs | 313 +++++ 8 files changed, 1300 insertions(+), 1136 deletions(-) delete mode 100644 nodedb/src/control/cluster/calvin/scheduler/lock/manager.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/lock/manager/acquire.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/lock/manager/introspection.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/lock/manager/mod.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/lock/manager/release.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/lock/manager/try_acquire.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/lock/manager/types.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/lock/manager/wound_wait.rs diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager.rs deleted file mode 100644 index 258d6c328..000000000 --- a/nodedb/src/control/cluster/calvin/scheduler/lock/manager.rs +++ /dev/null @@ -1,1136 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Deterministic lock manager for the Calvin scheduler. -//! -//! # Design -//! -//! The lock manager provides a deterministic, totally-ordered lock table over -//! per-key entries keyed by [`LockKey`]. Locks come in two modes: `Exclusive` -//! (one holder, excludes all others) and `Shared` (many compatible holders). -//! The Calvin batch acquire path takes every key in a transaction's -//! `read_set ∪ write_set` as an `Exclusive` lock; single-key `Shared` locks are -//! available via [`LockManager::acquire_shared`]. -//! -//! # Determinism -//! -//! `BTreeMap` is used throughout (not `HashMap`) so that iteration order is -//! deterministic and reproducible across replicas. This is a correctness -//! requirement, not a style preference. - -use std::collections::btree_map::Entry; -use std::collections::{BTreeMap, BTreeSet, VecDeque}; - -use smallvec::smallvec; - -use super::lock_entry::{AcquireOutcome, LockEntry, LockMode}; -use super::lock_key::{LockKey, TxnId}; - -// ── LockManager ─────────────────────────────────────────────────────────────── - -/// Deterministic Calvin lock manager for one vshard. -/// -/// Manages an in-memory lock table keyed by [`LockKey`]. The table is held in -/// a `BTreeMap` so iteration is always deterministic. -/// -/// # Key sets tracked per transaction -/// -/// - `held_locks`: key sets for transactions that are a current holder on ALL -/// their keys and are actively executing (i.e. dispatched to the Data Plane). -/// - `pending_keys`: key sets for transactions that are blocked waiting for at -/// least one key. When `release` promotes a blocked txn to holder on every -/// one of its keys, the entry moves from `pending_keys` to `held_locks`. -pub struct LockManager { - /// Per-key lock entries. Uses `BTreeMap` for deterministic iteration. - /// `pub(super)` so the sibling `reap` module can scan entries for - /// lease-expired reservations without a public accessor. - pub(super) table: BTreeMap, - /// Per-transaction set of currently held keys for **dispatched** txns. - /// Used by `release` to iterate the key set without a full table scan. - /// `pub(super)` — see `table`. - pub(super) held_locks: BTreeMap>, - /// Key sets for **blocked** (not-yet-dispatched) txns. Populated when - /// `acquire` returns `Blocked`; cleared (moved to `held_locks`) when all - /// keys have been acquired on the promotion path inside `release`. - pending_keys: BTreeMap>, -} - -/// Outcome of inspecting a single key during [`LockManager::acquire_shared`]. -enum SharedGrant { - /// The shared lock was granted (key was free or already held shared). - Granted, - /// The key is held exclusively by another txn; the request was enqueued. - Blocked, -} - -/// The wound-wait decision for an exclusive requester that meets a conflict. -enum ExclusiveWait { - /// Every conflicting holder is a shared reservation and the requester is - /// older than all of them: wound (revoke) those shared holders and proceed. - Wound, - /// The requester must block: a conflicting holder is exclusive, or the - /// requester is younger than some conflicting shared holder. - Block, -} - -/// The waiters promoted off one key when its holders drained, together with the -/// action to take on the now-empty entry. -enum Promotion { - /// No waiters remained; the entry should be removed entirely. - Freed, - /// These waiters were installed as the new holders. - Promoted(Vec), -} - -impl LockManager { - /// Create an empty lock manager. - pub fn new() -> Self { - Self { - table: BTreeMap::new(), - held_locks: BTreeMap::new(), - pending_keys: BTreeMap::new(), - } - } - - /// Attempt to acquire **exclusive** locks on all keys for `txn`. - /// - /// If every key is free, already held exclusively by `txn` (promoted from - /// waiter), or held **shared solely by `txn`** (a reservation this txn placed - /// earlier, now upgraded in place to exclusive), records `txn` as sole holder - /// of each key and returns [`AcquireOutcome::Ready`]. - /// - /// Otherwise the whole key set is classified under the WOUND-WAIT discipline - /// (see [`Self::wound_or_block`]). `TxnId` order (`(epoch, position)`) is - /// the replicated total order, so "older" means a smaller id and the - /// decision is a pure function of the lock table plus the replicated ids — - /// every replica computes it identically: - /// - Any conflicting holder is **exclusive** → `txn` waits (an exclusive - /// holder is executing/applied work and is never wounded). - /// - All conflicting holders are **shared** reservations and `txn` is older - /// than every one of them → **wound** them all (revoke to plain OCC, no - /// notification) and take every key. Wounding is silent. - /// - Otherwise (`txn` younger than some conflicting shared holder) → wait. - /// - /// On the wait path `txn` is enqueued as an exclusive waiter on every - /// conflicting key and holds none; its key set is stored in `pending_keys` - /// so `release` can promote it atomically when all keys become available. - /// The whole key set is evaluated before any mutation, so acquisition stays - /// all-keys-or-none — the manager never partially wounds and then blocks. - pub fn acquire(&mut self, txn: TxnId, keys: BTreeSet) -> AcquireOutcome { - // First pass: determine whether any key is held by a *different* txn. - // A key already held exclusively by `txn` (promoted via release) or held - // shared solely by `txn` (an earlier reservation) counts as available — - // the former is a no-op re-acquire, the latter a self-upgrade to exclusive. - let all_available = keys.iter().all(|k| { - self.table.get(k).is_none_or(|entry| { - entry.held_exclusively_by(txn) || entry.held_shared_solely_by(txn) - }) - }); - - if all_available { - // Acquire all keys. For keys not yet in the table (free), insert a - // new exclusive entry. For keys already held by this txn (promoted - // waiter), leave the entry unchanged — the waiter queue is intact. - for key in &keys { - match self.table.get_mut(key) { - None => { - self.table.insert( - key.clone(), - LockEntry { - mode: LockMode::Exclusive, - holders: smallvec![txn], - waiters: VecDeque::new(), - }, - ); - } - Some(entry) => { - // A key this txn already holds shared-solely is upgraded - // to exclusive in place (holders is exactly `[txn]`, so no - // holder change and any waiter queue stays intact). A key - // already held exclusively by `txn` is left unchanged. - if entry.held_shared_solely_by(txn) { - entry.mode = LockMode::Exclusive; - } - } - } - } - // Move out of pending (if the txn was previously blocked on this - // same key set) and into held_locks. - self.pending_keys.remove(&txn); - self.held_locks.insert(txn, keys); - return AcquireOutcome::Ready; - } - - // A conflict exists. Classify the WHOLE key set before mutating so the - // wound / block decision is atomic (never partially wound then block). - match self.wound_or_block(txn, &keys) { - ExclusiveWait::Wound => { - // Take every key exclusively. A conflicting shared entry has its - // holders revoked (the wounded readers, all younger than `txn`, - // degrade to plain OCC); `txn` becomes the sole holder while any - // existing waiters remain queued behind it. - for key in &keys { - match self.table.get_mut(key) { - Some(entry) => { - entry.mode = LockMode::Exclusive; - entry.holders.clear(); - entry.holders.push(txn); - } - None => { - self.table.insert( - key.clone(), - LockEntry { - mode: LockMode::Exclusive, - holders: smallvec![txn], - waiters: VecDeque::new(), - }, - ); - } - } - } - self.pending_keys.remove(&txn); - self.held_locks.insert(txn, keys); - AcquireOutcome::Ready - } - ExclusiveWait::Block => { - // Enqueue as an exclusive waiter on every key held by a - // different txn. - for key in &keys { - if let Some(entry) = self.table.get_mut(key) { - // No conflict on this key means it is held solely by `txn` - // (a shared reservation to be upgraded, or an exclusive - // re-acquire) — leave it untouched; `txn` keeps the key and - // upgrades it once its conflicting keys are free. Same - // predicate as the `all_available` check above. - if entry.held_exclusively_by(txn) || entry.held_shared_solely_by(txn) { - continue; - } - // Real conflict on this key. If `txn` also holds it shared - // (an upgrade that must wait behind an OLDER shared holder), - // drop its own shared hold — degrading that read to plain - // OCC, never worse than today — so the key can drain to - // empty and normal promotion can grant `txn` the exclusive - // lock later. Without this, `txn` would occupy the key - // forever and its own exclusive request could never fire. - entry.holders.retain(|h| *h != txn); - if !entry.has_waiter(txn) { - entry.waiters.push_back((txn, LockMode::Exclusive)); - } - } - // Free keys: no entry exists; the txn acquires them on the - // re-acquire path after all conflicting keys are released. - } - // Store the full key set so that release can promote this txn - // atomically once all its keys become available. - self.pending_keys.insert(txn, keys); - AcquireOutcome::Blocked - } - } - } - - /// Classify the wound-wait decision for an exclusive requester `txn` over - /// `keys`, given that at least one key already conflicts. - /// - /// Pure read over the lock table: any exclusive conflict forces - /// [`ExclusiveWait::Block`] (an exclusive holder is never wounded, so a mix - /// of exclusive and shared conflicts blocks too). Otherwise all conflicting - /// holders are shared reservations, and `txn` wounds them only when it is - /// older than every one (`txn < h` for each conflicting shared holder `h`); - /// if it is younger than any, it blocks. A key held only by `txn` itself is - /// not a conflict. - fn wound_or_block(&self, txn: TxnId, keys: &BTreeSet) -> ExclusiveWait { - let mut shared_conflicts: Vec = Vec::new(); - for key in keys { - if let Some(entry) = self.table.get(key) { - match entry.mode { - LockMode::Exclusive => { - // Exclusive entries have exactly one holder; a holder - // other than `txn` is an exclusive conflict. - if !entry.holders.contains(&txn) { - return ExclusiveWait::Block; - } - } - LockMode::Shared => { - for holder in &entry.holders { - if *holder != txn { - shared_conflicts.push(*holder); - } - } - } - } - } - } - // Wound only when there is a shared conflict AND `txn` is older than - // every conflicting shared holder; otherwise block. `shared_conflicts` - // only ever holds *other* txns' shared holders (a key held shared solely - // by `txn` never reaches here — it takes the self-upgrade path in - // `acquire`), so an empty set here means every conflict was exclusive. - if !shared_conflicts.is_empty() && shared_conflicts.iter().all(|holder| txn < *holder) { - ExclusiveWait::Wound - } else { - ExclusiveWait::Block - } - } - - /// Attempt to acquire a **shared** lock on a single `key` for `txn`. - /// - /// - Key free → create a shared entry holding `txn`, return - /// [`AcquireOutcome::Ready`]. - /// - Key held shared → add `txn` to the holders, return - /// [`AcquireOutcome::Ready`]. - /// - Key held exclusively by another txn → enqueue `txn` as a shared waiter - /// (FIFO) and return [`AcquireOutcome::Blocked`]. - /// - /// A shared request that meets an exclusive holder blocks FIFO for now; - /// wound-wait priority resolution lands in a following change. - pub fn acquire_shared(&mut self, txn: TxnId, key: LockKey) -> AcquireOutcome { - // Inspect / mutate the entry via the `Entry` API (which takes the key by - // value, sidestepping a get-then-insert borrow conflict) inside a scoped - // borrow so the map-level bookkeeping below can re-borrow `self`. - let grant = match self.table.entry(key.clone()) { - Entry::Vacant(slot) => { - slot.insert(LockEntry { - mode: LockMode::Shared, - holders: smallvec![txn], - waiters: VecDeque::new(), - }); - SharedGrant::Granted - } - Entry::Occupied(mut slot) => { - let entry = slot.get_mut(); - if entry.mode == LockMode::Shared { - if !entry.holders.contains(&txn) { - entry.holders.push(txn); - } - SharedGrant::Granted - } else { - // Held exclusively by another txn: block FIFO. - if !entry.has_waiter(txn) { - entry.waiters.push_back((txn, LockMode::Shared)); - } - SharedGrant::Blocked - } - } - }; - - match grant { - SharedGrant::Granted => { - self.pending_keys.remove(&txn); - self.held_locks.entry(txn).or_default().insert(key); - AcquireOutcome::Ready - } - SharedGrant::Blocked => { - let mut pending = BTreeSet::new(); - pending.insert(key); - self.pending_keys.insert(txn, pending); - AcquireOutcome::Blocked - } - } - } - - /// Non-blocking exclusive acquire: take all `keys` for `txn` iff every one is - /// free (or already held by `txn`), returning `true`; otherwise return - /// `false` WITHOUT enqueuing a waiter or recording any pending state. - /// - /// This is the fast path's probe. Unlike [`acquire`](Self::acquire), the - /// contended (`false`) path touches NOTHING — no holder, no `pending_keys`, - /// no waiter `VecDeque` — so a caller that does not intend to block (an - /// autocommit point write that will instead route to the scheduler) never - /// leaves an orphaned waiter that a later `release` would promote to an - /// unowned holder. It also never perturbs the FIFO ordering that Calvin - /// transactions depend on. - pub fn try_acquire(&mut self, txn: TxnId, keys: BTreeSet) -> bool { - if !self.is_ready(txn, &keys) { - // Contended: leave the table, waiter queues, and pending_keys - // completely untouched. - return false; - } - // Every key is free or already held by `txn`, so `acquire` takes its - // all-available path — it inserts the holder and never enqueues. - let outcome = self.acquire(txn, keys); - debug_assert_eq!( - outcome, - AcquireOutcome::Ready, - "try_acquire: is_ready was true but acquire returned Blocked" - ); - true - } - - /// Release all locks held by `txn`. - /// - /// `txn` is removed from every entry's holder set. When an entry's holders - /// drain to empty, its FIFO waiters are promoted mode-aware: a leading run - /// of shared waiters is promoted together, or a single leading exclusive - /// waiter is promoted alone. A waiter that becomes holder on ALL its - /// pending keys is moved from `pending_keys` to `held_locks` immediately. - /// - /// Returns the set of `TxnId`s that have been fully promoted (i.e. moved - /// into `held_locks`). The caller may use this list to dispatch those - /// transactions. - pub fn release(&mut self, txn: TxnId) -> Vec { - let held = match self.held_locks.remove(&txn) { - Some(h) => h, - None => return Vec::new(), - }; - - let mut newly_promoted: BTreeSet = BTreeSet::new(); - - for key in &held { - // Drop `txn` from this key's holders. If other (shared) holders - // remain, the key stays held and there is nothing to promote. - let now_empty = match self.table.get_mut(key) { - Some(entry) => { - entry.holders.retain(|h| *h != txn); - entry.holders.is_empty() - } - None => continue, - }; - if now_empty { - self.promote_waiters(key, &mut newly_promoted); - } - } - - newly_promoted.into_iter().collect() - } - - /// Promote the front of `key`'s waiter queue after its holders drained. - /// - /// A leading run of shared waiters is granted together; a single leading - /// exclusive waiter is granted alone; an empty queue frees the entry. Any - /// promoted txn that is now holder on all of its pending keys is moved into - /// `held_locks` and recorded in `newly_promoted`. - fn promote_waiters(&mut self, key: &LockKey, newly_promoted: &mut BTreeSet) { - // Decide the promotion inside a scoped borrow so the readiness sweep - // below can re-borrow the table. - let decision = match self.table.get_mut(key) { - Some(entry) => match entry.waiters.front().map(|(_, mode)| *mode) { - None => Promotion::Freed, - Some(LockMode::Exclusive) => match entry.waiters.pop_front() { - Some((next, _)) => { - entry.mode = LockMode::Exclusive; - entry.holders.clear(); - entry.holders.push(next); - Promotion::Promoted(vec![next]) - } - None => Promotion::Freed, - }, - Some(LockMode::Shared) => { - entry.mode = LockMode::Shared; - entry.holders.clear(); - let mut promoted = Vec::new(); - while matches!(entry.waiters.front(), Some((_, LockMode::Shared))) { - if let Some((next, _)) = entry.waiters.pop_front() { - entry.holders.push(next); - promoted.push(next); - } - } - Promotion::Promoted(promoted) - } - }, - None => return, - }; - - let promoted = match decision { - Promotion::Freed => { - self.table.remove(key); - return; - } - Promotion::Promoted(promoted) => promoted, - }; - - // For each promoted txn, check whether it is now holder on ALL of its - // pending keys. If so, it is fully ready — move to held_locks. Remove - // first (rather than `get` + a follow-up `remove`) so there is no - // unwrap/expect on a "just confirmed Some" invariant: the owned - // `pending` set is reinserted on the not-yet-ready path. - for next in promoted { - if let Some(pending) = self.pending_keys.remove(&next) { - let all_held = pending - .iter() - .all(|k| self.table.get(k).is_none_or(|e| e.holders.contains(&next))); - if all_held { - self.held_locks.insert(next, pending); - newly_promoted.insert(next); - } else { - self.pending_keys.insert(next, pending); - } - } - } - } - - /// Check whether a previously-blocked transaction is now ready. - /// - /// A transaction is ready when for every key in its key set, the key is - /// either: - /// - Not present in the lock table (free), or - /// - Present in the lock table with `txn` among the current holders - /// (shared or exclusive). - /// - /// This is called after `release` returns `txn_id` in the unblocked set. - /// If `is_ready` returns `true`, the caller calls `acquire` again which - /// will succeed on the all-available path (because the waiter was promoted). - pub fn is_ready(&self, txn: TxnId, keys: &BTreeSet) -> bool { - keys.iter().all(|key| { - match self.table.get(key) { - None => true, // key is free - Some(entry) => entry.holders.contains(&txn), // txn is a current holder - } - }) - } - - /// Number of holders of `key` that hold it as a Calvin read reservation - /// (a `TxnId` in the reservation position band), or 0 when the key is - /// unlocked or held only by non-reservation transactions. Used to observe - /// reservation install/release from outside the scheduler. - pub fn reservation_holder_count(&self, key: &LockKey) -> usize { - self.table - .get(key) - .map(|e| e.holders.iter().filter(|h| h.is_reservation()).count()) - .unwrap_or(0) - } - - /// Number of currently-held locks (entries in the lock table). - #[cfg(test)] - pub fn lock_count(&self) -> usize { - self.table.len() - } - - /// Number of transactions currently holding at least one lock. - #[cfg(test)] - pub fn holder_count(&self) -> usize { - self.held_locks.len() - } -} - -impl LockEntry { - /// Whether this entry is held exclusively by exactly `txn` (the self - /// re-acquire case on the exclusive path). - fn held_exclusively_by(&self, txn: TxnId) -> bool { - self.mode == LockMode::Exclusive && self.holders.len() == 1 && self.holders[0] == txn - } - - /// Whether this entry is held **shared** by exactly `txn` and no one else — - /// the self-upgrade case: `txn` may take the key exclusively because it is - /// the sole current holder. - fn held_shared_solely_by(&self, txn: TxnId) -> bool { - self.mode == LockMode::Shared && self.holders.len() == 1 && self.holders[0] == txn - } - - /// Whether `txn` is already enqueued as a waiter on this entry. - fn has_waiter(&self, txn: TxnId) -> bool { - self.waiters.iter().any(|(w, _)| *w == txn) - } -} - -impl Default for LockManager { - fn default() -> Self { - Self::new() - } -} - -// ── Tests ───────────────────────────────────────────────────────────────────── - -#[cfg(test)] -mod tests { - use super::*; - use std::sync::Arc; - - fn key(name: &str) -> LockKey { - LockKey::Surrogate { - collection: Arc::from(name), - surrogate: 1, - } - } - - fn keyset(names: &[&str]) -> BTreeSet { - names.iter().map(|n| key(n)).collect() - } - - fn txn(epoch: u64, pos: u32) -> TxnId { - TxnId::new(epoch, pos) - } - - #[test] - fn acquire_free_keys_returns_ready() { - let mut lm = LockManager::new(); - let t = txn(1, 0); - let outcome = lm.acquire(t, keyset(&["a", "b"])); - assert_eq!(outcome, AcquireOutcome::Ready); - assert_eq!(lm.lock_count(), 2); - } - - #[test] - fn acquire_held_key_returns_blocked_and_enqueues_waiter() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - lm.acquire(t1, keyset(&["x"])); - - let outcome = lm.acquire(t2, keyset(&["x"])); - assert_eq!(outcome, AcquireOutcome::Blocked); - - // t2 should be in the waiter queue for "x". - assert!(lm.table.get(&key("x")).unwrap().has_waiter(t2)); - } - - #[test] - fn release_returns_unblocked_waiter_ids() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - lm.acquire(t1, keyset(&["x"])); - lm.acquire(t2, keyset(&["x"])); - - let unblocked = lm.release(t1); - assert!(unblocked.contains(&t2)); - } - - #[test] - fn autocommit_holder_release_promotes_and_returns_scheduler_waiter() { - // Mirrors the write-admission fast path: an autocommit-band holder takes - // an uncontended key, a normal-band scheduler txn then blocks behind it, - // and the holder's release promotes that scheduler txn AND returns its id - // — the value the fast-path guard forwards to the scheduler on drop - // (previously discarded, stranding the promoted txn as a zombie holder). - let mut lm = LockManager::new(); - let autocommit = txn(TxnId::AUTOCOMMIT_EPOCH, 0); - let scheduler_txn = txn(9, 0); - - assert!( - lm.try_acquire(autocommit, keyset(&["k"])), - "the fast-path holder takes the uncontended key" - ); - assert_eq!( - lm.acquire(scheduler_txn, keyset(&["k"])), - AcquireOutcome::Blocked, - "the scheduler txn queues behind the fast-path holder" - ); - - let promoted = lm.release(autocommit); - assert_eq!( - promoted, - vec![scheduler_txn], - "release must return the promoted scheduler waiter" - ); - assert!( - lm.is_ready(scheduler_txn, &keyset(&["k"])), - "the promoted scheduler txn is now holder of the freed key" - ); - } - - #[test] - fn release_preserves_fifo_waiter_order() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - let t3 = txn(1, 2); - lm.acquire(t1, keyset(&["x"])); - lm.acquire(t2, keyset(&["x"])); - lm.acquire(t3, keyset(&["x"])); - - // Release t1 — t2 should become holder (FIFO). - lm.release(t1); - let holder = lm.table.get(&key("x")).unwrap().holders[0]; - assert_eq!(holder, t2); - - // Release t2 — t3 should become holder. - lm.release(t2); - let holder = lm.table.get(&key("x")).unwrap().holders[0]; - assert_eq!(holder, t3); - } - - #[test] - fn multi_key_txn_releases_all_atomically() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - lm.acquire(t1, keyset(&["a", "b", "c"])); - assert_eq!(lm.lock_count(), 3); - - lm.release(t1); - assert_eq!(lm.lock_count(), 0); - assert_eq!(lm.holder_count(), 0); - } - - #[test] - fn is_ready_returns_true_when_all_keys_free_or_self_at_front() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - lm.acquire(t1, keyset(&["x", "y"])); - lm.acquire(t2, keyset(&["x", "y"])); - - // t2 is not ready while t1 holds. - assert!(!lm.is_ready(t2, &keyset(&["x", "y"]))); - - // Release t1 — t2 becomes holder on both keys. - lm.release(t1); - // After release, t2 is promoted to holder on both keys. - assert!(lm.is_ready(t2, &keyset(&["x", "y"]))); - } - - #[test] - fn shared_shared_compatible() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire_shared(t1, key("s")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t2, key("s")), AcquireOutcome::Ready); - - let entry = lm.table.get(&key("s")).unwrap(); - assert_eq!(entry.mode, LockMode::Shared); - assert!(entry.holders.contains(&t1)); - assert!(entry.holders.contains(&t2)); - } - - #[test] - fn shared_blocks_exclusive() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); - assert_eq!( - lm.acquire(t2, keyset(&["k"])), - AcquireOutcome::Blocked, - "an exclusive request must block behind a shared holder" - ); - assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); - } - - #[test] - fn exclusive_blocks_shared() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); - assert_eq!( - lm.acquire_shared(t2, key("k")), - AcquireOutcome::Blocked, - "a shared request must block behind an exclusive holder" - ); - assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); - } - - #[test] - fn release_promotes_shared_run_together() { - let mut lm = LockManager::new(); - let holder = txn(1, 0); - let s1 = txn(2, 0); - let s2 = txn(2, 1); - - // Exclusive holder, two shared waiters queued behind it. - assert_eq!(lm.acquire(holder, keyset(&["k"])), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(s1, key("k")), AcquireOutcome::Blocked); - assert_eq!(lm.acquire_shared(s2, key("k")), AcquireOutcome::Blocked); - - // Releasing the exclusive holder promotes the whole run of shared - // waiters together. - let promoted = lm.release(holder); - assert!(promoted.contains(&s1)); - assert!(promoted.contains(&s2)); - - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Shared); - assert!(entry.holders.contains(&s1)); - assert!(entry.holders.contains(&s2)); - } - - #[test] - fn release_promotes_single_exclusive_waiter() { - let mut lm = LockManager::new(); - let holder = txn(1, 0); - let x1 = txn(2, 0); - let x2 = txn(2, 1); - - assert_eq!(lm.acquire(holder, keyset(&["k"])), AcquireOutcome::Ready); - assert_eq!(lm.acquire(x1, keyset(&["k"])), AcquireOutcome::Blocked); - assert_eq!(lm.acquire(x2, keyset(&["k"])), AcquireOutcome::Blocked); - - // Only the single leading exclusive waiter is promoted. - let promoted = lm.release(holder); - assert_eq!(promoted, vec![x1]); - - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert_eq!(entry.holders.len(), 1); - assert_eq!(entry.holders[0], x1); - // x2 is still waiting behind x1. - assert!(entry.has_waiter(x2)); - } - - #[test] - fn multi_holder_release() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t2, key("k")), AcquireOutcome::Ready); - assert_eq!(lm.lock_count(), 1); - - // Releasing one shared holder leaves the other holding the key. - lm.release(t1); - let entry = lm.table.get(&key("k")).unwrap(); - assert!(!entry.holders.contains(&t1)); - assert!(entry.holders.contains(&t2)); - assert_eq!(lm.lock_count(), 1); - - // Releasing the last shared holder frees the key. - lm.release(t2); - assert_eq!(lm.lock_count(), 0); - } - - #[test] - fn older_writer_wounds_shared() { - let mut lm = LockManager::new(); - let t2 = txn(1, 2); // shared holder - let t1 = txn(1, 1); // exclusive requester, older than t2 - - assert_eq!(lm.acquire_shared(t2, key("k")), AcquireOutcome::Ready); - // The older writer wounds the younger shared holder and proceeds. - assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); - - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert!(entry.holders.contains(&t1), "R is now the exclusive holder"); - assert!( - !entry.holders.contains(&t2), - "the wounded shared holder is gone" - ); - } - - #[test] - fn younger_writer_waits() { - let mut lm = LockManager::new(); - let t1 = txn(1, 1); // shared holder - let t2 = txn(1, 2); // exclusive requester, younger than t1 - - assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); - // The younger writer must not wound; it waits behind the shared holder. - assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Blocked); - - let entry = lm.table.get(&key("k")).unwrap(); - assert!(entry.holders.contains(&t1), "the shared holder still holds"); - assert!(!entry.holders.contains(&t2), "R holds nothing"); - assert!(entry.has_waiter(t2), "R is enqueued as an exclusive waiter"); - } - - #[test] - fn exclusive_waits_on_exclusive() { - let mut lm = LockManager::new(); - let t1 = txn(1, 0); - let t2 = txn(1, 1); - - assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); - assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Blocked); - assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); - } - - #[test] - fn exclusive_waits_on_exclusive_regardless_of_age() { - let mut lm = LockManager::new(); - let t2 = txn(1, 2); // exclusive holder (younger) - let t1 = txn(1, 1); // exclusive requester (older) - - assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Ready); - // An exclusive holder is NEVER wounded, even by an older writer. - assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Blocked); - - let entry = lm.table.get(&key("k")).unwrap(); - assert!( - entry.holders.contains(&t2), - "the exclusive holder is intact" - ); - assert!(!entry.holders.contains(&t1)); - assert!(entry.has_waiter(t1)); - } - - #[test] - fn multi_key_atomic_wound_takes_both() { - let mut lm = LockManager::new(); - let s1 = txn(1, 5); // shared holder on k1, younger than R - let s2 = txn(1, 6); // shared holder on k2, younger than R - let r = txn(1, 1); // exclusive requester, older than both - - assert_eq!(lm.acquire_shared(s1, key("k1")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(s2, key("k2")), AcquireOutcome::Ready); - - assert_eq!(lm.acquire(r, keyset(&["k1", "k2"])), AcquireOutcome::Ready); - - for k in ["k1", "k2"] { - let entry = lm.table.get(&key(k)).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert!(entry.holders.contains(&r), "R holds {k}"); - } - assert!(!lm.table.get(&key("k1")).unwrap().holders.contains(&s1)); - assert!(!lm.table.get(&key("k2")).unwrap().holders.contains(&s2)); - } - - #[test] - fn multi_key_atomic_wait_holds_none() { - let mut lm = LockManager::new(); - let s1 = txn(1, 5); // shared holder on k1, younger than R - let s2 = txn(1, 0); // shared holder on k2, OLDER than R - let r = txn(1, 1); // exclusive requester - - assert_eq!(lm.acquire_shared(s1, key("k1")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(s2, key("k2")), AcquireOutcome::Ready); - - // R is younger than the holder on k2, so it must wait on BOTH keys and - // hold neither (all-or-nothing). - assert_eq!( - lm.acquire(r, keyset(&["k1", "k2"])), - AcquireOutcome::Blocked - ); - - assert!( - !lm.table.get(&key("k1")).unwrap().holders.contains(&r), - "R holds no key" - ); - assert!(!lm.table.get(&key("k2")).unwrap().holders.contains(&r)); - // The older shared holder on k2 is untouched. - assert!(lm.table.get(&key("k2")).unwrap().holders.contains(&s2)); - } - - #[test] - fn crossed_reservations_are_acyclic() { - // T1 holds shared K1 and wants exclusive K2; T2 holds shared K2 and - // wants exclusive K1. The older writer's exclusive acquire wounds the - // younger's shared holding, breaking the cycle — no deadlock. - let mut lm = LockManager::new(); - let t1 = txn(1, 1); // older - let t2 = txn(1, 2); // younger - - assert_eq!(lm.acquire_shared(t1, key("k1")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t2, key("k2")), AcquireOutcome::Ready); - - // T1 (older) acquires exclusive K2: wounds T2's shared holding and - // proceeds. - assert_eq!(lm.acquire(t1, keyset(&["k2"])), AcquireOutcome::Ready); - - let k2 = lm.table.get(&key("k2")).unwrap(); - assert_eq!(k2.mode, LockMode::Exclusive); - assert!(k2.holders.contains(&t1), "the older writer proceeds"); - assert!( - !k2.holders.contains(&t2), - "the younger's reservation is wounded away" - ); - } - - #[test] - fn shared_reservation_self_upgrades_to_exclusive() { - let mut lm = LockManager::new(); - let t = txn(1, 0); - - assert_eq!(lm.acquire_shared(t, key("k")), AcquireOutcome::Ready); - // The txn re-acquires its own shared reservation exclusively — this must - // NOT self-deadlock by blocking on its own held key. - assert_eq!( - lm.acquire(t, keyset(&["k"])), - AcquireOutcome::Ready, - "self-upgrade from shared to exclusive must not block" - ); - - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert_eq!(entry.holders.len(), 1); - assert_eq!(entry.holders[0], t); - } - - #[test] - fn self_upgrade_with_other_shared_holder_blocks_or_wounds() { - // T_old is older than T_young: T_old's self-upgrade must wound T_young. - let mut lm = LockManager::new(); - let t_old = txn(1, 0); - let t_young = txn(1, 1); - - assert_eq!(lm.acquire_shared(t_old, key("k")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t_young, key("k")), AcquireOutcome::Ready); - - assert_eq!( - lm.acquire(t_old, keyset(&["k"])), - AcquireOutcome::Ready, - "the older self-upgrader wounds the younger shared holder" - ); - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert_eq!(entry.holders.len(), 1); - assert_eq!(entry.holders[0], t_old); - - // Symmetric case: the YOUNGER of the two self-upgrades and must block. - let mut lm = LockManager::new(); - let t_old = txn(1, 0); - let t_young = txn(1, 1); - - assert_eq!(lm.acquire_shared(t_old, key("k")), AcquireOutcome::Ready); - assert_eq!(lm.acquire_shared(t_young, key("k")), AcquireOutcome::Ready); - - assert_eq!( - lm.acquire(t_young, keyset(&["k"])), - AcquireOutcome::Blocked, - "the younger self-upgrader must wait behind the older shared holder" - ); - // t_young drops its own shared hold (degrading to plain OCC) so the key - // can drain to empty and its exclusive request can later be promoted; - // t_old remains the sole shared holder, and t_young is enqueued as an - // exclusive waiter rather than left stuck as a non-waiting holder. - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Shared); - assert!(entry.holders.contains(&t_old)); - assert!(!entry.holders.contains(&t_young)); - assert!(entry.has_waiter(t_young)); - - // Once t_old releases, t_young is promoted to sole exclusive holder. - let unblocked = lm.release(t_old); - assert!(unblocked.contains(&t_young)); - let entry = lm.table.get(&key("k")).unwrap(); - assert_eq!(entry.mode, LockMode::Exclusive); - assert_eq!(entry.holders.len(), 1); - assert_eq!(entry.holders[0], t_young); - } - - #[test] - fn self_upgrade_mixed_with_conflict_on_other_key() { - let mut lm = LockManager::new(); - let t = txn(1, 0); - let u = txn(1, 1); - - // T reserves K1 shared; U holds K2 exclusively. - assert_eq!(lm.acquire_shared(t, key("k1")), AcquireOutcome::Ready); - assert_eq!(lm.acquire(u, keyset(&["k2"])), AcquireOutcome::Ready); - - // T tries to take both keys exclusively: K2 conflicts with U, so T must - // block on the whole set — and critically must NOT self-deadlock on K1. - assert_eq!( - lm.acquire(t, keyset(&["k1", "k2"])), - AcquireOutcome::Blocked, - "conflict on k2 blocks the whole set" - ); - - // After U releases K2, T's re-acquire succeeds and upgrades K1 in place. - lm.release(u); - assert_eq!( - lm.acquire(t, keyset(&["k1", "k2"])), - AcquireOutcome::Ready, - "once k2 frees up, t acquires both keys exclusively" - ); - let k1 = lm.table.get(&key("k1")).unwrap(); - assert_eq!(k1.mode, LockMode::Exclusive); - assert_eq!(k1.holders.len(), 1); - assert_eq!(k1.holders[0], t); - let k2 = lm.table.get(&key("k2")).unwrap(); - assert_eq!(k2.mode, LockMode::Exclusive); - assert_eq!(k2.holders.len(), 1); - assert_eq!(k2.holders[0], t); - } - - #[test] - fn two_non_conflicting_both_dispatch_immediately() { - let mut lm = LockManager::new(); - - let txn1 = TxnId::new(1, 0); - let txn2 = TxnId::new(1, 1); - - let keys1: BTreeSet = [LockKey::Surrogate { - collection: Arc::from("coll"), - surrogate: 1, - }] - .into(); - let keys2: BTreeSet = [LockKey::Surrogate { - collection: Arc::from("coll"), - surrogate: 2, - }] - .into(); - - let o1 = lm.acquire(txn1, keys1); - let o2 = lm.acquire(txn2, keys2); - - assert_eq!(o1, AcquireOutcome::Ready, "txn1 should be ready"); - assert_eq!( - o2, - AcquireOutcome::Ready, - "txn2 should be ready (disjoint keys)" - ); - } - - #[test] - fn two_conflicting_second_dispatches_after_first_completes() { - let mut lm = LockManager::new(); - - let txn1 = TxnId::new(1, 0); - let txn2 = TxnId::new(1, 1); - let shared_key: BTreeSet = [LockKey::Surrogate { - collection: Arc::from("coll"), - surrogate: 42, - }] - .into(); - - let o1 = lm.acquire(txn1, shared_key.clone()); - assert_eq!(o1, AcquireOutcome::Ready); - - let o2 = lm.acquire(txn2, shared_key.clone()); - assert_eq!(o2, AcquireOutcome::Blocked); - - let unblocked = lm.release(txn1); - assert!(unblocked.contains(&txn2)); - - assert!(lm.is_ready(txn2, &shared_key)); - } - - #[test] - fn many_mixed_deterministic_dispatch_order() { - let mut lm = LockManager::new(); - let mut dispatched: Vec = Vec::new(); - - let pairs = [(2, 0), (1, 1), (3, 0), (1, 0), (2, 1)]; - for (epoch, pos) in pairs { - let tid = TxnId::new(epoch, pos); - let keys: BTreeSet = [LockKey::Surrogate { - collection: Arc::from(format!("c_{epoch}_{pos}")), - surrogate: epoch as u32 * 10 + pos, - }] - .into(); - let outcome = lm.acquire(tid, keys); - if outcome == AcquireOutcome::Ready { - dispatched.push(tid); - } - } - - assert_eq!( - dispatched.len(), - 5, - "all non-conflicting txns should be ready" - ); - - let mut expected = pairs.map(|(e, p)| TxnId::new(e, p)).to_vec(); - expected.sort(); - let mut sorted_dispatched = dispatched.clone(); - sorted_dispatched.sort(); - assert_eq!(sorted_dispatched, expected); - } - - #[test] - fn cross_epoch_raw_blocks_correctly() { - let mut lm = LockManager::new(); - - let txn_n = TxnId::new(1, 0); - let txn_n1 = TxnId::new(2, 0); - - let key_k: BTreeSet = [LockKey::Surrogate { - collection: Arc::from("orders"), - surrogate: 100, - }] - .into(); - - let o1 = lm.acquire(txn_n, key_k.clone()); - assert_eq!(o1, AcquireOutcome::Ready); - - let o2 = lm.acquire(txn_n1, key_k.clone()); - assert_eq!(o2, AcquireOutcome::Blocked); - - let unblocked = lm.release(txn_n); - assert!(unblocked.contains(&txn_n1)); - assert!(lm.is_ready(txn_n1, &key_k)); - } -} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/acquire.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/acquire.rs new file mode 100644 index 000000000..f896c15c9 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/acquire.rs @@ -0,0 +1,378 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Exclusive lock acquisition and waiter queueing. + +use std::collections::{BTreeSet, VecDeque}; + +use smallvec::smallvec; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::{ + AcquireOutcome, LockEntry, LockMode, +}; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::{ExclusiveWait, LockManager}; + +impl LockManager { + /// Attempt to acquire **exclusive** locks on all keys for `txn`. + /// + /// If every key is free, already held exclusively by `txn` (promoted from + /// waiter), or held **shared solely by `txn`** (a reservation this txn placed + /// earlier, now upgraded in place to exclusive), records `txn` as sole holder + /// of each key and returns [`AcquireOutcome::Ready`]. + /// + /// Otherwise the whole key set is classified under the WOUND-WAIT discipline + /// (see [`Self::wound_or_block`]). `TxnId` order (`(epoch, position)`) is + /// the replicated total order, so "older" means a smaller id and the + /// decision is a pure function of the lock table plus the replicated ids — + /// every replica computes it identically: + /// - Any conflicting holder is **exclusive** → `txn` waits (an exclusive + /// holder is executing/applied work and is never wounded). + /// - All conflicting holders are **shared** reservations and `txn` is older + /// than every one of them → **wound** them all (revoke to plain OCC, no + /// notification) and take every key. Wounding is silent. + /// - Otherwise (`txn` younger than some conflicting shared holder) → wait. + /// + /// On the wait path `txn` is enqueued as an exclusive waiter on every + /// conflicting key and holds none; its key set is stored in `pending_keys` + /// so `release` can promote it atomically when all keys become available. + /// The whole key set is evaluated before any mutation, so acquisition stays + /// all-keys-or-none — the manager never partially wounds and then blocks. + pub fn acquire(&mut self, txn: TxnId, keys: BTreeSet) -> AcquireOutcome { + // First pass: determine whether any key is held by a *different* txn. + // A key already held exclusively by `txn` (promoted via release) or held + // shared solely by `txn` (an earlier reservation) counts as available — + // the former is a no-op re-acquire, the latter a self-upgrade to exclusive. + let all_available = keys.iter().all(|k| { + self.table.get(k).is_none_or(|entry| { + entry.held_exclusively_by(txn) || entry.held_shared_solely_by(txn) + }) + }); + + if all_available { + // Acquire all keys. For keys not yet in the table (free), insert a + // new exclusive entry. For keys already held by this txn (promoted + // waiter), leave the entry unchanged — the waiter queue is intact. + for key in &keys { + match self.table.get_mut(key) { + None => { + self.table.insert( + key.clone(), + LockEntry { + mode: LockMode::Exclusive, + holders: smallvec![txn], + waiters: VecDeque::new(), + }, + ); + } + Some(entry) => { + // A key this txn already holds shared-solely is upgraded + // to exclusive in place (holders is exactly `[txn]`, so no + // holder change and any waiter queue stays intact). A key + // already held exclusively by `txn` is left unchanged. + if entry.held_shared_solely_by(txn) { + entry.mode = LockMode::Exclusive; + } + } + } + } + // Move out of pending (if the txn was previously blocked on this + // same key set) and into held_locks. + self.pending_keys.remove(&txn); + self.held_locks.insert(txn, keys); + return AcquireOutcome::Ready; + } + + // A conflict exists. Classify the WHOLE key set before mutating so the + // wound / block decision is atomic (never partially wound then block). + match self.wound_or_block(txn, &keys) { + ExclusiveWait::Wound => { + // Take every key exclusively. A conflicting shared entry has its + // holders revoked (the wounded readers, all younger than `txn`, + // degrade to plain OCC); `txn` becomes the sole holder while any + // existing waiters remain queued behind it. + for key in &keys { + match self.table.get_mut(key) { + Some(entry) => { + entry.mode = LockMode::Exclusive; + entry.holders.clear(); + entry.holders.push(txn); + } + None => { + self.table.insert( + key.clone(), + LockEntry { + mode: LockMode::Exclusive, + holders: smallvec![txn], + waiters: VecDeque::new(), + }, + ); + } + } + } + self.pending_keys.remove(&txn); + self.held_locks.insert(txn, keys); + AcquireOutcome::Ready + } + ExclusiveWait::Block => { + // Enqueue as an exclusive waiter on every key held by a + // different txn. + for key in &keys { + if let Some(entry) = self.table.get_mut(key) { + // No conflict on this key means it is held solely by `txn` + // (a shared reservation to be upgraded, or an exclusive + // re-acquire) — leave it untouched; `txn` keeps the key and + // upgrades it once its conflicting keys are free. Same + // predicate as the `all_available` check above. + if entry.held_exclusively_by(txn) || entry.held_shared_solely_by(txn) { + continue; + } + // Real conflict on this key. If `txn` also holds it shared + // (an upgrade that must wait behind an OLDER shared holder), + // drop its own shared hold — degrading that read to plain + // OCC, never worse than today — so the key can drain to + // empty and normal promotion can grant `txn` the exclusive + // lock later. Without this, `txn` would occupy the key + // forever and its own exclusive request could never fire. + entry.holders.retain(|h| *h != txn); + if !entry.has_waiter(txn) { + entry.waiters.push_back((txn, LockMode::Exclusive)); + } + } + // Free keys: no entry exists; the txn acquires them on the + // re-acquire path after all conflicting keys are released. + } + // Store the full key set so that release can promote this txn + // atomically once all its keys become available. + self.pending_keys.insert(txn, keys); + AcquireOutcome::Blocked + } + } + } +} + +// ── Tests ───────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + fn key(name: &str) -> LockKey { + LockKey::Surrogate { + collection: Arc::from(name), + surrogate: 1, + } + } + + fn keyset(names: &[&str]) -> BTreeSet { + names.iter().map(|n| key(n)).collect() + } + + fn txn(epoch: u64, pos: u32) -> TxnId { + TxnId::new(epoch, pos) + } + + #[test] + fn acquire_free_keys_returns_ready() { + let mut lm = LockManager::new(); + let t = txn(1, 0); + let outcome = lm.acquire(t, keyset(&["a", "b"])); + assert_eq!(outcome, AcquireOutcome::Ready); + assert_eq!(lm.lock_count(), 2); + } + + #[test] + fn acquire_held_key_returns_blocked_and_enqueues_waiter() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + lm.acquire(t1, keyset(&["x"])); + + let outcome = lm.acquire(t2, keyset(&["x"])); + assert_eq!(outcome, AcquireOutcome::Blocked); + + // t2 should be in the waiter queue for "x". + assert!(lm.table.get(&key("x")).unwrap().has_waiter(t2)); + } + + #[test] + fn exclusive_waits_on_exclusive() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); + assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Blocked); + assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); + } + + #[test] + fn shared_reservation_self_upgrades_to_exclusive() { + let mut lm = LockManager::new(); + let t = txn(1, 0); + + assert_eq!(lm.acquire_shared(t, key("k")), AcquireOutcome::Ready); + // The txn re-acquires its own shared reservation exclusively — this must + // NOT self-deadlock by blocking on its own held key. + assert_eq!( + lm.acquire(t, keyset(&["k"])), + AcquireOutcome::Ready, + "self-upgrade from shared to exclusive must not block" + ); + + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert_eq!(entry.holders.len(), 1); + assert_eq!(entry.holders[0], t); + } + + #[test] + fn self_upgrade_with_other_shared_holder_blocks_or_wounds() { + // T_old is older than T_young: T_old's self-upgrade must wound T_young. + let mut lm = LockManager::new(); + let t_old = txn(1, 0); + let t_young = txn(1, 1); + + assert_eq!(lm.acquire_shared(t_old, key("k")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t_young, key("k")), AcquireOutcome::Ready); + + assert_eq!( + lm.acquire(t_old, keyset(&["k"])), + AcquireOutcome::Ready, + "the older self-upgrader wounds the younger shared holder" + ); + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert_eq!(entry.holders.len(), 1); + assert_eq!(entry.holders[0], t_old); + + // Symmetric case: the YOUNGER of the two self-upgrades and must block. + let mut lm = LockManager::new(); + let t_old = txn(1, 0); + let t_young = txn(1, 1); + + assert_eq!(lm.acquire_shared(t_old, key("k")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t_young, key("k")), AcquireOutcome::Ready); + + assert_eq!( + lm.acquire(t_young, keyset(&["k"])), + AcquireOutcome::Blocked, + "the younger self-upgrader must wait behind the older shared holder" + ); + // t_young drops its own shared hold (degrading to plain OCC) so the key + // can drain to empty and its exclusive request can later be promoted; + // t_old remains the sole shared holder, and t_young is enqueued as an + // exclusive waiter rather than left stuck as a non-waiting holder. + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Shared); + assert!(entry.holders.contains(&t_old)); + assert!(!entry.holders.contains(&t_young)); + assert!(entry.has_waiter(t_young)); + + // Once t_old releases, t_young is promoted to sole exclusive holder. + let unblocked = lm.release(t_old); + assert!(unblocked.contains(&t_young)); + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert_eq!(entry.holders.len(), 1); + assert_eq!(entry.holders[0], t_young); + } + + #[test] + fn self_upgrade_mixed_with_conflict_on_other_key() { + let mut lm = LockManager::new(); + let t = txn(1, 0); + let u = txn(1, 1); + + // T reserves K1 shared; U holds K2 exclusively. + assert_eq!(lm.acquire_shared(t, key("k1")), AcquireOutcome::Ready); + assert_eq!(lm.acquire(u, keyset(&["k2"])), AcquireOutcome::Ready); + + // T tries to take both keys exclusively: K2 conflicts with U, so T must + // block on the whole set — and critically must NOT self-deadlock on K1. + assert_eq!( + lm.acquire(t, keyset(&["k1", "k2"])), + AcquireOutcome::Blocked, + "conflict on k2 blocks the whole set" + ); + + // After U releases K2, T's re-acquire succeeds and upgrades K1 in place. + lm.release(u); + assert_eq!( + lm.acquire(t, keyset(&["k1", "k2"])), + AcquireOutcome::Ready, + "once k2 frees up, t acquires both keys exclusively" + ); + let k1 = lm.table.get(&key("k1")).unwrap(); + assert_eq!(k1.mode, LockMode::Exclusive); + assert_eq!(k1.holders.len(), 1); + assert_eq!(k1.holders[0], t); + let k2 = lm.table.get(&key("k2")).unwrap(); + assert_eq!(k2.mode, LockMode::Exclusive); + assert_eq!(k2.holders.len(), 1); + assert_eq!(k2.holders[0], t); + } + + #[test] + fn two_non_conflicting_both_dispatch_immediately() { + let mut lm = LockManager::new(); + + let txn1 = TxnId::new(1, 0); + let txn2 = TxnId::new(1, 1); + + let keys1: BTreeSet = [LockKey::Surrogate { + collection: Arc::from("coll"), + surrogate: 1, + }] + .into(); + let keys2: BTreeSet = [LockKey::Surrogate { + collection: Arc::from("coll"), + surrogate: 2, + }] + .into(); + + let o1 = lm.acquire(txn1, keys1); + let o2 = lm.acquire(txn2, keys2); + + assert_eq!(o1, AcquireOutcome::Ready, "txn1 should be ready"); + assert_eq!( + o2, + AcquireOutcome::Ready, + "txn2 should be ready (disjoint keys)" + ); + } + + #[test] + fn many_mixed_deterministic_dispatch_order() { + let mut lm = LockManager::new(); + let mut dispatched: Vec = Vec::new(); + + let pairs = [(2, 0), (1, 1), (3, 0), (1, 0), (2, 1)]; + for (epoch, pos) in pairs { + let tid = TxnId::new(epoch, pos); + let keys: BTreeSet = [LockKey::Surrogate { + collection: Arc::from(format!("c_{epoch}_{pos}")), + surrogate: epoch as u32 * 10 + pos, + }] + .into(); + let outcome = lm.acquire(tid, keys); + if outcome == AcquireOutcome::Ready { + dispatched.push(tid); + } + } + + assert_eq!( + dispatched.len(), + 5, + "all non-conflicting txns should be ready" + ); + + let mut expected = pairs.map(|(e, p)| TxnId::new(e, p)).to_vec(); + expected.sort(); + let mut sorted_dispatched = dispatched.clone(); + sorted_dispatched.sort(); + assert_eq!(sorted_dispatched, expected); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/introspection.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/introspection.rs new file mode 100644 index 000000000..ef9f4e13d --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/introspection.rs @@ -0,0 +1,96 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Read-only inspection of lock manager state — readiness checks and the +//! test-only counters used to assert on table/holder sizes. + +use std::collections::BTreeSet; + +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::LockManager; + +impl LockManager { + /// Check whether a previously-blocked transaction is now ready. + /// + /// A transaction is ready when for every key in its key set, the key is + /// either: + /// - Not present in the lock table (free), or + /// - Present in the lock table with `txn` among the current holders + /// (shared or exclusive). + /// + /// This is called after `release` returns `txn_id` in the unblocked set. + /// If `is_ready` returns `true`, the caller calls `acquire` again which + /// will succeed on the all-available path (because the waiter was promoted). + pub fn is_ready(&self, txn: TxnId, keys: &BTreeSet) -> bool { + keys.iter().all(|key| { + match self.table.get(key) { + None => true, // key is free + Some(entry) => entry.holders.contains(&txn), // txn is a current holder + } + }) + } + + /// Number of holders of `key` that hold it as a Calvin read reservation + /// (a `TxnId` in the reservation position band), or 0 when the key is + /// unlocked or held only by non-reservation transactions. Used to observe + /// reservation install/release from outside the scheduler. + pub fn reservation_holder_count(&self, key: &LockKey) -> usize { + self.table + .get(key) + .map(|e| e.holders.iter().filter(|h| h.is_reservation()).count()) + .unwrap_or(0) + } + + /// Number of currently-held locks (entries in the lock table). + #[cfg(test)] + pub fn lock_count(&self) -> usize { + self.table.len() + } + + /// Number of transactions currently holding at least one lock. + #[cfg(test)] + pub fn holder_count(&self) -> usize { + self.held_locks.len() + } +} + +// ── Tests ───────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + fn key(name: &str) -> LockKey { + LockKey::Surrogate { + collection: Arc::from(name), + surrogate: 1, + } + } + + fn keyset(names: &[&str]) -> BTreeSet { + names.iter().map(|n| key(n)).collect() + } + + fn txn(epoch: u64, pos: u32) -> TxnId { + TxnId::new(epoch, pos) + } + + #[test] + fn is_ready_returns_true_when_all_keys_free_or_self_at_front() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + lm.acquire(t1, keyset(&["x", "y"])); + lm.acquire(t2, keyset(&["x", "y"])); + + // t2 is not ready while t1 holds. + assert!(!lm.is_ready(t2, &keyset(&["x", "y"]))); + + // Release t1 — t2 becomes holder on both keys. + lm.release(t1); + // After release, t2 is promoted to holder on both keys. + assert!(lm.is_ready(t2, &keyset(&["x", "y"]))); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/mod.rs new file mode 100644 index 000000000..e59e665b9 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/mod.rs @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Deterministic lock manager for the Calvin scheduler. +//! +//! # Design +//! +//! The lock manager provides a deterministic, totally-ordered lock table over +//! per-key entries keyed by +//! [`LockKey`](crate::control::cluster::calvin::scheduler::lock::LockKey). +//! Locks come in two modes: `Exclusive` (one holder, excludes all others) and +//! `Shared` (many compatible holders). The Calvin batch acquire path takes +//! every key in a transaction's `read_set ∪ write_set` as an `Exclusive` +//! lock; single-key `Shared` locks are available via +//! [`LockManager::acquire_shared`]. +//! +//! # Determinism +//! +//! `BTreeMap` is used throughout (not `HashMap`) so that iteration order is +//! deterministic and reproducible across replicas. This is a correctness +//! requirement, not a style preference. +//! +//! Split by concern: +//! - [`types`]: the lock table struct and its internal decision enums. +//! - [`acquire`]: exclusive lock acquisition and waiter queueing. +//! - [`wound_wait`]: shared-lock reservations and wound-wait conflict +//! resolution. +//! - [`release`]: lock release and FIFO/shared waiter promotion. +//! - [`try_acquire`]: the non-blocking exclusive fast path. +//! - [`introspection`]: readiness checks and test counters. + +mod acquire; +mod introspection; +mod release; +mod try_acquire; +mod types; +mod wound_wait; + +pub use types::LockManager; diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/release.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/release.rs new file mode 100644 index 000000000..1ba0550d5 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/release.rs @@ -0,0 +1,334 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Lock release and FIFO/shared waiter promotion. + +use std::collections::BTreeSet; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::LockMode; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::{LockManager, Promotion}; + +impl LockManager { + /// Release all locks held by `txn`. + /// + /// `txn` is removed from every entry's holder set. When an entry's holders + /// drain to empty, its FIFO waiters are promoted mode-aware: a leading run + /// of shared waiters is promoted together, or a single leading exclusive + /// waiter is promoted alone. A waiter that becomes holder on ALL its + /// pending keys is moved from `pending_keys` to `held_locks` immediately. + /// + /// Returns the set of `TxnId`s that have been fully promoted (i.e. moved + /// into `held_locks`). The caller may use this list to dispatch those + /// transactions. + pub fn release(&mut self, txn: TxnId) -> Vec { + let held = match self.held_locks.remove(&txn) { + Some(h) => h, + None => return Vec::new(), + }; + + let mut newly_promoted: BTreeSet = BTreeSet::new(); + + for key in &held { + // Drop `txn` from this key's holders. If other (shared) holders + // remain, the key stays held and there is nothing to promote. + let now_empty = match self.table.get_mut(key) { + Some(entry) => { + entry.holders.retain(|h| *h != txn); + entry.holders.is_empty() + } + None => continue, + }; + if now_empty { + self.promote_waiters(key, &mut newly_promoted); + } + } + + newly_promoted.into_iter().collect() + } + + /// Promote the front of `key`'s waiter queue after its holders drained. + /// + /// A leading run of shared waiters is granted together; a single leading + /// exclusive waiter is granted alone; an empty queue frees the entry. Any + /// promoted txn that is now holder on all of its pending keys is moved into + /// `held_locks` and recorded in `newly_promoted`. + fn promote_waiters(&mut self, key: &LockKey, newly_promoted: &mut BTreeSet) { + // Decide the promotion inside a scoped borrow so the readiness sweep + // below can re-borrow the table. + let decision = match self.table.get_mut(key) { + Some(entry) => match entry.waiters.front().map(|(_, mode)| *mode) { + None => Promotion::Freed, + Some(LockMode::Exclusive) => match entry.waiters.pop_front() { + Some((next, _)) => { + entry.mode = LockMode::Exclusive; + entry.holders.clear(); + entry.holders.push(next); + Promotion::Promoted(vec![next]) + } + None => Promotion::Freed, + }, + Some(LockMode::Shared) => { + entry.mode = LockMode::Shared; + entry.holders.clear(); + let mut promoted = Vec::new(); + while matches!(entry.waiters.front(), Some((_, LockMode::Shared))) { + if let Some((next, _)) = entry.waiters.pop_front() { + entry.holders.push(next); + promoted.push(next); + } + } + Promotion::Promoted(promoted) + } + }, + None => return, + }; + + let promoted = match decision { + Promotion::Freed => { + self.table.remove(key); + return; + } + Promotion::Promoted(promoted) => promoted, + }; + + // For each promoted txn, check whether it is now holder on ALL of its + // pending keys. If so, it is fully ready — move to held_locks. Remove + // first (rather than `get` + a follow-up `remove`) so there is no + // unwrap/expect on a "just confirmed Some" invariant: the owned + // `pending` set is reinserted on the not-yet-ready path. + for next in promoted { + if let Some(pending) = self.pending_keys.remove(&next) { + let all_held = pending + .iter() + .all(|k| self.table.get(k).is_none_or(|e| e.holders.contains(&next))); + if all_held { + self.held_locks.insert(next, pending); + newly_promoted.insert(next); + } else { + self.pending_keys.insert(next, pending); + } + } + } + } +} + +// ── Tests ───────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::control::cluster::calvin::scheduler::lock::lock_entry::AcquireOutcome; + + fn key(name: &str) -> LockKey { + LockKey::Surrogate { + collection: Arc::from(name), + surrogate: 1, + } + } + + fn keyset(names: &[&str]) -> BTreeSet { + names.iter().map(|n| key(n)).collect() + } + + fn txn(epoch: u64, pos: u32) -> TxnId { + TxnId::new(epoch, pos) + } + + #[test] + fn release_returns_unblocked_waiter_ids() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + lm.acquire(t1, keyset(&["x"])); + lm.acquire(t2, keyset(&["x"])); + + let unblocked = lm.release(t1); + assert!(unblocked.contains(&t2)); + } + + #[test] + fn autocommit_holder_release_promotes_and_returns_scheduler_waiter() { + // Mirrors the write-admission fast path: an autocommit-band holder takes + // an uncontended key, a normal-band scheduler txn then blocks behind it, + // and the holder's release promotes that scheduler txn AND returns its id + // — the value the fast-path guard forwards to the scheduler on drop + // (previously discarded, stranding the promoted txn as a zombie holder). + let mut lm = LockManager::new(); + let autocommit = txn(TxnId::AUTOCOMMIT_EPOCH, 0); + let scheduler_txn = txn(9, 0); + + assert!( + lm.try_acquire(autocommit, keyset(&["k"])), + "the fast-path holder takes the uncontended key" + ); + assert_eq!( + lm.acquire(scheduler_txn, keyset(&["k"])), + AcquireOutcome::Blocked, + "the scheduler txn queues behind the fast-path holder" + ); + + let promoted = lm.release(autocommit); + assert_eq!( + promoted, + vec![scheduler_txn], + "release must return the promoted scheduler waiter" + ); + assert!( + lm.is_ready(scheduler_txn, &keyset(&["k"])), + "the promoted scheduler txn is now holder of the freed key" + ); + } + + #[test] + fn release_preserves_fifo_waiter_order() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + let t3 = txn(1, 2); + lm.acquire(t1, keyset(&["x"])); + lm.acquire(t2, keyset(&["x"])); + lm.acquire(t3, keyset(&["x"])); + + // Release t1 — t2 should become holder (FIFO). + lm.release(t1); + let holder = lm.table.get(&key("x")).unwrap().holders[0]; + assert_eq!(holder, t2); + + // Release t2 — t3 should become holder. + lm.release(t2); + let holder = lm.table.get(&key("x")).unwrap().holders[0]; + assert_eq!(holder, t3); + } + + #[test] + fn multi_key_txn_releases_all_atomically() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + lm.acquire(t1, keyset(&["a", "b", "c"])); + assert_eq!(lm.lock_count(), 3); + + lm.release(t1); + assert_eq!(lm.lock_count(), 0); + assert_eq!(lm.holder_count(), 0); + } + + #[test] + fn release_promotes_shared_run_together() { + let mut lm = LockManager::new(); + let holder = txn(1, 0); + let s1 = txn(2, 0); + let s2 = txn(2, 1); + + // Exclusive holder, two shared waiters queued behind it. + assert_eq!(lm.acquire(holder, keyset(&["k"])), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(s1, key("k")), AcquireOutcome::Blocked); + assert_eq!(lm.acquire_shared(s2, key("k")), AcquireOutcome::Blocked); + + // Releasing the exclusive holder promotes the whole run of shared + // waiters together. + let promoted = lm.release(holder); + assert!(promoted.contains(&s1)); + assert!(promoted.contains(&s2)); + + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Shared); + assert!(entry.holders.contains(&s1)); + assert!(entry.holders.contains(&s2)); + } + + #[test] + fn release_promotes_single_exclusive_waiter() { + let mut lm = LockManager::new(); + let holder = txn(1, 0); + let x1 = txn(2, 0); + let x2 = txn(2, 1); + + assert_eq!(lm.acquire(holder, keyset(&["k"])), AcquireOutcome::Ready); + assert_eq!(lm.acquire(x1, keyset(&["k"])), AcquireOutcome::Blocked); + assert_eq!(lm.acquire(x2, keyset(&["k"])), AcquireOutcome::Blocked); + + // Only the single leading exclusive waiter is promoted. + let promoted = lm.release(holder); + assert_eq!(promoted, vec![x1]); + + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert_eq!(entry.holders.len(), 1); + assert_eq!(entry.holders[0], x1); + // x2 is still waiting behind x1. + assert!(entry.has_waiter(x2)); + } + + #[test] + fn multi_holder_release() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t2, key("k")), AcquireOutcome::Ready); + assert_eq!(lm.lock_count(), 1); + + // Releasing one shared holder leaves the other holding the key. + lm.release(t1); + let entry = lm.table.get(&key("k")).unwrap(); + assert!(!entry.holders.contains(&t1)); + assert!(entry.holders.contains(&t2)); + assert_eq!(lm.lock_count(), 1); + + // Releasing the last shared holder frees the key. + lm.release(t2); + assert_eq!(lm.lock_count(), 0); + } + + #[test] + fn two_conflicting_second_dispatches_after_first_completes() { + let mut lm = LockManager::new(); + + let txn1 = TxnId::new(1, 0); + let txn2 = TxnId::new(1, 1); + let shared_key: BTreeSet = [LockKey::Surrogate { + collection: Arc::from("coll"), + surrogate: 42, + }] + .into(); + + let o1 = lm.acquire(txn1, shared_key.clone()); + assert_eq!(o1, AcquireOutcome::Ready); + + let o2 = lm.acquire(txn2, shared_key.clone()); + assert_eq!(o2, AcquireOutcome::Blocked); + + let unblocked = lm.release(txn1); + assert!(unblocked.contains(&txn2)); + + assert!(lm.is_ready(txn2, &shared_key)); + } + + #[test] + fn cross_epoch_raw_blocks_correctly() { + let mut lm = LockManager::new(); + + let txn_n = TxnId::new(1, 0); + let txn_n1 = TxnId::new(2, 0); + + let key_k: BTreeSet = [LockKey::Surrogate { + collection: Arc::from("orders"), + surrogate: 100, + }] + .into(); + + let o1 = lm.acquire(txn_n, key_k.clone()); + assert_eq!(o1, AcquireOutcome::Ready); + + let o2 = lm.acquire(txn_n1, key_k.clone()); + assert_eq!(o2, AcquireOutcome::Blocked); + + let unblocked = lm.release(txn_n); + assert!(unblocked.contains(&txn_n1)); + assert!(lm.is_ready(txn_n1, &key_k)); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/try_acquire.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/try_acquire.rs new file mode 100644 index 000000000..0dee86f99 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/try_acquire.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Non-blocking exclusive acquire fast path. + +use std::collections::BTreeSet; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::AcquireOutcome; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::LockManager; + +impl LockManager { + /// Non-blocking exclusive acquire: take all `keys` for `txn` iff every one is + /// free (or already held by `txn`), returning `true`; otherwise return + /// `false` WITHOUT enqueuing a waiter or recording any pending state. + /// + /// This is the fast path's probe. Unlike [`acquire`](Self::acquire), the + /// contended (`false`) path touches NOTHING — no holder, no `pending_keys`, + /// no waiter `VecDeque` — so a caller that does not intend to block (an + /// autocommit point write that will instead route to the scheduler) never + /// leaves an orphaned waiter that a later `release` would promote to an + /// unowned holder. It also never perturbs the FIFO ordering that Calvin + /// transactions depend on. + pub fn try_acquire(&mut self, txn: TxnId, keys: BTreeSet) -> bool { + if !self.is_ready(txn, &keys) { + // Contended: leave the table, waiter queues, and pending_keys + // completely untouched. + return false; + } + // Every key is free or already held by `txn`, so `acquire` takes its + // all-available path — it inserts the holder and never enqueues. + let outcome = self.acquire(txn, keys); + debug_assert_eq!( + outcome, + AcquireOutcome::Ready, + "try_acquire: is_ready was true but acquire returned Blocked" + ); + true + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/types.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/types.rs new file mode 100644 index 000000000..93e6862bd --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/types.rs @@ -0,0 +1,101 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The lock table struct, its internal decision enums, and the small +//! per-entry predicates the acquire/release paths share. + +use std::collections::{BTreeMap, BTreeSet}; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::{LockEntry, LockMode}; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +/// Deterministic Calvin lock manager for one vshard. +/// +/// Manages an in-memory lock table keyed by [`LockKey`]. The table is held in +/// a `BTreeMap` so iteration is always deterministic. +/// +/// # Key sets tracked per transaction +/// +/// - `held_locks`: key sets for transactions that are a current holder on ALL +/// their keys and are actively executing (i.e. dispatched to the Data Plane). +/// - `pending_keys`: key sets for transactions that are blocked waiting for at +/// least one key. When `release` promotes a blocked txn to holder on every +/// one of its keys, the entry moves from `pending_keys` to `held_locks`. +pub struct LockManager { + /// Per-key lock entries. Uses `BTreeMap` for deterministic iteration. + /// Visible to the whole `lock` module so the sibling `reap` module can + /// scan entries for lease-expired reservations without a public accessor. + pub(in crate::control::cluster::calvin::scheduler::lock) table: BTreeMap, + /// Per-transaction set of currently held keys for **dispatched** txns. + /// Used by `release` to iterate the key set without a full table scan. + /// Visible to the whole `lock` module — see `table`. + pub(in crate::control::cluster::calvin::scheduler::lock) held_locks: + BTreeMap>, + /// Key sets for **blocked** (not-yet-dispatched) txns. Populated when + /// `acquire` returns `Blocked`; cleared (moved to `held_locks`) when all + /// keys have been acquired on the promotion path inside `release`. + pub(super) pending_keys: BTreeMap>, +} + +/// Outcome of inspecting a single key during [`LockManager::acquire_shared`]. +pub(super) enum SharedGrant { + /// The shared lock was granted (key was free or already held shared). + Granted, + /// The key is held exclusively by another txn; the request was enqueued. + Blocked, +} + +/// The wound-wait decision for an exclusive requester that meets a conflict. +pub(super) enum ExclusiveWait { + /// Every conflicting holder is a shared reservation and the requester is + /// older than all of them: wound (revoke) those shared holders and proceed. + Wound, + /// The requester must block: a conflicting holder is exclusive, or the + /// requester is younger than some conflicting shared holder. + Block, +} + +/// The waiters promoted off one key when its holders drained, together with the +/// action to take on the now-empty entry. +pub(super) enum Promotion { + /// No waiters remained; the entry should be removed entirely. + Freed, + /// These waiters were installed as the new holders. + Promoted(Vec), +} + +impl LockManager { + /// Create an empty lock manager. + pub fn new() -> Self { + Self { + table: BTreeMap::new(), + held_locks: BTreeMap::new(), + pending_keys: BTreeMap::new(), + } + } +} + +impl Default for LockManager { + fn default() -> Self { + Self::new() + } +} + +impl LockEntry { + /// Whether this entry is held exclusively by exactly `txn` (the self + /// re-acquire case on the exclusive path). + pub(super) fn held_exclusively_by(&self, txn: TxnId) -> bool { + self.mode == LockMode::Exclusive && self.holders.len() == 1 && self.holders[0] == txn + } + + /// Whether this entry is held **shared** by exactly `txn` and no one else — + /// the self-upgrade case: `txn` may take the key exclusively because it is + /// the sole current holder. + pub(super) fn held_shared_solely_by(&self, txn: TxnId) -> bool { + self.mode == LockMode::Shared && self.holders.len() == 1 && self.holders[0] == txn + } + + /// Whether `txn` is already enqueued as a waiter on this entry. + pub(super) fn has_waiter(&self, txn: TxnId) -> bool { + self.waiters.iter().any(|(w, _)| *w == txn) + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/lock/manager/wound_wait.rs b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/wound_wait.rs new file mode 100644 index 000000000..2a77b1a45 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/lock/manager/wound_wait.rs @@ -0,0 +1,313 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared-lock reservations and the wound-wait conflict resolution used by +//! exclusive acquisition. + +use std::collections::btree_map::Entry; +use std::collections::{BTreeSet, VecDeque}; + +use smallvec::smallvec; + +use crate::control::cluster::calvin::scheduler::lock::lock_entry::{ + AcquireOutcome, LockEntry, LockMode, +}; +use crate::control::cluster::calvin::scheduler::lock::lock_key::{LockKey, TxnId}; + +use super::types::{ExclusiveWait, LockManager, SharedGrant}; + +impl LockManager { + /// Classify the wound-wait decision for an exclusive requester `txn` over + /// `keys`, given that at least one key already conflicts. + /// + /// Pure read over the lock table: any exclusive conflict forces + /// [`ExclusiveWait::Block`] (an exclusive holder is never wounded, so a mix + /// of exclusive and shared conflicts blocks too). Otherwise all conflicting + /// holders are shared reservations, and `txn` wounds them only when it is + /// older than every one (`txn < h` for each conflicting shared holder `h`); + /// if it is younger than any, it blocks. A key held only by `txn` itself is + /// not a conflict. + pub(super) fn wound_or_block(&self, txn: TxnId, keys: &BTreeSet) -> ExclusiveWait { + let mut shared_conflicts: Vec = Vec::new(); + for key in keys { + if let Some(entry) = self.table.get(key) { + match entry.mode { + LockMode::Exclusive => { + // Exclusive entries have exactly one holder; a holder + // other than `txn` is an exclusive conflict. + if !entry.holders.contains(&txn) { + return ExclusiveWait::Block; + } + } + LockMode::Shared => { + for holder in &entry.holders { + if *holder != txn { + shared_conflicts.push(*holder); + } + } + } + } + } + } + // Wound only when there is a shared conflict AND `txn` is older than + // every conflicting shared holder; otherwise block. `shared_conflicts` + // only ever holds *other* txns' shared holders (a key held shared solely + // by `txn` never reaches here — it takes the self-upgrade path in + // `acquire`), so an empty set here means every conflict was exclusive. + if !shared_conflicts.is_empty() && shared_conflicts.iter().all(|holder| txn < *holder) { + ExclusiveWait::Wound + } else { + ExclusiveWait::Block + } + } + + /// Attempt to acquire a **shared** lock on a single `key` for `txn`. + /// + /// - Key free → create a shared entry holding `txn`, return + /// [`AcquireOutcome::Ready`]. + /// - Key held shared → add `txn` to the holders, return + /// [`AcquireOutcome::Ready`]. + /// - Key held exclusively by another txn → enqueue `txn` as a shared waiter + /// (FIFO) and return [`AcquireOutcome::Blocked`]. + /// + /// A shared request that meets an exclusive holder blocks FIFO for now; + /// wound-wait priority resolution lands in a following change. + pub fn acquire_shared(&mut self, txn: TxnId, key: LockKey) -> AcquireOutcome { + // Inspect / mutate the entry via the `Entry` API (which takes the key by + // value, sidestepping a get-then-insert borrow conflict) inside a scoped + // borrow so the map-level bookkeeping below can re-borrow `self`. + let grant = match self.table.entry(key.clone()) { + Entry::Vacant(slot) => { + slot.insert(LockEntry { + mode: LockMode::Shared, + holders: smallvec![txn], + waiters: VecDeque::new(), + }); + SharedGrant::Granted + } + Entry::Occupied(mut slot) => { + let entry = slot.get_mut(); + if entry.mode == LockMode::Shared { + if !entry.holders.contains(&txn) { + entry.holders.push(txn); + } + SharedGrant::Granted + } else { + // Held exclusively by another txn: block FIFO. + if !entry.has_waiter(txn) { + entry.waiters.push_back((txn, LockMode::Shared)); + } + SharedGrant::Blocked + } + } + }; + + match grant { + SharedGrant::Granted => { + self.pending_keys.remove(&txn); + self.held_locks.entry(txn).or_default().insert(key); + AcquireOutcome::Ready + } + SharedGrant::Blocked => { + let mut pending = BTreeSet::new(); + pending.insert(key); + self.pending_keys.insert(txn, pending); + AcquireOutcome::Blocked + } + } + } +} + +// ── Tests ───────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + fn key(name: &str) -> LockKey { + LockKey::Surrogate { + collection: Arc::from(name), + surrogate: 1, + } + } + + fn keyset(names: &[&str]) -> BTreeSet { + names.iter().map(|n| key(n)).collect() + } + + fn txn(epoch: u64, pos: u32) -> TxnId { + TxnId::new(epoch, pos) + } + + #[test] + fn shared_shared_compatible() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire_shared(t1, key("s")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t2, key("s")), AcquireOutcome::Ready); + + let entry = lm.table.get(&key("s")).unwrap(); + assert_eq!(entry.mode, LockMode::Shared); + assert!(entry.holders.contains(&t1)); + assert!(entry.holders.contains(&t2)); + } + + #[test] + fn shared_blocks_exclusive() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); + assert_eq!( + lm.acquire(t2, keyset(&["k"])), + AcquireOutcome::Blocked, + "an exclusive request must block behind a shared holder" + ); + assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); + } + + #[test] + fn exclusive_blocks_shared() { + let mut lm = LockManager::new(); + let t1 = txn(1, 0); + let t2 = txn(1, 1); + + assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); + assert_eq!( + lm.acquire_shared(t2, key("k")), + AcquireOutcome::Blocked, + "a shared request must block behind an exclusive holder" + ); + assert!(lm.table.get(&key("k")).unwrap().has_waiter(t2)); + } + + #[test] + fn older_writer_wounds_shared() { + let mut lm = LockManager::new(); + let t2 = txn(1, 2); // shared holder + let t1 = txn(1, 1); // exclusive requester, older than t2 + + assert_eq!(lm.acquire_shared(t2, key("k")), AcquireOutcome::Ready); + // The older writer wounds the younger shared holder and proceeds. + assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Ready); + + let entry = lm.table.get(&key("k")).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert!(entry.holders.contains(&t1), "R is now the exclusive holder"); + assert!( + !entry.holders.contains(&t2), + "the wounded shared holder is gone" + ); + } + + #[test] + fn younger_writer_waits() { + let mut lm = LockManager::new(); + let t1 = txn(1, 1); // shared holder + let t2 = txn(1, 2); // exclusive requester, younger than t1 + + assert_eq!(lm.acquire_shared(t1, key("k")), AcquireOutcome::Ready); + // The younger writer must not wound; it waits behind the shared holder. + assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Blocked); + + let entry = lm.table.get(&key("k")).unwrap(); + assert!(entry.holders.contains(&t1), "the shared holder still holds"); + assert!(!entry.holders.contains(&t2), "R holds nothing"); + assert!(entry.has_waiter(t2), "R is enqueued as an exclusive waiter"); + } + + #[test] + fn exclusive_waits_on_exclusive_regardless_of_age() { + let mut lm = LockManager::new(); + let t2 = txn(1, 2); // exclusive holder (younger) + let t1 = txn(1, 1); // exclusive requester (older) + + assert_eq!(lm.acquire(t2, keyset(&["k"])), AcquireOutcome::Ready); + // An exclusive holder is NEVER wounded, even by an older writer. + assert_eq!(lm.acquire(t1, keyset(&["k"])), AcquireOutcome::Blocked); + + let entry = lm.table.get(&key("k")).unwrap(); + assert!( + entry.holders.contains(&t2), + "the exclusive holder is intact" + ); + assert!(!entry.holders.contains(&t1)); + assert!(entry.has_waiter(t1)); + } + + #[test] + fn multi_key_atomic_wound_takes_both() { + let mut lm = LockManager::new(); + let s1 = txn(1, 5); // shared holder on k1, younger than R + let s2 = txn(1, 6); // shared holder on k2, younger than R + let r = txn(1, 1); // exclusive requester, older than both + + assert_eq!(lm.acquire_shared(s1, key("k1")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(s2, key("k2")), AcquireOutcome::Ready); + + assert_eq!(lm.acquire(r, keyset(&["k1", "k2"])), AcquireOutcome::Ready); + + for k in ["k1", "k2"] { + let entry = lm.table.get(&key(k)).unwrap(); + assert_eq!(entry.mode, LockMode::Exclusive); + assert!(entry.holders.contains(&r), "R holds {k}"); + } + assert!(!lm.table.get(&key("k1")).unwrap().holders.contains(&s1)); + assert!(!lm.table.get(&key("k2")).unwrap().holders.contains(&s2)); + } + + #[test] + fn multi_key_atomic_wait_holds_none() { + let mut lm = LockManager::new(); + let s1 = txn(1, 5); // shared holder on k1, younger than R + let s2 = txn(1, 0); // shared holder on k2, OLDER than R + let r = txn(1, 1); // exclusive requester + + assert_eq!(lm.acquire_shared(s1, key("k1")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(s2, key("k2")), AcquireOutcome::Ready); + + // R is younger than the holder on k2, so it must wait on BOTH keys and + // hold neither (all-or-nothing). + assert_eq!( + lm.acquire(r, keyset(&["k1", "k2"])), + AcquireOutcome::Blocked + ); + + assert!( + !lm.table.get(&key("k1")).unwrap().holders.contains(&r), + "R holds no key" + ); + assert!(!lm.table.get(&key("k2")).unwrap().holders.contains(&r)); + // The older shared holder on k2 is untouched. + assert!(lm.table.get(&key("k2")).unwrap().holders.contains(&s2)); + } + + #[test] + fn crossed_reservations_are_acyclic() { + // T1 holds shared K1 and wants exclusive K2; T2 holds shared K2 and + // wants exclusive K1. The older writer's exclusive acquire wounds the + // younger's shared holding, breaking the cycle — no deadlock. + let mut lm = LockManager::new(); + let t1 = txn(1, 1); // older + let t2 = txn(1, 2); // younger + + assert_eq!(lm.acquire_shared(t1, key("k1")), AcquireOutcome::Ready); + assert_eq!(lm.acquire_shared(t2, key("k2")), AcquireOutcome::Ready); + + // T1 (older) acquires exclusive K2: wounds T2's shared holding and + // proceeds. + assert_eq!(lm.acquire(t1, keyset(&["k2"])), AcquireOutcome::Ready); + + let k2 = lm.table.get(&key("k2")).unwrap(); + assert_eq!(k2.mode, LockMode::Exclusive); + assert!(k2.holders.contains(&t1), "the older writer proceeds"); + assert!( + !k2.holders.contains(&t2), + "the younger's reservation is wounded away" + ); + } +} From 06429757ef99555b5d1728676a6eec9a02cd3761 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 18:52:10 +0800 Subject: [PATCH 10/64] refactor(distributed_applier): split apply_loop into a directory module apply_loop.rs held the per-batch driver, Array/Calvin/write dispatch paths, applied-floor bookkeeping, and shared helpers in one file. It becomes an apply_loop/ directory with driver.rs, array_dispatch.rs, calvin_read_result.rs, write_dispatch.rs, bookkeeping.rs, and helpers.rs, each keeping its existing logic. --- .../control/distributed_applier/apply_loop.rs | 530 ------------------ .../apply_loop/array_dispatch.rs | 53 ++ .../apply_loop/bookkeeping.rs | 62 ++ .../apply_loop/calvin_read_result.rs | 105 ++++ .../distributed_applier/apply_loop/driver.rs | 236 ++++++++ .../distributed_applier/apply_loop/helpers.rs | 39 ++ .../distributed_applier/apply_loop/mod.rs | 30 + .../apply_loop/write_dispatch.rs | 206 +++++++ 8 files changed, 731 insertions(+), 530 deletions(-) delete mode 100644 nodedb/src/control/distributed_applier/apply_loop.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/array_dispatch.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/bookkeeping.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/driver.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/helpers.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/mod.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs diff --git a/nodedb/src/control/distributed_applier/apply_loop.rs b/nodedb/src/control/distributed_applier/apply_loop.rs deleted file mode 100644 index 1ec317e77..000000000 --- a/nodedb/src/control/distributed_applier/apply_loop.rs +++ /dev/null @@ -1,530 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Background apply loop — reads committed Raft entries from the mpsc channel, -//! submits them through the shared Control-Plane write funnel (which appends -//! each entry's redo record on THIS replica before the enqueue), and resolves -//! propose waiters with the result. -//! -//! Each batch advances the group's durable applied floor (see -//! [`super::applied_index`]) to its highest contiguous successfully-applied -//! entry, so the next boot replays only above it and no entry is applied by -//! both WAL replay and Raft log replay. - -use std::sync::Arc; - -use tokio::sync::mpsc; -use tracing::debug; - -use crate::bridge::envelope::{PhysicalPlan, Status}; -use crate::control::array_sync::raft_apply::{ - AppliedPosition, ArrayCellTarget, apply_array_cell_write, apply_array_op, apply_array_schema, -}; -use crate::control::cluster::calvin::ReadResultEvent; -use crate::control::server::dispatch_utils::{ - ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, -}; -use crate::control::state::SharedState; -use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite, from_replicated_entry}; -use crate::types::{DatabaseId, TenantId, TraceId}; -use nodedb_physical::physical_plan::ArrayOp; - -use super::applied_index::{AppliedPrefix, save_applied_index}; -use super::applier::ApplyBatch; -use super::propose_tracker::{AppliedWrite, ProposeTracker}; - -fn committed_response_result( - response: &crate::bridge::envelope::Response, -) -> crate::Result { - if response.status == Status::Ok { - return Ok(AppliedWrite::from_response(response)); - } - // The typed code carries the client's classification (constraint, authz, - // conflict); stringifying it here would leave the caller only XX000. - match response.error_code.as_deref() { - Some(code) => { - tracing::warn!(reason = ?code, "applying committed write failed"); - Err(crate::Error::DataPlane(code.clone())) - } - None => { - tracing::warn!( - reason = "execution error", - "applying committed write failed" - ); - Err(crate::Error::Internal { - detail: "execution error".to_owned(), - }) - } - } -} - -fn deterministic_crdt_fence_noop(result: &crate::Result) -> bool { - matches!( - result, - Err(crate::Error::DataPlane( - crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { .. } - )) - ) -} - -/// Run the background loop that applies committed Raft entries to the local Data Plane. -/// -/// This task reads from the apply channel, deserializes each entry, dispatches -/// the write to the Data Plane via SPSC, and notifies proposers. -pub async fn run_apply_loop( - mut apply_rx: mpsc::Receiver, - state: Arc, - tracker: Arc, - calvin_read_result_senders: Arc< - std::sync::Mutex>>, - >, -) { - while let Some(batch) = apply_rx.recv().await { - // The floor is saved ONCE per batch, after the loop — never per entry. - // `save_applied_index` lands a redb transaction, and redb commits at - // `Durability::Immediate`, so a per-entry save puts one synchronous - // fsync per applied entry directly on the raft apply path. That stalls - // the raft loop hard enough to delay heartbeats and keep elections from - // stabilizing under a multi-node write load. One fsync per batch - // amortizes the cost across every entry in it and keeps the critical - // path free. - // - // `AppliedPrefix` computes WHICH index is safe to save: the highest - // contiguous successfully-applied entry, stopping at the first failure - // and never advancing past it. Every branch below must therefore report - // its outcome — `record` for the ones whose success means a durable - // redo record, `skip` for the ones that apply no durable state at all. - let mut prefix = AppliedPrefix::new(); - for entry in &batch.entries { - // Decode once; reused for both the idempotency key and the - // Array/Calvin fast-path match below. Returns 0 for - // unparseable / pre-key entries; the tracker treats 0 as - // "no key" (no mismatch detection). - let replicated_opt = ReplicatedEntry::from_bytes(&entry.data); - let applied_key = replicated_opt - .as_ref() - .map(|e| e.idempotency_key) - .unwrap_or(0); - - // Database scope for the entry, read from the wire. `0` decodes to - // `DatabaseId::DEFAULT` (the pre-`database_id` legacy shape). The - // generic decode path (`from_replicated_entry`) returns only - // `(tenant, vshard, plan, resolved_now_ms)`, so the scope is taken - // from the entry itself — a WAL redo appended under the wrong - // database scope replays into the wrong catalog namespace. - let database_id = replicated_opt - .as_ref() - .map(|e| DatabaseId::new(e.database_id)) - .unwrap_or(DatabaseId::DEFAULT); - - // ── Array CRDT variants — handled on the Control Plane, bypass Data Plane ── - if let Some(replicated) = replicated_opt { - let target_vshard = replicated.vshard_id; - match replicated.write { - ReplicatedWrite::ArrayOp { - ref array, - ref op_bytes, - ref provenance, - .. - } => { - let applied_ok = apply_array_op( - &state, - &tracker, - AppliedPosition { - group_id: batch.group_id, - log_index: entry.index, - applied_key, - }, - crate::control::array_sync::ArrayOpTarget { - tenant_id: TenantId::new(replicated.tenant_id), - database_id: DatabaseId::new(replicated.database_id), - array, - }, - op_bytes, - provenance.as_deref(), - ) - .await; - // Advance the durable prefix only when the op durably - // applied — same safe-watermark rule as the Data Plane - // write path below, and the same funnel: the op path - // submits through `submit_write`, so its redo is fsynced - // before it reports success. A failure breaks the - // prefix: the entry must stay replayable. - prefix.record(entry.index, applied_ok); - continue; - } - ReplicatedWrite::ArraySchema { - ref array, - ref snapshot_payload, - schema_hlc_bytes, - } => { - let applied_ok = apply_array_schema( - &state, - &tracker, - AppliedPosition { - group_id: batch.group_id, - log_index: entry.index, - applied_key, - }, - crate::control::array_sync::raft_apply::ArraySchemaPayload { - tenant_id: TenantId::new(replicated.tenant_id), - database_id: DatabaseId::new(replicated.database_id), - array, - snapshot_payload, - schema_hlc_bytes, - }, - ); - // Advance the durable prefix only when the schema - // snapshot durably imported. - // - // This is the one applied branch that mints no WAL redo - // record, and it needs none: its entire effect is two - // fsync-committed redb transactions — the schema - // registry's snapshot row and the array catalog's entry - // — both written before it reports success. The floor's - // invariant ("this entry's state survives a restart, so - // Raft need not redeliver it") is therefore already met - // by the registries themselves. The cell paths have no - // such durable store behind them: their state lives in - // Data-Plane memtables and exists on disk only as the - // redo record the funnel appends, which is why they must - // route through `submit_write`. - prefix.record(entry.index, applied_ok); - continue; - } - ReplicatedWrite::CalvinReadResult { - epoch, - position, - passive_vshard, - tenant_id, - ref values, - } => { - let decoded_values: Vec<( - nodedb_physical::physical_plan::meta::PassiveReadKeyId, - nodedb_types::Value, - )> = match zerompk::from_msgpack(values) { - Ok(decoded) => decoded, - Err(e) => { - tracing::warn!( - group_id = batch.group_id, - index = entry.index, - error = %e, - "failed to decode CalvinReadResult payload" - ); - tracker.complete( - batch.group_id, - entry.index, - applied_key, - Err(crate::Error::Internal { - detail: format!("decode CalvinReadResult payload: {e}"), - }), - ); - // Prefix-neutral, like the forward below: a read - // result mints no durable state either way, so - // there is nothing a re-delivery could restore - // and nothing later entries must wait behind. - prefix.skip(); - continue; - } - }; - - let event = ReadResultEvent { - epoch, - position, - passive_vshard, - tenant_id: TenantId::new(tenant_id), - values: decoded_values, - }; - - let send_result = calvin_read_result_senders - .lock() - .unwrap_or_else(|p| p.into_inner()) - .get(&target_vshard) - .cloned() - .map(|sender| sender.try_send(event)); - - if let Some(Err(e)) = send_result { - tracing::warn!( - group_id = batch.group_id, - index = entry.index, - error = %e, - "failed to forward CalvinReadResult to scheduler" - ); - } - tracker.complete( - batch.group_id, - entry.index, - applied_key, - Ok(AppliedWrite::unversioned(Vec::new())), - ); - // A read result is forwarded to an in-memory Calvin - // scheduler and writes nothing durable, so it neither - // advances the prefix nor breaks it. Advancing on it - // would assert a redo record that does not exist; - // breaking on it would stall the floor behind an entry - // that a re-delivery could not usefully replay anyway — - // the epoch it belongs to does not survive a restart — - // and force every later write in the batch to be applied - // twice on the next boot. - prefix.skip(); - continue; - } - _ => {} - } - } - - let decoded = - from_replicated_entry(&entry.data, Some(state.surrogate_assigner.as_ref())); - let (tenant_id, vshard_id, plan, resolved_now_ms) = match decoded { - Ok(Some(t)) => t, - Ok(None) => { - // Couldn't deserialize — might be a different format or corrupted. - debug!( - group_id = batch.group_id, - index = entry.index, - "skipping non-ReplicatedEntry commit" - ); - tracker.complete( - batch.group_id, - entry.index, - applied_key, - Ok(AppliedWrite::unversioned(Vec::new())), - ); - // Prefix-neutral. This is a pure shape check over - // `entry.data`, so a re-delivery on the next boot decodes to - // `None` again and skips again — stalling the floor behind - // it buys nothing and costs a double-apply of every later - // write in the batch. It applied no state, so it must not - // advance the floor either. - prefix.skip(); - continue; - } - Err(e) => { - tracing::warn!( - group_id = batch.group_id, - index = entry.index, - error = %e, - "failed to decode replicated entry (surrogate bind error)" - ); - tracker.complete( - batch.group_id, - entry.index, - applied_key, - Err(crate::Error::Internal { - detail: format!("decode replicated entry: {e}"), - }), - ); - // Breaks the prefix, unlike the `Ok(None)` skip above: this - // IS a write, and it failed against live surrogate-assigner - // state rather than on its own bytes, so a re-delivery can - // legitimately succeed. Holding the floor below it is what - // keeps it replayable. - prefix.record(entry.index, false); - continue; - } - }; - - // Raft-native array cell writes (`ArrayCellPut` / `ArrayCellDelete`) - // decode to `PhysicalPlan::Array(Put | Delete)`. A follower must - // OPEN the array on the Data Plane before applying, so these route - // through the array-open bootstrap first — and then through the same - // write funnel as the generic branch below, which is what gives them - // a redo record and the fsync the applied floor asserts. No other - // `ReplicatedWrite` variant decodes to a `PhysicalPlan::Array`, so - // this match is exact. - if matches!( - plan, - PhysicalPlan::Array(ArrayOp::Put { .. } | ArrayOp::Delete { .. }) - ) { - let applied_ok = apply_array_cell_write( - &state, - &tracker, - AppliedPosition { - group_id: batch.group_id, - log_index: entry.index, - applied_key, - }, - ArrayCellTarget { - tenant_id, - database_id, - vshard: vshard_id, - resolved_now_ms, - }, - plan, - ) - .await; - prefix.record(entry.index, applied_ok); - continue; - } - - let submitted = submit_write( - &state, - SubmitWrite { - tenant_id, - database_id, - vshard_id, - plan, - trace_id: TraceId::generate(), - // Cluster mode has exactly ONE write-apply path — this loop; - // the proposing node does not execute locally before commit - // either. Tagging these `RaftFollower` would mean AFTER - // triggers, DML audit, and CRDT packaging never fire anywhere - // in cluster mode, so the committed write keeps the `User` - // source its proposer had. - event_source: crate::event::EventSource::User, - txn_id: None, - // Auth ran on the proposing node before the entry was - // proposed; the committed entry carries no session user. - user_id: None, - // The redo record is appended HERE, on this replica, from the - // committed plan — the leader's WAL LSN is deliberately not - // carried on the wire, and the memory-only engines have no - // other durability path. `now_override` pins a TTL-bearing KV - // write's `expire_at_ms` to the instant the proposing node - // resolved, so this replica's redo record and its live apply - // install the byte-identical value every other replica does. - durability: WalDurability::AppendHere { - now_override: resolved_now_ms, - }, - // Raft committed this entry at a fixed log index; every - // replica applies it in that order. Re-entering the - // write-admission gate would re-decide an ordering that is - // already final. - ordering: WriteOrdering::AlreadyOrdered, - // This loop runs on EVERY replica, so it must not publish: - // the node that proposed this entry already published the - // write's change event once, after commit + apply. Emitting - // here would give each subscriber one copy per replica plus - // a NOTIFY fan-out from each. See [`ChangeFeedOwner`]. - change_feed: ChangeFeedOwner::Unowned, - }, - ) - .await - .map(|outcome| outcome.response); - - // The funnel returns an error-status response as `Ok`; a committed - // entry that failed to apply must surface to the propose waiter as a - // failure, not as an empty success. - let result = match submitted { - // The response carries this replica's post-write - // `coll_write_lsn` for the written collection, which the - // proposer needs as its read-your-writes floor: the version is - // minted here (the funnel's WAL append) and never travels on the - // wire, so the propose waiter is the only place it can be - // handed back. - Ok(resp) if resp.status == Status::Ok => Ok(AppliedWrite::from_response(&resp)), - Ok(resp) => committed_response_result(&resp), - Err(e) => { - tracing::warn!( - group_id = batch.group_id, - index = entry.index, - error = %e, - "applying committed write failed" - ); - // Passed through: the typed error already carries the - // caller's classification. - Err(e) - } - }; - - let applied_ok = result.is_ok() || deterministic_crdt_fence_noop(&result); - tracker.complete(batch.group_id, entry.index, applied_key, result); - - // Extend the batch's durable prefix. On success `submit_write`'s - // durable-at-ack barrier has already fsynced this entry's redo, - // which is exactly the fact the floor asserts — `entry.index` is - // the data-plane applied watermark here, NOT raft's commit index. - // On failure the engines did not persist this index, so it is - // neither a safe compaction boundary nor a safe restart floor; - // breaking the prefix is what keeps a genuinely failed apply - // replayable rather than silently skipped. - prefix.record(entry.index, applied_ok); - } - - // One save + one compaction check per batch, against the contiguous - // prefix. Compaction is deliberately driven by the same index the floor - // was just saved at — never the batch's last delivered index — so it can - // never discard an entry the next boot still has to replay. Compacting - // on the raft commit index while the SPSC apply lags would likewise let - // the `SnapshotBuilder` serialize incomplete engine state and corrupt a - // lagging follower's snapshot. - if let Some(applied_index) = prefix.floor() { - record_durable_apply(&state, batch.group_id, applied_index); - } - } -} - -/// Record entry `applied_index` of `group_id` as durably applied: persist the -/// group's durable applied floor, then fire the compaction trigger against it. -/// -/// `applied_index` MUST be the highest CONTIGUOUS successfully-applied entry — -/// [`AppliedPrefix::floor`] — not merely some entry that happened to succeed. -/// Everything at and below it must have applied with its redo record already -/// WAL-fsync-durable, because that is the fact the floor asserts and the next -/// boot resumes Raft delivery above it on the strength of it. -/// -/// Called once per apply batch: each call is a redb commit and therefore an -/// fsync, and one per entry is slow enough to stall the raft loop it runs on. -/// -/// Order is load-bearing: the floor lands first because compaction is itself -/// gated on the floor (it may only discard entries the engines can no longer -/// need the log for). Compacting first would either be refused or, if the gate -/// used the delivery watermark, discard entries whose redo is not yet fsynced. -fn record_durable_apply(state: &Arc, group_id: u64, applied_index: u64) { - save_applied_index(state, group_id, applied_index); - maybe_compact_log(state, group_id, applied_index); -} - -/// Fire the Raft log-compaction trigger for `group_id` up to the -/// data-plane applied index `applied_index`, if a compactor is wired. -/// -/// Gated by the caller on data-plane apply completion. A no-op when no -/// compactor is installed (single-node mode) or when the group's -/// `log_compaction_threshold` is `None`. -fn maybe_compact_log(state: &Arc, group_id: u64, applied_index: u64) { - let Some(compactor) = state.raft_compactor.get() else { - return; - }; - match compactor(group_id, applied_index) { - Ok(true) => { - debug!( - group_id, - applied_index, "raft log compacted past data-plane applied watermark" - ); - } - Ok(false) => {} - Err(e) => { - tracing::warn!( - group_id, - applied_index, - error = %e, - "raft log compaction failed" - ); - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn fenced_frontier_mismatch_completes_retry_and_advances_durable_prefix() { - let result: crate::Result = Err(crate::Error::DataPlane( - crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { - expected: [1; 32], - actual: [2; 32], - }, - )); - assert!(deterministic_crdt_fence_noop(&result)); - assert!(matches!( - result, - Err(crate::Error::DataPlane( - crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { .. } - )) - )); - - let mut prefix = AppliedPrefix::new(); - prefix.record(17, deterministic_crdt_fence_noop(&result)); - assert_eq!(prefix.floor(), Some(17)); - } -} diff --git a/nodedb/src/control/distributed_applier/apply_loop/array_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/array_dispatch.rs new file mode 100644 index 000000000..a0f9a3b81 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/array_dispatch.rs @@ -0,0 +1,53 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Array CRDT variants — handled on the Control Plane, bypass the Data Plane. + +use std::sync::Arc; + +use crate::control::array_sync::ArrayOpTarget; +use crate::control::array_sync::raft_apply::{ + AppliedPosition, ArraySchemaPayload, apply_array_op, apply_array_schema, +}; +use crate::control::distributed_applier::propose_tracker::ProposeTracker; +use crate::control::state::SharedState; + +/// Apply a committed `ReplicatedWrite::ArrayOp` entry. +/// +/// Advances the durable prefix only when the op durably applied — same +/// safe-watermark rule as the Data Plane write path, and the same funnel: the +/// op path submits through `submit_write`, so its redo is fsynced before it +/// reports success. A failure breaks the prefix: the entry must stay +/// replayable. +pub(super) async fn apply_array_op_entry( + state: &Arc, + tracker: &Arc, + pos: AppliedPosition, + target: ArrayOpTarget<'_>, + op_bytes: &[u8], + provenance: Option<&[u8]>, +) -> bool { + apply_array_op(state, tracker, pos, target, op_bytes, provenance).await +} + +/// Apply a committed `ReplicatedWrite::ArraySchema` entry. +/// +/// Advances the durable prefix only when the schema snapshot durably +/// imported. +/// +/// This is the one applied branch that mints no WAL redo record, and it needs +/// none: its entire effect is two fsync-committed redb transactions — the +/// schema registry's snapshot row and the array catalog's entry — both +/// written before it reports success. The floor's invariant ("this entry's +/// state survives a restart, so Raft need not redeliver it") is therefore +/// already met by the registries themselves. The cell paths have no such +/// durable store behind them: their state lives in Data-Plane memtables and +/// exists on disk only as the redo record the funnel appends, which is why +/// they must route through `submit_write`. +pub(super) fn apply_array_schema_entry( + state: &Arc, + tracker: &Arc, + pos: AppliedPosition, + payload: ArraySchemaPayload<'_>, +) -> bool { + apply_array_schema(state, tracker, pos, payload) +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/bookkeeping.rs b/nodedb/src/control/distributed_applier/apply_loop/bookkeeping.rs new file mode 100644 index 000000000..1de50ce95 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/bookkeeping.rs @@ -0,0 +1,62 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Applied-prefix bookkeeping: persist the group's durable applied floor, +//! then fire the Raft log-compaction trigger against it. + +use std::sync::Arc; + +use tracing::debug; + +use crate::control::distributed_applier::applied_index::save_applied_index; +use crate::control::state::SharedState; + +/// Record entry `applied_index` of `group_id` as durably applied: persist the +/// group's durable applied floor, then fire the compaction trigger against it. +/// +/// `applied_index` MUST be the highest CONTIGUOUS successfully-applied entry — +/// [`crate::control::distributed_applier::applied_index::AppliedPrefix::floor`] +/// — not merely some entry that happened to succeed. Everything at and below +/// it must have applied with its redo record already WAL-fsync-durable, +/// because that is the fact the floor asserts and the next boot resumes Raft +/// delivery above it on the strength of it. +/// +/// Called once per apply batch: each call is a redb commit and therefore an +/// fsync, and one per entry is slow enough to stall the raft loop it runs on. +/// +/// Order is load-bearing: the floor lands first because compaction is itself +/// gated on the floor (it may only discard entries the engines can no longer +/// need the log for). Compacting first would either be refused or, if the gate +/// used the delivery watermark, discard entries whose redo is not yet fsynced. +pub(super) fn record_durable_apply(state: &Arc, group_id: u64, applied_index: u64) { + save_applied_index(state, group_id, applied_index); + maybe_compact_log(state, group_id, applied_index); +} + +/// Fire the Raft log-compaction trigger for `group_id` up to the +/// data-plane applied index `applied_index`, if a compactor is wired. +/// +/// Gated by the caller on data-plane apply completion. A no-op when no +/// compactor is installed (single-node mode) or when the group's +/// `log_compaction_threshold` is `None`. +fn maybe_compact_log(state: &Arc, group_id: u64, applied_index: u64) { + let Some(compactor) = state.raft_compactor.get() else { + return; + }; + match compactor(group_id, applied_index) { + Ok(true) => { + debug!( + group_id, + applied_index, "raft log compacted past data-plane applied watermark" + ); + } + Ok(false) => {} + Err(e) => { + tracing::warn!( + group_id, + applied_index, + error = %e, + "raft log compaction failed" + ); + } + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs b/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs new file mode 100644 index 000000000..1b4ff743a --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs @@ -0,0 +1,105 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Forward a committed `ReplicatedWrite::CalvinReadResult` entry to the local +//! Calvin scheduler's read-result channel for its target vShard. +//! +//! A read result is forwarded to an in-memory Calvin scheduler and writes +//! nothing durable, so the caller neither advances the applied prefix nor +//! breaks it for this entry. Advancing on it would assert a redo record that +//! does not exist; breaking on it would stall the floor behind an entry that a +//! re-delivery could not usefully replay anyway — the epoch it belongs to does +//! not survive a restart — and force every later write in the batch to be +//! applied twice on the next boot. + +use std::collections::BTreeMap; +use std::sync::{Arc, Mutex}; + +use tokio::sync::mpsc; + +use crate::control::array_sync::raft_apply::AppliedPosition; +use crate::control::cluster::calvin::ReadResultEvent; +use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; +use crate::types::TenantId; + +/// Fields extracted from a `ReplicatedWrite::CalvinReadResult` entry. +pub(super) struct CalvinReadResultFields<'a> { + pub target_vshard: u32, + pub epoch: u64, + pub position: u32, + pub passive_vshard: u32, + pub tenant_id: u64, + pub values: &'a [u8], +} + +/// Decode `fields.values`, forward the resulting [`ReadResultEvent`] to the +/// scheduler registered for `fields.target_vshard`, and complete the propose +/// waiter. +pub(super) fn forward_calvin_read_result( + tracker: &Arc, + calvin_read_result_senders: &Arc>>>, + pos: AppliedPosition, + fields: CalvinReadResultFields<'_>, +) { + let AppliedPosition { + group_id, + log_index, + applied_key, + } = pos; + + let decoded_values: Vec<( + nodedb_physical::physical_plan::meta::PassiveReadKeyId, + nodedb_types::Value, + )> = match zerompk::from_msgpack(fields.values) { + Ok(decoded) => decoded, + Err(e) => { + tracing::warn!( + group_id, + index = log_index, + error = %e, + "failed to decode CalvinReadResult payload" + ); + tracker.complete( + group_id, + log_index, + applied_key, + Err(crate::Error::Internal { + detail: format!("decode CalvinReadResult payload: {e}"), + }), + ); + // Prefix-neutral, like the forward below: a read result mints no + // durable state either way, so there is nothing a re-delivery + // could restore and nothing later entries must wait behind. + return; + } + }; + + let event = ReadResultEvent { + epoch: fields.epoch, + position: fields.position, + passive_vshard: fields.passive_vshard, + tenant_id: TenantId::new(fields.tenant_id), + values: decoded_values, + }; + + let send_result = calvin_read_result_senders + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&fields.target_vshard) + .cloned() + .map(|sender| sender.try_send(event)); + + if let Some(Err(e)) = send_result { + tracing::warn!( + group_id, + index = log_index, + error = %e, + "failed to forward CalvinReadResult to scheduler" + ); + } + tracker.complete( + group_id, + log_index, + applied_key, + Ok(AppliedWrite::unversioned(Vec::new())), + ); +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/driver.rs b/nodedb/src/control/distributed_applier/apply_loop/driver.rs new file mode 100644 index 000000000..c97f09e52 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/driver.rs @@ -0,0 +1,236 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-batch loop driver: pulls batches off the apply channel, dispatches +//! each entry to the Array CRDT / Calvin fast path or the generic write +//! path, and lands the batch's durable applied floor once every entry in it +//! has been applied. + +use std::sync::Arc; + +use tokio::sync::mpsc; + +use crate::control::array_sync::ArrayOpTarget; +use crate::control::array_sync::raft_apply::{AppliedPosition, ArraySchemaPayload}; +use crate::control::cluster::calvin::ReadResultEvent; +use crate::control::distributed_applier::applied_index::AppliedPrefix; +use crate::control::distributed_applier::applier::ApplyBatch; +use crate::control::distributed_applier::propose_tracker::ProposeTracker; +use crate::control::state::SharedState; +use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite}; +use crate::types::{DatabaseId, TenantId}; + +use super::array_dispatch::{apply_array_op_entry, apply_array_schema_entry}; +use super::bookkeeping::record_durable_apply; +use super::calvin_read_result::{CalvinReadResultFields, forward_calvin_read_result}; +use super::write_dispatch::apply_generic_entry; + +/// Run the background loop that applies committed Raft entries to the local Data Plane. +/// +/// This task reads from the apply channel, deserializes each entry, dispatches +/// the write to the Data Plane via SPSC, and notifies proposers. +pub async fn run_apply_loop( + mut apply_rx: mpsc::Receiver, + state: Arc, + tracker: Arc, + calvin_read_result_senders: Arc< + std::sync::Mutex>>, + >, +) { + while let Some(batch) = apply_rx.recv().await { + // The floor is saved ONCE per batch, after the loop — never per entry. + // `save_applied_index` lands a redb transaction, and redb commits at + // `Durability::Immediate`, so a per-entry save puts one synchronous + // fsync per applied entry directly on the raft apply path. That stalls + // the raft loop hard enough to delay heartbeats and keep elections from + // stabilizing under a multi-node write load. One fsync per batch + // amortizes the cost across every entry in it and keeps the critical + // path free. + // + // `AppliedPrefix` computes WHICH index is safe to save: the highest + // contiguous successfully-applied entry, stopping at the first failure + // and never advancing past it. Every branch below must therefore report + // its outcome — `record` for the ones whose success means a durable + // redo record, `skip` for the ones that apply no durable state at all. + let mut prefix = AppliedPrefix::new(); + for entry in &batch.entries { + // Decode once; reused for both the idempotency key and the + // Array/Calvin fast-path match below. Returns 0 for + // unparseable / pre-key entries; the tracker treats 0 as + // "no key" (no mismatch detection). + let replicated_opt = ReplicatedEntry::from_bytes(&entry.data); + let applied_key = replicated_opt + .as_ref() + .map(|e| e.idempotency_key) + .unwrap_or(0); + + // Database scope for the entry, read from the wire. `0` decodes to + // `DatabaseId::DEFAULT` (the pre-`database_id` legacy shape). The + // generic decode path (`from_replicated_entry`) returns only + // `(tenant, vshard, plan, resolved_now_ms)`, so the scope is taken + // from the entry itself — a WAL redo appended under the wrong + // database scope replays into the wrong catalog namespace. + let database_id = replicated_opt + .as_ref() + .map(|e| DatabaseId::new(e.database_id)) + .unwrap_or(DatabaseId::DEFAULT); + + // ── Array CRDT variants — handled on the Control Plane, bypass Data Plane ── + if let Some(replicated) = replicated_opt { + let target_vshard = replicated.vshard_id; + let pos = AppliedPosition { + group_id: batch.group_id, + log_index: entry.index, + applied_key, + }; + match replicated.write { + ReplicatedWrite::ArrayOp { + ref array, + ref op_bytes, + ref provenance, + .. + } => { + let applied_ok = apply_array_op_entry( + &state, + &tracker, + pos, + ArrayOpTarget { + tenant_id: TenantId::new(replicated.tenant_id), + database_id: DatabaseId::new(replicated.database_id), + array, + }, + op_bytes, + provenance.as_deref(), + ) + .await; + // Advance the durable prefix only when the op durably + // applied — same safe-watermark rule as the Data Plane + // write path below, and the same funnel: the op path + // submits through `submit_write`, so its redo is fsynced + // before it reports success. A failure breaks the + // prefix: the entry must stay replayable. + prefix.record(entry.index, applied_ok); + continue; + } + ReplicatedWrite::ArraySchema { + ref array, + ref snapshot_payload, + schema_hlc_bytes, + } => { + let applied_ok = apply_array_schema_entry( + &state, + &tracker, + pos, + ArraySchemaPayload { + tenant_id: TenantId::new(replicated.tenant_id), + database_id: DatabaseId::new(replicated.database_id), + array, + snapshot_payload, + schema_hlc_bytes, + }, + ); + // Advance the durable prefix only when the schema + // snapshot durably imported. + // + // This is the one applied branch that mints no WAL redo + // record, and it needs none: its entire effect is two + // fsync-committed redb transactions — the schema + // registry's snapshot row and the array catalog's entry + // — both written before it reports success. The floor's + // invariant ("this entry's state survives a restart, so + // Raft need not redeliver it") is therefore already met + // by the registries themselves. The cell paths have no + // such durable store behind them: their state lives in + // Data-Plane memtables and exists on disk only as the + // redo record the funnel appends, which is why they must + // route through `submit_write`. + prefix.record(entry.index, applied_ok); + continue; + } + ReplicatedWrite::CalvinReadResult { + epoch, + position, + passive_vshard, + tenant_id, + ref values, + } => { + forward_calvin_read_result( + &tracker, + &calvin_read_result_senders, + pos, + CalvinReadResultFields { + target_vshard, + epoch, + position, + passive_vshard, + tenant_id, + values, + }, + ); + // A read result is forwarded to an in-memory Calvin + // scheduler and writes nothing durable, so it neither + // advances the prefix nor breaks it. Advancing on it + // would assert a redo record that does not exist; + // breaking on it would stall the floor behind an entry + // that a re-delivery could not usefully replay anyway — + // the epoch it belongs to does not survive a restart — + // and force every later write in the batch to be applied + // twice on the next boot. + prefix.skip(); + continue; + } + _ => {} + } + } + + apply_generic_entry( + &state, + &tracker, + &mut prefix, + batch.group_id, + entry, + applied_key, + database_id, + ) + .await; + } + + // One save + one compaction check per batch, against the contiguous + // prefix. Compaction is deliberately driven by the same index the floor + // was just saved at — never the batch's last delivered index — so it can + // never discard an entry the next boot still has to replay. Compacting + // on the raft commit index while the SPSC apply lags would likewise let + // the `SnapshotBuilder` serialize incomplete engine state and corrupt a + // lagging follower's snapshot. + if let Some(applied_index) = prefix.floor() { + record_durable_apply(&state, batch.group_id, applied_index); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::distributed_applier::apply_loop::helpers::deterministic_crdt_fence_noop; + use crate::control::distributed_applier::propose_tracker::AppliedWrite; + + #[test] + fn fenced_frontier_mismatch_completes_retry_and_advances_durable_prefix() { + let result: crate::Result = Err(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { + expected: [1; 32], + actual: [2; 32], + }, + )); + assert!(deterministic_crdt_fence_noop(&result)); + assert!(matches!( + result, + Err(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { .. } + )) + )); + + let mut prefix = AppliedPrefix::new(); + prefix.record(17, deterministic_crdt_fence_noop(&result)); + assert_eq!(prefix.floor(), Some(17)); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/helpers.rs b/nodedb/src/control/distributed_applier/apply_loop/helpers.rs new file mode 100644 index 000000000..976e9d881 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/helpers.rs @@ -0,0 +1,39 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared classification helpers for committed-write apply results. + +use crate::control::distributed_applier::propose_tracker::AppliedWrite; + +pub(super) fn committed_response_result( + response: &crate::bridge::envelope::Response, +) -> crate::Result { + if response.status == crate::bridge::envelope::Status::Ok { + return Ok(AppliedWrite::from_response(response)); + } + // The typed code carries the client's classification (constraint, authz, + // conflict); stringifying it here would leave the caller only XX000. + match response.error_code.as_deref() { + Some(code) => { + tracing::warn!(reason = ?code, "applying committed write failed"); + Err(crate::Error::DataPlane(code.clone())) + } + None => { + tracing::warn!( + reason = "execution error", + "applying committed write failed" + ); + Err(crate::Error::Internal { + detail: "execution error".to_owned(), + }) + } + } +} + +pub(super) fn deterministic_crdt_fence_noop(result: &crate::Result) -> bool { + matches!( + result, + Err(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { .. } + )) + ) +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/mod.rs b/nodedb/src/control/distributed_applier/apply_loop/mod.rs new file mode 100644 index 000000000..f8ce17289 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/mod.rs @@ -0,0 +1,30 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Background apply loop — reads committed Raft entries from the mpsc channel, +//! submits them through the shared Control-Plane write funnel (which appends +//! each entry's redo record on THIS replica before the enqueue), and resolves +//! propose waiters with the result. +//! +//! Each batch advances the group's durable applied floor (see +//! [`super::applied_index`]) to its highest contiguous successfully-applied +//! entry, so the next boot replays only above it and no entry is applied by +//! both WAL replay and Raft log replay. +//! +//! Split by concern: +//! - [`driver`]: the per-batch loop that reads off the apply channel and +//! dispatches each entry, then lands the batch's durable applied floor. +//! - [`array_dispatch`]: the Array CRDT `ArrayOp` / `ArraySchema` apply path. +//! - [`calvin_read_result`]: forwards a committed `CalvinReadResult` entry to +//! the local Calvin scheduler. +//! - [`write_dispatch`]: the generic decode + Data-Plane `submit_write` path. +//! - [`bookkeeping`]: applied-floor persistence + Raft log compaction trigger. +//! - [`helpers`]: shared response/result classification helpers. + +mod array_dispatch; +mod bookkeeping; +mod calvin_read_result; +mod driver; +mod helpers; +mod write_dispatch; + +pub use driver::run_apply_loop; diff --git a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs new file mode 100644 index 000000000..3ea436d53 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs @@ -0,0 +1,206 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Generic per-entry apply path: decode the replicated entry, route +//! Raft-native array cell writes through the array-open bootstrap, and +//! dispatch everything else through the shared Control-Plane write funnel. + +use std::sync::Arc; + +use tracing::debug; + +use nodedb_physical::physical_plan::ArrayOp; +use nodedb_raft::message::LogEntry; + +use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::control::array_sync::raft_apply::{ + AppliedPosition, ArrayCellTarget, apply_array_cell_write, +}; +use crate::control::distributed_applier::applied_index::AppliedPrefix; +use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; +use crate::control::server::dispatch_utils::{ + ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, +}; +use crate::control::state::SharedState; +use crate::control::wal_replication::from_replicated_entry; +use crate::types::{DatabaseId, TraceId}; + +use super::helpers::{committed_response_result, deterministic_crdt_fence_noop}; + +/// Decode `entry` and apply it: Raft-native array cell writes route through +/// the array-open bootstrap and the write funnel; everything else dispatches +/// through the write funnel directly. Records the outcome into `prefix`. +pub(super) async fn apply_generic_entry( + state: &Arc, + tracker: &Arc, + prefix: &mut AppliedPrefix, + group_id: u64, + entry: &LogEntry, + applied_key: u64, + database_id: DatabaseId, +) { + let decoded = from_replicated_entry(&entry.data, Some(state.surrogate_assigner.as_ref())); + let (tenant_id, vshard_id, plan, resolved_now_ms) = match decoded { + Ok(Some(t)) => t, + Ok(None) => { + // Couldn't deserialize — might be a different format or corrupted. + debug!( + group_id, + index = entry.index, + "skipping non-ReplicatedEntry commit" + ); + tracker.complete( + group_id, + entry.index, + applied_key, + Ok(AppliedWrite::unversioned(Vec::new())), + ); + // Prefix-neutral. This is a pure shape check over + // `entry.data`, so a re-delivery on the next boot decodes to + // `None` again and skips again — stalling the floor behind + // it buys nothing and costs a double-apply of every later + // write in the batch. It applied no state, so it must not + // advance the floor either. + prefix.skip(); + return; + } + Err(e) => { + tracing::warn!( + group_id, + index = entry.index, + error = %e, + "failed to decode replicated entry (surrogate bind error)" + ); + tracker.complete( + group_id, + entry.index, + applied_key, + Err(crate::Error::Internal { + detail: format!("decode replicated entry: {e}"), + }), + ); + // Breaks the prefix, unlike the `Ok(None)` skip above: this + // IS a write, and it failed against live surrogate-assigner + // state rather than on its own bytes, so a re-delivery can + // legitimately succeed. Holding the floor below it is what + // keeps it replayable. + prefix.record(entry.index, false); + return; + } + }; + + // Raft-native array cell writes (`ArrayCellPut` / `ArrayCellDelete`) + // decode to `PhysicalPlan::Array(Put | Delete)`. A follower must + // OPEN the array on the Data Plane before applying, so these route + // through the array-open bootstrap first — and then through the same + // write funnel as the generic branch below, which is what gives them + // a redo record and the fsync the applied floor asserts. No other + // `ReplicatedWrite` variant decodes to a `PhysicalPlan::Array`, so + // this match is exact. + if matches!( + plan, + PhysicalPlan::Array(ArrayOp::Put { .. } | ArrayOp::Delete { .. }) + ) { + let applied_ok = apply_array_cell_write( + state, + tracker, + AppliedPosition { + group_id, + log_index: entry.index, + applied_key, + }, + ArrayCellTarget { + tenant_id, + database_id, + vshard: vshard_id, + resolved_now_ms, + }, + plan, + ) + .await; + prefix.record(entry.index, applied_ok); + return; + } + + let submitted = submit_write( + state, + SubmitWrite { + tenant_id, + database_id, + vshard_id, + plan, + trace_id: TraceId::generate(), + // Cluster mode has exactly ONE write-apply path — this loop; + // the proposing node does not execute locally before commit + // either. Tagging these `RaftFollower` would mean AFTER + // triggers, DML audit, and CRDT packaging never fire anywhere + // in cluster mode, so the committed write keeps the `User` + // source its proposer had. + event_source: crate::event::EventSource::User, + txn_id: None, + // Auth ran on the proposing node before the entry was + // proposed; the committed entry carries no session user. + user_id: None, + // The redo record is appended HERE, on this replica, from the + // committed plan — the leader's WAL LSN is deliberately not + // carried on the wire, and the memory-only engines have no + // other durability path. `now_override` pins a TTL-bearing KV + // write's `expire_at_ms` to the instant the proposing node + // resolved, so this replica's redo record and its live apply + // install the byte-identical value every other replica does. + durability: WalDurability::AppendHere { + now_override: resolved_now_ms, + }, + // Raft committed this entry at a fixed log index; every + // replica applies it in that order. Re-entering the + // write-admission gate would re-decide an ordering that is + // already final. + ordering: WriteOrdering::AlreadyOrdered, + // This loop runs on EVERY replica, so it must not publish: + // the node that proposed this entry already published the + // write's change event once, after commit + apply. Emitting + // here would give each subscriber one copy per replica plus + // a NOTIFY fan-out from each. See [`ChangeFeedOwner`]. + change_feed: ChangeFeedOwner::Unowned, + }, + ) + .await + .map(|outcome| outcome.response); + + // The funnel returns an error-status response as `Ok`; a committed + // entry that failed to apply must surface to the propose waiter as a + // failure, not as an empty success. + let result = match submitted { + // The response carries this replica's post-write + // `coll_write_lsn` for the written collection, which the + // proposer needs as its read-your-writes floor: the version is + // minted here (the funnel's WAL append) and never travels on the + // wire, so the propose waiter is the only place it can be + // handed back. + Ok(resp) if resp.status == Status::Ok => Ok(AppliedWrite::from_response(&resp)), + Ok(resp) => committed_response_result(&resp), + Err(e) => { + tracing::warn!( + group_id, + index = entry.index, + error = %e, + "applying committed write failed" + ); + // Passed through: the typed error already carries the + // caller's classification. + Err(e) + } + }; + + let applied_ok = result.is_ok() || deterministic_crdt_fence_noop(&result); + tracker.complete(group_id, entry.index, applied_key, result); + + // Extend the batch's durable prefix. On success `submit_write`'s + // durable-at-ack barrier has already fsynced this entry's redo, + // which is exactly the fact the floor asserts — `entry.index` is + // the data-plane applied watermark here, NOT raft's commit index. + // On failure the engines did not persist this index, so it is + // neither a safe compaction boundary nor a safe restart floor; + // breaking the prefix is what keeps a genuinely failed apply + // replayable rather than silently skipped. + prefix.record(entry.index, applied_ok); +} From 138bd279111f8a1a796aaa5e6987ef91b8fd45c3 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 18:52:15 +0800 Subject: [PATCH 11/64] refactor(dispatch_utils): split the write funnel into a directory module funnel.rs held write admission, WAL redo append, Data-Plane dispatch, and response collection/post-apply steps in one file. It becomes a funnel/ directory with admission.rs, wal_append.rs, dispatch.rs, driver.rs, and response.rs, each keeping its existing logic. --- .../dispatch_utils/submit_write/funnel.rs | 450 ------------------ .../submit_write/funnel/admission.rs | 94 ++++ .../submit_write/funnel/dispatch.rs | 130 +++++ .../submit_write/funnel/driver.rs | 159 +++++++ .../dispatch_utils/submit_write/funnel/mod.rs | 41 ++ .../submit_write/funnel/response.rs | 245 ++++++++++ .../submit_write/funnel/wal_append.rs | 110 +++++ 7 files changed, 779 insertions(+), 450 deletions(-) delete mode 100644 nodedb/src/control/server/dispatch_utils/submit_write/funnel.rs create mode 100644 nodedb/src/control/server/dispatch_utils/submit_write/funnel/admission.rs create mode 100644 nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs create mode 100644 nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs create mode 100644 nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs create mode 100644 nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs create mode 100644 nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel.rs deleted file mode 100644 index a7e9d3796..000000000 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel.rs +++ /dev/null @@ -1,450 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! THE single Control-Plane write funnel. -//! -//! Every Control-Plane path that puts a write on the SPSC bridge routes through -//! [`submit_write`]: the autocommit / internal funnel (`dispatch.rs`), the -//! pgwire local-dispatch path (`pgwire::handler::submit`), and the Raft -//! apply loop (`distributed_applier::apply_loop`). The funnel owns write -//! admission, the WAL redo append, the enqueue, the bounded response collect, -//! the post-apply redo, the durable-at-ack barrier, and — for the caller that -//! owns it (see [`ChangeFeedOwner`]) — the CDC publish, in that order, which is -//! the correctness contract. -//! -//! A path that reimplements these steps drifts silently: it is not a compile -//! error to omit the redo append or the change-event publish, and the omission -//! only surfaces as lost data after a crash, or as a change stream that never -//! fires. Add the step here, once, and every caller gets it. -//! -//! It also owns the mirror of the redo append: when it appended the record -//! itself and the Data Plane then REFUSED the write, it cancels that record -//! before returning the error. See [`super::super::write_abort`] for which verdicts -//! qualify, the residual crash window it does not close, and the latency it -//! costs a rejection. - -use std::time::Instant; - -use crate::bridge::envelope::{Priority, Request, Status}; -use crate::control::server::shared::session::statement_deadline; -use crate::control::server::wal_dispatch::{self, WalAppendRequest}; -use crate::control::state::SharedState; -use crate::types::ReadConsistency; - -use super::super::change_events::{extract_write_change_set, publish_change_set}; -use super::super::collect::{DispatchCollectError, collect_bounded_response}; -use super::ambiguous_ddl::preserve_ambiguous_array_ddl; -use super::params::{ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering}; - -/// Admit, make durable, enqueue, collect, and publish one write. -/// -/// See [`SubmitOutcome`] for what comes back. -pub(crate) async fn submit_write( - shared: &SharedState, - params: SubmitWrite, -) -> crate::Result { - let SubmitWrite { - tenant_id, - database_id, - vshard_id, - mut plan, - trace_id, - event_source, - txn_id, - user_id, - durability, - ordering, - change_feed, - } = params; - - // The running statement's deadline, pinned once at the session boundary and - // shared by every request the statement fans out into. Used for both the - // envelope the Data Plane enforces and the Control-Plane collect below, so - // the two halves cannot disagree about when this statement expires. - let deadline = statement_deadline(shared.tuning.network.default_deadline_secs); - - // Change metadata is derived from the plan HERE, before it is moved into - // the request — the publish itself happens after apply, once the response - // (which carries the event's LSN) exists, by which point the plan is gone. - // Extraction is a pure match that clones out collection / document - // identity, so a caller whose change feed is `Unowned` skips it rather than - // allocating tuples nothing will read. - let change_set = match change_feed { - ChangeFeedOwner::Funnel => Some(extract_write_change_set(&plan, tenant_id)), - ChangeFeedOwner::Unowned => None, - }; - - // Post-apply redo classification, computed before `plan` is moved (the - // RouteToCalvin admit arm moves it). For a write whose autocommit WAL path - // mints no redo of its own but whose effect must survive a WAL-only restart - // (a document PointUpdate on a collection carrying a secondary vector - // index), the durable redo is minted AFTER apply from the surrogate + - // post-image the Data Plane returns in `Response::write_set`. - // `Some(collection)` for such a write, else `None`. - let post_apply = wal_dispatch::plan_post_apply_redo(&plan); - let appends_here = matches!(&durability, WalDurability::AppendHere { .. }); - - // Durable-at-ack obligation, also computed before `plan` moves. `Some` only - // for a write whose redo record THIS funnel is required to mint; a caller - // that appended upstream (or declared durability owned elsewhere) is not - // held to it, because the LSN it does or does not supply is its own - // contract. See `durability_barrier` for why this is narrower than - // "write-class plan with no LSN". - let funnel_redo_engine = if appends_here { - super::super::durability_barrier::funnel_minted_redo_engine(&plan) - } else { - None - }; - - // Write-admission gate: every write-class plan whose ordering is not already - // final passes here. An uncontended point write takes the fast path holding - // its per-vShard deterministic locks; a contended or bulk write is submitted - // through the deterministic scheduler and its applied response is surfaced - // here; reads / control ops are `Exempt`. - // - // Ordering (fast path): the guard is acquired FIRST, then — for a write that - // owns its durability (`AppendHere`) — the WAL append happens below, under - // the guard, minting the LSN just before the enqueue. The guard is released - // immediately after the enqueue (not across the response await). - use crate::control::server::shared::write_admission::{ - WriteAdmission, WriteTarget, admit, bare_ok_response, route_write_to_calvin, - }; - let (admission, admission_guard, order_guard) = match ordering { - WriteOrdering::AlreadyOrdered => ( - crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::AlreadyOrdered, - ), - None, - None, - ), - WriteOrdering::Gate => match admit( - shared, - &WriteTarget { - tenant_id, - database_id, - vshard_id, - plan: &plan, - }, - ) { - WriteAdmission::ExemptRead => ( - crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::Read, - ), - None, - None, - ), - WriteAdmission::FastPath { guard } => { - (crate::bridge::envelope::Admission::Admitted, guard, None) - } - WriteAdmission::FastPathBlocking { key, keyed_lock } => { - // Single-node serialization point: acquire the per-key FIFO - // order-lock FIRST, before the WAL append and enqueue below. - // `tokio::sync::Mutex` is fair, so concurrent same-key writers - // are admitted in arrival order — the WAL append + enqueue then - // happen in that order, giving WAL-LSN order == enqueue order == - // apply order per key. Distinct keys use distinct per-key mutexes - // and never contend. - let order_guard = keyed_lock.lock_owned(key).await; - ( - crate::bridge::envelope::Admission::Admitted, - None, - Some(order_guard), - ) - } - WriteAdmission::RouteToCalvin => { - // The deterministic scheduler applies the write (emitting its own - // WriteEvents) and returns the applied response; a plain write with - // no RETURNING rows yields `None`, synthesized into a bare `Ok`. - // Calvin owns durability on this route (the sequenced TxClass plus - // its own `CalvinApplied` WAL record), so no local append happens. - let routed = - route_write_to_calvin(shared, tenant_id, database_id, vshard_id, plan).await?; - return Ok(SubmitOutcome { - response: routed - .unwrap_or_else(|| bare_ok_response(crate::types::RequestId::new(0))), - wal_lsn: None, - }); - } - }, - }; - - // Array DDL conversion is intentionally read-only. Once this task has - // passed authorization and admission, install its durable catalog state - // immediately before the Data-Plane dispatch. The mirror is changed only - // after the redb transaction commits. - let ddl_transition = crate::control::array_catalog::ddl::apply_authorized_ddl( - shared, - tenant_id, - database_id, - &plan, - )?; - macro_rules! rollback_ddl { - ($result:expr) => { - match $result { - Ok(value) => value, - Err(error) => { - let _ = ddl_transition.rollback(shared); - return Err(error.into()); - } - } - }; - } - - // Durability, under the guard, immediately before the enqueue: the LSN is - // minted in the same order the request is about to be enqueued. - let (wal_lsn, resolved_now_ms) = match durability { - WalDurability::AppendHere { now_override } => { - let outcome = rollback_ddl!(wal_dispatch::wal_append(WalAppendRequest { - wal: &shared.wal, - tenant_id, - vshard_id, - database_id, - plan: &plan, - credentials: None, - now_override, - })); - (outcome.lsn, outcome.resolved_now_ms) - } - WalDurability::CallerSupplied { - wal_lsn, - resolved_now_ms, - } => (wal_lsn, resolved_now_ms), - }; - - // Write the resolved LSN back into the plan itself. The envelope's - // `wal_lsn` is where most engines read their committed version from, but the - // array engine stamps its tile versions from the LSN carried in the plan - // while replay stamps them from the record header — so the plan the Data - // Plane is about to execute must name the record that reproduces it. This is - // the only place that knows both, and it knows them for every caller: no - // upstream path may allocate an LSN of its own and hope it matches. - if let Some(lsn) = wal_lsn { - wal_dispatch::stamp_minted_lsn(&mut plan, lsn); - } - - // Per-vShard QPS + latency timer. `dispatch_started` marks the wall-clock - // moment the request enters the Control Plane dispatch site; observation - // happens on every exit path (success, budget over-run, timeout) so the - // histogram captures the true end-to-end shape of the work routed to this - // vshard. - let dispatch_started = Instant::now(); - let vshard_u32 = vshard_id.as_u32(); - let observe = |shared: &SharedState| { - let latency_us = dispatch_started.elapsed().as_micros().min(u64::MAX as u128) as u64; - shared.per_vshard_metrics.observe(vshard_u32, latency_us); - }; - - let request_id = shared.next_request_id(); - let request = Request { - request_id, - tenant_id, - database_id, - vshard_id, - plan, - deadline, - priority: Priority::Normal, - trace_id, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source, - user_roles: Vec::new(), - user_id, - statement_digest: None, - txn_id, - wal_lsn, - resolved_now_ms, - admission, - }; - - let mut rx = shared.tracker.register(request_id); - - match shared.dispatcher.lock() { - Ok(mut d) => rollback_ddl!(d.dispatch(request)), - Err(poisoned) => rollback_ddl!(poisoned.into_inner().dispatch(request)), - }; - - // Release the write-admission guards immediately after the enqueue, before - // the Data-Plane round-trip. The per-database WFQ is strict FIFO, so once LSN - // order equals enqueue order the apply order follows from the queue alone; - // holding the guards across the response await would only serialize same-key - // throughput needlessly. - // - // EXCEPTION — a post-apply-redo write (`post_apply.is_some()`) mints its - // durable redo AFTER apply, from the write-set on the response; the guards - // MUST stay held across the response collect + that append so two concurrent - // same-surrogate writes cannot reorder their redo appends. Both guard types - // are `Send`, so holding them across the `.await` is sound. Moved into an - // `Option` so the release is a single, unconditional `drop` below regardless - // of which path took it. (`None` guard slots when no lock manager was - // registered / for the exempt-read / Calvin / already-ordered cases.) - let deferred_guards = if post_apply.is_some() { - Some((admission_guard, order_guard)) - } else { - drop(admission_guard); - drop(order_guard); - None - }; - - // Collect response(s). For non-streaming queries, exactly one arrives. - // For streaming queries, multiple partial chunks arrive before the final. - // The mpsc channel is bounded (see `RequestTracker::register`); here we - // additionally cap the *total* accumulated payload so a runaway scan - // can't pin Control-Plane RAM — any query whose combined result exceeds - // `tuning.network.max_query_result_bytes` is cancelled with a typed - // `ExecutionLimitExceeded` error. - let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; - // The same instant the envelope carries. The Data Plane normally answers - // with `DeadlineExceeded` first; this bounds the wait when it is inside a - // stage that carries no safe point yet. - let response = match tokio::time::timeout_at( - tokio::time::Instant::from_std(deadline), - collect_bounded_response(&mut rx, max_result_bytes), - ) - .await - { - Ok(response) => response, - Err(_) => { - observe(shared); - // Dispatch completed, but the Data Plane may have applied CREATE - // or ALTER before this deadline. Never roll that catalog state - // back on an ambiguous post-enqueue outcome. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); - } - return Err(crate::Error::DeadlineExceeded { request_id }); - } - }; - - let response = match response { - Ok(r) => r, - Err(DispatchCollectError::OverBudget { bytes }) => { - shared.tracker.cancel(&request_id); - observe(shared); - // A partial response proves dispatch began but not whether an - // Array DDL completed; preserve CREATE/ALTER and fail-stop. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); - } - return Err(crate::Error::ExecutionLimitExceeded { - detail: format!( - "query result exceeded max_query_result_bytes \ - ({bytes} > {max_result_bytes} bytes)" - ), - }); - } - Err(DispatchCollectError::ChannelClosed) => { - observe(shared); - // The producer can close after applying but before sending its - // response. CREATE/ALTER must remain catalog-finalized here. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); - } - // A producer that stopped after the deadline stopped because the - // statement ran out of time. Reporting the closure would hand the - // client an internal error for its own timeout. - if std::time::Instant::now() >= deadline { - return Err(crate::Error::DeadlineExceeded { request_id }); - } - return Err(crate::Error::Dispatch { - detail: "response channel closed".into(), - }); - } - }; - - if response.status != Status::Ok { - let _ = ddl_transition.rollback(shared); - super::super::write_abort::abort_refused_write( - shared, - super::super::write_abort::AbortTarget { - tenant_id, - database_id, - vshard_id, - wal_lsn, - appends_here, - }, - &response, - ) - .await?; - } - - // Mint the post-apply redo record while the guards are still held, then - // release them. A PointUpdate whose collection carries a secondary vector - // index returns its surrogate + post-image in `write_set`; without this - // durable `Put` a WAL-only restart rebuilds the HNSW from the pre-update body - // and resurrects the old embedding. - let post_apply_lsn = if let Some(collection) = &post_apply - && appends_here - && response.status == Status::Ok - { - rollback_ddl!(wal_dispatch::append_write_set_redo( - &shared.wal, - tenant_id, - vshard_id, - database_id, - collection, - &response.write_set, - )) - } else { - None - }; - drop(deferred_guards); - - // Durable-at-ack barrier: an acknowledged write must be WAL-fsync-durable - // before this response (the client ack) returns. `WalManager::append_*` only - // buffers the record and mints its `Lsn`; without this barrier a `kill -9` - // loses the buffered bytes, which is invisible for engines whose rows are - // committed durably by redb but silently destroys every engine whose only - // durability path is WAL replay: the KV hash tables, the HNSW graphs, the - // columnar / timeseries memtables, the graph node labels, the CRDT states, - // and the FTS index. `wal_lsn` is the forward write's LSN — minted above - // under the admission guard for a write that owns its durability, or supplied - // by a caller that appended upstream (procedural batch flush, - // interactive-COMMIT transaction redo). `post_apply_lsn` covers the - // post-apply redo appended just above. Both records are already buffered in - // the shared WAL; one group-commit fsync coalesces concurrent writers (see - // `WalManager::wait_durable`), and it runs here — after the admission guards - // are released — so it never serializes same-key throughput. Reads / control - // ops / trigger / staged-write dispatch carry no LSN and skip the barrier; - // `durability_barrier` decides which of those skips are legitimate and makes - // the rest loud instead of letting them ack a write nothing can recover. - if response.status == Status::Ok { - let durable_target = match (wal_lsn, post_apply_lsn) { - (Some(a), Some(b)) => Some(a.max(b)), - (a, b) => a.or(b), - }; - match durable_target { - Some(lsn) => rollback_ddl!(shared.wal.wait_durable(lsn).await), - // Nothing to fsync. Legitimate for most plans, but if this funnel - // was the one required to mint the record, the ack below promises - // durability the engine cannot deliver — silent until a `kill -9` - // proves it, hence the check. - None => super::super::durability_barrier::assert_durable_before_ack(funnel_redo_engine), - } - } - - if response.status == Status::Ok { - ddl_transition.finalize(shared)?; - } - - // Publish change events for successful writes whose change feed this funnel - // owns. `None` is a caller whose change feed is `Unowned` — see - // [`ChangeFeedOwner`] for why the node that applies those writes is not the - // node that publishes them. - if response.status == Status::Ok { - if let Some(change_set) = change_set { - publish_change_set(shared, tenant_id, database_id, change_set, &response); - } - - // Advance the tenant's observed write-HLC high-water on any successful - // dispatch. Used by the RESTORE staleness gate. Advancing on every - // success (not just writes) is intentionally conservative — - // envelope.watermark is captured AFTER fan-out so it always dominates - // the tenant_wm of a fresh backup. - shared.advance_tenant_write_hlc(tenant_id.as_u64()); - } - - observe(shared); - Ok(SubmitOutcome { response, wal_lsn }) -} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/admission.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/admission.rs new file mode 100644 index 000000000..7315b095e --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/admission.rs @@ -0,0 +1,94 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Write-admission gate dispatch for the funnel. + +use tokio::sync::OwnedMutexGuard; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::shared::write_admission::{ + WriteAdmission, WriteAdmissionGuard, WriteTarget, admit, +}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, VShardId}; + +use super::super::params::WriteOrdering; + +/// Outcome of the write-admission phase. +pub(super) enum AdmissionOutcome { + /// Proceed with the funnel's remaining phases under these guards. + Proceed { + admission: crate::bridge::envelope::Admission, + admission_guard: Option, + order_guard: Option>, + }, + /// The deterministic Calvin scheduler must apply the write. + RouteToCalvin, +} + +/// Write-admission gate: every write-class plan whose ordering is not already +/// final passes here. An uncontended point write takes the fast path holding +/// its per-vShard deterministic locks; a contended or bulk write is submitted +/// through the deterministic scheduler and its applied response is surfaced +/// here; reads / control ops are `Exempt`. +/// +/// Ordering (fast path): the guard is acquired FIRST, then — for a write that +/// owns its durability (`AppendHere`) — the WAL append happens after this +/// call returns, under the guard, minting the LSN just before the enqueue. +/// The guard is released immediately after the enqueue (not across the +/// response await). +pub(super) async fn admit_write( + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: &PhysicalPlan, + ordering: WriteOrdering, +) -> AdmissionOutcome { + match ordering { + WriteOrdering::AlreadyOrdered => AdmissionOutcome::Proceed { + admission: crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::AlreadyOrdered, + ), + admission_guard: None, + order_guard: None, + }, + WriteOrdering::Gate => match admit( + shared, + &WriteTarget { + tenant_id, + database_id, + vshard_id, + plan, + }, + ) { + WriteAdmission::ExemptRead => AdmissionOutcome::Proceed { + admission: crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::Read, + ), + admission_guard: None, + order_guard: None, + }, + WriteAdmission::FastPath { guard } => AdmissionOutcome::Proceed { + admission: crate::bridge::envelope::Admission::Admitted, + admission_guard: guard, + order_guard: None, + }, + WriteAdmission::FastPathBlocking { key, keyed_lock } => { + // Single-node serialization point: acquire the per-key FIFO + // order-lock FIRST, before the WAL append and enqueue below. + // `tokio::sync::Mutex` is fair, so concurrent same-key writers + // are admitted in arrival order — the WAL append + enqueue then + // happen in that order, giving WAL-LSN order == enqueue order == + // apply order per key. Distinct keys use distinct per-key mutexes + // and never contend. + let order_guard = keyed_lock.lock_owned(key).await; + AdmissionOutcome::Proceed { + admission: crate::bridge::envelope::Admission::Admitted, + admission_guard: None, + order_guard: Some(order_guard), + } + } + WriteAdmission::RouteToCalvin => AdmissionOutcome::RouteToCalvin, + }, + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs new file mode 100644 index 000000000..e17e7a632 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs @@ -0,0 +1,130 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Build the wire `Request` and hand it to the Data-Plane dispatcher. + +use std::time::Instant; + +use tokio::sync::{OwnedMutexGuard, mpsc}; + +use crate::bridge::envelope::{Admission, PhysicalPlan, Priority, Request, Response}; +use crate::control::array_catalog::ddl::AuthorizedDdlTransition; +use crate::control::server::shared::write_admission::WriteAdmissionGuard; +use crate::control::state::SharedState; +use crate::types::{ + DatabaseId, Lsn, ReadConsistency, RequestId, TenantId, TraceId, TxnId, VShardId, +}; + +use super::wal_append::rollback_on_err; + +/// The write-admission guards a dispatched write must hold across the +/// response await — deferred (rather than dropped at enqueue) only when the +/// write's redo is minted post-apply. +pub(super) type DeferredGuards = Option<(Option, Option>)>; + +/// Everything [`dispatch_to_data_plane`] needs to build the wire `Request`. +pub(super) struct DispatchTarget { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard_id: VShardId, + pub plan: PhysicalPlan, + pub deadline: Instant, + pub trace_id: TraceId, + pub event_source: crate::event::EventSource, + pub user_id: Option>, + pub txn_id: Option, + pub wal_lsn: Option, + pub resolved_now_ms: Option, + pub admission: Admission, +} + +/// What the dispatch phase produced: the id the response is tracked under, +/// the receiver it arrives on, the per-vShard latency timer's start instant, +/// and the write-admission guards deferred across the response await (if +/// any). +pub(super) struct DispatchOutcome { + pub request_id: RequestId, + pub rx: mpsc::Receiver, + pub dispatch_started: Instant, + pub deferred_guards: DeferredGuards, +} + +/// Build the `Request` envelope, register its response channel, and hand it +/// to the Data-Plane dispatcher — then release the write-admission guards +/// (or defer them, for a write whose redo mints post-apply). +/// +/// The per-vShard QPS + latency timer starts here, at the wall-clock moment +/// the request enters dispatch; the caller observes it on every exit path +/// (success, budget over-run, timeout) so the histogram captures the true +/// end-to-end shape of the work routed to this vshard. +pub(super) fn dispatch_to_data_plane( + shared: &SharedState, + ddl_transition: &AuthorizedDdlTransition, + target: DispatchTarget, + admission_guard: Option, + order_guard: Option>, + post_apply_pending: bool, +) -> crate::Result { + let dispatch_started = Instant::now(); + + let request_id = shared.next_request_id(); + let request = Request { + request_id, + tenant_id: target.tenant_id, + database_id: target.database_id, + vshard_id: target.vshard_id, + plan: target.plan, + deadline: target.deadline, + priority: Priority::Normal, + trace_id: target.trace_id, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: target.event_source, + user_roles: Vec::new(), + user_id: target.user_id, + statement_digest: None, + txn_id: target.txn_id, + wal_lsn: target.wal_lsn, + resolved_now_ms: target.resolved_now_ms, + admission: target.admission, + }; + + let rx = shared.tracker.register(request_id); + + match shared.dispatcher.lock() { + Ok(mut d) => rollback_on_err(shared, ddl_transition, d.dispatch(request))?, + Err(poisoned) => rollback_on_err( + shared, + ddl_transition, + poisoned.into_inner().dispatch(request), + )?, + }; + + // Release the write-admission guards immediately after the enqueue, before + // the Data-Plane round-trip. The per-database WFQ is strict FIFO, so once LSN + // order equals enqueue order the apply order follows from the queue alone; + // holding the guards across the response await would only serialize same-key + // throughput needlessly. + // + // EXCEPTION — a post-apply-redo write mints its durable redo AFTER apply, + // from the write-set on the response; the guards MUST stay held across the + // response collect + that append so two concurrent same-surrogate writes + // cannot reorder their redo appends. Both guard types are `Send`, so + // holding them across the `.await` is sound. Moved into an `Option` so the + // release is a single, unconditional `drop` below regardless of which path + // took it. (`None` guard slots when no lock manager was registered / for + // the exempt-read / Calvin / already-ordered cases.) + let deferred_guards = if post_apply_pending { + Some((admission_guard, order_guard)) + } else { + drop(admission_guard); + drop(order_guard); + None + }; + + Ok(DispatchOutcome { + request_id, + rx, + dispatch_started, + deferred_guards, + }) +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs new file mode 100644 index 000000000..f41c02435 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -0,0 +1,159 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The funnel's orchestrator: runs admission, WAL append, dispatch, and +//! response classification in that fixed order for one write. + +use crate::control::server::dispatch_utils::change_events::extract_write_change_set; +use crate::control::server::dispatch_utils::durability_barrier::funnel_minted_redo_engine; +use crate::control::server::shared::session::statement_deadline; +use crate::control::server::shared::write_admission::{bare_ok_response, route_write_to_calvin}; +use crate::control::server::wal_dispatch; +use crate::control::state::SharedState; + +use super::super::params::{ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability}; +use super::admission::{AdmissionOutcome, admit_write}; +use super::dispatch::{DispatchTarget, dispatch_to_data_plane}; +use super::response::{ResponsePhaseInput, collect_classify_and_finish}; +use super::wal_append::authorize_and_append; + +/// Admit, make durable, enqueue, collect, and publish one write. +/// +/// See [`SubmitOutcome`] for what comes back. +pub(crate) async fn submit_write( + shared: &SharedState, + params: SubmitWrite, +) -> crate::Result { + let SubmitWrite { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + event_source, + txn_id, + user_id, + durability, + ordering, + change_feed, + } = params; + + // The running statement's deadline, pinned once at the session boundary and + // shared by every request the statement fans out into. Used for both the + // envelope the Data Plane enforces and the Control-Plane collect below, so + // the two halves cannot disagree about when this statement expires. + let deadline = statement_deadline(shared.tuning.network.default_deadline_secs); + + // Change metadata is derived from the plan HERE, before it is moved into + // the request — the publish itself happens after apply, once the response + // (which carries the event's LSN) exists, by which point the plan is gone. + // Extraction is a pure match that clones out collection / document + // identity, so a caller whose change feed is `Unowned` skips it rather than + // allocating tuples nothing will read. + let change_set = match change_feed { + ChangeFeedOwner::Funnel => Some(extract_write_change_set(&plan, tenant_id)), + ChangeFeedOwner::Unowned => None, + }; + + // Post-apply redo classification, computed before `plan` is moved (the + // RouteToCalvin admit arm moves it). For a write whose autocommit WAL path + // mints no redo of its own but whose effect must survive a WAL-only restart + // (a document PointUpdate on a collection carrying a secondary vector + // index), the durable redo is minted AFTER apply from the surrogate + + // post-image the Data Plane returns in `Response::write_set`. + // `Some(collection)` for such a write, else `None`. + let post_apply = wal_dispatch::plan_post_apply_redo(&plan); + let appends_here = matches!(&durability, WalDurability::AppendHere { .. }); + + // Durable-at-ack obligation, also computed before `plan` moves. `Some` only + // for a write whose redo record THIS funnel is required to mint; a caller + // that appended upstream (or declared durability owned elsewhere) is not + // held to it, because the LSN it does or does not supply is its own + // contract. See `durability_barrier` for why this is narrower than + // "write-class plan with no LSN". + let funnel_redo_engine = if appends_here { + funnel_minted_redo_engine(&plan) + } else { + None + }; + + // Write-admission gate: every write-class plan whose ordering is not already + // final passes here. On the Calvin route the deterministic scheduler applies + // the write, emits its own WriteEvents, and owns durability (the sequenced + // TxClass plus its own `CalvinApplied` WAL record), so no local WAL append or + // enqueue happens. A plain write with no RETURNING rows yields `None`, + // synthesized into a bare `Ok`. + let (admission, admission_guard, order_guard) = + match admit_write(shared, tenant_id, database_id, vshard_id, &plan, ordering).await { + AdmissionOutcome::Proceed { + admission, + admission_guard, + order_guard, + } => (admission, admission_guard, order_guard), + AdmissionOutcome::RouteToCalvin => { + let routed = + route_write_to_calvin(shared, tenant_id, database_id, vshard_id, plan).await?; + return Ok(SubmitOutcome { + response: routed + .unwrap_or_else(|| bare_ok_response(crate::types::RequestId::new(0))), + wal_lsn: None, + }); + } + }; + + // Array DDL authorization + durability, under the admission guard, + // immediately before the enqueue below. + let wal_append_outcome = + authorize_and_append(shared, tenant_id, database_id, vshard_id, plan, durability)?; + let ddl_transition = wal_append_outcome.ddl_transition; + let plan = wal_append_outcome.plan; + let wal_lsn = wal_append_outcome.wal_lsn; + let resolved_now_ms = wal_append_outcome.resolved_now_ms; + + // Build the wire request and hand it to the Data-Plane dispatcher. + let dispatch_outcome = dispatch_to_data_plane( + shared, + &ddl_transition, + DispatchTarget { + tenant_id, + database_id, + vshard_id, + plan, + deadline, + trace_id, + event_source, + user_id, + txn_id, + wal_lsn, + resolved_now_ms, + admission, + }, + admission_guard, + order_guard, + post_apply.is_some(), + )?; + + // Collect response(s), classify the outcome, and run the post-apply steps + // a successful write still owes. + let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; + collect_classify_and_finish( + shared, + max_result_bytes, + ResponsePhaseInput { + request_id: dispatch_outcome.request_id, + rx: dispatch_outcome.rx, + deadline, + dispatch_started: dispatch_outcome.dispatch_started, + tenant_id, + database_id, + vshard_id, + wal_lsn, + appends_here, + post_apply, + funnel_redo_engine, + change_set, + ddl_transition, + deferred_guards: dispatch_outcome.deferred_guards, + }, + ) + .await +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs new file mode 100644 index 000000000..843372a53 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs @@ -0,0 +1,41 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! THE single Control-Plane write funnel. +//! +//! Every Control-Plane path that puts a write on the SPSC bridge routes through +//! [`submit_write`]: the autocommit / internal funnel +//! ([`crate::control::server::dispatch_utils::dispatch`]), the pgwire +//! local-dispatch path (`pgwire::handler::submit`), and the Raft apply loop +//! (`distributed_applier::apply_loop`). The funnel owns write admission, the +//! WAL redo append, the enqueue, the bounded response collect, the post-apply +//! redo, the durable-at-ack barrier, and — for the caller that owns it (see +//! [`super::params::ChangeFeedOwner`]) — the CDC publish, in that order, which +//! is the correctness contract. +//! +//! A path that reimplements these steps drifts silently: it is not a compile +//! error to omit the redo append or the change-event publish, and the omission +//! only surfaces as lost data after a crash, or as a change stream that never +//! fires. Add the step here, once, and every caller gets it. +//! +//! It also owns the mirror of the redo append: when it appended the record +//! itself and the Data Plane then REFUSED the write, it cancels that record +//! before returning the error. See +//! [`crate::control::server::dispatch_utils::write_abort`] for which verdicts +//! qualify, the residual crash window it does not close, and the latency it +//! costs a rejection. +//! +//! Split by concern, run in this fixed order by [`driver::submit_write`]: +//! - [`admission`]: the write-admission gate. +//! - [`wal_append`]: Array DDL authorization and the WAL redo append/stamp. +//! - [`dispatch`]: building the wire `Request` and handing it to the Data +//! Plane. +//! - [`response`]: collecting the response, classifying the outcome, and the +//! post-apply steps a successful write still owes. + +mod admission; +mod dispatch; +mod driver; +mod response; +mod wal_append; + +pub(crate) use driver::submit_write; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs new file mode 100644 index 000000000..12268b818 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -0,0 +1,245 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Collect the Data Plane's response, classify the outcome, and run the +//! post-apply steps a successful write still owes: the post-apply redo, the +//! durable-at-ack barrier, DDL finalization, and the change-event publish. + +use std::time::Instant; + +use tokio::sync::mpsc; + +use crate::bridge::envelope::{Response, Status}; +use crate::control::array_catalog::ddl::AuthorizedDdlTransition; +use crate::control::server::dispatch_utils::change_events::{WriteChangeSet, publish_change_set}; +use crate::control::server::dispatch_utils::collect::{ + DispatchCollectError, collect_bounded_response, +}; +use crate::control::server::dispatch_utils::durability_barrier::assert_durable_before_ack; +use crate::control::server::dispatch_utils::submit_write::ambiguous_ddl::preserve_ambiguous_array_ddl; +use crate::control::server::dispatch_utils::write_abort::{AbortTarget, abort_refused_write}; +use crate::control::server::wal_dispatch; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, RequestId, TenantId, VShardId}; + +use super::super::params::SubmitOutcome; +use super::wal_append::rollback_on_err; + +/// Everything the response phase needs, gathered from the admission, WAL +/// append, and dispatch phases that ran before it. +pub(super) struct ResponsePhaseInput { + pub request_id: RequestId, + pub rx: mpsc::Receiver, + pub deadline: Instant, + pub dispatch_started: Instant, + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard_id: VShardId, + pub wal_lsn: Option, + pub appends_here: bool, + pub post_apply: Option, + pub funnel_redo_engine: Option<&'static str>, + pub change_set: Option, + pub ddl_transition: AuthorizedDdlTransition, + pub deferred_guards: super::dispatch::DeferredGuards, +} + +/// Collect the response(s), classify the outcome, and run every step a +/// completed write still owes before the funnel returns. +/// +/// For non-streaming queries, exactly one response arrives. For streaming +/// queries, multiple partial chunks arrive before the final. The mpsc channel +/// is bounded (see `RequestTracker::register`); here the *total* accumulated +/// payload is additionally capped so a runaway scan can't pin Control-Plane +/// RAM — any query whose combined result exceeds +/// `tuning.network.max_query_result_bytes` is cancelled with a typed +/// `ExecutionLimitExceeded` error. +pub(super) async fn collect_classify_and_finish( + shared: &SharedState, + max_result_bytes: usize, + input: ResponsePhaseInput, +) -> crate::Result { + let ResponsePhaseInput { + request_id, + mut rx, + deadline, + dispatch_started, + tenant_id, + database_id, + vshard_id, + wal_lsn, + appends_here, + post_apply, + funnel_redo_engine, + change_set, + ddl_transition, + deferred_guards, + } = input; + + let vshard_u32 = vshard_id.as_u32(); + let observe = |shared: &SharedState| { + let latency_us = dispatch_started.elapsed().as_micros().min(u64::MAX as u128) as u64; + shared.per_vshard_metrics.observe(vshard_u32, latency_us); + }; + + // The same instant the envelope carries. The Data Plane normally answers + // with `DeadlineExceeded` first; this bounds the wait when it is inside a + // stage that carries no safe point yet. + let response = match tokio::time::timeout_at( + tokio::time::Instant::from_std(deadline), + collect_bounded_response(&mut rx, max_result_bytes), + ) + .await + { + Ok(response) => response, + Err(_) => { + observe(shared); + // Dispatch completed, but the Data Plane may have applied CREATE + // or ALTER before this deadline. Never roll that catalog state + // back on an ambiguous post-enqueue outcome. + preserve_ambiguous_array_ddl(shared, &ddl_transition); + if !ddl_transition.preserves_on_ambiguous_apply() { + let _ = ddl_transition.rollback(shared); + } + return Err(crate::Error::DeadlineExceeded { request_id }); + } + }; + + let response = match response { + Ok(r) => r, + Err(DispatchCollectError::OverBudget { bytes }) => { + shared.tracker.cancel(&request_id); + observe(shared); + // A partial response proves dispatch began but not whether an + // Array DDL completed; preserve CREATE/ALTER and fail-stop. + preserve_ambiguous_array_ddl(shared, &ddl_transition); + if !ddl_transition.preserves_on_ambiguous_apply() { + let _ = ddl_transition.rollback(shared); + } + return Err(crate::Error::ExecutionLimitExceeded { + detail: format!( + "query result exceeded max_query_result_bytes \ + ({bytes} > {max_result_bytes} bytes)" + ), + }); + } + Err(DispatchCollectError::ChannelClosed) => { + observe(shared); + // The producer can close after applying but before sending its + // response. CREATE/ALTER must remain catalog-finalized here. + preserve_ambiguous_array_ddl(shared, &ddl_transition); + if !ddl_transition.preserves_on_ambiguous_apply() { + let _ = ddl_transition.rollback(shared); + } + // A producer that stopped after the deadline stopped because the + // statement ran out of time. Reporting the closure would hand the + // client an internal error for its own timeout. + if std::time::Instant::now() >= deadline { + return Err(crate::Error::DeadlineExceeded { request_id }); + } + return Err(crate::Error::Dispatch { + detail: "response channel closed".into(), + }); + } + }; + + if response.status != Status::Ok { + let _ = ddl_transition.rollback(shared); + abort_refused_write( + shared, + AbortTarget { + tenant_id, + database_id, + vshard_id, + wal_lsn, + appends_here, + }, + &response, + ) + .await?; + } + + // Mint the post-apply redo record while the guards are still held, then + // release them. A PointUpdate whose collection carries a secondary vector + // index returns its surrogate + post-image in `write_set`; without this + // durable `Put` a WAL-only restart rebuilds the HNSW from the pre-update body + // and resurrects the old embedding. + let post_apply_lsn = if let Some(collection) = &post_apply + && appends_here + && response.status == Status::Ok + { + rollback_on_err( + shared, + &ddl_transition, + wal_dispatch::append_write_set_redo( + &shared.wal, + tenant_id, + vshard_id, + database_id, + collection, + &response.write_set, + ), + )? + } else { + None + }; + drop(deferred_guards); + + // Durable-at-ack barrier: an acknowledged write must be WAL-fsync-durable + // before this response (the client ack) returns. `WalManager::append_*` only + // buffers the record and mints its `Lsn`; without this barrier a `kill -9` + // loses the buffered bytes, which is invisible for engines whose rows are + // committed durably by redb but silently destroys every engine whose only + // durability path is WAL replay: the KV hash tables, the HNSW graphs, the + // columnar / timeseries memtables, the graph node labels, the CRDT states, + // and the FTS index. `wal_lsn` is the forward write's LSN — minted above + // under the admission guard for a write that owns its durability, or supplied + // by a caller that appended upstream (procedural batch flush, + // interactive-COMMIT transaction redo). `post_apply_lsn` covers the + // post-apply redo appended just above. Both records are already buffered in + // the shared WAL; one group-commit fsync coalesces concurrent writers (see + // `WalManager::wait_durable`), and it runs here — after the admission guards + // are released — so it never serializes same-key throughput. Reads / control + // ops / trigger / staged-write dispatch carry no LSN and skip the barrier; + // `durability_barrier` decides which of those skips are legitimate and makes + // the rest loud instead of letting them ack a write nothing can recover. + if response.status == Status::Ok { + let durable_target = match (wal_lsn, post_apply_lsn) { + (Some(a), Some(b)) => Some(a.max(b)), + (a, b) => a.or(b), + }; + match durable_target { + Some(lsn) => { + rollback_on_err(shared, &ddl_transition, shared.wal.wait_durable(lsn).await)? + } + // Nothing to fsync. Legitimate for most plans, but if this funnel + // was the one required to mint the record, the ack below promises + // durability the engine cannot deliver — silent until a `kill -9` + // proves it, hence the check. + None => assert_durable_before_ack(funnel_redo_engine), + } + } + + if response.status == Status::Ok { + ddl_transition.finalize(shared)?; + } + + // Publish change events for successful writes whose change feed this funnel + // owns. `None` is a caller whose change feed is `Unowned` — see + // [`super::super::params::ChangeFeedOwner`] for why the node that applies those + // writes is not the node that publishes them. + if response.status == Status::Ok { + if let Some(change_set) = change_set { + publish_change_set(shared, tenant_id, database_id, change_set, &response); + } + + // Advance the tenant's observed write-HLC high-water on any successful + // dispatch. Used by the RESTORE staleness gate. Advancing on every + // success (not just writes) is intentionally conservative — + // envelope.watermark is captured AFTER fan-out so it always dominates + // the tenant_wm of a fresh backup. + shared.advance_tenant_write_hlc(tenant_id.as_u64()); + } + + observe(shared); + Ok(SubmitOutcome { response, wal_lsn }) +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs new file mode 100644 index 000000000..40a93089f --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Array DDL authorization and WAL redo append/stamp for the funnel. +//! +//! Array DDL conversion is intentionally read-only. Once a task has passed +//! authorization and admission, its durable catalog state installs +//! immediately before the Data-Plane dispatch; the mirror changes only after +//! the redb transaction commits. The DDL transition this creates must be +//! rolled back by every later phase that can fail before the Data Plane has +//! proved it applied the write — see [`rollback_on_err`]. + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::array_catalog::ddl::AuthorizedDdlTransition; +use crate::control::server::wal_dispatch::{self, WalAppendRequest}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + +use super::super::params::WalDurability; + +/// What the authorize-and-append phase produced: the DDL transition every +/// later phase must roll back on failure, the plan (stamped with its minted +/// LSN, if any), and the resolved durability values. +pub(super) struct WalAppendOutcome { + pub ddl_transition: AuthorizedDdlTransition, + pub plan: PhysicalPlan, + pub wal_lsn: Option, + pub resolved_now_ms: Option, +} + +/// Roll `ddl_transition` back and convert `result`'s error, or pass a success +/// through unchanged. Every phase after DDL authorization that can fail calls +/// this instead of returning its error directly, so an authorized Array +/// CREATE/DROP/ALTER never survives a failure later in the sequence. +pub(super) fn rollback_on_err( + shared: &SharedState, + ddl_transition: &AuthorizedDdlTransition, + result: crate::Result, +) -> crate::Result { + match result { + Ok(value) => Ok(value), + Err(error) => { + let _ = ddl_transition.rollback(shared); + Err(error) + } + } +} + +/// Install the plan's Array DDL catalog transition, then make the write +/// durable: append its WAL redo record here (under the write-admission guard +/// the caller already holds) or take the LSN a caller supplied upstream. +/// +/// Durability, under the guard, immediately before the enqueue: the LSN is +/// minted in the same order the request is about to be enqueued. +/// +/// Writes the resolved LSN back into the plan itself. The envelope's +/// `wal_lsn` is where most engines read their committed version from, but the +/// array engine stamps its tile versions from the LSN carried in the plan +/// while replay stamps them from the record header — so the plan the Data +/// Plane is about to execute must name the record that reproduces it. This is +/// the only place that knows both, and it knows them for every caller: no +/// upstream path may allocate an LSN of its own and hope it matches. +pub(super) fn authorize_and_append( + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + mut plan: PhysicalPlan, + durability: WalDurability, +) -> crate::Result { + let ddl_transition = crate::control::array_catalog::ddl::apply_authorized_ddl( + shared, + tenant_id, + database_id, + &plan, + )?; + + let (wal_lsn, resolved_now_ms) = match durability { + WalDurability::AppendHere { now_override } => { + let outcome = rollback_on_err( + shared, + &ddl_transition, + wal_dispatch::wal_append(WalAppendRequest { + wal: &shared.wal, + tenant_id, + vshard_id, + database_id, + plan: &plan, + credentials: None, + now_override, + }), + )?; + (outcome.lsn, outcome.resolved_now_ms) + } + WalDurability::CallerSupplied { + wal_lsn, + resolved_now_ms, + } => (wal_lsn, resolved_now_ms), + }; + + if let Some(lsn) = wal_lsn { + wal_dispatch::stamp_minted_lsn(&mut plan, lsn); + } + + Ok(WalAppendOutcome { + ddl_transition, + plan, + wal_lsn, + resolved_now_ms, + }) +} From 08486cd735f0ce71c6a536a1317eb04084733fc8 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 18:52:20 +0800 Subject: [PATCH 12/64] refactor(pgwire): split handler dispatch into a directory module dispatch.rs held the dispatch entry points, per-task routing decision, Raft-replicated and local Data Plane submission paths, and the authorization helper in one file. It becomes a dispatch/ directory with entry.rs, routing.rs, replicated.rs, local.rs, and authorize.rs, each keeping its existing logic. --- .../control/server/pgwire/handler/dispatch.rs | 485 ------------------ .../pgwire/handler/dispatch/authorize.rs | 87 ++++ .../server/pgwire/handler/dispatch/entry.rs | 84 +++ .../server/pgwire/handler/dispatch/local.rs | 75 +++ .../server/pgwire/handler/dispatch/mod.rs | 22 + .../pgwire/handler/dispatch/replicated.rs | 65 +++ .../server/pgwire/handler/dispatch/routing.rs | 233 +++++++++ 7 files changed, 566 insertions(+), 485 deletions(-) delete mode 100644 nodedb/src/control/server/pgwire/handler/dispatch.rs create mode 100644 nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs create mode 100644 nodedb/src/control/server/pgwire/handler/dispatch/entry.rs create mode 100644 nodedb/src/control/server/pgwire/handler/dispatch/local.rs create mode 100644 nodedb/src/control/server/pgwire/handler/dispatch/mod.rs create mode 100644 nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs create mode 100644 nodedb/src/control/server/pgwire/handler/dispatch/routing.rs diff --git a/nodedb/src/control/server/pgwire/handler/dispatch.rs b/nodedb/src/control/server/pgwire/handler/dispatch.rs deleted file mode 100644 index 22941a6b7..000000000 --- a/nodedb/src/control/server/pgwire/handler/dispatch.rs +++ /dev/null @@ -1,485 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Core dispatch mechanics: single-task dispatch, Raft replication, and local Data Plane submission. - -use std::sync::Arc; - -use crate::bridge::envelope::Response; -use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::dispatch_utils::{WalDurability, publish_origin_change_events}; -use crate::control::server::exchange::resolve::{ - DistributedReadCapture, Resolved, resolve_and_materialize, -}; -use crate::types::{Lsn, ReadConsistency, TraceId, VShardId}; -use nodedb_physical::physical_plan::{CrdtOp, PhysicalPlan}; -use nodedb_physical::physical_task::PhysicalTask; - -use super::core::NodeDbPgHandler; -use super::submit::SubmitArgs; - -/// Inputs for [`NodeDbPgHandler::dispatch_replicated_write`]: the entry to -/// propose, the proposer, and the identity + plan its origin CDC publish needs. -struct ReplicatedWrite<'a> { - entry: crate::control::wal_replication::ReplicatedEntry, - proposer: &'a Arc, - authorized: crate::control::server::shared::authorization::AuthorizedTask, -} - -impl NodeDbPgHandler { - fn authorize_for_dispatch( - &self, - identity: &AuthenticatedIdentity, - task: &PhysicalTask, - ) -> crate::Result { - let emitter = - crate::control::security::audit::ArcAuditEmitter(Arc::clone(&self.state.audit)); - crate::control::server::shared::authorization::authorize_task_set( - identity, - std::slice::from_ref(task), - &self.state.permissions, - &self.state.roles, - &emitter, - ) - .map_err(crate::Error::from)? - .into_tasks() - .into_iter() - .next() - .ok_or_else(|| crate::Error::Internal { - detail: "pgwire authorization returned no capability".into(), - }) - } - - /// Dispatch a single physical task and wait for the response. - /// - /// In cluster mode, writes propose to Raft first and execute only after - /// quorum commit; reads bypass Raft. `identity` must be passed for every - /// externally derived task. - pub(super) async fn dispatch_authorized_task( - &self, - task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - ) -> crate::Result { - let mut shard_watermarks = Vec::new(); - let mut distributed_reads = Vec::new(); - self.dispatch_task_hlc( - task, - user_id, - identity, - &mut shard_watermarks, - &mut distributed_reads, - ) - .await - } - - /// Dispatch a task and return the response, per-shard watermark LSNs a fan - /// gather observed, and per-side read captures a shuffle JOIN produced. - /// Used by the transactional read-recording seam. - pub(super) async fn dispatch_authorized_task_with_watermarks( - &self, - task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - ) -> crate::Result<(Response, Vec<(VShardId, Lsn)>, Vec)> { - let mut shard_watermarks = Vec::new(); - let mut distributed_reads = Vec::new(); - let resp = self - .dispatch_task_hlc( - task, - user_id, - identity, - &mut shard_watermarks, - &mut distributed_reads, - ) - .await?; - Ok((resp, shard_watermarks, distributed_reads)) - } - - async fn dispatch_task_hlc( - &self, - task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - shard_watermarks: &mut Vec<(VShardId, Lsn)>, - distributed_reads: &mut Vec, - ) -> crate::Result { - let tenant_id = task.tenant_id; - let result = self - .dispatch_task_inner(task, user_id, identity, shard_watermarks, distributed_reads) - .await; - // Advances per-tenant write-HLC on any successful dispatch; used by RESTORE's - // staleness gate. Backup captures its watermark after fan-out, so it dominates. - if let Ok(ref resp) = result - && resp.status == crate::bridge::envelope::Status::Ok - { - self.state.advance_tenant_write_hlc(tenant_id.as_u64()); - } - result - } - - async fn dispatch_task_inner( - &self, - mut task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - shard_watermarks: &mut Vec<(VShardId, Lsn)>, - distributed_reads: &mut Vec, - ) -> crate::Result { - // Reject user writes against a database frozen by a clone materializer sweep. - // Reads/DDL pass through. - use crate::control::security::identity::{Permission, required_permission}; - let perm = required_permission(&task.plan); - if matches!(perm, Permission::Write | Permission::Admin) - && self.state.materialize_freeze.is_frozen(task.database_id) - { - return Err(crate::Error::SourceFrozen { - database_id: task.database_id, - }); - } - - // Mirror enforcement: writes reject on non-promoted mirrors; reads gate by - // ReadConsistency. Catalog lookup skipped for db id=0 to stay allocation-free. - let catalog = self.state.credentials.catalog(); - if task.database_id.as_u64() != 0 - && let Ok(Some(descriptor)) = catalog.get_database(task.database_id) - && let Some(origin) = descriptor.mirror_origin.as_ref() - && !matches!(origin.status, nodedb_types::MirrorStatus::Promoted) - { - if matches!(perm, Permission::Write | Permission::Admin) { - return Err(crate::Error::MirrorReadOnly { - database: descriptor.name.clone(), - }); - } - - use crate::control::server::pgwire::ddl::database::{ - MirrorReadOutcome, check_mirror_read_consistency, - }; - // Defaults to Strong: mirrors aren't the source leader, so reads reject - // unless the session opted into BoundedStaleness or Eventual. - let now_ms = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap_or(std::time::Duration::ZERO) - .as_millis() as u64; - let outcome = check_mirror_read_consistency( - catalog, - task.database_id, - origin, - ReadConsistency::Strong, - now_ms, - ); - if let MirrorReadOutcome::Reject { message, .. } = outcome { - return Err(crate::Error::StaleReadNotLeader { - database: descriptor.name.clone(), - source_cluster: origin.source_cluster.clone(), - detail: message, - }); - } - } - - if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. } - ) - ) { - let authorized = self.authorize_for_dispatch(identity, &task)?; - return crate::control::insert_select::run_authorized_insert_select( - &self.state, - authorized, - ) - .await; - } - - // Autocommit `MERGE` orchestrates on the Control Plane (`control::merge_orchestrator`). - // In-transaction MERGE buffers for COMMIT replay and never reaches this method. - if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::Merge { - resolved_inserts: None, - .. - } - ) - ) { - let authorized = self.authorize_for_dispatch(identity, &task)?; - return crate::control::merge_orchestrator::run_authorized_merge( - &self.state, - authorized, - ) - .await; - } - - // Scans the source on its own core and ships raw rows into the plan (source's - // vShard can differ). In-transaction buffers for COMMIT replay instead. - if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { - source_rows: None, - .. - } - ) - ) { - let authorized = self.authorize_for_dispatch(identity, &task)?; - return crate::control::update_from_join_orchestrator::run_authorized_update_from_join( - &self.state, - authorized, - ) - .await; - } - - // Can't replicate bare over Raft — a follower has no writing identity to decide - // `$auth.*` against. `write_resolve` resolves it while the identity is live. - if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) - && self.state.async_raft_proposer().is_some() - { - let authorized = self.authorize_for_dispatch(identity, &task)?; - return crate::control::write_resolve::run_authorized_write_resolve( - &self.state, - authorized, - resolver, - ) - .await; - } - - // `DROP ARRAY` reaches every core so each releases its store and segment dir — - // otherwise a follow-up `CREATE ARRAY` carries stale state. - if matches!( - task.plan, - crate::bridge::envelope::PhysicalPlan::Array( - nodedb_physical::physical_plan::ArrayOp::DropArray { .. } - ) - ) { - // Broadcast bypasses the write funnel, so a denied DROP must not - // delete catalog rows or surrogate bindings. - let authorized = self.authorize_for_dispatch(identity, &task)?; - let task = authorized.into_physical_task(); - return crate::control::array_catalog::ddl::run_authorized_drop( - &self.state, - task.tenant_id, - task.database_id, - task.plan, - TraceId::ZERO, - ) - .await; - } - - // Clone-read must run first: resolving derived Exchange plans below - // dispatches straight to the Data Plane, bypassing the clone check. - if let Some(resp) = self - .maybe_intercept_clone_read_early(&task, identity, perm) - .await? - { - return Ok(resp); - } - - // Resolve derived Exchange plans before authorizing the dispatched task. - match resolve_and_materialize( - &self.state, - identity, - task.database_id, - task.tenant_id, - task.plan, - TraceId::ZERO, - task.txn_id, - ) - .await? - { - Resolved::Gathered(resp, wms, caps) => { - *shard_watermarks = wms; - *distributed_reads = caps; - return Ok(resp); - } - Resolved::Plan(resolved_plan) => { - let resolved_plan = *resolved_plan; - task.plan = resolved_plan; - } - Resolved::Stream(stream) => { - return crate::control::server::exchange::gather::stream_to_response(stream).await; - } - } - - reject_unadmitted_crdt_apply(&task.plan)?; - let checked = self - .intercept_and_authorize_for_dispatch(identity, task) - .await?; - let checked = match checked { - crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => { - return Ok(resp); - } - crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(t) => t, - }; - if let Some(async_proposer) = self.state.async_raft_proposer() - && let Some(entry) = crate::control::wal_replication::to_replicated_entry( - checked.tenant_id(), - checked.database_id(), - checked.vshard_id(), - &crate::control::wal_replication::ReplicableWrite::decide_for_replication( - checked.plan(), - )?, - )? - { - return self - .dispatch_replicated_write(ReplicatedWrite { - entry, - proposer: async_proposer, - authorized: checked.into_authorized(), - }) - .await; - } - self.dispatch_local(checked, user_id).await - } - - /// Dispatch a write through Raft: propose → register waiter → await apply. - /// `ProposeTracker` is race-safe against an entry applying before register. - /// - /// Also the origin CDC publish site; replicas publish nothing (`ChangeFeedOwner::Unowned`). - async fn dispatch_replicated_write( - &self, - args: ReplicatedWrite<'_>, - ) -> crate::Result { - let ReplicatedWrite { - entry, - proposer, - authorized, - } = args; - let task = authorized.into_physical_task(); - let tenant_id = task.tenant_id; - let database_id = task.database_id; - let plan = task.plan; - let request_id = self.next_request_id(); - - // `write_version` is the post-write `coll_write_lsn`, surfaced so the session - // can floor a later read-set at it (read-your-writes for cross-shard OCC). - let (payload, write_version) = - crate::control::wal_replication::propose_replicated_entry(&self.state, proposer, entry) - .await?; - - let response = Response { - request_id, - status: crate::bridge::envelope::Status::Ok, - attempt: 1, - partial: false, - payload: payload.into(), - // Authoritative participant WAL LSN — CDC ordering must use it, not zero. - watermark_lsn: write_version, - error_code: None, - read_set_valid: None, - read_version_lsn: write_version, - write_set: Vec::new(), - }; - - // Propose returned: entry is committed and applied. Publish once, from this plan. - publish_origin_change_events(&self.state, tenant_id, database_id, &plan, &response); - - Ok(response) - } - - /// Dispatch a task directly to the local Data Plane (single-node or reads). - /// - /// WAL append happens inside the write funnel, under the admission guard just - /// before enqueue, so LSN order equals apply order. Reads bypass the WAL entirely. - async fn dispatch_local( - &self, - checked: crate::control::server::shared::clone_write::CloneCheckedTask, - user_id: Option>, - ) -> crate::Result { - self.submit_authorized_to_data_plane( - checked, - user_id, - WalDurability::AppendHere { now_override: None }, - ) - .await - } - - /// Dispatch a task to the Data Plane WITHOUT individual WAL append. - /// - /// Used by COMMIT after the transaction is written as one `RecordType::Transaction` - /// record — per-task WAL would double-write. - pub(super) async fn dispatch_task_no_wal( - &self, - task: PhysicalTask, - user_id: Option>, - wal_lsn: Option, - ) -> crate::Result { - // Without this, a transaction begun before the freeze could COMMIT mid-scan and - // break the as-of contract. - use crate::control::security::identity::{Permission, required_permission}; - let perm = required_permission(&task.plan); - if matches!(perm, Permission::Write | Permission::Admin) - && self.state.materialize_freeze.is_frozen(task.database_id) - { - return Err(crate::Error::SourceFrozen { - database_id: task.database_id, - }); - } - reject_unadmitted_crdt_apply(&task.plan)?; - let txn_id = task.txn_id; - // Writes were durably recorded under one `RecordType::Transaction` record at - // COMMIT; per-task WAL append is skipped. `wal_lsn` stamps that record's LSN. - self.submit_to_data_plane(SubmitArgs { - tenant_id: task.tenant_id, - vshard_id: task.vshard_id, - database_id: task.database_id, - plan: task.plan, - user_id, - txn_id, - // No per-task TTL instant (see `flush_transaction_buffer`), so a TTL-bearing - // KV write falls back to `epoch_system_ms` at apply time. - durability: WalDurability::CallerSupplied { - wal_lsn, - resolved_now_ms: None, - }, - }) - .await - } -} - -fn reject_unadmitted_crdt_apply(plan: &PhysicalPlan) -> crate::Result<()> { - if matches!( - plan, - PhysicalPlan::Crdt( - CrdtOp::Apply { .. } - | CrdtOp::ApplyAuthenticated { .. } - | CrdtOp::ImportSnapshot { .. } - ) - ) { - return Err(crate::Error::CrdtApplyRequiresAdmission); - } - Ok(()) -} - -#[cfg(test)] -mod tests { - use nodedb_types::Surrogate; - - use super::*; - - #[test] - fn generic_pgwire_dispatch_rejects_unadmitted_apply() { - let plan = PhysicalPlan::Crdt(CrdtOp::Apply { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "docs", - ), - document_id: "doc-1".into(), - delta: Vec::new(), - peer_id: 1, - mutation_id: 1, - surrogate: Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }); - assert!(matches!( - reject_unadmitted_crdt_apply(&plan), - Err(crate::Error::CrdtApplyRequiresAdmission) - )); - } - - #[test] - fn dispatch_task_compile_check() { - // Confirms the dispatch module compiles. - let _: () = (); - } -} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs b/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs new file mode 100644 index 000000000..594bc47b6 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Task authorization and the CRDT admission-gate check shared by every +//! dispatch routing path. + +use std::sync::Arc; + +use nodedb_physical::physical_plan::{CrdtOp, PhysicalPlan}; +use nodedb_physical::physical_task::PhysicalTask; + +use crate::control::security::identity::AuthenticatedIdentity; + +use super::super::core::NodeDbPgHandler; + +impl NodeDbPgHandler { + pub(super) fn authorize_for_dispatch( + &self, + identity: &AuthenticatedIdentity, + task: &PhysicalTask, + ) -> crate::Result { + let emitter = + crate::control::security::audit::ArcAuditEmitter(Arc::clone(&self.state.audit)); + crate::control::server::shared::authorization::authorize_task_set( + identity, + std::slice::from_ref(task), + &self.state.permissions, + &self.state.roles, + &emitter, + ) + .map_err(crate::Error::from)? + .into_tasks() + .into_iter() + .next() + .ok_or_else(|| crate::Error::Internal { + detail: "pgwire authorization returned no capability".into(), + }) + } +} + +pub(super) fn reject_unadmitted_crdt_apply(plan: &PhysicalPlan) -> crate::Result<()> { + if matches!( + plan, + PhysicalPlan::Crdt( + CrdtOp::Apply { .. } + | CrdtOp::ApplyAuthenticated { .. } + | CrdtOp::ImportSnapshot { .. } + ) + ) { + return Err(crate::Error::CrdtApplyRequiresAdmission); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + + #[test] + fn generic_pgwire_dispatch_rejects_unadmitted_apply() { + let plan = PhysicalPlan::Crdt(CrdtOp::Apply { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "docs", + ), + document_id: "doc-1".into(), + delta: Vec::new(), + peer_id: 1, + mutation_id: 1, + surrogate: Surrogate::ZERO, + provenance: None, + constraint_version_required: 0, + expected_frontier_digest: None, + }); + assert!(matches!( + reject_unadmitted_crdt_apply(&plan), + Err(crate::Error::CrdtApplyRequiresAdmission) + )); + } + + #[test] + fn dispatch_task_compile_check() { + // Confirms the dispatch module compiles. + let _: () = (); + } +} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs b/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs new file mode 100644 index 000000000..67c5fe2aa --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs @@ -0,0 +1,84 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Public dispatch entry points, and the write-HLC bookkeeping wrapper around +//! them. + +use std::sync::Arc; + +use crate::bridge::envelope::Response; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::exchange::resolve::DistributedReadCapture; +use crate::types::{Lsn, VShardId}; +use nodedb_physical::physical_task::PhysicalTask; + +use super::super::core::NodeDbPgHandler; + +impl NodeDbPgHandler { + /// Dispatch a single physical task and wait for the response. + /// + /// In cluster mode, writes propose to Raft first and execute only after + /// quorum commit; reads bypass Raft. `identity` must be passed for every + /// externally derived task. + pub(in crate::control::server::pgwire::handler) async fn dispatch_authorized_task( + &self, + task: PhysicalTask, + user_id: Option>, + identity: &AuthenticatedIdentity, + ) -> crate::Result { + let mut shard_watermarks = Vec::new(); + let mut distributed_reads = Vec::new(); + self.dispatch_task_hlc( + task, + user_id, + identity, + &mut shard_watermarks, + &mut distributed_reads, + ) + .await + } + + /// Dispatch a task and return the response, per-shard watermark LSNs a fan + /// gather observed, and per-side read captures a shuffle JOIN produced. + /// Used by the transactional read-recording seam. + pub(in crate::control::server::pgwire::handler) async fn dispatch_authorized_task_with_watermarks( + &self, + task: PhysicalTask, + user_id: Option>, + identity: &AuthenticatedIdentity, + ) -> crate::Result<(Response, Vec<(VShardId, Lsn)>, Vec)> { + let mut shard_watermarks = Vec::new(); + let mut distributed_reads = Vec::new(); + let resp = self + .dispatch_task_hlc( + task, + user_id, + identity, + &mut shard_watermarks, + &mut distributed_reads, + ) + .await?; + Ok((resp, shard_watermarks, distributed_reads)) + } + + async fn dispatch_task_hlc( + &self, + task: PhysicalTask, + user_id: Option>, + identity: &AuthenticatedIdentity, + shard_watermarks: &mut Vec<(VShardId, Lsn)>, + distributed_reads: &mut Vec, + ) -> crate::Result { + let tenant_id = task.tenant_id; + let result = self + .dispatch_task_inner(task, user_id, identity, shard_watermarks, distributed_reads) + .await; + // Advances per-tenant write-HLC on any successful dispatch; used by RESTORE's + // staleness gate. Backup captures its watermark after fan-out, so it dominates. + if let Ok(ref resp) = result + && resp.status == crate::bridge::envelope::Status::Ok + { + self.state.advance_tenant_write_hlc(tenant_id.as_u64()); + } + result + } +} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs new file mode 100644 index 000000000..d8be234d7 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Dispatch a task directly to the local Data Plane, with or without a +//! per-task WAL append. + +use std::sync::Arc; + +use crate::bridge::envelope::Response; +use crate::control::server::dispatch_utils::WalDurability; +use nodedb_physical::physical_task::PhysicalTask; + +use super::super::core::NodeDbPgHandler; +use super::super::submit::SubmitArgs; +use super::authorize::reject_unadmitted_crdt_apply; + +impl NodeDbPgHandler { + /// Dispatch a task directly to the local Data Plane (single-node or reads). + /// + /// WAL append happens inside the write funnel, under the admission guard just + /// before enqueue, so LSN order equals apply order. Reads bypass the WAL entirely. + pub(super) async fn dispatch_local( + &self, + checked: crate::control::server::shared::clone_write::CloneCheckedTask, + user_id: Option>, + ) -> crate::Result { + self.submit_authorized_to_data_plane( + checked, + user_id, + WalDurability::AppendHere { now_override: None }, + ) + .await + } + + /// Dispatch a task to the Data Plane WITHOUT individual WAL append. + /// + /// Used by COMMIT after the transaction is written as one `RecordType::Transaction` + /// record — per-task WAL would double-write. + pub(in crate::control::server::pgwire::handler) async fn dispatch_task_no_wal( + &self, + task: PhysicalTask, + user_id: Option>, + wal_lsn: Option, + ) -> crate::Result { + // Without this, a transaction begun before the freeze could COMMIT mid-scan and + // break the as-of contract. + use crate::control::security::identity::{Permission, required_permission}; + let perm = required_permission(&task.plan); + if matches!(perm, Permission::Write | Permission::Admin) + && self.state.materialize_freeze.is_frozen(task.database_id) + { + return Err(crate::Error::SourceFrozen { + database_id: task.database_id, + }); + } + reject_unadmitted_crdt_apply(&task.plan)?; + let txn_id = task.txn_id; + // Writes were durably recorded under one `RecordType::Transaction` record at + // COMMIT; per-task WAL append is skipped. `wal_lsn` stamps that record's LSN. + self.submit_to_data_plane(SubmitArgs { + tenant_id: task.tenant_id, + vshard_id: task.vshard_id, + database_id: task.database_id, + plan: task.plan, + user_id, + txn_id, + // No per-task TTL instant (see `flush_transaction_buffer`), so a TTL-bearing + // KV write falls back to `epoch_system_ms` at apply time. + durability: WalDurability::CallerSupplied { + wal_lsn, + resolved_now_ms: None, + }, + }) + .await + } +} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs b/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs new file mode 100644 index 000000000..dd3957b21 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs @@ -0,0 +1,22 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Core dispatch mechanics: single-task dispatch, Raft replication, and local Data Plane submission. +//! +//! Split by concern: +//! - [`entry`]: the public dispatch entry points and the write-HLC +//! bookkeeping wrapper around them. +//! - [`routing`]: the per-task routing decision — freeze/mirror checks, +//! orchestrated DML, exchange resolution, and the replicated-vs-local +//! choice. +//! - [`replicated`]: proposing a write to Raft and shaping the response once +//! it applies. +//! - [`local`]: dispatching a task directly to the local Data Plane, with or +//! without a per-task WAL append. +//! - [`authorize`]: the task-authorization helper and the CRDT +//! admission-gate check shared by every routing path. + +mod authorize; +mod entry; +mod local; +mod replicated; +mod routing; diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs b/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs new file mode 100644 index 000000000..8b98e7479 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Propose a write to Raft and shape the response once it applies. + +use std::sync::Arc; + +use crate::bridge::envelope::Response; +use crate::control::server::dispatch_utils::publish_origin_change_events; + +use super::super::core::NodeDbPgHandler; + +/// Inputs for [`NodeDbPgHandler::dispatch_replicated_write`]: the entry to +/// propose, the proposer, and the identity + plan its origin CDC publish needs. +pub(super) struct ReplicatedWrite<'a> { + pub(super) entry: crate::control::wal_replication::ReplicatedEntry, + pub(super) proposer: &'a Arc, + pub(super) authorized: crate::control::server::shared::authorization::AuthorizedTask, +} + +impl NodeDbPgHandler { + /// Dispatch a write through Raft: propose → register waiter → await apply. + /// `ProposeTracker` is race-safe against an entry applying before register. + /// + /// Also the origin CDC publish site; replicas publish nothing (`ChangeFeedOwner::Unowned`). + pub(super) async fn dispatch_replicated_write( + &self, + args: ReplicatedWrite<'_>, + ) -> crate::Result { + let ReplicatedWrite { + entry, + proposer, + authorized, + } = args; + let task = authorized.into_physical_task(); + let tenant_id = task.tenant_id; + let database_id = task.database_id; + let plan = task.plan; + let request_id = self.next_request_id(); + + // `write_version` is the post-write `coll_write_lsn`, surfaced so the session + // can floor a later read-set at it (read-your-writes for cross-shard OCC). + let (payload, write_version) = + crate::control::wal_replication::propose_replicated_entry(&self.state, proposer, entry) + .await?; + + let response = Response { + request_id, + status: crate::bridge::envelope::Status::Ok, + attempt: 1, + partial: false, + payload: payload.into(), + // Authoritative participant WAL LSN — CDC ordering must use it, not zero. + watermark_lsn: write_version, + error_code: None, + read_set_valid: None, + read_version_lsn: write_version, + write_set: Vec::new(), + }; + + // Propose returned: entry is committed and applied. Publish once, from this plan. + publish_origin_change_events(&self.state, tenant_id, database_id, &plan, &response); + + Ok(response) + } +} diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs b/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs new file mode 100644 index 000000000..33fbb2be2 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs @@ -0,0 +1,233 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-task routing: freeze/mirror checks, orchestrated DML, exchange +//! resolution, and the replicated-vs-local dispatch choice. + +use std::sync::Arc; + +use crate::bridge::envelope::Response; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::exchange::resolve::{ + DistributedReadCapture, Resolved, resolve_and_materialize, +}; +use crate::types::{Lsn, ReadConsistency, TraceId, VShardId}; +use nodedb_physical::physical_task::PhysicalTask; + +use super::super::core::NodeDbPgHandler; +use super::authorize::reject_unadmitted_crdt_apply; +use super::replicated::ReplicatedWrite; + +impl NodeDbPgHandler { + pub(super) async fn dispatch_task_inner( + &self, + mut task: PhysicalTask, + user_id: Option>, + identity: &AuthenticatedIdentity, + shard_watermarks: &mut Vec<(VShardId, Lsn)>, + distributed_reads: &mut Vec, + ) -> crate::Result { + // Reject user writes against a database frozen by a clone materializer sweep. + // Reads/DDL pass through. + use crate::control::security::identity::{Permission, required_permission}; + let perm = required_permission(&task.plan); + if matches!(perm, Permission::Write | Permission::Admin) + && self.state.materialize_freeze.is_frozen(task.database_id) + { + return Err(crate::Error::SourceFrozen { + database_id: task.database_id, + }); + } + + // Mirror enforcement: writes reject on non-promoted mirrors; reads gate by + // ReadConsistency. Catalog lookup skipped for db id=0 to stay allocation-free. + let catalog = self.state.credentials.catalog(); + if task.database_id.as_u64() != 0 + && let Ok(Some(descriptor)) = catalog.get_database(task.database_id) + && let Some(origin) = descriptor.mirror_origin.as_ref() + && !matches!(origin.status, nodedb_types::MirrorStatus::Promoted) + { + if matches!(perm, Permission::Write | Permission::Admin) { + return Err(crate::Error::MirrorReadOnly { + database: descriptor.name.clone(), + }); + } + + use crate::control::server::pgwire::ddl::database::{ + MirrorReadOutcome, check_mirror_read_consistency, + }; + // Defaults to Strong: mirrors aren't the source leader, so reads reject + // unless the session opted into BoundedStaleness or Eventual. + let now_ms = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or(std::time::Duration::ZERO) + .as_millis() as u64; + let outcome = check_mirror_read_consistency( + catalog, + task.database_id, + origin, + ReadConsistency::Strong, + now_ms, + ); + if let MirrorReadOutcome::Reject { message, .. } = outcome { + return Err(crate::Error::StaleReadNotLeader { + database: descriptor.name.clone(), + source_cluster: origin.source_cluster.clone(), + detail: message, + }); + } + } + + if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. } + ) + ) { + let authorized = self.authorize_for_dispatch(identity, &task)?; + return crate::control::insert_select::run_authorized_insert_select( + &self.state, + authorized, + ) + .await; + } + + // Autocommit `MERGE` orchestrates on the Control Plane (`control::merge_orchestrator`). + // In-transaction MERGE buffers for COMMIT replay and never reaches this method. + if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::Merge { + resolved_inserts: None, + .. + } + ) + ) { + let authorized = self.authorize_for_dispatch(identity, &task)?; + return crate::control::merge_orchestrator::run_authorized_merge( + &self.state, + authorized, + ) + .await; + } + + // Scans the source on its own core and ships raw rows into the plan (source's + // vShard can differ). In-transaction buffers for COMMIT replay instead. + if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { + source_rows: None, + .. + } + ) + ) { + let authorized = self.authorize_for_dispatch(identity, &task)?; + return crate::control::update_from_join_orchestrator::run_authorized_update_from_join( + &self.state, + authorized, + ) + .await; + } + + // Can't replicate bare over Raft — a follower has no writing identity to decide + // `$auth.*` against. `write_resolve` resolves it while the identity is live. + if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) + && self.state.async_raft_proposer().is_some() + { + let authorized = self.authorize_for_dispatch(identity, &task)?; + return crate::control::write_resolve::run_authorized_write_resolve( + &self.state, + authorized, + resolver, + ) + .await; + } + + // `DROP ARRAY` reaches every core so each releases its store and segment dir — + // otherwise a follow-up `CREATE ARRAY` carries stale state. + if matches!( + task.plan, + crate::bridge::envelope::PhysicalPlan::Array( + nodedb_physical::physical_plan::ArrayOp::DropArray { .. } + ) + ) { + // Broadcast bypasses the write funnel, so a denied DROP must not + // delete catalog rows or surrogate bindings. + let authorized = self.authorize_for_dispatch(identity, &task)?; + let task = authorized.into_physical_task(); + return crate::control::array_catalog::ddl::run_authorized_drop( + &self.state, + task.tenant_id, + task.database_id, + task.plan, + TraceId::ZERO, + ) + .await; + } + + // Clone-read must run first: resolving derived Exchange plans below + // dispatches straight to the Data Plane, bypassing the clone check. + if let Some(resp) = self + .maybe_intercept_clone_read_early(&task, identity, perm) + .await? + { + return Ok(resp); + } + + // Resolve derived Exchange plans before authorizing the dispatched task. + match resolve_and_materialize( + &self.state, + identity, + task.database_id, + task.tenant_id, + task.plan, + TraceId::ZERO, + task.txn_id, + ) + .await? + { + Resolved::Gathered(resp, wms, caps) => { + *shard_watermarks = wms; + *distributed_reads = caps; + return Ok(resp); + } + Resolved::Plan(resolved_plan) => { + let resolved_plan = *resolved_plan; + task.plan = resolved_plan; + } + Resolved::Stream(stream) => { + return crate::control::server::exchange::gather::stream_to_response(stream).await; + } + } + + reject_unadmitted_crdt_apply(&task.plan)?; + let checked = self + .intercept_and_authorize_for_dispatch(identity, task) + .await?; + let checked = match checked { + crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => { + return Ok(resp); + } + crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(t) => t, + }; + if let Some(async_proposer) = self.state.async_raft_proposer() + && let Some(entry) = crate::control::wal_replication::to_replicated_entry( + checked.tenant_id(), + checked.database_id(), + checked.vshard_id(), + &crate::control::wal_replication::ReplicableWrite::decide_for_replication( + checked.plan(), + )?, + )? + { + return self + .dispatch_replicated_write(ReplicatedWrite { + entry, + proposer: async_proposer, + authorized: checked.into_authorized(), + }) + .await; + } + self.dispatch_local(checked, user_id).await + } +} From 5df1dc67be76992f4ec9244e20eb7337a32e3447 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Wed, 23 Sep 2026 18:52:29 +0800 Subject: [PATCH 13/64] refactor(executor): split Calvin handlers into a directory module calvin.rs held the static-set, passive-participant, active-participant, flush, and discard handler logic in one file. It becomes a calvin/ directory with static_stage.rs, active_passive.rs, flush.rs, discard.rs, and shared.rs, each keeping its existing logic. --- .../data/executor/handlers/control/calvin.rs | 1085 ----------------- .../handlers/control/calvin/active_passive.rs | 313 +++++ .../handlers/control/calvin/discard.rs | 79 ++ .../executor/handlers/control/calvin/flush.rs | 141 +++ .../executor/handlers/control/calvin/mod.rs | 38 + .../handlers/control/calvin/shared.rs | 180 +++ .../handlers/control/calvin/static_stage.rs | 471 +++++++ 7 files changed, 1222 insertions(+), 1085 deletions(-) delete mode 100644 nodedb/src/data/executor/handlers/control/calvin.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin/active_passive.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin/discard.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin/flush.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin/mod.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin/shared.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin/static_stage.rs diff --git a/nodedb/src/data/executor/handlers/control/calvin.rs b/nodedb/src/data/executor/handlers/control/calvin.rs deleted file mode 100644 index d7470a9c0..000000000 --- a/nodedb/src/data/executor/handlers/control/calvin.rs +++ /dev/null @@ -1,1085 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Calvin deterministic executor handlers. -//! -//! Handler entry points: -//! -//! - [`CoreLoop::execute_calvin_execute_static`]: static-set multi-shard txn -//! (the common case). It VALIDATES the read-set to compute the local commit -//! vote and STAGES the transaction's plans into the commit-pending buffer -//! WITHOUT mutating base or firing side effects, then returns the vote. -//! [`CoreLoop::execute_calvin_flush`] later replays the staged plans through -//! the durable apply funnel, or [`CoreLoop::execute_calvin_drop`] discards -//! them. -//! -//! - [`CoreLoop::execute_calvin_execute_passive`]: passive participant for a -//! dependent-read txn. Reads each declared key from the local engine and -//! returns a msgpack-encoded `Vec<(PassiveReadKeyId, Value)>` payload. The -//! Control Plane scheduler proposes a `CalvinReadResult` Raft entry after -//! receiving this response. -//! -//! - [`CoreLoop::execute_calvin_execute_active`]: active participant for a -//! dependent-read txn. Executes the physical plans with the injected read -//! values already resolved. Performs an OLLP verification hook: if the -//! active participant detects that the declared predicate no longer matches -//! the current engine state, it returns `OllpRetryRequired` WITHOUT writing. -//! The OLLP orchestrator on the Control Plane retries via `Inbox::submit`. -//! -//! The `CalvinApplied` WAL record is written on the Control Plane side (in the -//! scheduler's response path) after a successful response is received through -//! the SPSC bridge; not here in the Data Plane. - -use std::panic::{AssertUnwindSafe, catch_unwind}; - -use tracing::{debug, info_span}; - -use nodedb_cluster::calvin::types::PassiveReadKey; -use nodedb_types::Value; -use nodedb_types::calvin::VersionedReadEntry; - -use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::core_loop::commit_pending::PendingCommit; -use crate::data::executor::handlers::transaction::overlay::BitemporalStamp; -use crate::data::executor::response_codec; -use crate::data::executor::task::ExecutionTask; -use crate::types::TenantId; - -use super::calvin_txn_id::calvin_synthetic_txn_id; -use nodedb_physical::physical_plan::PhysicalPlan; -use nodedb_physical::physical_plan::meta::PassiveReadKeyId; - -use std::collections::BTreeMap; - -/// Execution context shared by both static and active Calvin handler variants. -/// -/// Bundles the epoch-scoped parameters that repeat across -/// `execute_calvin_execute_static` and `execute_calvin_execute_active`, -/// keeping each function's argument count within the lint budget. -pub(in crate::data::executor) struct CalvinExecCtx { - pub epoch: u64, - pub position: u32, - pub epoch_system_ms: i64, - pub is_group_leader: bool, -} - -impl CoreLoop { - /// Validate a static-set Calvin transaction and stage it for commit. - /// - /// Computes the local commit vote by checking whether this participant's - /// slice of the transaction's LSN-versioned read-set is still current - /// against the per-core write versions, then STAGES the write plans into - /// the commit-pending buffer keyed by `(epoch, position)`. It performs NO - /// base mutation and fires NO side effects — nothing is observable until a - /// subsequent [`CoreLoop::execute_calvin_flush`] replays the staged plans - /// (or [`CoreLoop::execute_calvin_drop`] discards them). The response - /// carries the vote on `read_set_valid`; the deterministic time anchor and - /// leadership scope are captured with the staged plans and restored at - /// flush time (when the actual apply — and any time-dependent writes — run). - pub(in crate::data::executor) fn execute_calvin_execute_static( - &mut self, - task: &ExecutionTask, - ctx: CalvinExecCtx, - tenant_id: &TenantId, - plans: &[PhysicalPlan], - versioned_reads: &[VersionedReadEntry], - ) -> Response { - let CalvinExecCtx { - epoch, - position, - epoch_system_ms, - is_group_leader, - } = ctx; - let vshard_id = task.request.vshard_id.as_u32(); - debug!( - core = self.core_id, - epoch, - position, - epoch_system_ms, - vshard_id, - is_group_leader, - plan_count = plans.len(), - read_count = versioned_reads.len(), - "calvin stage for commit" - ); - let _stage_span = info_span!( - "executor_stage", - epoch, - position, - vshard = vshard_id, - tenant_id = tenant_id.as_u64(), - trace_id = ?task.request.trace_id, - ) - .entered(); - - // Derive the synthetic transaction identity before ANY mutation. A - // representational failure must not leave either a pending buffer or an - // overlay behind for a transaction that cannot later be resolved. - let synthetic_txn_id = match calvin_synthetic_txn_id(epoch, position, vshard_id) { - Ok(id) => id, - Err(error) => { - return self.calvin_stage_failure(task, epoch, position, vshard_id, error); - } - }; - - // Stage every plan before publishing `PendingCommit`. Panic isolation - // cleans both staging representations after any failure. - let stage_result = catch_unwind(AssertUnwindSafe(|| { - for plan in plans { - self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan)?; - // Test-only fault boundary after a potentially-mutating stage. - crate::fail_point!("calvin_static::during_overlay_stage"); - } - Ok::<(), ErrorCode>(()) - })); - match stage_result { - Ok(Ok(())) => {} - Ok(Err(error)) => { - return self.calvin_stage_failure(task, epoch, position, vshard_id, error); - } - Err(payload) => { - return self.calvin_stage_failure( - task, - epoch, - position, - vshard_id, - ErrorCode::Internal { - detail: format!( - "panic while staging static Calvin transaction: {}", - calvin_panic_payload_to_string(payload.as_ref()) - ), - }, - ); - } - } - - // Local commit vote: is this participant's slice of the read-set still - // current against the local write versions? Empty read-set is vacuously - // current. Read-only — no base mutation here. A stale-read false vote - // retains the fully staged state until the durable global verdict. - let vote = self.read_set_still_current(task, tenant_id.as_u64(), versioned_reads); - - // Publish only a fully staged transaction. The verdict-driven flush - // replays this raw plan buffer; a global abort drops it and its overlay. - self.commit_pending.insert( - (epoch, position, vshard_id), - PendingCommit { - plans: plans.to_vec(), - tenant_id: *tenant_id, - epoch_system_ms, - is_group_leader, - }, - ); - - Response { - request_id: task.request_id(), - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: self.watermark, - error_code: None, - read_set_valid: Some(vote), - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - } - } - - /// Clean failed static staging and return an explicit abort vote. - /// Defensive removal clears document/KV and graph overlays plus their gauge. - fn calvin_stage_failure( - &mut self, - task: &ExecutionTask, - epoch: u64, - position: u32, - vshard_id: u32, - error: E, - ) -> Response - where - E: Into, - { - self.commit_pending.remove(&(epoch, position, vshard_id)); - self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); - let mut response = self.response_error(task, error.into()); - // Scheduler treats this as a durable local abort vote and still waits - // for the authoritative global verdict before issuing any drop. - response.read_set_valid = Some(false); - response - } - - /// Flush a staged Calvin transaction to base storage. - /// - /// Pops the plans staged by [`CoreLoop::execute_calvin_execute_static`] - /// under `(epoch, position)` and replays them through the durable apply - /// funnel (`execute_transaction_batch`) — the same funnel the single-shard - /// commit and recovery use — so base mutation, side effects, and - /// version recording all run exactly once here. The deterministic epoch - /// time anchor and leadership scope captured at stage time are restored - /// around the apply so time-dependent writes stay identical across - /// replicas. An absent key (already flushed or dropped, e.g. a duplicate - /// dispatch) is an idempotent no-op returning `Ok`. - pub(in crate::data::executor) fn execute_calvin_flush( - &mut self, - task: &ExecutionTask, - epoch: u64, - position: u32, - ) -> Response { - let vshard_id = task.request.vshard_id.as_u32(); - // Capture the resolve-time bitemporal stamps (if `CalvinResolve` staged - // them into the synthetic overlay) BEFORE the overlay is dropped, so the - // base install below reuses the exact stamp the redo carries rather than - // minting a fresh one. Empty when this transaction wrote no bitemporal - // document rows or resolve never ran. - let synthetic_txn_id = calvin_synthetic_txn_id(epoch, position, vshard_id).ok(); - let bitemporal_stamps: Vec<(u32, BitemporalStamp)> = synthetic_txn_id - .and_then(|synthetic| self.txn_overlays.get(&synthetic)) - .map(|overlay| overlay.all_bitemporal_stamps().collect()) - .unwrap_or_default(); - let graph_system_from = synthetic_txn_id - .and_then(|synthetic| self.graph_txn_overlays.get(&synthetic)) - .and_then(|overlay| overlay.resolved_system_from()); - // Drop the synthetic overlay entry staged by - // `execute_calvin_execute_static` unconditionally, before the apply - // below: idempotent no-op on a duplicate dispatch. - self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); - let Some(pending) = self.commit_pending.remove(&(epoch, position, vshard_id)) else { - debug!( - core = self.core_id, - epoch, position, vshard_id, "calvin flush: no staged commit (already resolved)" - ); - return self.response_ok(task); - }; - let _apply_span = info_span!( - "executor_apply", - epoch, - position, - vshard = vshard_id, - tenant_id = pending.tenant_id.as_u64(), - trace_id = ?task.request.trace_id, - ) - .entered(); - const NANOS_PER_MS: i64 = 1_000_000; - self.hlc - .update_from_remote(pending.epoch_system_ms.saturating_mul(NANOS_PER_MS)); - self.epoch_system_ms = Some(pending.epoch_system_ms); - // Scope OLLP verification to this participant's staged leadership for the - // batch, then restore the resting (authoritative) state. - let prev_group_leader = self.ollp_is_group_leader; - self.ollp_is_group_leader = pending.is_group_leader; - // Install the captured resolve-time stamps into apply scratch; the - // batch consumes them for its bitemporal document puts and clears the - // scratch when it returns. `txn_id = None`: the synthetic overlay was - // already dropped above, so the stamps are threaded in directly here. - for (surrogate, stamp) in bitemporal_stamps { - self.active_bitemporal_stamps.insert(surrogate, stamp); - } - self.active_graph_system_from = graph_system_from; - // The read-set was already validated at stage time and drives the - // flush/drop decision; the replay itself carries no read-set to re-check. - // Scope the flush key so `record_batch_index_write_values` stages this - // batch's index tuples (the apply carries `wal_lsn: None`); the post-apply - // `RecordCalvinWriteVersions` op drains them at the replicated applied LSN. - self.calvin_flush_key = Some((epoch, position, vshard_id)); - let result = self.execute_transaction_batch( - task, - pending.tenant_id.as_u64(), - &pending.plans, - &[], - None, - ); - self.calvin_flush_key = None; - self.ollp_is_group_leader = prev_group_leader; - self.epoch_system_ms = None; - result - } - - /// Discard a staged Calvin transaction. - /// - /// Removes the plans staged under `(epoch, position, vshard)` from the - /// commit-pending buffer and fires nothing — no base mutation, no side - /// effects. An - /// absent key (already flushed or dropped) is an idempotent no-op. - pub(in crate::data::executor) fn execute_calvin_drop( - &mut self, - task: &ExecutionTask, - epoch: u64, - position: u32, - ) -> Response { - let vshard_id = task.request.vshard_id.as_u32(); - let existed = self - .commit_pending - .remove(&(epoch, position, vshard_id)) - .is_some(); - // Discard the synthetic overlay entry alongside the raw plan buffer; - // idempotent no-op if it was never staged or already removed. - self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); - debug!( - core = self.core_id, - epoch, position, vshard_id, existed, "calvin drop: discarding staged commit" - ); - self.response_ok(task) - } - - /// Execute a passive-participant dependent-read Calvin txn. - /// - /// Reads each key from the local engine state and returns a - /// msgpack-encoded `Vec<(PassiveReadKeyId, Value)>` as the response - /// payload. The Control Plane scheduler collects these values and - /// proposes a `ReplicatedWrite::CalvinReadResult` entry to the - /// per-vshard Raft group so all replicas see the same read results. - /// - /// `Instant::now()` is intentionally absent here — this is a - /// synchronous Data Plane read with no timer interaction. - pub(in crate::data::executor) fn execute_calvin_execute_passive( - &mut self, - task: &ExecutionTask, - epoch: u64, - position: u32, - tenant_id: &TenantId, - keys_to_read: &[PassiveReadKey], - ) -> Response { - debug!( - core = self.core_id, - epoch, - position, - vshard_id = task.request.vshard_id.as_u32(), - key_count = keys_to_read.len(), - "calvin execute passive: reading keys" - ); - - let mut results: Vec<(PassiveReadKeyId, Value)> = Vec::with_capacity(keys_to_read.len()); - - for passive_key in keys_to_read { - // Build a PassiveReadKeyId for each surrogate in the engine key set. - // For this v1 handler the engine key set carries single surrogates per - // key (as specified in the design); we iterate all surrogates to be safe. - let values = self.read_passive_key(tenant_id, &passive_key.engine_key); - results.extend(values); - } - - match response_codec::encode_serde(&results) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error( - task, - ErrorCode::Internal { - detail: format!("calvin passive read encode: {e}"), - }, - ), - } - } - - /// Stage an active-participant dependent-read Calvin txn for commit. - /// - /// Mirrors [`CoreLoop::execute_calvin_execute_static`]: it performs NO base - /// mutation and fires NO side effects — it buffers the write plans in - /// `commit_pending` and stages each into `txn_overlays` under the synthetic - /// `TxnId`, so a subsequent `CalvinResolve` reconstitutes them as one - /// replayable `RedoRecord` and [`CoreLoop::execute_calvin_flush`] applies - /// them. This restores WAL-only-restart durability for the dependent-read - /// path, which previously applied directly with `wal_lsn: None` (only a - /// non-replayable `CalvinApplied` marker survived). - /// - /// The one divergence from the static path: OLLP predicate verification - /// (leader-only) runs HERE, before staging, via - /// [`CoreLoop::verify_calvin_active_ollp`]. The dependent-read path has no - /// LSN-versioned read-set to vote on; its conflict detector is the OLLP - /// `actual != predicted` re-check. Running it at stage time (not flush) - /// ensures a mismatch returns `OllpRetryRequired` and stages nothing — - /// otherwise a stale redo would be WAL-appended before the flush-time check - /// (whose retry signal is swallowed as a degraded shard). The Control Plane - /// scheduler releases locks and re-recons on `OllpRetryRequired`. - /// - /// `injected_reads` is retained on the wire for future plan variants that - /// reference resolved read values by `PassiveReadKeyId`; in v1 the - /// coordinator baked the read values into concrete point ops / the predicted - /// surrogate set at recon, so the plans are self-contained and stage - /// byte-identically to the static path. - pub(in crate::data::executor) fn execute_calvin_execute_active( - &mut self, - task: &ExecutionTask, - ctx: CalvinExecCtx, - tenant_id: &TenantId, - plans: &[PhysicalPlan], - injected_reads: &BTreeMap, - ) -> Response { - let CalvinExecCtx { - epoch, - position, - epoch_system_ms, - is_group_leader, - } = ctx; - let vshard_id = task.request.vshard_id.as_u32(); - debug!( - core = self.core_id, - epoch, - position, - epoch_system_ms, - vshard_id, - is_group_leader, - plan_count = plans.len(), - injected_count = injected_reads.len(), - "calvin execute active" - ); - let _stage_span = info_span!( - "executor_stage", - epoch, - position, - vshard = vshard_id, - tenant_id = tenant_id.as_u64(), - trace_id = ?task.request.trace_id, - ) - .entered(); - - // OLLP verification runs HERE, before staging, so a predicate-drift - // mismatch surfaces on THIS stage response (where the scheduler releases - // locks and re-recons) and nothing is staged, resolved, or WAL-appended. - // Scoped to this replica's staged leadership for the check, then the - // resting (authoritative) state is restored. A read-only scan needs no - // time anchor, so `epoch_system_ms`/`hlc` stay unset until flush - // (mirroring the static path, where they ride `PendingCommit`). - let prev_group_leader = self.ollp_is_group_leader; - self.ollp_is_group_leader = is_group_leader; - let verified = self.verify_calvin_active_ollp(task, tenant_id.as_u64(), plans); - self.ollp_is_group_leader = prev_group_leader; - match verified { - Ok(true) => {} - Ok(false) => return self.response_error(task, ErrorCode::OllpRetryRequired), - Err(e) => return self.response_error(task, e), - } - - // Stage exactly like `execute_calvin_execute_static`: buffer the plans in - // `commit_pending` (the sole durable apply the flush replays) and stage - // each write into `txn_overlays` under the synthetic `TxnId` (producer - // side for `CalvinResolve`). No base mutation, no side effects; the time - // anchor + leadership scope captured here are restored at flush time. - self.commit_pending.insert( - (epoch, position, vshard_id), - PendingCommit { - plans: plans.to_vec(), - tenant_id: *tenant_id, - epoch_system_ms, - is_group_leader, - }, - ); - let synthetic_txn_id = match calvin_synthetic_txn_id(epoch, position, vshard_id) { - Ok(id) => id, - Err(e) => return self.response_error(task, e), - }; - for plan in plans { - if let Err(e) = self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan) { - return self.response_error(task, e); - } - } - - Response { - request_id: task.request_id(), - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: self.watermark, - error_code: None, - // The dependent-read path carries no versioned read-set; `None` maps - // to "commit" in `resolve_staged_commit` (`read_set_valid != Some(false)`). - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - } - } -} -fn calvin_panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String { - if let Some(message) = payload.downcast_ref::<&'static str>() { - (*message).to_owned() - } else if let Some(message) = payload.downcast_ref::() { - message.clone() - } else { - "".to_owned() - } -} - -#[cfg(test)] -mod tests { - use std::time::{Duration, Instant}; - - use nodedb_physical::physical_plan::{DocumentOp, TimeseriesOp}; - use nodedb_types::{QualifiedCollection, Surrogate}; - - use super::*; - use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request}; - use crate::data::executor::core_loop::tests::make_core_with_dir; - use crate::data::executor::doc_format; - use crate::data::executor::handlers::transaction::overlay::Staged; - use crate::types::{DatabaseId, RequestId, TraceId, VShardId}; - - /// A minimal `ExecutionTask` homing to vShard 0, tenant 1, database - /// DEFAULT -- everything a Calvin static-execute handler needs beyond - /// its explicit `CalvinExecCtx` / `tenant_id` / `plans` arguments. - fn make_task() -> ExecutionTask { - let plan = PhysicalPlan::Document(DocumentOp::PointGet { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), - document_id: "y".into(), - surrogate: Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }); - let request = Request { - request_id: RequestId::new(1), - tenant_id: TenantId::new(1), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - plan, - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: crate::types::ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: Admission::Exempt(ExemptReason::Read), - }; - ExecutionTask::new(request) - } - - fn doc_value(field: &str, val: &str) -> Vec { - let mut obj = std::collections::HashMap::new(); - obj.insert(field.to_string(), Value::String(val.into())); - zerompk::to_msgpack_vec(&Value::Object(obj)).unwrap() - } - - fn point_insert_plan(collection: &str, document_id: &str, surrogate: u32) -> PhysicalPlan { - PhysicalPlan::Document(DocumentOp::PointInsert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, collection), - document_id: document_id.to_string(), - value: doc_value("a", "1"), - if_absent: false, - surrogate: Surrogate::new(surrogate), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - deferred_sum_targets: Vec::new(), - }) - } - - fn canonical_ilp_plan(collection: &str, lines: Vec<&str>, tokens: Vec) -> PhysicalPlan { - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, collection), - payload: zerompk::to_msgpack_vec(&lines).expect("canonical ILP payload"), - format: "ilp-msgpack".to_owned(), - wal_lsn: None, - surrogates: tokens.into_iter().map(Surrogate::new).collect(), - provenance: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }) - } - - fn bulk_delete_plan(collection: &str, predicted: Option>) -> PhysicalPlan { - PhysicalPlan::Document(DocumentOp::BulkDelete { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, collection), - filters: Vec::new(), - returning: None, - ollp_predicted_surrogates: predicted, - ollp_predicted_edges: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - resolved_sum_targets: Vec::new(), - declared_primary_key: None, - }) - } - - /// Seed a row directly into base storage (bypassing Calvin staging), the - /// pre-existing state the active-path OLLP verifier scans against. - fn seed_row(core: &mut CoreLoop, collection: &str, surrogate: u32) { - let doc_id = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); - let body = doc_format::canonicalize_document_for_storage(&doc_value("a", "1")); - core.sparse - .put(DatabaseId::DEFAULT.as_u64(), 1, collection, &doc_id, &body) - .expect("seed row"); - } - - #[test] - fn calvin_execute_static_stages_point_insert_into_overlay() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - - let task = make_task(); - let tenant_id = TenantId::new(1); - let plans = vec![point_insert_plan("orders", "o1", 7)]; - let ctx = CalvinExecCtx { - epoch: 1, - position: 0, - epoch_system_ms: 0, - is_group_leader: true, - }; - - let resp = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); - assert_eq!(resp.status, Status::Ok); - - let vshard_id = task.request.vshard_id.as_u32(); - - // `commit_pending` is unchanged -- it still holds the raw plans that - // drive the base install at flush time. - assert!( - core.commit_pending.contains_key(&(1, 0, vshard_id)), - "commit_pending must still be populated exactly as before this unit" - ); - - // The synthetic overlay entry additionally holds the resolved - // post-image for the concrete point-write plan. - let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); - let coll_key = (DatabaseId::DEFAULT, tenant_id, "orders".to_string()); - let expected_body = doc_format::canonicalize_document_for_storage(&doc_value("a", "1")); - assert_eq!( - core.txn_overlays - .get(&synthetic) - .and_then(|o| o.get(&coll_key, 7)), - Some(&Staged::Put(expected_body)), - "the Calvin write plan must be staged into the synthetic-TxnId overlay" - ); - } - - #[test] - fn synthetic_id_failure_leaves_no_pending_or_overlay_state() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_task(); - let tenant_id = TenantId::new(1); - let epoch_outside_synthetic_range = 1_u64 << 33; - - let response = core.execute_calvin_execute_static( - &task, - CalvinExecCtx { - epoch: epoch_outside_synthetic_range, - position: 0, - epoch_system_ms: 0, - is_group_leader: true, - }, - &tenant_id, - &[point_insert_plan("orders", "o1", 7)], - &[], - ); - - assert_eq!(response.status, Status::Error); - assert_eq!(response.read_set_valid, Some(false)); - assert!(core.commit_pending.is_empty()); - assert!(core.txn_overlays.is_empty()); - assert!(core.graph_txn_overlays.is_empty()); - } - - #[test] - fn static_stage_error_cleans_all_prior_overlay_state_before_voting_abort() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let metrics = std::sync::Arc::new(crate::control::metrics::SystemMetrics::new()); - core.metrics = Some(metrics.clone()); - let gauge = || { - metrics - .active_txn_overlays - .load(std::sync::atomic::Ordering::Relaxed) - }; - let baseline = gauge(); - let task = make_task(); - let tenant_id = TenantId::new(1); - let ctx = CalvinExecCtx { - epoch: 9, - position: 3, - epoch_system_ms: 0, - is_group_leader: true, - }; - let vshard = task.request.vshard_id.as_u32(); - let synthetic = calvin_synthetic_txn_id(9, 3, vshard).unwrap(); - core.graph_txn_overlay_mut(synthetic); - assert_eq!( - gauge(), - baseline + 1, - "graph staging must increment the gauge" - ); - // The first plan adds a document overlay; the second is invalid without - // an OLLP prediction and must clean both overlays atomically. - let plans = vec![ - point_insert_plan("orders", "o1", 7), - bulk_delete_plan("orders", None), - ]; - - let response = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); - - assert_eq!(response.status, Status::Error); - assert_eq!(response.read_set_valid, Some(false)); - assert!(!core.commit_pending.contains_key(&(9, 3, vshard))); - assert!(!core.txn_overlays.contains_key(&synthetic)); - assert!(!core.graph_txn_overlays.contains_key(&synthetic)); - assert_eq!( - gauge(), - baseline, - "failed staging must restore the overlay gauge" - ); - } - - #[test] - fn static_stage_error_also_cleans_existing_graph_overlay_for_synthetic_id() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_task(); - let tenant_id = TenantId::new(1); - let vshard = task.request.vshard_id.as_u32(); - let synthetic = calvin_synthetic_txn_id(9, 5, vshard).unwrap(); - // Use the graph-overlay choke point so this regression also covers its - // gauge accounting without requiring a graph catalog fixture. - core.graph_txn_overlay_mut(synthetic); - - let response = core.execute_calvin_execute_static( - &task, - CalvinExecCtx { - epoch: 9, - position: 5, - epoch_system_ms: 0, - is_group_leader: true, - }, - &tenant_id, - &[bulk_delete_plan("orders", None)], - &[], - ); - - assert_eq!(response.read_set_valid, Some(false)); - assert!(!core.graph_txn_overlays.contains_key(&synthetic)); - } - - #[cfg(feature = "failpoints")] - #[test] - fn static_stage_panic_cleans_prior_overlay_state_before_voting_abort() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let metrics = std::sync::Arc::new(crate::control::metrics::SystemMetrics::new()); - core.metrics = Some(metrics.clone()); - let gauge = || { - metrics - .active_txn_overlays - .load(std::sync::atomic::Ordering::Relaxed) - }; - let baseline = gauge(); - let task = make_task(); - let tenant_id = TenantId::new(1); - let ctx = CalvinExecCtx { - epoch: 9, - position: 4, - epoch_system_ms: 0, - is_group_leader: true, - }; - let vshard = task.request.vshard_id.as_u32(); - let synthetic = calvin_synthetic_txn_id(9, 4, vshard).unwrap(); - core.graph_txn_overlay_mut(synthetic); - assert_eq!( - gauge(), - baseline + 1, - "graph staging must increment the gauge" - ); - let _fail = crate::fail_point::FailGuard::install( - "calvin_static::during_overlay_stage", - crate::fail_point::FailAction::Panic, - ); - - let response = core.execute_calvin_execute_static( - &task, - ctx, - &tenant_id, - &[point_insert_plan("orders", "o1", 7)], - &[], - ); - - assert_eq!(response.status, Status::Error); - assert_eq!(response.read_set_valid, Some(false)); - assert!(!core.commit_pending.contains_key(&(9, 4, vshard))); - assert!(!core.txn_overlays.contains_key(&synthetic)); - assert!(!core.graph_txn_overlays.contains_key(&synthetic)); - assert_eq!( - gauge(), - baseline, - "panic cleanup must restore the overlay gauge" - ); - } - - #[test] - fn static_calvin_ilp_staging_is_all_or_abort() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_task(); - let tenant = TenantId::new(1); - let vshard = task.request.vshard_id.as_u32(); - let synthetic = calvin_synthetic_txn_id(21, 1, vshard).expect("synthetic transaction id"); - - let malformed = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cpu"), - payload: vec![0xc1], - format: "ilp-msgpack".to_owned(), - wal_lsn: None, - surrogates: vec![Surrogate::new(1)], - provenance: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }); - let failed = core.execute_calvin_execute_static( - &task, - CalvinExecCtx { - epoch: 21, - position: 1, - epoch_system_ms: 0, - is_group_leader: true, - }, - &tenant, - &[malformed], - &[], - ); - assert_eq!(failed.read_set_valid, Some(false)); - assert!(matches!( - failed.error_code.as_deref(), - Some(ErrorCode::RejectedPrevalidation { .. }) - )); - assert!(!core.commit_pending.contains_key(&(21, 1, vshard))); - assert!(!core.txn_overlays.contains_key(&synthetic)); - - let valid = canonical_ilp_plan("cpu", vec!["cpu value=1i", "cpu value=2i"], vec![1, 2]); - let staged = core.execute_calvin_execute_static( - &task, - CalvinExecCtx { - epoch: 21, - position: 1, - epoch_system_ms: 0, - is_group_leader: true, - }, - &tenant, - &[valid], - &[], - ); - assert_eq!(staged.read_set_valid, Some(true)); - assert_eq!( - core.txn_overlays - .get(&synthetic) - .map(|overlay| overlay.len()), - Some(2) - ); - - let mismatch = canonical_ilp_plan("cpu", vec!["memory value=1i"], vec![3]); - let failed = core.execute_calvin_execute_static( - &task, - CalvinExecCtx { - epoch: 21, - position: 2, - epoch_system_ms: 0, - is_group_leader: true, - }, - &tenant, - &[mismatch], - &[], - ); - let mismatch_id = calvin_synthetic_txn_id(21, 2, vshard).expect("synthetic transaction id"); - assert_eq!(failed.read_set_valid, Some(false)); - assert!(!core.commit_pending.contains_key(&(21, 2, vshard))); - assert!(!core.txn_overlays.contains_key(&mismatch_id)); - - core.ts_tuning.max_tag_cardinality = 1; - let overflow = canonical_ilp_plan( - "cpu", - vec!["cpu,host=a value=1i", "cpu,host=b value=2i"], - vec![4, 5], - ); - let failed = core.execute_calvin_execute_static( - &task, - CalvinExecCtx { - epoch: 21, - position: 3, - epoch_system_ms: 0, - is_group_leader: true, - }, - &tenant, - &[overflow], - &[], - ); - let overflow_id = calvin_synthetic_txn_id(21, 3, vshard).expect("synthetic transaction id"); - assert_eq!(failed.status, Status::Error); - assert_eq!(failed.read_set_valid, Some(false)); - assert!(matches!( - failed.error_code.as_deref(), - Some(ErrorCode::RejectedPrevalidation { .. }) - )); - assert!( - !core.commit_pending.contains_key(&(21, 3, vshard)), - "a rejected stage cannot reach the TransactionRedo-producing flush path" - ); - assert!(!core.txn_overlays.contains_key(&overflow_id)); - } - - #[test] - fn calvin_flush_drops_synthetic_overlay() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - - let task = make_task(); - let tenant_id = TenantId::new(1); - let plans = vec![point_insert_plan("orders", "o1", 7)]; - let ctx = CalvinExecCtx { - epoch: 1, - position: 0, - epoch_system_ms: 0, - is_group_leader: true, - }; - let resp = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); - assert_eq!(resp.status, Status::Ok); - - let vshard_id = task.request.vshard_id.as_u32(); - let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); - assert!(core.txn_overlays.contains_key(&synthetic)); - - let flush_resp = core.execute_calvin_flush(&task, 1, 0); - assert_eq!(flush_resp.status, Status::Ok); - - assert!( - !core.txn_overlays.contains_key(&synthetic), - "flush must drop the synthetic overlay entry alongside commit_pending" - ); - } - - #[test] - fn calvin_drop_discards_synthetic_overlay() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - - let task = make_task(); - let tenant_id = TenantId::new(1); - let plans = vec![point_insert_plan("orders", "o1", 7)]; - let ctx = CalvinExecCtx { - epoch: 1, - position: 0, - epoch_system_ms: 0, - is_group_leader: true, - }; - let resp = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); - assert_eq!(resp.status, Status::Ok); - - let vshard_id = task.request.vshard_id.as_u32(); - let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); - assert!(core.txn_overlays.contains_key(&synthetic)); - - let drop_resp = core.execute_calvin_drop(&task, 1, 0); - assert_eq!(drop_resp.status, Status::Ok); - - assert!( - !core.txn_overlays.contains_key(&synthetic), - "drop must discard the synthetic overlay entry alongside commit_pending" - ); - } - - /// The dependent-read ACTIVE path STAGES its writes (into `commit_pending` + - /// the synthetic overlay) instead of applying them to base directly. This is - /// the direct regression guard for U-CAL5: before it, this handler called - /// `execute_transaction_batch` inline (`wal_lsn: None`), so a Calvin-committed - /// dependent-read write left only a non-replayable `CalvinApplied` marker and - /// was lost on a WAL-only restart. Staging routes it through the same - /// resolve → redo → flush the static path uses. - #[test] - fn calvin_execute_active_stages_point_insert_into_overlay() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - - let task = make_task(); - let tenant_id = TenantId::new(1); - let plans = vec![point_insert_plan("orders", "o1", 7)]; - let ctx = CalvinExecCtx { - epoch: 1, - position: 0, - epoch_system_ms: 0, - is_group_leader: true, - }; - let injected = BTreeMap::new(); - - let resp = core.execute_calvin_execute_active(&task, ctx, &tenant_id, &plans, &injected); - assert_eq!(resp.status, Status::Ok); - // The dependent-read path carries no versioned read-set; `None` maps to - // "commit" in `resolve_staged_commit`. - assert_eq!(resp.read_set_valid, None); - - let vshard_id = task.request.vshard_id.as_u32(); - - // STAGED, not applied: the plans are buffered for the flush replay. - assert!( - core.commit_pending.contains_key(&(1, 0, vshard_id)), - "active-path write must be STAGED into commit_pending, not applied directly" - ); - - // No base mutation at stage time — the row appears only after flush. - let doc_id = nodedb_types::StorageKey::for_surrogate(Surrogate::new(7)); - assert!( - core.sparse - .get(DatabaseId::DEFAULT.as_u64(), 1, "orders", &doc_id) - .expect("base get") - .is_none(), - "staging the active-path write must NOT mutate base storage" - ); - - // The synthetic overlay holds the resolved post-image (producer side for - // `CalvinResolve` → redo). - let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); - let coll_key = (DatabaseId::DEFAULT, tenant_id, "orders".to_string()); - let expected_body = doc_format::canonicalize_document_for_storage(&doc_value("a", "1")); - assert_eq!( - core.txn_overlays - .get(&synthetic) - .and_then(|o| o.get(&coll_key, 7)), - Some(&Staged::Put(expected_body)), - "the active Calvin write plan must be staged into the synthetic-TxnId overlay" - ); - } - - /// On the data-group leader, a predicate-write plan whose carried OLLP - /// predicted set no longer matches live state returns `OllpRetryRequired` - /// BEFORE staging anything — so no stale redo is WAL-appended and the - /// coordinator can re-recon under a fresh attempt. Verifying at stage time - /// (not flush) is the one divergence from the static path. - #[test] - fn calvin_execute_active_ollp_drift_returns_retry_and_stages_nothing() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - - // Live match-all set is {10, 11}; the plan predicts only {10} → drift. - seed_row(&mut core, "orders", 10); - seed_row(&mut core, "orders", 11); - - let task = make_task(); - let tenant_id = TenantId::new(1); - let plans = vec![bulk_delete_plan("orders", Some(vec![10]))]; - let ctx = CalvinExecCtx { - epoch: 1, - position: 0, - epoch_system_ms: 0, - is_group_leader: true, - }; - let injected = BTreeMap::new(); - - let resp = core.execute_calvin_execute_active(&task, ctx, &tenant_id, &plans, &injected); - assert_eq!(resp.status, Status::Error); - assert_eq!( - resp.error_code.as_deref(), - Some(&ErrorCode::OllpRetryRequired) - ); - - // Drift stages NOTHING — neither the raw buffer nor the overlay. - let vshard_id = task.request.vshard_id.as_u32(); - assert!( - !core.commit_pending.contains_key(&(1, 0, vshard_id)), - "an OLLP-drift retry must not leave a staged commit buffer" - ); - let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); - assert!( - !core.txn_overlays.contains_key(&synthetic), - "an OLLP-drift retry must not leave a staged overlay" - ); - } -} diff --git a/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs b/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs new file mode 100644 index 000000000..673800c52 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs @@ -0,0 +1,313 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Passive and active participants for a dependent-read Calvin transaction. + +use std::collections::BTreeMap; + +use tracing::{debug, info_span}; + +use nodedb_cluster::calvin::types::PassiveReadKey; +use nodedb_types::Value; + +use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::commit_pending::PendingCommit; +use crate::data::executor::response_codec; +use crate::data::executor::task::ExecutionTask; +use crate::types::TenantId; +use nodedb_physical::physical_plan::PhysicalPlan; +use nodedb_physical::physical_plan::meta::PassiveReadKeyId; + +use crate::data::executor::handlers::control::calvin_txn_id::calvin_synthetic_txn_id; + +use super::shared::CalvinExecCtx; + +impl CoreLoop { + /// Execute a passive-participant dependent-read Calvin txn. + /// + /// Reads each key from the local engine state and returns a + /// msgpack-encoded `Vec<(PassiveReadKeyId, Value)>` as the response + /// payload. The Control Plane scheduler collects these values and + /// proposes a `ReplicatedWrite::CalvinReadResult` entry to the + /// per-vshard Raft group so all replicas see the same read results. + /// + /// `Instant::now()` is intentionally absent here — this is a + /// synchronous Data Plane read with no timer interaction. + pub(in crate::data::executor) fn execute_calvin_execute_passive( + &mut self, + task: &ExecutionTask, + epoch: u64, + position: u32, + tenant_id: &TenantId, + keys_to_read: &[PassiveReadKey], + ) -> Response { + debug!( + core = self.core_id, + epoch, + position, + vshard_id = task.request.vshard_id.as_u32(), + key_count = keys_to_read.len(), + "calvin execute passive: reading keys" + ); + + let mut results: Vec<(PassiveReadKeyId, Value)> = Vec::with_capacity(keys_to_read.len()); + + for passive_key in keys_to_read { + // Build a PassiveReadKeyId for each surrogate in the engine key set. + // For this v1 handler the engine key set carries single surrogates per + // key (as specified in the design); we iterate all surrogates to be safe. + let values = self.read_passive_key(tenant_id, &passive_key.engine_key); + results.extend(values); + } + + match response_codec::encode_serde(&results) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => self.response_error( + task, + ErrorCode::Internal { + detail: format!("calvin passive read encode: {e}"), + }, + ), + } + } + + /// Stage an active-participant dependent-read Calvin txn for commit. + /// + /// Mirrors [`CoreLoop::execute_calvin_execute_static`]: it performs NO base + /// mutation and fires NO side effects — it buffers the write plans in + /// `commit_pending` and stages each into `txn_overlays` under the synthetic + /// `TxnId`, so a subsequent `CalvinResolve` reconstitutes them as one + /// replayable `RedoRecord` and [`CoreLoop::execute_calvin_flush`] applies + /// them. This restores WAL-only-restart durability for the dependent-read + /// path, which previously applied directly with `wal_lsn: None` (only a + /// non-replayable `CalvinApplied` marker survived). + /// + /// The one divergence from the static path: OLLP predicate verification + /// (leader-only) runs HERE, before staging, via + /// [`CoreLoop::verify_calvin_active_ollp`]. The dependent-read path has no + /// LSN-versioned read-set to vote on; its conflict detector is the OLLP + /// `actual != predicted` re-check. Running it at stage time (not flush) + /// ensures a mismatch returns `OllpRetryRequired` and stages nothing — + /// otherwise a stale redo would be WAL-appended before the flush-time check + /// (whose retry signal is swallowed as a degraded shard). The Control Plane + /// scheduler releases locks and re-recons on `OllpRetryRequired`. + /// + /// `injected_reads` is retained on the wire for future plan variants that + /// reference resolved read values by `PassiveReadKeyId`; in v1 the + /// coordinator baked the read values into concrete point ops / the predicted + /// surrogate set at recon, so the plans are self-contained and stage + /// byte-identically to the static path. + pub(in crate::data::executor) fn execute_calvin_execute_active( + &mut self, + task: &ExecutionTask, + ctx: CalvinExecCtx, + tenant_id: &TenantId, + plans: &[PhysicalPlan], + injected_reads: &BTreeMap, + ) -> Response { + let CalvinExecCtx { + epoch, + position, + epoch_system_ms, + is_group_leader, + } = ctx; + let vshard_id = task.request.vshard_id.as_u32(); + debug!( + core = self.core_id, + epoch, + position, + epoch_system_ms, + vshard_id, + is_group_leader, + plan_count = plans.len(), + injected_count = injected_reads.len(), + "calvin execute active" + ); + let _stage_span = info_span!( + "executor_stage", + epoch, + position, + vshard = vshard_id, + tenant_id = tenant_id.as_u64(), + trace_id = ?task.request.trace_id, + ) + .entered(); + + // OLLP verification runs HERE, before staging, so a predicate-drift + // mismatch surfaces on THIS stage response (where the scheduler releases + // locks and re-recons) and nothing is staged, resolved, or WAL-appended. + // Scoped to this replica's staged leadership for the check, then the + // resting (authoritative) state is restored. A read-only scan needs no + // time anchor, so `epoch_system_ms`/`hlc` stay unset until flush + // (mirroring the static path, where they ride `PendingCommit`). + let prev_group_leader = self.ollp_is_group_leader; + self.ollp_is_group_leader = is_group_leader; + let verified = self.verify_calvin_active_ollp(task, tenant_id.as_u64(), plans); + self.ollp_is_group_leader = prev_group_leader; + match verified { + Ok(true) => {} + Ok(false) => return self.response_error(task, ErrorCode::OllpRetryRequired), + Err(e) => return self.response_error(task, e), + } + + // Stage exactly like `execute_calvin_execute_static`: buffer the plans in + // `commit_pending` (the sole durable apply the flush replays) and stage + // each write into `txn_overlays` under the synthetic `TxnId` (producer + // side for `CalvinResolve`). No base mutation, no side effects; the time + // anchor + leadership scope captured here are restored at flush time. + self.commit_pending.insert( + (epoch, position, vshard_id), + PendingCommit { + plans: plans.to_vec(), + tenant_id: *tenant_id, + epoch_system_ms, + is_group_leader, + }, + ); + let synthetic_txn_id = match calvin_synthetic_txn_id(epoch, position, vshard_id) { + Ok(id) => id, + Err(e) => return self.response_error(task, e), + }; + for plan in plans { + if let Err(e) = self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan) { + return self.response_error(task, e); + } + } + + Response { + request_id: task.request_id(), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: self.watermark, + error_code: None, + // The dependent-read path carries no versioned read-set; `None` maps + // to "commit" in `resolve_staged_commit` (`read_set_valid != Some(false)`). + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + } + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::data::executor::doc_format; + use crate::data::executor::handlers::transaction::overlay::Staged; + use crate::types::DatabaseId; + + use super::super::shared::test_support::{ + bulk_delete_plan, doc_value, make_task, point_insert_plan, seed_row, + }; + + /// The dependent-read ACTIVE path STAGES its writes (into `commit_pending` + + /// the synthetic overlay) instead of applying them to base directly. This is + /// the direct regression guard for U-CAL5: before it, this handler called + /// `execute_transaction_batch` inline (`wal_lsn: None`), so a Calvin-committed + /// dependent-read write left only a non-replayable `CalvinApplied` marker and + /// was lost on a WAL-only restart. Staging routes it through the same + /// resolve → redo → flush the static path uses. + #[test] + fn calvin_execute_active_stages_point_insert_into_overlay() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + + let task = make_task(); + let tenant_id = TenantId::new(1); + let plans = vec![point_insert_plan("orders", "o1", 7)]; + let ctx = CalvinExecCtx { + epoch: 1, + position: 0, + epoch_system_ms: 0, + is_group_leader: true, + }; + let injected = BTreeMap::new(); + + let resp = core.execute_calvin_execute_active(&task, ctx, &tenant_id, &plans, &injected); + assert_eq!(resp.status, Status::Ok); + // The dependent-read path carries no versioned read-set; `None` maps to + // "commit" in `resolve_staged_commit`. + assert_eq!(resp.read_set_valid, None); + + let vshard_id = task.request.vshard_id.as_u32(); + + // STAGED, not applied: the plans are buffered for the flush replay. + assert!( + core.commit_pending.contains_key(&(1, 0, vshard_id)), + "active-path write must be STAGED into commit_pending, not applied directly" + ); + + // No base mutation at stage time — the row appears only after flush. + let doc_id = nodedb_types::StorageKey::for_surrogate(Surrogate::new(7)); + assert!( + core.sparse + .get(DatabaseId::DEFAULT.as_u64(), 1, "orders", &doc_id) + .expect("base get") + .is_none(), + "staging the active-path write must NOT mutate base storage" + ); + + // The synthetic overlay holds the resolved post-image (producer side for + // `CalvinResolve` → redo). + let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); + let coll_key = (DatabaseId::DEFAULT, tenant_id, "orders".to_string()); + let expected_body = doc_format::canonicalize_document_for_storage(&doc_value("a", "1")); + assert_eq!( + core.txn_overlays + .get(&synthetic) + .and_then(|o| o.get(&coll_key, 7)), + Some(&Staged::Put(expected_body)), + "the active Calvin write plan must be staged into the synthetic-TxnId overlay" + ); + } + + /// On the data-group leader, a predicate-write plan whose carried OLLP + /// predicted set no longer matches live state returns `OllpRetryRequired` + /// BEFORE staging anything — so no stale redo is WAL-appended and the + /// coordinator can re-recon under a fresh attempt. Verifying at stage time + /// (not flush) is the one divergence from the static path. + #[test] + fn calvin_execute_active_ollp_drift_returns_retry_and_stages_nothing() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + + // Live match-all set is {10, 11}; the plan predicts only {10} → drift. + seed_row(&mut core, "orders", 10); + seed_row(&mut core, "orders", 11); + + let task = make_task(); + let tenant_id = TenantId::new(1); + let plans = vec![bulk_delete_plan("orders", Some(vec![10]))]; + let ctx = CalvinExecCtx { + epoch: 1, + position: 0, + epoch_system_ms: 0, + is_group_leader: true, + }; + let injected = BTreeMap::new(); + + let resp = core.execute_calvin_execute_active(&task, ctx, &tenant_id, &plans, &injected); + assert_eq!(resp.status, Status::Error); + assert_eq!( + resp.error_code.as_deref(), + Some(&ErrorCode::OllpRetryRequired) + ); + + // Drift stages NOTHING — neither the raw buffer nor the overlay. + let vshard_id = task.request.vshard_id.as_u32(); + assert!( + !core.commit_pending.contains_key(&(1, 0, vshard_id)), + "an OLLP-drift retry must not leave a staged commit buffer" + ); + let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); + assert!( + !core.txn_overlays.contains_key(&synthetic), + "an OLLP-drift retry must not leave a staged overlay" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin/discard.rs b/nodedb/src/data/executor/handlers/control/calvin/discard.rs new file mode 100644 index 000000000..420aaded3 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin/discard.rs @@ -0,0 +1,79 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Discard a staged Calvin transaction. + +use tracing::debug; + +use crate::bridge::envelope::Response; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; + +impl CoreLoop { + /// Discard a staged Calvin transaction. + /// + /// Removes the plans staged under `(epoch, position, vshard)` from the + /// commit-pending buffer and fires nothing — no base mutation, no side + /// effects. An + /// absent key (already flushed or dropped) is an idempotent no-op. + pub(in crate::data::executor) fn execute_calvin_drop( + &mut self, + task: &ExecutionTask, + epoch: u64, + position: u32, + ) -> Response { + let vshard_id = task.request.vshard_id.as_u32(); + let existed = self + .commit_pending + .remove(&(epoch, position, vshard_id)) + .is_some(); + // Discard the synthetic overlay entry alongside the raw plan buffer; + // idempotent no-op if it was never staged or already removed. + self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); + debug!( + core = self.core_id, + epoch, position, vshard_id, existed, "calvin drop: discarding staged commit" + ); + self.response_ok(task) + } +} + +#[cfg(test)] +mod tests { + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::types::TenantId; + + use super::super::shared::CalvinExecCtx; + use super::super::shared::test_support::{make_task, point_insert_plan}; + use crate::data::executor::handlers::control::calvin_txn_id::calvin_synthetic_txn_id; + + #[test] + fn calvin_drop_discards_synthetic_overlay() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + + let task = make_task(); + let tenant_id = TenantId::new(1); + let plans = vec![point_insert_plan("orders", "o1", 7)]; + let ctx = CalvinExecCtx { + epoch: 1, + position: 0, + epoch_system_ms: 0, + is_group_leader: true, + }; + let resp = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); + assert_eq!(resp.status, Status::Ok); + + let vshard_id = task.request.vshard_id.as_u32(); + let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); + assert!(core.txn_overlays.contains_key(&synthetic)); + + let drop_resp = core.execute_calvin_drop(&task, 1, 0); + assert_eq!(drop_resp.status, Status::Ok); + + assert!( + !core.txn_overlays.contains_key(&synthetic), + "drop must discard the synthetic overlay entry alongside commit_pending" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin/flush.rs b/nodedb/src/data/executor/handlers/control/calvin/flush.rs new file mode 100644 index 000000000..8c3126608 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin/flush.rs @@ -0,0 +1,141 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Flush a staged Calvin transaction to base storage. + +use tracing::{debug, info_span}; + +use crate::bridge::envelope::Response; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::transaction::overlay::BitemporalStamp; +use crate::data::executor::task::ExecutionTask; + +use crate::data::executor::handlers::control::calvin_txn_id::calvin_synthetic_txn_id; + +impl CoreLoop { + /// Flush a staged Calvin transaction to base storage. + /// + /// Pops the plans staged by [`CoreLoop::execute_calvin_execute_static`] + /// under `(epoch, position)` and replays them through the durable apply + /// funnel (`execute_transaction_batch`) — the same funnel the single-shard + /// commit and recovery use — so base mutation, side effects, and + /// version recording all run exactly once here. The deterministic epoch + /// time anchor and leadership scope captured at stage time are restored + /// around the apply so time-dependent writes stay identical across + /// replicas. An absent key (already flushed or dropped, e.g. a duplicate + /// dispatch) is an idempotent no-op returning `Ok`. + pub(in crate::data::executor) fn execute_calvin_flush( + &mut self, + task: &ExecutionTask, + epoch: u64, + position: u32, + ) -> Response { + let vshard_id = task.request.vshard_id.as_u32(); + // Capture the resolve-time bitemporal stamps (if `CalvinResolve` staged + // them into the synthetic overlay) BEFORE the overlay is dropped, so the + // base install below reuses the exact stamp the redo carries rather than + // minting a fresh one. Empty when this transaction wrote no bitemporal + // document rows or resolve never ran. + let synthetic_txn_id = calvin_synthetic_txn_id(epoch, position, vshard_id).ok(); + let bitemporal_stamps: Vec<(u32, BitemporalStamp)> = synthetic_txn_id + .and_then(|synthetic| self.txn_overlays.get(&synthetic)) + .map(|overlay| overlay.all_bitemporal_stamps().collect()) + .unwrap_or_default(); + let graph_system_from = synthetic_txn_id + .and_then(|synthetic| self.graph_txn_overlays.get(&synthetic)) + .and_then(|overlay| overlay.resolved_system_from()); + // Drop the synthetic overlay entry staged by + // `execute_calvin_execute_static` unconditionally, before the apply + // below: idempotent no-op on a duplicate dispatch. + self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); + let Some(pending) = self.commit_pending.remove(&(epoch, position, vshard_id)) else { + debug!( + core = self.core_id, + epoch, position, vshard_id, "calvin flush: no staged commit (already resolved)" + ); + return self.response_ok(task); + }; + let _apply_span = info_span!( + "executor_apply", + epoch, + position, + vshard = vshard_id, + tenant_id = pending.tenant_id.as_u64(), + trace_id = ?task.request.trace_id, + ) + .entered(); + const NANOS_PER_MS: i64 = 1_000_000; + self.hlc + .update_from_remote(pending.epoch_system_ms.saturating_mul(NANOS_PER_MS)); + self.epoch_system_ms = Some(pending.epoch_system_ms); + // Scope OLLP verification to this participant's staged leadership for the + // batch, then restore the resting (authoritative) state. + let prev_group_leader = self.ollp_is_group_leader; + self.ollp_is_group_leader = pending.is_group_leader; + // Install the captured resolve-time stamps into apply scratch; the + // batch consumes them for its bitemporal document puts and clears the + // scratch when it returns. `txn_id = None`: the synthetic overlay was + // already dropped above, so the stamps are threaded in directly here. + for (surrogate, stamp) in bitemporal_stamps { + self.active_bitemporal_stamps.insert(surrogate, stamp); + } + self.active_graph_system_from = graph_system_from; + // The read-set was already validated at stage time and drives the + // flush/drop decision; the replay itself carries no read-set to re-check. + // Scope the flush key so `record_batch_index_write_values` stages this + // batch's index tuples (the apply carries `wal_lsn: None`); the post-apply + // `RecordCalvinWriteVersions` op drains them at the replicated applied LSN. + self.calvin_flush_key = Some((epoch, position, vshard_id)); + let result = self.execute_transaction_batch( + task, + pending.tenant_id.as_u64(), + &pending.plans, + &[], + None, + ); + self.calvin_flush_key = None; + self.ollp_is_group_leader = prev_group_leader; + self.epoch_system_ms = None; + result + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::types::TenantId; + + use super::super::shared::CalvinExecCtx; + use super::super::shared::test_support::{make_task, point_insert_plan}; + + #[test] + fn calvin_flush_drops_synthetic_overlay() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + + let task = make_task(); + let tenant_id = TenantId::new(1); + let plans = vec![point_insert_plan("orders", "o1", 7)]; + let ctx = CalvinExecCtx { + epoch: 1, + position: 0, + epoch_system_ms: 0, + is_group_leader: true, + }; + let resp = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); + assert_eq!(resp.status, Status::Ok); + + let vshard_id = task.request.vshard_id.as_u32(); + let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); + assert!(core.txn_overlays.contains_key(&synthetic)); + + let flush_resp = core.execute_calvin_flush(&task, 1, 0); + assert_eq!(flush_resp.status, Status::Ok); + + assert!( + !core.txn_overlays.contains_key(&synthetic), + "flush must drop the synthetic overlay entry alongside commit_pending" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin/mod.rs b/nodedb/src/data/executor/handlers/control/calvin/mod.rs new file mode 100644 index 000000000..ead7dde45 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin/mod.rs @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Calvin deterministic executor handlers. +//! +//! Handler entry points (`CoreLoop` = [`crate::data::executor::core_loop::CoreLoop`]): +//! +//! - `CoreLoop::execute_calvin_execute_static`: static-set multi-shard txn +//! (the common case). It VALIDATES the read-set to compute the local commit +//! vote and STAGES the transaction's plans into the commit-pending buffer +//! WITHOUT mutating base or firing side effects, then returns the vote. +//! `CoreLoop::execute_calvin_flush` later replays the staged plans through +//! the durable apply funnel, or `CoreLoop::execute_calvin_drop` discards +//! them. +//! +//! - `CoreLoop::execute_calvin_execute_passive`: passive participant for a +//! dependent-read txn. Reads each declared key from the local engine and +//! returns a msgpack-encoded `Vec<(PassiveReadKeyId, Value)>` payload. The +//! Control Plane scheduler proposes a `CalvinReadResult` Raft entry after +//! receiving this response. +//! +//! - `CoreLoop::execute_calvin_execute_active`: active participant for a +//! dependent-read txn. Executes the physical plans with the injected read +//! values already resolved. Performs an OLLP verification hook: if the +//! active participant detects that the declared predicate no longer matches +//! the current engine state, it returns `OllpRetryRequired` WITHOUT writing. +//! The OLLP orchestrator on the Control Plane retries via `Inbox::submit`. +//! +//! The `CalvinApplied` WAL record is written on the Control Plane side (in the +//! scheduler's response path) after a successful response is received through +//! the SPSC bridge; not here in the Data Plane. + +mod active_passive; +mod discard; +mod flush; +mod shared; +mod static_stage; + +pub(in crate::data::executor) use shared::CalvinExecCtx; diff --git a/nodedb/src/data/executor/handlers/control/calvin/shared.rs b/nodedb/src/data/executor/handlers/control/calvin/shared.rs new file mode 100644 index 000000000..c8b566932 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin/shared.rs @@ -0,0 +1,180 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `CalvinExecCtx`, the helpers the static and active Calvin stage paths +//! share, and the test fixtures every concern file's tests build on. + +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; + +/// Execution context shared by both static and active Calvin handler variants. +/// +/// Bundles the epoch-scoped parameters that repeat across +/// `execute_calvin_execute_static` and `execute_calvin_execute_active`, +/// keeping each function's argument count within the lint budget. +pub(in crate::data::executor) struct CalvinExecCtx { + pub epoch: u64, + pub position: u32, + pub epoch_system_ms: i64, + pub is_group_leader: bool, +} + +impl CoreLoop { + /// Clean failed static staging and return an explicit abort vote. + /// Defensive removal clears document/KV and graph overlays plus their gauge. + pub(super) fn calvin_stage_failure( + &mut self, + task: &ExecutionTask, + epoch: u64, + position: u32, + vshard_id: u32, + error: E, + ) -> Response + where + E: Into, + { + self.commit_pending.remove(&(epoch, position, vshard_id)); + self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); + let mut response = self.response_error(task, error.into()); + // Scheduler treats this as a durable local abort vote and still waits + // for the authoritative global verdict before issuing any drop. + response.read_set_valid = Some(false); + response + } +} + +pub(super) fn calvin_panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String { + if let Some(message) = payload.downcast_ref::<&'static str>() { + (*message).to_owned() + } else if let Some(message) = payload.downcast_ref::() { + message.clone() + } else { + "".to_owned() + } +} + +#[cfg(test)] +pub(in crate::data::executor::handlers::control::calvin) mod test_support { + use std::time::{Duration, Instant}; + + use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan, TimeseriesOp}; + use nodedb_types::{QualifiedCollection, Surrogate, Value}; + + use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request}; + use crate::data::executor::core_loop::CoreLoop; + use crate::data::executor::doc_format; + use crate::data::executor::task::ExecutionTask; + use crate::types::{DatabaseId, RequestId, TenantId, TraceId, VShardId}; + + /// A minimal `ExecutionTask` homing to vShard 0, tenant 1, database + /// DEFAULT -- everything a Calvin static-execute handler needs beyond + /// its explicit `CalvinExecCtx` / `tenant_id` / `plans` arguments. + pub(in crate::data::executor::handlers::control::calvin) fn make_task() -> ExecutionTask { + let plan = PhysicalPlan::Document(DocumentOp::PointGet { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), + document_id: "y".into(), + surrogate: Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }); + let request = Request { + request_id: RequestId::new(1), + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan, + deadline: Instant::now() + Duration::from_secs(5), + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: crate::types::ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + }; + ExecutionTask::new(request) + } + + pub(in crate::data::executor::handlers::control::calvin) fn doc_value( + field: &str, + val: &str, + ) -> Vec { + let mut obj = std::collections::HashMap::new(); + obj.insert(field.to_string(), Value::String(val.into())); + zerompk::to_msgpack_vec(&Value::Object(obj)).unwrap() + } + + pub(in crate::data::executor::handlers::control::calvin) fn point_insert_plan( + collection: &str, + document_id: &str, + surrogate: u32, + ) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, collection), + document_id: document_id.to_string(), + value: doc_value("a", "1"), + if_absent: false, + surrogate: Surrogate::new(surrogate), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }) + } + + pub(in crate::data::executor::handlers::control::calvin) fn canonical_ilp_plan( + collection: &str, + lines: Vec<&str>, + tokens: Vec, + ) -> PhysicalPlan { + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, collection), + payload: zerompk::to_msgpack_vec(&lines).expect("canonical ILP payload"), + format: "ilp-msgpack".to_owned(), + wal_lsn: None, + surrogates: tokens.into_iter().map(Surrogate::new).collect(), + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }) + } + + pub(in crate::data::executor::handlers::control::calvin) fn bulk_delete_plan( + collection: &str, + predicted: Option>, + ) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::BulkDelete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, collection), + filters: Vec::new(), + returning: None, + ollp_predicted_surrogates: predicted, + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }) + } + + /// Seed a row directly into base storage (bypassing Calvin staging), the + /// pre-existing state the active-path OLLP verifier scans against. + pub(in crate::data::executor::handlers::control::calvin) fn seed_row( + core: &mut CoreLoop, + collection: &str, + surrogate: u32, + ) { + let doc_id = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + let body = doc_format::canonicalize_document_for_storage(&doc_value("a", "1")); + core.sparse + .put(DatabaseId::DEFAULT.as_u64(), 1, collection, &doc_id, &body) + .expect("seed row"); + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs b/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs new file mode 100644 index 000000000..bea3184d0 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs @@ -0,0 +1,471 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Static-set multi-shard Calvin transaction staging. + +use std::panic::{AssertUnwindSafe, catch_unwind}; + +use tracing::{debug, info_span}; + +use nodedb_types::calvin::VersionedReadEntry; + +use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::commit_pending::PendingCommit; +use crate::data::executor::task::ExecutionTask; +use crate::types::TenantId; +use nodedb_physical::physical_plan::PhysicalPlan; + +use crate::data::executor::handlers::control::calvin_txn_id::calvin_synthetic_txn_id; + +use super::shared::{CalvinExecCtx, calvin_panic_payload_to_string}; + +impl CoreLoop { + /// Validate a static-set Calvin transaction and stage it for commit. + /// + /// Computes the local commit vote by checking whether this participant's + /// slice of the transaction's LSN-versioned read-set is still current + /// against the per-core write versions, then STAGES the write plans into + /// the commit-pending buffer keyed by `(epoch, position)`. It performs NO + /// base mutation and fires NO side effects — nothing is observable until a + /// subsequent [`CoreLoop::execute_calvin_flush`] replays the staged plans + /// (or [`CoreLoop::execute_calvin_drop`] discards them). The response + /// carries the vote on `read_set_valid`; the deterministic time anchor and + /// leadership scope are captured with the staged plans and restored at + /// flush time (when the actual apply — and any time-dependent writes — run). + pub(in crate::data::executor) fn execute_calvin_execute_static( + &mut self, + task: &ExecutionTask, + ctx: CalvinExecCtx, + tenant_id: &TenantId, + plans: &[PhysicalPlan], + versioned_reads: &[VersionedReadEntry], + ) -> Response { + let CalvinExecCtx { + epoch, + position, + epoch_system_ms, + is_group_leader, + } = ctx; + let vshard_id = task.request.vshard_id.as_u32(); + debug!( + core = self.core_id, + epoch, + position, + epoch_system_ms, + vshard_id, + is_group_leader, + plan_count = plans.len(), + read_count = versioned_reads.len(), + "calvin stage for commit" + ); + let _stage_span = info_span!( + "executor_stage", + epoch, + position, + vshard = vshard_id, + tenant_id = tenant_id.as_u64(), + trace_id = ?task.request.trace_id, + ) + .entered(); + + // Derive the synthetic transaction identity before ANY mutation. A + // representational failure must not leave either a pending buffer or an + // overlay behind for a transaction that cannot later be resolved. + let synthetic_txn_id = match calvin_synthetic_txn_id(epoch, position, vshard_id) { + Ok(id) => id, + Err(error) => { + return self.calvin_stage_failure(task, epoch, position, vshard_id, error); + } + }; + + // Stage every plan before publishing `PendingCommit`. Panic isolation + // cleans both staging representations after any failure. + let stage_result = catch_unwind(AssertUnwindSafe(|| { + for plan in plans { + self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan)?; + // Test-only fault boundary after a potentially-mutating stage. + crate::fail_point!("calvin_static::during_overlay_stage"); + } + Ok::<(), ErrorCode>(()) + })); + match stage_result { + Ok(Ok(())) => {} + Ok(Err(error)) => { + return self.calvin_stage_failure(task, epoch, position, vshard_id, error); + } + Err(payload) => { + return self.calvin_stage_failure( + task, + epoch, + position, + vshard_id, + ErrorCode::Internal { + detail: format!( + "panic while staging static Calvin transaction: {}", + calvin_panic_payload_to_string(payload.as_ref()) + ), + }, + ); + } + } + + // Local commit vote: is this participant's slice of the read-set still + // current against the local write versions? Empty read-set is vacuously + // current. Read-only — no base mutation here. A stale-read false vote + // retains the fully staged state until the durable global verdict. + let vote = self.read_set_still_current(task, tenant_id.as_u64(), versioned_reads); + + // Publish only a fully staged transaction. The verdict-driven flush + // replays this raw plan buffer; a global abort drops it and its overlay. + self.commit_pending.insert( + (epoch, position, vshard_id), + PendingCommit { + plans: plans.to_vec(), + tenant_id: *tenant_id, + epoch_system_ms, + is_group_leader, + }, + ); + + Response { + request_id: task.request_id(), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: self.watermark, + error_code: None, + read_set_valid: Some(vote), + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + } + } +} + +#[cfg(test)] +mod tests { + use nodedb_physical::physical_plan::TimeseriesOp; + use nodedb_types::{QualifiedCollection, Surrogate}; + + use super::*; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::data::executor::doc_format; + use crate::data::executor::handlers::transaction::overlay::Staged; + use crate::types::DatabaseId; + + use super::super::shared::test_support::{ + bulk_delete_plan, canonical_ilp_plan, doc_value, make_task, point_insert_plan, + }; + + #[test] + fn calvin_execute_static_stages_point_insert_into_overlay() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + + let task = make_task(); + let tenant_id = TenantId::new(1); + let plans = vec![point_insert_plan("orders", "o1", 7)]; + let ctx = CalvinExecCtx { + epoch: 1, + position: 0, + epoch_system_ms: 0, + is_group_leader: true, + }; + + let resp = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); + assert_eq!(resp.status, Status::Ok); + + let vshard_id = task.request.vshard_id.as_u32(); + + // `commit_pending` is unchanged -- it still holds the raw plans that + // drive the base install at flush time. + assert!( + core.commit_pending.contains_key(&(1, 0, vshard_id)), + "commit_pending must still be populated exactly as before this unit" + ); + + // The synthetic overlay entry additionally holds the resolved + // post-image for the concrete point-write plan. + let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); + let coll_key = (DatabaseId::DEFAULT, tenant_id, "orders".to_string()); + let expected_body = doc_format::canonicalize_document_for_storage(&doc_value("a", "1")); + assert_eq!( + core.txn_overlays + .get(&synthetic) + .and_then(|o| o.get(&coll_key, 7)), + Some(&Staged::Put(expected_body)), + "the Calvin write plan must be staged into the synthetic-TxnId overlay" + ); + } + + #[test] + fn synthetic_id_failure_leaves_no_pending_or_overlay_state() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let task = make_task(); + let tenant_id = TenantId::new(1); + let epoch_outside_synthetic_range = 1_u64 << 33; + + let response = core.execute_calvin_execute_static( + &task, + CalvinExecCtx { + epoch: epoch_outside_synthetic_range, + position: 0, + epoch_system_ms: 0, + is_group_leader: true, + }, + &tenant_id, + &[point_insert_plan("orders", "o1", 7)], + &[], + ); + + assert_eq!(response.status, Status::Error); + assert_eq!(response.read_set_valid, Some(false)); + assert!(core.commit_pending.is_empty()); + assert!(core.txn_overlays.is_empty()); + assert!(core.graph_txn_overlays.is_empty()); + } + + #[test] + fn static_stage_error_cleans_all_prior_overlay_state_before_voting_abort() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let metrics = std::sync::Arc::new(crate::control::metrics::SystemMetrics::new()); + core.metrics = Some(metrics.clone()); + let gauge = || { + metrics + .active_txn_overlays + .load(std::sync::atomic::Ordering::Relaxed) + }; + let baseline = gauge(); + let task = make_task(); + let tenant_id = TenantId::new(1); + let ctx = CalvinExecCtx { + epoch: 9, + position: 3, + epoch_system_ms: 0, + is_group_leader: true, + }; + let vshard = task.request.vshard_id.as_u32(); + let synthetic = calvin_synthetic_txn_id(9, 3, vshard).unwrap(); + core.graph_txn_overlay_mut(synthetic); + assert_eq!( + gauge(), + baseline + 1, + "graph staging must increment the gauge" + ); + // The first plan adds a document overlay; the second is invalid without + // an OLLP prediction and must clean both overlays atomically. + let plans = vec![ + point_insert_plan("orders", "o1", 7), + bulk_delete_plan("orders", None), + ]; + + let response = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); + + assert_eq!(response.status, Status::Error); + assert_eq!(response.read_set_valid, Some(false)); + assert!(!core.commit_pending.contains_key(&(9, 3, vshard))); + assert!(!core.txn_overlays.contains_key(&synthetic)); + assert!(!core.graph_txn_overlays.contains_key(&synthetic)); + assert_eq!( + gauge(), + baseline, + "failed staging must restore the overlay gauge" + ); + } + + #[test] + fn static_stage_error_also_cleans_existing_graph_overlay_for_synthetic_id() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let task = make_task(); + let tenant_id = TenantId::new(1); + let vshard = task.request.vshard_id.as_u32(); + let synthetic = calvin_synthetic_txn_id(9, 5, vshard).unwrap(); + // Use the graph-overlay choke point so this regression also covers its + // gauge accounting without requiring a graph catalog fixture. + core.graph_txn_overlay_mut(synthetic); + + let response = core.execute_calvin_execute_static( + &task, + CalvinExecCtx { + epoch: 9, + position: 5, + epoch_system_ms: 0, + is_group_leader: true, + }, + &tenant_id, + &[bulk_delete_plan("orders", None)], + &[], + ); + + assert_eq!(response.read_set_valid, Some(false)); + assert!(!core.graph_txn_overlays.contains_key(&synthetic)); + } + + #[cfg(feature = "failpoints")] + #[test] + fn static_stage_panic_cleans_prior_overlay_state_before_voting_abort() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let metrics = std::sync::Arc::new(crate::control::metrics::SystemMetrics::new()); + core.metrics = Some(metrics.clone()); + let gauge = || { + metrics + .active_txn_overlays + .load(std::sync::atomic::Ordering::Relaxed) + }; + let baseline = gauge(); + let task = make_task(); + let tenant_id = TenantId::new(1); + let ctx = CalvinExecCtx { + epoch: 9, + position: 4, + epoch_system_ms: 0, + is_group_leader: true, + }; + let vshard = task.request.vshard_id.as_u32(); + let synthetic = calvin_synthetic_txn_id(9, 4, vshard).unwrap(); + core.graph_txn_overlay_mut(synthetic); + assert_eq!( + gauge(), + baseline + 1, + "graph staging must increment the gauge" + ); + let _fail = crate::fail_point::FailGuard::install( + "calvin_static::during_overlay_stage", + crate::fail_point::FailAction::Panic, + ); + + let response = core.execute_calvin_execute_static( + &task, + ctx, + &tenant_id, + &[point_insert_plan("orders", "o1", 7)], + &[], + ); + + assert_eq!(response.status, Status::Error); + assert_eq!(response.read_set_valid, Some(false)); + assert!(!core.commit_pending.contains_key(&(9, 4, vshard))); + assert!(!core.txn_overlays.contains_key(&synthetic)); + assert!(!core.graph_txn_overlays.contains_key(&synthetic)); + assert_eq!( + gauge(), + baseline, + "panic cleanup must restore the overlay gauge" + ); + } + + #[test] + fn static_calvin_ilp_staging_is_all_or_abort() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let task = make_task(); + let tenant = TenantId::new(1); + let vshard = task.request.vshard_id.as_u32(); + let synthetic = calvin_synthetic_txn_id(21, 1, vshard).expect("synthetic transaction id"); + + let malformed = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cpu"), + payload: vec![0xc1], + format: "ilp-msgpack".to_owned(), + wal_lsn: None, + surrogates: vec![Surrogate::new(1)], + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }); + let failed = core.execute_calvin_execute_static( + &task, + CalvinExecCtx { + epoch: 21, + position: 1, + epoch_system_ms: 0, + is_group_leader: true, + }, + &tenant, + &[malformed], + &[], + ); + assert_eq!(failed.read_set_valid, Some(false)); + assert!(matches!( + failed.error_code.as_deref(), + Some(ErrorCode::RejectedPrevalidation { .. }) + )); + assert!(!core.commit_pending.contains_key(&(21, 1, vshard))); + assert!(!core.txn_overlays.contains_key(&synthetic)); + + let valid = canonical_ilp_plan("cpu", vec!["cpu value=1i", "cpu value=2i"], vec![1, 2]); + let staged = core.execute_calvin_execute_static( + &task, + CalvinExecCtx { + epoch: 21, + position: 1, + epoch_system_ms: 0, + is_group_leader: true, + }, + &tenant, + &[valid], + &[], + ); + assert_eq!(staged.read_set_valid, Some(true)); + assert_eq!( + core.txn_overlays + .get(&synthetic) + .map(|overlay| overlay.len()), + Some(2) + ); + + let mismatch = canonical_ilp_plan("cpu", vec!["memory value=1i"], vec![3]); + let failed = core.execute_calvin_execute_static( + &task, + CalvinExecCtx { + epoch: 21, + position: 2, + epoch_system_ms: 0, + is_group_leader: true, + }, + &tenant, + &[mismatch], + &[], + ); + let mismatch_id = calvin_synthetic_txn_id(21, 2, vshard).expect("synthetic transaction id"); + assert_eq!(failed.read_set_valid, Some(false)); + assert!(!core.commit_pending.contains_key(&(21, 2, vshard))); + assert!(!core.txn_overlays.contains_key(&mismatch_id)); + + core.ts_tuning.max_tag_cardinality = 1; + let overflow = canonical_ilp_plan( + "cpu", + vec!["cpu,host=a value=1i", "cpu,host=b value=2i"], + vec![4, 5], + ); + let failed = core.execute_calvin_execute_static( + &task, + CalvinExecCtx { + epoch: 21, + position: 3, + epoch_system_ms: 0, + is_group_leader: true, + }, + &tenant, + &[overflow], + &[], + ); + let overflow_id = calvin_synthetic_txn_id(21, 3, vshard).expect("synthetic transaction id"); + assert_eq!(failed.status, Status::Error); + assert_eq!(failed.read_set_valid, Some(false)); + assert!(matches!( + failed.error_code.as_deref(), + Some(ErrorCode::RejectedPrevalidation { .. }) + )); + assert!( + !core.commit_pending.contains_key(&(21, 3, vshard)), + "a rejected stage cannot reach the TransactionRedo-producing flush path" + ); + assert!(!core.txn_overlays.contains_key(&overflow_id)); + } +} From 712b1b9179a3aae904a64ffeadcf2ada64daccf7 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 03:11:12 +0800 Subject: [PATCH 14/64] feat(wal): apply each replicated proposal exactly once A proposal re-proposed after a leader change can commit at two Raft log indexes. Applying both copies double-counts every non-idempotent effect: a materialized-sum fold, a columnar append, a timeseries ingest. RecordHeader gains an `apply_key` field (replacing the unused `reserved` bytes) carrying the idempotency key of the proposal whose apply appended the record, durable in the same write and covered by the CRC. A new `ProposalApplied` record marks an apply that writes no record of its own. `ProposalLedger` tracks applied keys per data group, bounded in memory and recovered from WAL replay on restart, so the apply loop can recognize and skip a duplicate copy and answer its waiter with the first copy's outcome. Building on this, committed transactions now replicate as a single `TransactionRedo` record carrying post-images, applied identically on every replica via `ApplyTransactionRedo` instead of re-resolving each sub-plan per node. Materialized-sum resolution travels with the redo as `RedoSumTargets` so replicas fold source writes into their targets without re-running the join. Columnar row updates now carry the old row's cross-engine surrogate forward to the replacement row, and the vector, columnar, and array engines gain rollback/truncate primitives so a partially-applied write can be withdrawn cleanly. The `resolve/columnar.rs` transaction resolver is renamed to `resolve/timeseries.rs` to match the timeseries-specific resolution it now holds, with the columnar resolution split out separately. A new cluster test verifies a proposal committed twice applies its effect once. --- .../tests/cluster_common/calvin_test_node.rs | 34 +- .../tests/cluster_common/mod.rs | 2 +- .../cases/calvin_3node_normal.rs | 12 +- .../cases/calvin_3node_shard_failover.rs | 10 +- .../cluster_suite/cases/calvin_e2e_ollp.rs | 21 +- .../cluster_suite/cases/calvin_e2e_pgwire.rs | 12 +- .../tests/common_suite/cases/mod.rs | 1 + .../cases/multi_replica_data_groups.rs | 243 ++++ .../proposal_committed_twice_applies_once.rs | 160 +++ nodedb-columnar/src/mutation/engine.rs | 58 + nodedb-columnar/src/mutation/write.rs | 24 +- .../src/physical_plan/document/mod.rs | 2 +- .../src/physical_plan/document/sum_target.rs | 25 + nodedb-physical/src/physical_plan/meta.rs | 18 + nodedb-physical/src/physical_plan/mod.rs | 4 +- .../node/lifecycle/spawn_full.rs | 1 + nodedb-test-support/src/core_loop_runner.rs | 10 +- .../src/native_harness/server.rs | 1 + .../src/pgwire_harness/multicore.rs | 1 + .../src/pgwire_harness/restart.rs | 2 + .../src/pgwire_harness/start.rs | 2 + nodedb-types/src/columnar/dml_wal_record.rs | 1 + nodedb-types/src/columnar/image_wal_record.rs | 114 ++ nodedb-types/src/columnar/mod.rs | 2 + nodedb-types/src/columnar/wal_record.rs | 9 + nodedb-vector/src/collection/mod.rs | 2 + nodedb-vector/src/collection/payload_index.rs | 16 + nodedb-vector/src/collection/rollback.rs | 244 ++++ nodedb-vector/src/flat.rs | 15 + nodedb-vector/src/ivf.rs | 22 + nodedb-wal/src/lazy_reader.rs | 2 +- nodedb-wal/src/lib.rs | 5 +- nodedb-wal/src/mmap_reader/reader.rs | 4 +- nodedb-wal/src/reader.rs | 2 +- nodedb-wal/src/record/header.rs | 35 +- nodedb-wal/src/record/mod.rs | 2 +- nodedb-wal/src/record/types.rs | 16 + nodedb-wal/src/record/wal_record.rs | 27 +- nodedb-wal/src/segmented.rs | 25 +- nodedb-wal/src/writer/core.rs | 51 +- .../src/control/array_sync/raft_apply/cell.rs | 1 + .../control/array_sync/raft_apply/common.rs | 5 + .../src/control/array_sync/raft_apply/op.rs | 1 + .../control/cluster/array_executor/write.rs | 5 +- .../distributed_applier/apply_loop/driver.rs | 62 +- .../distributed_applier/apply_loop/mod.rs | 6 + .../apply_loop/proposal_gate.rs | 100 ++ .../apply_loop/transaction_redo.rs | 92 ++ .../apply_loop/write_dispatch.rs | 30 +- nodedb/src/control/distributed_applier/mod.rs | 2 + .../distributed_applier/proposal_ledger.rs | 260 +++++ .../distributed_applier/propose_tracker.rs | 2 +- nodedb/src/control/gateway/router.rs | 82 +- .../src/control/planner/rls_injection/meta.rs | 3 +- .../rls_injection/permission_tree/meta.rs | 3 +- .../security/identity/plan_permission.rs | 3 + .../control/server/dispatch_utils/dispatch.rs | 15 +- .../dispatch_utils/durability_barrier.rs | 23 +- .../src/control/server/dispatch_utils/mod.rs | 1 + .../submit_write/funnel/driver.rs | 21 + .../submit_write/funnel/response.rs | 27 +- .../submit_write/funnel/wal_append.rs | 23 +- .../dispatch_utils/submit_write/params.rs | 10 +- .../control/server/dispatch_utils/types.rs | 10 +- .../server/dispatch_utils/write_abort.rs | 63 +- .../server/exchange/all_cores/dispatch.rs | 3 +- .../server/native/dispatch/raw_dispatch.rs | 10 +- .../server/native/dispatch/sql_gateway.rs | 10 +- .../server/native/dispatch/transaction.rs | 4 + .../server/pgwire/handler/dispatch/local.rs | 5 +- .../pgwire/handler/transaction_cmds/commit.rs | 4 + .../control/server/shared/returning/inject.rs | 3 +- .../shared/session/commit/single_shard.rs | 187 +-- .../server/shared/session/lifecycle.rs | 4 + .../control/server/shared/session/outcome.rs | 4 + .../server/shared/session/savepoint_ops.rs | 4 + .../predicate/txn_buffering/classify.rs | 3 +- .../control/server/wal_dispatch/columnar.rs | 23 +- .../src/control/server/wal_dispatch/core.rs | 87 +- .../src/control/server/wal_dispatch/crdt.rs | 71 +- nodedb/src/control/server/wal_dispatch/mod.rs | 8 +- .../src/control/server/wal_dispatch/text.rs | 46 +- .../control/server/wal_dispatch/timeseries.rs | 256 ++++- .../server/wal_dispatch/write_set_redo.rs | 8 +- .../surrogate/assign/bind_plan/binder.rs | 59 +- .../surrogate/assign/bind_plan/carried.rs | 77 ++ .../control/surrogate/assign/bind_plan/mod.rs | 2 + nodedb/src/control/surrogate/assign/mod.rs | 5 +- nodedb/src/control/surrogate/mod.rs | 3 +- nodedb/src/control/system_txn/data_plane.rs | 4 + nodedb/src/control/wal_catchup.rs | 59 +- .../control/wal_replication/decode/entry.rs | 26 +- .../decode/entry_columnar_family.rs | 49 + .../src/control/wal_replication/decode/mod.rs | 3 + .../decode/transaction_redo.rs | 132 +++ .../wal_replication/encode/columnar.rs | 37 +- .../encode/entry_columnar_family.rs | 13 +- .../src/control/wal_replication/encode/mod.rs | 4 + .../encode/transaction_redo.rs | 42 + nodedb/src/control/wal_replication/mod.rs | 8 +- .../wal_replication/transaction_redo/apply.rs | 73 ++ .../transaction_redo/collections.rs | 68 ++ .../wal_replication/transaction_redo/mod.rs | 17 + .../transaction_redo/payload.rs | 66 ++ .../transaction_redo/sum_targets.rs | 176 +++ .../src/control/wal_replication/types/mod.rs | 3 + .../wal_replication/types/replicated_write.rs | 24 + .../types/transaction_redo_wire.rs | 75 ++ nodedb/src/data/executor/array_checkpoint.rs | 69 ++ .../data/executor/columnar_checkpoint/load.rs | 1 + .../executor/columnar_checkpoint/write.rs | 7 +- .../core_loop/checkpoint_floors/init.rs | 4 + .../core_loop/checkpoint_floors/state.rs | 19 + .../src/data/executor/core_loop/event_emit.rs | 35 +- .../data/executor/core_loop/maintenance.rs | 8 + nodedb/src/data/executor/core_loop/open.rs | 2 + nodedb/src/data/executor/core_loop/state.rs | 17 +- nodedb/src/data/executor/dispatch/meta.rs | 18 + .../handlers/columnar_mutation_apply.rs | 96 +- .../handlers/control/calvin/active_passive.rs | 14 +- .../handlers/control/calvin/static_stage.rs | 6 +- .../handlers/control/calvin_overlay_stage.rs | 45 +- .../handlers/control/calvin_resolve.rs | 82 +- .../executor/handlers/control/crdt_doc.rs | 35 +- .../handlers/control/crdt_materialize.rs | 72 +- .../src/data/executor/handlers/kv/atomic.rs | 118 +- .../handlers/kv/resolve/atomic_ops.rs | 39 +- nodedb/src/data/executor/handlers/mod.rs | 1 + .../executor/handlers/point/apply_delete.rs | 16 +- .../handlers/point/apply_put/enforce.rs | 20 +- .../executor/handlers/point/apply_put/mod.rs | 3 +- .../handlers/point/apply_put/vector/types.rs | 1 + .../executor/handlers/timeseries/ingest.rs | 46 +- .../handlers/timeseries/ingest_dispatch.rs | 27 +- .../data/executor/handlers/timeseries/mod.rs | 2 + .../handlers/timeseries/redo_ingest.rs | 64 ++ .../handlers/timeseries/resolve_ingest.rs | 76 +- .../data/executor/handlers/timeseries_wal.rs | 522 ++++----- .../handlers/timeseries_wal_decode.rs | 83 +- .../handlers/timeseries_wal_payload.rs | 258 +++++ .../executor/handlers/transaction/batch.rs | 7 +- .../transaction/index_write_values.rs | 11 + .../data/executor/handlers/transaction/mod.rs | 5 +- .../handlers/transaction/overlay/staged.rs | 11 + .../transaction/overlay/staged_sidecar.rs | 73 +- .../handlers/transaction/redo_apply/cover.rs | 379 ++++++ .../transaction/redo_apply/document.rs | 341 ++++++ .../handlers/transaction/redo_apply/entry.rs | 580 ++++++++++ .../handlers/transaction/redo_apply/events.rs | 159 +++ .../handlers/transaction/redo_apply/mod.rs | 28 + .../handlers/transaction/redo_apply/passes.rs | 151 +++ .../handlers/transaction/redo_apply/settle.rs | 148 +++ .../handlers/transaction/redo_apply/state.rs | 167 +++ .../transaction/redo_apply/sub_ops.rs | 239 ++++ .../transaction/redo_apply/validate.rs | 263 +++++ .../handlers/transaction/resolve/classify.rs | 171 +++ .../handlers/transaction/resolve/columnar.rs | 195 ---- .../transaction/resolve/columnar_image.rs | 317 ++++++ .../handlers/transaction/resolve/crdt.rs | 80 ++ .../handlers/transaction/resolve/document.rs | 89 +- .../handlers/transaction/resolve/entry.rs | 1011 +++++++++++------ .../handlers/transaction/resolve/graph.rs | 11 +- .../handlers/transaction/resolve/mod.rs | 10 +- .../handlers/transaction/resolve/text.rs | 48 + .../transaction/resolve/timeseries.rs | 120 ++ .../handlers/transaction/resolve/vector.rs | 211 ++-- .../transaction/resolve/vector_direct.rs | 174 +++ .../transaction/resolve/vector_primary.rs | 194 ++++ .../transaction/stage_write/dispatch.rs | 4 + .../handlers/transaction/stage_write/mod.rs | 2 + .../transaction/stage_write/stage_columnar.rs | 106 +- .../stage_write/stage_columnar_base_key.rs | 70 ++ .../stage_write/stage_columnar_dml.rs | 32 +- .../stage_write/stage_columnar_family.rs | 4 +- .../stage_columnar_resolved_dml.rs | 18 +- .../stage_write/stage_kv_atomic.rs | 45 +- .../stage_write/stage_timeseries.rs | 21 +- .../stage_write/stage_timeseries_now.rs | 34 + .../transaction/stage_write/stage_vector.rs | 5 +- .../transaction/sub_plan_doc/delete.rs | 120 +- .../handlers/transaction/sub_plan_doc/put.rs | 90 +- .../handlers/transaction/sub_plan_kv.rs | 89 +- .../transaction/sub_plan_kv_atomics.rs | 16 +- .../transaction/sub_plan_kv_writes.rs | 75 +- .../transaction/undo/crdt_collection.rs | 78 ++ .../transaction/undo/document_outcome.rs | 168 +++ .../handlers/transaction/undo/entry.rs | 95 +- .../handlers/transaction/undo/fts_doc.rs | 82 ++ .../handlers/transaction/undo/graph_node.rs | 25 + .../executor/handlers/transaction/undo/kv.rs | 270 +++-- .../executor/handlers/transaction/undo/mod.rs | 7 + .../handlers/transaction/undo/rollback.rs | 56 + .../handlers/transaction/undo/spatial_row.rs | 133 +++ .../handlers/transaction/undo/sync_hwm.rs | 51 + .../handlers/transaction/undo/timeseries.rs | 34 +- .../transaction/undo/vector_truncate.rs | 120 ++ .../handlers/transaction/undo/vector_write.rs | 224 ++++ nodedb/src/data/executor/handlers/vector.rs | 8 +- .../executor/handlers/vector_direct_row.rs | 5 +- .../data/executor/handlers/vector_multi.rs | 5 +- .../data/executor/handlers/vector_write.rs | 5 +- .../src/data/executor/kv_checkpoint/load.rs | 1 + .../src/data/executor/kv_checkpoint/write.rs | 6 +- nodedb/src/data/executor/mod.rs | 7 + nodedb/src/data/executor/replay_policy.rs | 181 +++ nodedb/src/data/executor/replay_task.rs | 51 + nodedb/src/data/executor/task.rs | 20 +- .../data/executor/vector_checkpoint/load.rs | 1 + .../data/executor/vector_checkpoint/write.rs | 9 +- nodedb/src/data/executor/wal_replay/array.rs | 209 +++- .../src/data/executor/wal_replay/crdt_doc.rs | 23 +- .../src/data/executor/wal_replay/crdt_list.rs | 100 +- .../data/executor/wal_replay/crdt_ordered.rs | 63 +- nodedb/src/data/executor/wal_replay/kv.rs | 47 +- nodedb/src/data/executor/wal_replay/kv_put.rs | 19 +- .../data/executor/wal_replay_columnar_dml.rs | 78 +- .../executor/wal_replay_columnar_image.rs | 330 ++++++ .../executor/wal_replay_columnar_truncate.rs | 108 +- nodedb/src/data/executor/wal_replay_fts.rs | 131 ++- .../data/executor/wal_replay_graph_labels.rs | 64 +- .../src/data/executor/wal_replay_kv_atomic.rs | 59 +- .../data/executor/wal_replay_redo_document.rs | 198 +++- .../data/executor/wal_replay_redo_graph.rs | 38 +- .../src/data/executor/wal_replay_spatial.rs | 169 ++- nodedb/src/data/executor/wal_replay_vector.rs | 270 ++--- .../data/executor/wal_replay_vector_delete.rs | 149 +++ .../data/executor/wal_replay_vector_direct.rs | 107 +- .../executor/wal_replay_vector_extended.rs | 192 ++-- .../data/executor/wal_replay_vector_redo.rs | 129 +++ .../executor/wal_replay_vector_resolved.rs | 49 +- .../data/executor/wal_replay_vector_sparse.rs | 163 +++ .../data/executor/wal_replay_vector_task.rs | 52 + nodedb/src/data/runtime/spawn.rs | 3 + .../src/engine/array/memtable/tile_buffer.rs | 17 + nodedb/src/engine/array/mod.rs | 2 + nodedb/src/engine/array/rollback.rs | 92 ++ nodedb/src/engine/kv/engine/mod.rs | 1 + nodedb/src/engine/kv/engine/reads.rs | 102 +- nodedb/src/engine/kv/engine_atomic.rs | 129 ++- nodedb/src/engine/kv/engine_atomic_compute.rs | 463 +++++--- nodedb/src/engine/kv/mod.rs | 8 +- .../src/engine/sparse/inverted/doc_image.rs | 130 +++ nodedb/src/engine/sparse/inverted/indexing.rs | 2 +- nodedb/src/engine/sparse/inverted/mod.rs | 2 + nodedb/src/engine/vector/sparse/index.rs | 93 ++ nodedb/src/engine/vector/sparse/mod.rs | 2 +- nodedb/src/event/wal_replay.rs | 4 +- nodedb/src/wal/manager/append.rs | 55 +- nodedb/src/wal/manager/append_metadata.rs | 28 +- nodedb/src/wal/manager/append_transaction.rs | 13 + nodedb/src/wal/mod.rs | 4 + nodedb/src/wal/redo/replay.rs | 21 +- nodedb/src/wal/replay/surrogate/dispatch.rs | 4 +- nodedb/src/wal/replay/sync_hwm/advance.rs | 4 +- nodedb/src/wal/timeseries_batch_payload.rs | 172 +++ .../cases/native_txn_commit_visibility.rs | 10 +- nodedb/tests/wire/cases/mod.rs | 1 + ...ql_transactions_commit_point_visibility.rs | 77 ++ .../cases/transactional_ddl_compensation.rs | 10 +- 259 files changed, 14393 insertions(+), 3027 deletions(-) create mode 100644 nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs create mode 100644 nodedb-types/src/columnar/image_wal_record.rs create mode 100644 nodedb-vector/src/collection/rollback.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs create mode 100644 nodedb/src/control/distributed_applier/proposal_ledger.rs create mode 100644 nodedb/src/control/surrogate/assign/bind_plan/carried.rs create mode 100644 nodedb/src/control/wal_replication/decode/transaction_redo.rs create mode 100644 nodedb/src/control/wal_replication/encode/transaction_redo.rs create mode 100644 nodedb/src/control/wal_replication/transaction_redo/apply.rs create mode 100644 nodedb/src/control/wal_replication/transaction_redo/collections.rs create mode 100644 nodedb/src/control/wal_replication/transaction_redo/mod.rs create mode 100644 nodedb/src/control/wal_replication/transaction_redo/payload.rs create mode 100644 nodedb/src/control/wal_replication/transaction_redo/sum_targets.rs create mode 100644 nodedb/src/control/wal_replication/types/transaction_redo_wire.rs create mode 100644 nodedb/src/data/executor/handlers/timeseries/redo_ingest.rs create mode 100644 nodedb/src/data/executor/handlers/timeseries_wal_payload.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/sub_ops.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/validate.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/classify.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/columnar.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/columnar_image.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/crdt.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/text.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/timeseries.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/vector_direct.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/vector_primary.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_base_key.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries_now.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/crdt_collection.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/fts_doc.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/spatial_row.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/sync_hwm.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/vector_truncate.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs create mode 100644 nodedb/src/data/executor/replay_policy.rs create mode 100644 nodedb/src/data/executor/replay_task.rs create mode 100644 nodedb/src/data/executor/wal_replay_columnar_image.rs create mode 100644 nodedb/src/data/executor/wal_replay_vector_delete.rs create mode 100644 nodedb/src/data/executor/wal_replay_vector_redo.rs create mode 100644 nodedb/src/data/executor/wal_replay_vector_sparse.rs create mode 100644 nodedb/src/data/executor/wal_replay_vector_task.rs create mode 100644 nodedb/src/engine/array/rollback.rs create mode 100644 nodedb/src/engine/sparse/inverted/doc_image.rs create mode 100644 nodedb/src/wal/timeseries_batch_payload.rs create mode 100644 nodedb/tests/wire/cases/sql_transactions_commit_point_visibility.rs diff --git a/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs b/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs index 1827537c2..b421c72a3 100644 --- a/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs +++ b/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs @@ -27,6 +27,8 @@ //! for shutdown. //! - `add_vshard_sender(vshard_id, sender)` — wire a per-vshard channel //! into the state machine so tests can assert fan-out. +//! - `try_recv_txn(rx)` — read the next sequenced txn a fan-out channel +//! holds. #![allow(dead_code)] // Not every test file uses every helper. @@ -130,24 +132,15 @@ impl CalvinTestNode { /// Register a per-vshard output sender so the state machine can /// fan out sequenced transactions to the receiving test code. - pub fn add_vshard_sender(&self, vshard_id: u32, sender: mpsc::Sender) { - // The state machine now fans out `SchedulerInput`; these tests assert on - // the sequenced-txn stream, so adapt: forward only `Txn` payloads to the - // caller's `SequencedTxn` channel (reservation inputs don't occur here). - let (adapt_tx, mut adapt_rx) = mpsc::channel::(512); - tokio::spawn(async move { - while let Some(input) = adapt_rx.recv().await { - if let SchedulerInput::Txn(txn) = input - && sender.send(txn).await.is_err() - { - break; - } - } - }); + /// + /// The state machine sends into `sender` itself, inside `apply`, before + /// it advances `last_applied_epoch`. A test that waits for the epoch can + /// then read the fan-out with `try_recv_txn` and no further wait. + pub fn add_vshard_sender(&self, vshard_id: u32, sender: mpsc::Sender) { self.state_machine .lock() .unwrap_or_else(|p| p.into_inner()) - .set_vshard_sender(vshard_id, adapt_tx); + .set_vshard_sender(vshard_id, sender); } /// Start the sequencer service epoch-ticker task on this node. @@ -417,3 +410,14 @@ pub async fn wait_for_sequencer_leader( tokio::time::sleep(step).await; } } + +/// The next sequenced txn `rx` holds, skipping any other scheduler input. +/// `None` when the channel holds no txn. +pub fn try_recv_txn(rx: &mut mpsc::Receiver) -> Option { + while let Ok(input) = rx.try_recv() { + if let SchedulerInput::Txn(txn) = input { + return Some(txn); + } + } + None +} diff --git a/nodedb-cluster-tests/tests/cluster_common/mod.rs b/nodedb-cluster-tests/tests/cluster_common/mod.rs index 4e828b2fc..4657be030 100644 --- a/nodedb-cluster-tests/tests/cluster_common/mod.rs +++ b/nodedb-cluster-tests/tests/cluster_common/mod.rs @@ -7,6 +7,6 @@ pub mod rebalancer; pub mod test_node; pub use calvin_test_node::{ - CalvinApplier, CalvinTestNode, spawn_with_sequencer, wait_for_sequencer_leader, + CalvinApplier, CalvinTestNode, spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader, }; pub use test_node::{NoopApplier, TestNode, test_transport, wait_for}; diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs index a052d53c9..8fbba3273 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs @@ -19,7 +19,7 @@ use std::time::Duration; use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, - types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet}, + types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; use nodedb_types::{ TenantId, @@ -27,7 +27,7 @@ use nodedb_types::{ }; use tokio::sync::mpsc; -use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; +use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; /// Find two collection names that hash to distinct vshards. fn two_distinct_collections() -> (String, String) { @@ -95,8 +95,8 @@ async fn sequencer_normal_path_commit_on_all_replicas() { let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &tx_a).as_u32(); let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b_name).as_u32(); - let mut vshard_rxs_a: Vec> = Vec::new(); - let mut vshard_rxs_b: Vec> = Vec::new(); + let mut vshard_rxs_a: Vec> = Vec::new(); + let mut vshard_rxs_b: Vec> = Vec::new(); for node in &nodes { let (tx_a_ch, rx_a) = mpsc::channel(64); let (tx_b_ch, rx_b) = mpsc::channel(64); @@ -149,8 +149,8 @@ async fn sequencer_normal_path_commit_on_all_replicas() { .zip(vshard_rxs_b.iter_mut()) .enumerate() { - let got_a = rx_a.try_recv().is_ok(); - let got_b = rx_b.try_recv().is_ok(); + let got_a = try_recv_txn(rx_a).is_some(); + let got_b = try_recv_txn(rx_b).is_some(); assert!( got_a || got_b, "node {}: neither vshard receiver got the txn fan-out", diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs index fe7ad0979..9238b8237 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs @@ -43,7 +43,7 @@ use std::time::Duration; use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, - types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet}, + types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; use nodedb_types::{ TenantId, @@ -51,7 +51,7 @@ use nodedb_types::{ }; use tokio::sync::mpsc; -use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; +use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; // ── Helpers ────────────────────────────────────────────────────────────────── @@ -133,8 +133,8 @@ async fn scheduler_catchup_via_raft_log_replay() { let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); - let mut vshard_rxs_a: Vec> = Vec::new(); - let mut vshard_rxs_b: Vec> = Vec::new(); + let mut vshard_rxs_a: Vec> = Vec::new(); + let mut vshard_rxs_b: Vec> = Vec::new(); for node in &nodes { let (tx_a, rx_a) = mpsc::channel(128); let (tx_b, rx_b) = mpsc::channel(128); @@ -246,7 +246,7 @@ async fn scheduler_catchup_via_raft_log_replay() { // Drain whatever arrived — we care that the routing worked, not the count. let mut total_received = 0usize; for rx in vshard_rxs_a.iter_mut().chain(vshard_rxs_b.iter_mut()) { - while rx.try_recv().is_ok() { + while try_recv_txn(rx).is_some() { total_received += 1; } } diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs index 1ff200f81..844158677 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs @@ -37,7 +37,7 @@ use std::time::Duration; use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, - types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet}, + types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; use nodedb_types::{ TenantId, @@ -45,7 +45,7 @@ use nodedb_types::{ }; use tokio::sync::mpsc; -use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; +use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; /// Find two collection names that hash to distinct vshards. /// @@ -104,14 +104,17 @@ fn make_ollp_tx_class( .expect("valid multi-vshard OLLP TxClass") } -/// Assert: one specific vshard channel received at least one SequencedTxn. +/// Assert: one specific vshard channel received at least one sequenced txn. +/// +/// The state machine sends into the channel inside `apply`, before it +/// advances the epoch the caller waited for, so the txn is already there. fn assert_fan_out_received( - rx: &mut mpsc::Receiver, + rx: &mut mpsc::Receiver, vshard_id: u32, replica_idx: usize, ) { assert!( - rx.try_recv().is_ok(), + try_recv_txn(rx).is_some(), "replica {replica_idx}: vshard {vshard_id} fan-out channel received no txn" ); } @@ -139,8 +142,8 @@ async fn ollp_bulk_update_txclass_admitted_and_fanned_out() { let vs_ollp = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_ollp).as_u32(); // Wire per-vshard fan-out receivers on every replica. - let mut rxs_static: Vec> = Vec::new(); - let mut rxs_ollp: Vec> = Vec::new(); + let mut rxs_static: Vec> = Vec::new(); + let mut rxs_ollp: Vec> = Vec::new(); for node in &nodes { let (tx_s, rx_s) = mpsc::channel(64); let (tx_o, rx_o) = mpsc::channel(64); @@ -191,8 +194,8 @@ async fn ollp_bulk_update_txclass_admitted_and_fanned_out() { // Simulate an OLLP retry: concurrent insert added surrogate 4 to col_ollp. // Re-wire fresh fan-out receivers and re-submit with the corrected set. - let mut retry_rxs_static: Vec> = Vec::new(); - let mut retry_rxs_ollp: Vec> = Vec::new(); + let mut retry_rxs_static: Vec> = Vec::new(); + let mut retry_rxs_ollp: Vec> = Vec::new(); for node in &nodes { let (tx_s, rx_s) = mpsc::channel(64); let (tx_o, rx_o) = mpsc::channel(64); diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs index 2be2ae5a6..e74290fff 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs @@ -20,7 +20,7 @@ use std::time::Duration; use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, - types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet}, + types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; use nodedb_types::{ TenantId, @@ -28,7 +28,7 @@ use nodedb_types::{ }; use tokio::sync::mpsc; -use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; +use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; /// Find two collection names that hash to distinct vshards. /// @@ -99,8 +99,8 @@ async fn multi_vshard_insert_via_sequencer_admitted_and_replicated() { let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); - let mut fan_out_rxs_a: Vec> = Vec::new(); - let mut fan_out_rxs_b: Vec> = Vec::new(); + let mut fan_out_rxs_a: Vec> = Vec::new(); + let mut fan_out_rxs_b: Vec> = Vec::new(); for node in &nodes { let (tx_a, rx_a) = mpsc::channel(64); let (tx_b, rx_b) = mpsc::channel(64); @@ -153,8 +153,8 @@ async fn multi_vshard_insert_via_sequencer_admitted_and_replicated() { .zip(fan_out_rxs_b.iter_mut()) .enumerate() { - let got_a = rx_a.try_recv().is_ok(); - let got_b = rx_b.try_recv().is_ok(); + let got_a = try_recv_txn(rx_a).is_some(); + let got_b = try_recv_txn(rx_b).is_some(); assert!( got_a || got_b, "replica {i}: neither vshard fan-out channel received the txn" diff --git a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs index bd7e362ef..7842c819a 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs @@ -69,6 +69,7 @@ mod node_labels_replicate_to_followers; mod pgwire_gateway_migration; mod planner_local_only; mod prepared_cache_invalidation; +mod proposal_committed_twice_applies_once; mod resp_gateway_migration; mod retention_policy_cross_node; mod scope_quota_cross_node; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs b/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs index 5e4f7f45a..6dbc6fa57 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs @@ -213,3 +213,246 @@ async fn data_group_is_multi_replica_and_survives_leader_loss() { node.shutdown().await; } } + +const TXN_COLL: &str = "mr_txn_commit"; + +/// Rows every replica holds once the explicit transaction commits. +const TXN_EXPECTED: [(&str, &str); 4] = [ + ("seed-0", "updated"), + ("seed-1", "seed"), + ("txn-0", "inserted-0"), + ("txn-1", "inserted-1"), +]; + +/// Leader of `group_id` as `node`'s routing table records it, or `0`. +fn routing_leader(node: &common::cluster_harness::TestClusterNode, group_id: u64) -> u64 { + let routing = node + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + routing.group_info(group_id).map(|i| i.leader).unwrap_or(0) +} + +/// True when `node` leads `group_id` by its own Raft state. +fn leads(node: &common::cluster_harness::TestClusterNode, group_id: u64) -> bool { + node.all_group_leaders().contains(&(group_id, node.node_id)) +} + +/// Index of the node that leads `group_id` by its own Raft state and by every +/// node's routing table. +async fn group_leader_index( + nodes: &[common::cluster_harness::TestClusterNode], + group_id: u64, +) -> usize { + let deadline = Instant::now() + Duration::from_secs(20); + loop { + let found = nodes.iter().position(|n| { + leads(n, group_id) + && nodes + .iter() + .all(|m| routing_leader(m, group_id) == n.node_id) + }); + if let Some(idx) = found { + return idx; + } + if Instant::now() >= deadline { + panic!("data group {group_id} has no leader agreed by every node within 20s"); + } + tokio::time::sleep(Duration::from_millis(100)).await; + } +} + +/// This node's local document entries for `TXN_COLL`, keyed by storage key. +async fn local_txn_documents( + node: &common::cluster_harness::TestClusterNode, +) -> std::collections::BTreeMap> { + let bytes = node + .create_tenant_snapshot(nodedb_types::TenantId::new(1)) + .await; + assert!( + !bytes.is_empty(), + "node {} returned an empty tenant snapshot", + node.node_id + ); + let snapshot: nodedb::types::TenantDataSnapshot = + zerompk::from_msgpack(&bytes).expect("decode TenantDataSnapshot"); + let marker = format!(":{TXN_COLL}:"); + snapshot + .documents + .into_iter() + .filter(|(key, _)| key.contains(&marker)) + .collect() +} + +/// `(id, payload)` rows of `TXN_COLL` served through `client`, sorted by id. +async fn served_txn_rows(client: &tokio_postgres::Client) -> Result, String> { + let msgs = client + .simple_query(&format!("SELECT id, payload FROM {TXN_COLL}")) + .await + .map_err(|e| pg_detail(&e))?; + let mut rows: Vec<(String, String)> = msgs + .iter() + .filter_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => Some(( + r.get("id").unwrap_or_default().to_owned(), + r.get("payload").unwrap_or_default().to_owned(), + )), + _ => None, + }) + .collect(); + rows.sort(); + Ok(rows) +} + +/// Spawn 3 nodes, seed `TXN_COLL`, and commit one single-shard explicit +/// transaction on the node that leads the collection's data group. Returns +/// the cluster, the group id, and the leader's index. +async fn commit_single_shard_txn_on_group_leader() -> (TestCluster, u64, usize) { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {TXN_COLL} WITH (engine='document_schemaless')" + )) + .await + .expect("CREATE COLLECTION"); + for id in ["seed-0", "seed-1"] { + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {TXN_COLL} (id, payload) VALUES ('{id}', 'seed')" + )) + .await + .unwrap_or_else(|e| panic!("seed {id}: {}", pg_detail(&e))); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let group_id = cluster.nodes[0] + .group_id_for_collection(TXN_COLL) + .expect("collection vshard mapped to a group"); + let leader_idx = group_leader_index(&cluster.nodes, group_id).await; + let leader = &cluster.nodes[leader_idx]; + + // One vShard, run on its leader: the commit takes the local single-shard path. + leader + .client + .simple_query(&format!( + "BEGIN; \ + INSERT INTO {TXN_COLL} (id, payload) VALUES ('txn-0', 'inserted-0'); \ + INSERT INTO {TXN_COLL} (id, payload) VALUES ('txn-1', 'inserted-1'); \ + UPDATE {TXN_COLL} SET payload = 'updated' WHERE id = 'seed-0'; \ + COMMIT" + )) + .await + .unwrap_or_else(|e| panic!("single-shard COMMIT on the leader: {}", pg_detail(&e))); + assert!( + leads(leader, group_id), + "node {} lost leadership of group {group_id} during the commit; \ + the commit did not run on the group leader", + leader.node_id + ); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + (cluster, group_id, leader_idx) +} + +/// A single-shard explicit transaction committed on the data-group leader +/// reaches every replica's local state. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn single_shard_txn_committed_on_group_leader_reaches_every_replica() { + let (cluster, _group_id, leader_idx) = commit_single_shard_txn_on_group_leader().await; + let leader = &cluster.nodes[leader_idx]; + + let expected: Vec<(String, String)> = TXN_EXPECTED + .iter() + .map(|(id, p)| ((*id).to_owned(), (*p).to_owned())) + .collect(); + let served = served_txn_rows(&leader.client) + .await + .expect("read committed rows on the leader"); + assert_eq!(served, expected, "the leader must serve the committed rows"); + let leader_docs = local_txn_documents(leader).await; + assert_eq!( + leader_docs.len(), + TXN_EXPECTED.len(), + "the leader must hold every committed row locally" + ); + + for follower in cluster.nodes.iter().filter(|n| n.node_id != leader.node_id) { + let deadline = Instant::now() + Duration::from_secs(15); + let follower_docs = loop { + let docs = local_txn_documents(follower).await; + if docs == leader_docs || Instant::now() >= deadline { + break docs; + } + tokio::time::sleep(Duration::from_millis(200)).await; + }; + let missing: Vec<&String> = leader_docs + .keys() + .filter(|k| follower_docs.get(*k) != leader_docs.get(*k)) + .collect(); + assert!( + missing.is_empty() && follower_docs.len() == leader_docs.len(), + "follower {} local state differs from leader {} after the committed \ + transaction; rows missing or different on the follower: {missing:?}", + follower.node_id, + leader.node_id + ); + } + + for node in cluster.nodes { + node.shutdown().await; + } +} + +/// A single-shard explicit transaction committed on the data-group leader +/// survives the loss of that leader. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn single_shard_txn_committed_on_group_leader_survives_leader_loss() { + let (cluster, group_id, leader_idx) = commit_single_shard_txn_on_group_leader().await; + + let mut nodes = cluster.nodes; + nodes.remove(leader_idx).shutdown().await; + + // A survivor takes over the data group. + let deadline = Instant::now() + Duration::from_secs(20); + while !nodes.iter().any(|n| leads(n, group_id)) { + if Instant::now() >= deadline { + panic!("no survivor took over data group {group_id} within 20s"); + } + tokio::time::sleep(Duration::from_millis(100)).await; + } + + let expected: Vec<(String, String)> = TXN_EXPECTED + .iter() + .map(|(id, p)| ((*id).to_owned(), (*p).to_owned())) + .collect(); + for node in &nodes { + let deadline = Instant::now() + Duration::from_secs(20); + let served = loop { + match served_txn_rows(&node.client).await { + Ok(rows) => break rows, + Err(e) if Instant::now() >= deadline => { + panic!("survivor {} could not serve rows: {e}", node.node_id) + } + Err(_) => tokio::time::sleep(Duration::from_millis(150)).await, + } + }; + assert_eq!( + served, expected, + "survivor {} must serve the committed transaction after the leader is lost", + node.node_id + ); + } + + for node in nodes { + node.shutdown().await; + } +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs new file mode 100644 index 000000000..1475731e1 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs @@ -0,0 +1,160 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! A proposal whose bytes commit twice applies once on every replica. +//! +//! A proposer re-proposes the same entry bytes after `RetryableLeaderChange`. +//! When the first copy also committed, the data group's log holds two copies +//! with one `idempotency_key`. The apply loop recognises the second copy by +//! that key and skips it. +//! +//! The test commits the same `KV_INCR` entry twice through the group leader's +//! raw proposer, which is exactly the log a double commit leaves behind. A +//! delta applied twice moves the counter twice, so every replica must read +//! the counter moved once. + +use crate::common; +use common::cluster_harness::TestCluster; + +use std::time::{Duration, Instant}; + +use nodedb::control::wal_replication::{ReplicableWrite, to_replicated_entry}; +use nodedb::types::{DatabaseId, TenantId, VShardId}; +use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; + +const COLL: &str = "dup_proposal_ctr"; +const TENANT: u64 = 1; + +fn pg_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn counter_on(client: &tokio_postgres::Client) -> Option { + let rows = client + .simple_query(&format!("SELECT n FROM {COLL} WHERE key = 'ctr'")) + .await + .unwrap_or_else(|e| panic!("read counter: {}", pg_detail(&e))); + rows.into_iter().find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get(0).map(str::to_string), + _ => None, + }) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_proposal_committed_twice_moves_the_counter_once() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLL} (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')" + )) + .await + .expect("create the counter collection"); + cluster.nodes[0] + .client + .simple_query(&format!("INSERT INTO {COLL} (key, n) VALUES ('ctr', 5)")) + .await + .unwrap_or_else(|e| panic!("seed the counter: {}", pg_detail(&e))); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, COLL); + let deadline = Instant::now() + Duration::from_secs(20); + let mut committed = 0; + let mut entry_bytes: Option> = None; + while committed < 2 { + assert!( + Instant::now() < deadline, + "could not commit both copies of the proposal through the group leader" + ); + let leader_id = { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let group = routing + .group_for_vshard(vshard.as_u32()) + .expect("the counter's vShard maps to a data group"); + routing + .group_info(group) + .map(|info| info.leader) + .unwrap_or(0) + }; + let Some(leader) = cluster.nodes.iter().find(|n| n.node_id == leader_id) else { + tokio::time::sleep(Duration::from_millis(100)).await; + continue; + }; + // One entry, built once: both copies carry the same bytes and so the + // same idempotency key, like a re-proposal after a leader change. + let bytes = match &entry_bytes { + Some(bytes) => bytes.clone(), + None => { + let surrogate = leader + .shared + .surrogate_assigner + .assign(DatabaseId::DEFAULT, TenantId::new(TENANT), COLL, b"ctr") + .expect("the seeded key has a surrogate"); + let plan = PhysicalPlan::Kv(KvOp::Incr { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLL), + key: b"ctr".to_vec(), + delta: 3, + ttl_ms: 0, + surrogate, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + }); + let write = + ReplicableWrite::decide_for_replication(&plan).expect("KV_INCR replicates"); + let entry = + to_replicated_entry(TenantId::new(TENANT), DatabaseId::DEFAULT, vshard, &write) + .expect("encode the proposal") + .expect("KV_INCR encodes to a replicated entry"); + let bytes = entry.to_bytes(); + entry_bytes = Some(bytes.clone()); + bytes + } + }; + let proposer = leader + .shared + .raft_proposer + .get() + .expect("the leader has a raft proposer"); + match proposer(vshard.as_u32(), bytes) { + Ok(_) => committed += 1, + Err(_) => tokio::time::sleep(Duration::from_millis(100)).await, + } + } + + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + for node in &cluster.nodes { + let deadline = Instant::now() + Duration::from_secs(15); + let mut last = None; + while Instant::now() < deadline { + last = counter_on(&node.client).await; + if last.as_deref() != Some("5") { + break; + } + tokio::time::sleep(Duration::from_millis(100)).await; + } + assert_eq!( + last.as_deref(), + Some("8"), + "node {} must hold the counter moved once by the proposal committed twice \ + (5 + 3); 11 means the second copy applied again", + node.node_id + ); + } + + for node in cluster.nodes { + node.shutdown().await; + } +} diff --git a/nodedb-columnar/src/mutation/engine.rs b/nodedb-columnar/src/mutation/engine.rs index 27987cccb..29534c539 100644 --- a/nodedb-columnar/src/mutation/engine.rs +++ b/nodedb-columnar/src/mutation/engine.rs @@ -442,6 +442,7 @@ mod tests { Value::String("Alice Updated".into()), Value::Float(0.75), ], + None, ) .expect("update"); @@ -616,6 +617,63 @@ mod tests { assert_eq!(engine.memtable_surrogates(), &[None]); } + #[test] + fn an_update_keeps_the_old_rows_surrogate() { + let mut engine = MutationEngine::new("col".into(), test_schema()); + engine + .insert_with_surrogate( + &[Value::Integer(1), Value::String("x".into()), Value::Null], + Surrogate(42), + ) + .expect("insert"); + engine + .update( + &Value::Integer(1), + &[Value::Integer(2), Value::String("y".into()), Value::Null], + Some(Surrogate(7)), + ) + .expect("update"); + let live: Vec> = engine + .scan_memtable_rows_with_surrogates() + .map(|(surrogate, _)| surrogate) + .collect(); + assert_eq!( + live, + vec![Some(Surrogate(42))], + "the replacement row carries the identity the memtable row had" + ); + } + + #[test] + fn an_update_of_a_flushed_row_carries_the_segment_surrogate() { + let mut engine = MutationEngine::new("col".into(), test_schema()); + engine + .insert_with_surrogate( + &[Value::Integer(1), Value::String("x".into()), Value::Null], + Surrogate(42), + ) + .expect("insert"); + let segment_id = engine.next_segment_id(); + let _drained = engine.memtable_mut().drain_optimized(); + engine.on_memtable_flushed(segment_id).expect("flush"); + engine + .update( + &Value::Integer(1), + &[Value::Integer(1), Value::String("y".into()), Value::Null], + Some(Surrogate(42)), + ) + .expect("update"); + let live: Vec> = engine + .scan_memtable_rows_with_surrogates() + .map(|(surrogate, _)| surrogate) + .collect(); + assert_eq!( + live, + vec![Some(Surrogate(42))], + "the replacement row carries the identity the flushed row had" + ); + } + #[test] fn flush_clears_surrogate_table() { let mut engine = MutationEngine::new("col".into(), test_schema()); diff --git a/nodedb-columnar/src/mutation/write.rs b/nodedb-columnar/src/mutation/write.rs index 827342c3f..b77341bf4 100644 --- a/nodedb-columnar/src/mutation/write.rs +++ b/nodedb-columnar/src/mutation/write.rs @@ -203,16 +203,36 @@ impl MutationEngine { /// /// NOTE: The caller must provide the full old row values for the re-insert. /// This method takes the complete new row (already merged with old values). + /// + /// The new row keeps the cross-engine surrogate the old row carried. A + /// memtable row's surrogate is in this engine's side table. A flushed + /// row's surrogate is in its segment's sidecar, outside this engine, so + /// the caller passes it as `flushed_surrogate`. It is read only when the + /// old row is flushed. pub fn update( &mut self, old_pk: &Value, new_values: &[Value], + flushed_surrogate: Option, ) -> Result { + let surrogate = match self.pk_index.get(&encode_pk(old_pk)) { + Some(loc) if loc.segment_id == self.memtable_segment_id => self + .memtable_surrogates + .get(loc.row_index as usize) + .copied() + .flatten(), + Some(_) => flushed_surrogate, + None => None, + }; + // Delete the old row. let delete_result = self.delete(old_pk)?; - // Insert the new row. - let insert_result = self.insert(new_values)?; + // Insert the new row under the old row's surrogate. + let insert_result = match surrogate { + Some(surrogate) => self.insert_with_surrogate(new_values, surrogate)?, + None => self.insert(new_values)?, + }; // Combine WAL records. let mut wal_records = delete_result.wal_records; diff --git a/nodedb-physical/src/physical_plan/document/mod.rs b/nodedb-physical/src/physical_plan/document/mod.rs index 6a1b2bbd2..9fb7cc651 100644 --- a/nodedb-physical/src/physical_plan/document/mod.rs +++ b/nodedb-physical/src/physical_plan/document/mod.rs @@ -19,7 +19,7 @@ pub use merge_types::{MergeActionOp, MergeClauseKind as MergeClauseKindOp, Merge pub use ollp_edge::OllpPredictedEdge; pub use op::DocumentOp; pub use resolved_mutation::{DocumentResolveOutcome, DocumentResolvedMutation}; -pub use sum_target::{ResolvedSumTarget, SumTargetKey, resolved_sum_surrogate}; +pub use sum_target::{RedoSumTargets, ResolvedSumTarget, SumTargetKey, resolved_sum_surrogate}; pub use timeseries_schema::TimeseriesSchema; pub use types::{ BalancedDef, EnforcementOptions, GeneratedColumnSpec, MaterializedSumBinding, PeriodLockConfig, diff --git a/nodedb-physical/src/physical_plan/document/sum_target.rs b/nodedb-physical/src/physical_plan/document/sum_target.rs index ad9b683a5..ac8e01179 100644 --- a/nodedb-physical/src/physical_plan/document/sum_target.rs +++ b/nodedb-physical/src/physical_plan/document/sum_target.rs @@ -122,6 +122,31 @@ impl ResolvedSumTarget { } } +/// The materialized-sum resolution one committed transaction's writes to one +/// SOURCE collection fold into their targets. +/// +/// A transaction redo record carries source post-images only. Every replica +/// applying it folds the source rows into their target rows, so the target +/// identities travel with the redo, keyed by source collection. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct RedoSumTargets { + /// SOURCE collection the transaction wrote. + pub collection: String, + /// Every target row the transaction's writes to `collection` resolved. + pub resolved: Vec, + /// TARGET collections whose delta travels on its own `ApplyBalanceDelta` + /// task, so the fold skips them. + pub deferred: Vec, +} + /// The surrogate `resolved` binds `target_collection`'s `join_value` to. /// /// The one lookup both planes use, so the Control Plane's "this one travels on diff --git a/nodedb-physical/src/physical_plan/meta.rs b/nodedb-physical/src/physical_plan/meta.rs index 265f23c4a..9b51156a3 100644 --- a/nodedb-physical/src/physical_plan/meta.rs +++ b/nodedb-physical/src/physical_plan/meta.rs @@ -573,4 +573,22 @@ pub enum MetaOp { /// base engine is touched during resolve. Wire-additive: appended last /// so older log entries decode unchanged. CalvinResolve { epoch: u64, position: u32 }, + + /// Apply one committed transaction's resolved redo record on the core that + /// owns its vShard. + /// + /// Every replica runs this from the vShard's data-group Raft log, in log + /// order, and installs the same post-images. The write funnel appends + /// `redo` to this node's WAL as one `TransactionRedo` record before the + /// dispatch, so restart replay reproduces the apply. + /// + /// `redo` is the zerompk-encoded redo record. `collections` names every + /// collection the transaction wrote; each gets a collection-floor write + /// version at the record's LSN. `sum_targets` is the materialized-sum + /// resolution the transaction's document writes fold into their targets. + ApplyTransactionRedo { + redo: Vec, + collections: Vec, + sum_targets: Vec, + }, } diff --git a/nodedb-physical/src/physical_plan/mod.rs b/nodedb-physical/src/physical_plan/mod.rs index 04e8a3ce3..22b64d0fc 100644 --- a/nodedb-physical/src/physical_plan/mod.rs +++ b/nodedb-physical/src/physical_plan/mod.rs @@ -40,8 +40,8 @@ pub use crdt::{CrdtOp, CrdtWriteVerb}; pub use document::{ BalancedDef, DocumentOp, DocumentResolveOutcome, DocumentResolvedMutation, EnforcementOptions, GeneratedColumnSpec, MaterializedSumBinding, OllpPredictedEdge, PeriodLockConfig, - RegisteredIndex, RegisteredIndexState, ResolvedSumTarget, ReturningColumns, ReturningItem, - ReturningSpec, StorageMode, SumTargetKey, TimeseriesSchema, UpdateValue, + RedoSumTargets, RegisteredIndex, RegisteredIndexState, ResolvedSumTarget, ReturningColumns, + ReturningItem, ReturningSpec, StorageMode, SumTargetKey, TimeseriesSchema, UpdateValue, resolved_sum_surrogate, }; pub use exchange::{ExchangeMode, ExchangeOp}; diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs index 06f6d6e6a..d36f1a4eb 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs @@ -253,6 +253,7 @@ impl TestClusterNode { let core_handle = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { idx, + num_cores, data_side, core_dir: data_dir_path.clone(), core_array_catalog: shared.array_catalog.clone(), diff --git a/nodedb-test-support/src/core_loop_runner.rs b/nodedb-test-support/src/core_loop_runner.rs index fab331129..a17642c8a 100644 --- a/nodedb-test-support/src/core_loop_runner.rs +++ b/nodedb-test-support/src/core_loop_runner.rs @@ -61,6 +61,10 @@ pub struct WalReplay { pub struct CoreLoopSpawn { /// Core index within the data plane (0-based). pub idx: usize, + /// Total number of Data-Plane cores in this node. The committed-redo apply + /// routes each record to `vshard_id % num_cores`, the same rule the + /// dispatcher routes requests by. + pub num_cores: usize, /// SPSC bridge endpoints for this core. pub data_side: CoreChannelDataSide, /// Storage directory shared with the rest of the harness. @@ -109,6 +113,7 @@ pub struct CoreLoopSpawn { pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { let CoreLoopSpawn { idx, + num_cores, data_side, core_dir, core_array_catalog, @@ -138,6 +143,7 @@ pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { ) .expect("CoreLoop::open_with_array_catalog"); core.set_event_producer(event_producer); + core.set_num_cores(num_cores); core.set_query_tuning(query_tuning); core.set_graph_tuning(graph_tuning); if let Some(m) = core_metrics { @@ -150,10 +156,10 @@ pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { if let Some(WalReplay { records, tombstones, - num_cores, + num_cores: replay_num_cores, }) = replay { - core.replay_all_wal(&records, num_cores, &tombstones); + core.replay_all_wal(&records, replay_num_cores, &tombstones); } while matches!( stop_rx.try_recv(), diff --git a/nodedb-test-support/src/native_harness/server.rs b/nodedb-test-support/src/native_harness/server.rs index e20af7293..cda0d1d71 100644 --- a/nodedb-test-support/src/native_harness/server.rs +++ b/nodedb-test-support/src/native_harness/server.rs @@ -98,6 +98,7 @@ impl NativeTestServer { ) .expect("open core"); core.set_event_producer(event_producer); + core.set_num_cores(1); if let Some(m) = core_metrics { core.set_metrics(m); } diff --git a/nodedb-test-support/src/pgwire_harness/multicore.rs b/nodedb-test-support/src/pgwire_harness/multicore.rs index caf2206ee..94421b3f8 100644 --- a/nodedb-test-support/src/pgwire_harness/multicore.rs +++ b/nodedb-test-support/src/pgwire_harness/multicore.rs @@ -61,6 +61,7 @@ impl TestServer { let core_handle = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { idx, + num_cores, data_side, core_dir: dir.path().to_path_buf(), core_array_catalog: shared.array_catalog.clone(), diff --git a/nodedb-test-support/src/pgwire_harness/restart.rs b/nodedb-test-support/src/pgwire_harness/restart.rs index cc815fcc3..cd93c0d26 100644 --- a/nodedb-test-support/src/pgwire_harness/restart.rs +++ b/nodedb-test-support/src/pgwire_harness/restart.rs @@ -234,6 +234,8 @@ impl TestServer { let core_handle = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { idx, + // Single-core harness (`Dispatcher::new(1, ..)`). + num_cores: 1, data_side, core_dir: dir_path.to_path_buf(), core_array_catalog: shared.array_catalog.clone(), diff --git a/nodedb-test-support/src/pgwire_harness/start.rs b/nodedb-test-support/src/pgwire_harness/start.rs index ccd2044df..4dae6b47e 100644 --- a/nodedb-test-support/src/pgwire_harness/start.rs +++ b/nodedb-test-support/src/pgwire_harness/start.rs @@ -254,6 +254,8 @@ impl TestServer { let core_handle = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { idx, + // Single-core harness (`Dispatcher::new(1, ..)`). + num_cores: 1, data_side, core_dir: dir.path().to_path_buf(), core_array_catalog: shared.array_catalog.clone(), diff --git a/nodedb-types/src/columnar/dml_wal_record.rs b/nodedb-types/src/columnar/dml_wal_record.rs index 277dded8c..09b06a9f9 100644 --- a/nodedb-types/src/columnar/dml_wal_record.rs +++ b/nodedb-types/src/columnar/dml_wal_record.rs @@ -109,6 +109,7 @@ mod tests { payload: vec![1, 2, 3], provenance: None, surrogates: Vec::new(), + conflict_policy: Vec::new(), }; let bytes = zerompk::to_msgpack_vec(&rec).expect("encode"); let as_dml_record: Result = zerompk::from_msgpack(&bytes); diff --git a/nodedb-types/src/columnar/image_wal_record.rs b/nodedb-types/src/columnar/image_wal_record.rs new file mode 100644 index 000000000..f67ef95a9 --- /dev/null +++ b/nodedb-types/src/columnar/image_wal_record.rs @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Columnar row-image WAL record payload. +//! +//! A committed transaction's columnar writes travel as the final row images +//! the transaction staged and showed its own reads. Each entry names one row +//! by its cross-engine surrogate. It carries the primary key of the base row +//! the transaction replaced or removed, and the image the row holds after the +//! commit. Replay installs the image verbatim. It never re-runs an +//! `ON CONFLICT` merge, a SET list or a predicate against the replaying +//! node's state. +//! +//! Rides `RecordType::TimeseriesBatch`, disambiguated from the other columnar +//! record shapes by `kind = "columnar_image"`. + +use serde::{Deserialize, Serialize}; + +/// The `kind` tag every [`ColumnarImageWalRecord`] carries. +pub const COLUMNAR_IMAGE_KIND: &str = "columnar_image"; + +/// One row inside a [`ColumnarImageWalRecord`]. +#[derive( + Debug, + Clone, + PartialEq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct ColumnarImageWalRow { + /// The row's cross-engine surrogate. + pub surrogate: u32, + /// MessagePack-encoded primary key of the base row this write replaces or + /// removes. Empty when the row had no base row the write displaced by key + /// (an insert or an `ON CONFLICT` upsert, which the image's own key + /// overwrites). + pub prior_pk_msgpack: Vec, + /// MessagePack-encoded post-image (`Value::Object`, column name to value, + /// bitemporal columns included). Empty for a delete. + pub image_msgpack: Vec, +} + +/// Map-encoded columnar row-image WAL record. +#[derive( + Debug, + Clone, + PartialEq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct ColumnarImageWalRecord { + /// Record kind tag. Always [`COLUMNAR_IMAGE_KIND`]. + pub kind: String, + /// Target collection name. + pub collection: String, + /// The catalog schema the writing plan carried (`ColumnarSchema`, + /// MessagePack). Empty when the plan carried none. Replay uses it to + /// create the engine on a node that holds no row of the collection yet. + pub schema_bytes: Vec, + /// Every row the transaction wrote, in surrogate order. + pub rows: Vec, +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::columnar::{ColumnarDmlWalRecord, ColumnarResolvedDmlWalRecord}; + + fn record() -> ColumnarImageWalRecord { + ColumnarImageWalRecord { + kind: COLUMNAR_IMAGE_KIND.to_string(), + collection: "events".to_string(), + schema_bytes: vec![9], + rows: vec![ + ColumnarImageWalRow { + surrogate: 7, + prior_pk_msgpack: vec![1], + image_msgpack: vec![2, 3], + }, + ColumnarImageWalRow { + surrogate: 8, + prior_pk_msgpack: vec![4], + image_msgpack: Vec::new(), + }, + ], + } + } + + #[test] + fn round_trips_every_row() { + let rec = record(); + let bytes = zerompk::to_msgpack_vec(&rec).expect("encode"); + let decoded: ColumnarImageWalRecord = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, rec); + } + + #[test] + fn does_not_decode_as_the_dml_record_shapes() { + let bytes = zerompk::to_msgpack_vec(&record()).expect("encode"); + let as_dml = zerompk::from_msgpack::(&bytes); + assert!(as_dml.map(|r| r.kind != "columnar_dml").unwrap_or(true)); + let as_resolved = zerompk::from_msgpack::(&bytes); + assert!( + as_resolved + .map(|r| r.kind != "columnar_resolved_dml") + .unwrap_or(true) + ); + } +} diff --git a/nodedb-types/src/columnar/mod.rs b/nodedb-types/src/columnar/mod.rs index f6534228b..5b2bca556 100644 --- a/nodedb-types/src/columnar/mod.rs +++ b/nodedb-types/src/columnar/mod.rs @@ -6,6 +6,7 @@ pub mod column_type; pub mod declared_type_keyword; pub mod dml_wal_record; pub mod float_width; +pub mod image_wal_record; pub mod int_width; pub mod profile; pub mod resolved_dml_wal_record; @@ -19,6 +20,7 @@ pub use column_type::ColumnType; pub use declared_type_keyword::declared_type_matches; pub use dml_wal_record::ColumnarDmlWalRecord; pub use float_width::FloatWidth; +pub use image_wal_record::{COLUMNAR_IMAGE_KIND, ColumnarImageWalRecord, ColumnarImageWalRow}; pub use int_width::IntWidth; pub use profile::{ColumnarProfile, DocumentMode}; pub use resolved_dml_wal_record::{ColumnarResolvedDmlWalRecord, ColumnarResolvedDmlWalRow}; diff --git a/nodedb-types/src/columnar/wal_record.rs b/nodedb-types/src/columnar/wal_record.rs index 5b5232013..856342d20 100644 --- a/nodedb-types/src/columnar/wal_record.rs +++ b/nodedb-types/src/columnar/wal_record.rs @@ -60,6 +60,13 @@ pub struct ColumnarWalRecord { #[serde(default)] #[msgpack(default)] pub surrogates: Vec, + /// What a row whose primary key already exists does, encoded by the + /// writer that knows the insert's conflict intent and `ON CONFLICT` + /// assignments. Empty for a plain insert, which replaces the row. Replay + /// decides each row the way the live insert did. + #[serde(default)] + #[msgpack(default)] + pub conflict_policy: Vec, } #[cfg(test)] @@ -80,6 +87,7 @@ mod tests { payload: vec![1, 2, 3, 4], provenance: Some(prov.clone()), surrogates: vec![Surrogate::new(10), Surrogate::new(11), Surrogate::new(12)], + conflict_policy: Vec::new(), }; let bytes = zerompk::to_msgpack_vec(&rec).expect("encode ColumnarWalRecord"); @@ -104,6 +112,7 @@ mod tests { payload: vec![9], provenance: None, surrogates: Vec::new(), + conflict_policy: Vec::new(), }; let bytes = zerompk::to_msgpack_vec(&rec).expect("encode"); let decoded: ColumnarWalRecord = zerompk::from_msgpack(&bytes).expect("decode"); diff --git a/nodedb-vector/src/collection/mod.rs b/nodedb-vector/src/collection/mod.rs index 88da448d2..28d8b3958 100644 --- a/nodedb-vector/src/collection/mod.rs +++ b/nodedb-vector/src/collection/mod.rs @@ -10,6 +10,7 @@ pub mod lifecycle_insert_ops; pub mod lifecycle_reindex; pub mod payload_index; pub mod quantize; +pub mod rollback; pub mod search; pub mod segment; pub mod stats; @@ -17,6 +18,7 @@ pub mod tier; pub use lifecycle::VectorCollection; pub use payload_index::{FilterPredicate, PayloadIndex, PayloadIndexKind, PayloadIndexSet}; +pub use rollback::VectorWriteMark; pub use segment::{ BuildComplete, BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment, }; diff --git a/nodedb-vector/src/collection/payload_index.rs b/nodedb-vector/src/collection/payload_index.rs index c7daeaba1..581111c31 100644 --- a/nodedb-vector/src/collection/payload_index.rs +++ b/nodedb-vector/src/collection/payload_index.rs @@ -295,6 +295,22 @@ impl PayloadIndexSet { } } + /// A set with the same registered fields and kinds and no rows. + pub fn definitions_only(&self) -> Self { + Self { + indexes: self + .indexes + .iter() + .map(|(field, index)| { + ( + field.clone(), + PayloadIndex::new(index.field.clone(), index.kind), + ) + }) + .collect(), + } + } + /// Drop every node from every index. The registered fields and their /// kinds stay, so the next `insert_row` indexes the same columns. pub fn clear_rows(&mut self) { diff --git a/nodedb-vector/src/collection/rollback.rs b/nodedb-vector/src/collection/rollback.rs new file mode 100644 index 000000000..49c7fe015 --- /dev/null +++ b/nodedb-vector/src/collection/rollback.rs @@ -0,0 +1,244 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Roll a collection back to the state it held before a write. +//! +//! A [`VectorWriteMark`] records what a write can change: the id counter, +//! the binding of every surrogate the write names, which of the named nodes +//! were live, and the multi-vector documents it names. Rolling back drops +//! every node inserted since the mark from the growing segment, so the id +//! counter returns to where it was and the next insert takes the same id it +//! would have taken without the write. Then it puts every binding and every +//! tombstone back. +//! +//! The inserted nodes must still sit in the growing segment: a seal between +//! the mark and the rollback moves them out, and the rollback then reports +//! that it cannot restore the mark. + +use nodedb_types::Surrogate; + +use super::lifecycle::VectorCollection; +use crate::flat::FlatIndex; + +/// A collection's state before a write. See the module docs. +#[derive(Debug, Clone)] +pub struct VectorWriteMark { + next_id: u32, + growing_base_id: u32, + /// Each named surrogate with the node bound to it, if any. + bindings: Vec<(Surrogate, Option)>, + /// Named nodes that were live. + live: Vec, + /// Each named multi-vector document with its node list, if any. + multi_docs: Vec<(Surrogate, Option>)>, +} + +impl VectorCollection { + /// Whether node `id` exists and is not soft-deleted, whichever segment + /// holds it. + pub fn is_live(&self, id: u32) -> bool { + if id >= self.growing_base_id { + let local = id - self.growing_base_id; + if (local as usize) < self.growing.len() { + return !self.growing.is_deleted(local); + } + } + for seg in &self.sealed { + if id >= seg.base_id { + let local = id - seg.base_id; + if (local as usize) < seg.index.len() { + return !seg.index.is_deleted(local); + } + } + } + for seg in &self.building { + if id >= seg.base_id { + let local = id - seg.base_id; + if (local as usize) < seg.flat.len() { + return !seg.flat.is_deleted(local); + } + } + } + false + } + + /// Mark the state a write naming `surrogates` and node `ids` can change. + pub fn write_mark(&self, surrogates: &[Surrogate], ids: &[u32]) -> VectorWriteMark { + let mut bindings = Vec::with_capacity(surrogates.len() + ids.len()); + let mut live = Vec::new(); + let mut multi_docs = Vec::with_capacity(surrogates.len()); + for &surrogate in surrogates { + let bound = self.surrogate_to_local.get(&surrogate).copied(); + bindings.push((surrogate, bound)); + if let Some(id) = bound + && self.is_live(id) + { + live.push(id); + } + let doc = self.multi_doc_map.get(&surrogate).cloned(); + if let Some(doc_ids) = &doc { + live.extend(doc_ids.iter().copied().filter(|id| self.is_live(*id))); + } + multi_docs.push((surrogate, doc)); + } + for &id in ids { + if !self.is_live(id) { + continue; + } + live.push(id); + if let Some(&surrogate) = self.surrogate_map.get(&id) { + bindings.push((surrogate, Some(id))); + } + } + VectorWriteMark { + next_id: self.next_id, + growing_base_id: self.growing_base_id, + bindings, + live, + multi_docs, + } + } + + /// Put the collection back to `mark`. Returns `false`, changing nothing, + /// when a seal moved the nodes inserted since the mark out of the growing + /// segment. + pub fn roll_back_to(&mut self, mark: VectorWriteMark) -> bool { + if self.growing_base_id != mark.growing_base_id || mark.next_id < self.growing_base_id { + return false; + } + for id in mark.next_id..self.next_id { + if let Some(surrogate) = self.surrogate_map.remove(&id) + && self.surrogate_to_local.get(&surrogate) == Some(&id) + { + self.surrogate_to_local.remove(&surrogate); + } + } + self.growing + .truncate((mark.next_id - self.growing_base_id) as usize); + self.next_id = mark.next_id; + + for (surrogate, prior) in mark.bindings { + if let Some(current) = self.surrogate_to_local.remove(&surrogate) + && Some(current) != prior + { + self.surrogate_map.remove(¤t); + } + if let Some(id) = prior { + self.surrogate_to_local.insert(surrogate, id); + self.surrogate_map.insert(id, surrogate); + } + } + for (surrogate, prior) in mark.multi_docs { + match prior { + Some(ids) => { + for id in &ids { + self.surrogate_map.insert(*id, surrogate); + } + self.multi_doc_map.insert(surrogate, ids); + } + None => { + self.multi_doc_map.remove(&surrogate); + } + } + } + for id in mark.live { + if !self.is_live(id) { + self.undelete(id); + } + } + true + } + + /// An empty collection with this one's configuration and id counters: + /// same dimension, parameters, index config, quantization, seal + /// threshold, storage settings, registered payload fields and + /// watermarks. The next insert takes the id this collection's next insert + /// would have taken. + pub fn detached_empty(&self) -> Self { + let mut fresh = Self::with_seal_threshold_and_config( + self.dim, + self.index_config.clone(), + self.seal_threshold, + ); + fresh.params = self.params.clone(); + fresh.growing = FlatIndex::new(self.dim, self.params.metric); + fresh.next_id = self.next_id; + fresh.growing_base_id = self.next_id; + fresh.next_segment_id = self.next_segment_id; + fresh.data_dir = self.data_dir.clone(); + fresh.ram_budget_bytes = self.ram_budget_bytes; + fresh.mmap_fallback_count = self.mmap_fallback_count; + fresh.quantization = self.quantization; + fresh.payload = self.payload.definitions_only(); + fresh.arena_index = self.arena_index; + fresh.checkpoint_wal_lsn = self.checkpoint_wal_lsn; + fresh.applied_wal_lsn = self.applied_wal_lsn; + fresh + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::hnsw::HnswParams; + + fn collection() -> VectorCollection { + VectorCollection::new(2, HnswParams::default()) + } + + #[test] + fn a_rolled_back_insert_returns_the_id_counter_and_the_binding() { + let mut coll = collection(); + coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(1)); + let mark = coll.write_mark(&[Surrogate::new(2)], &[]); + coll.insert_with_surrogate(vec![0.3, 0.4], Surrogate::new(2)); + assert!(coll.roll_back_to(mark)); + assert_eq!(coll.local_for_surrogate(Surrogate::new(2)), None); + assert_eq!( + coll.insert_with_surrogate(vec![0.5, 0.6], Surrogate::new(3)), + 1, + "the next insert takes the id the rolled-back insert took" + ); + } + + #[test] + fn a_rolled_back_rebind_restores_the_replaced_node() { + let mut coll = collection(); + let first = coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(7)); + let mark = coll.write_mark(&[Surrogate::new(7)], &[]); + coll.insert_with_surrogate(vec![0.3, 0.4], Surrogate::new(7)); + assert!(!coll.is_live(first)); + assert!(coll.roll_back_to(mark)); + assert!(coll.is_live(first)); + assert_eq!(coll.local_for_surrogate(Surrogate::new(7)), Some(first)); + } + + #[test] + fn a_rolled_back_delete_restores_the_node_and_its_binding() { + let mut coll = collection(); + let id = coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(4)); + let mark = coll.write_mark(&[], &[id]); + coll.delete(id); + assert!(coll.roll_back_to(mark)); + assert!(coll.is_live(id)); + assert_eq!(coll.local_for_surrogate(Surrogate::new(4)), Some(id)); + } + + #[test] + fn a_seal_after_the_mark_refuses_the_rollback() { + let mut coll = VectorCollection::with_seal_threshold(2, HnswParams::default(), 1); + let mark = coll.write_mark(&[Surrogate::new(1)], &[]); + coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(1)); + assert!(coll.seal("k").is_some()); + assert!(!coll.roll_back_to(mark)); + } + + #[test] + fn a_detached_empty_collection_continues_the_id_counter() { + let mut coll = collection(); + coll.insert(vec![0.1, 0.2]); + coll.insert(vec![0.3, 0.4]); + let mut fresh = coll.detached_empty(); + assert_eq!(fresh.live_count(), 0); + assert_eq!(fresh.insert(vec![0.5, 0.6]), 2); + } +} diff --git a/nodedb-vector/src/flat.rs b/nodedb-vector/src/flat.rs index 5ed0880f9..047c01e68 100644 --- a/nodedb-vector/src/flat.rs +++ b/nodedb-vector/src/flat.rs @@ -286,6 +286,21 @@ impl FlatIndex { self.deleted.len() } + /// Drop every vector at position `len` or later, as if it was never + /// inserted. A rollback uses it to withdraw the newest inserts. + pub fn truncate(&mut self, len: usize) { + if len >= self.deleted.len() { + return; + } + let dropped_live = self.deleted[len..] + .iter() + .filter(|deleted| !**deleted) + .count(); + self.live_count -= dropped_live; + self.deleted.truncate(len); + self.data.truncate(len * self.dim); + } + pub fn live_count(&self) -> usize { self.live_count } diff --git a/nodedb-vector/src/ivf.rs b/nodedb-vector/src/ivf.rs index bcc371de4..faba2c0df 100644 --- a/nodedb-vector/src/ivf.rs +++ b/nodedb-vector/src/ivf.rs @@ -123,6 +123,28 @@ impl IvfPqIndex { id } + /// Whether the index holds a trained codebook. + pub fn is_trained(&self) -> bool { + self.pq.is_some() + } + + /// Drop every vector added with id `count` or later. When `trained` is + /// `false` the training goes too, as the index held before its first + /// add. A rollback uses it to withdraw the newest adds. + pub fn roll_back_to(&mut self, count: u32, trained: bool) { + if !trained { + self.centroids.clear(); + self.pq = None; + self.cells.clear(); + self.count = 0; + return; + } + for cell in &mut self.cells { + cell.retain(|(id, _)| *id < count); + } + self.count = self.count.min(count); + } + /// Batch add vectors. pub fn add_batch(&mut self, vectors: &[&[f32]]) { for v in vectors { diff --git a/nodedb-wal/src/lazy_reader.rs b/nodedb-wal/src/lazy_reader.rs index 3e175dd72..72d916ca4 100644 --- a/nodedb-wal/src/lazy_reader.rs +++ b/nodedb-wal/src/lazy_reader.rs @@ -460,7 +460,7 @@ mod tests { vshard_id: 0, payload_len: (MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, - reserved: [0; 8], + apply_key: 0, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/lib.rs b/nodedb-wal/src/lib.rs index f61e167e8..8ba0ccaa0 100644 --- a/nodedb-wal/src/lib.rs +++ b/nodedb-wal/src/lib.rs @@ -59,8 +59,9 @@ pub use preamble::{ }; pub use reader::{StopReason, WalReader}; pub use record::{ - CalvinAppliedPayload, FtsDeletePayload, FtsIndexPayload, RecordHeader, RecordType, - SpatialDeletePayload, SpatialPutPayload, WalRecord, WalRecordArgs, WriteAbortedPayload, + CalvinAppliedPayload, FtsDeletePayload, FtsIndexPayload, RecordHeader, RecordTarget, + RecordType, SpatialDeletePayload, SpatialPutPayload, WalRecord, WalRecordArgs, + WriteAbortedPayload, }; pub use recovery::{RecoveryInfo, recover}; pub use replay::{ diff --git a/nodedb-wal/src/mmap_reader/reader.rs b/nodedb-wal/src/mmap_reader/reader.rs index 0ca585489..893495bb0 100644 --- a/nodedb-wal/src/mmap_reader/reader.rs +++ b/nodedb-wal/src/mmap_reader/reader.rs @@ -479,7 +479,7 @@ mod tests { vshard_id: 0, payload_len: 1, database_id: 0, - reserved: [0; 8], + apply_key: 0, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); @@ -501,7 +501,7 @@ mod tests { vshard_id: 0, payload_len: (crate::record::MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, - reserved: [0; 8], + apply_key: 0, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/reader.rs b/nodedb-wal/src/reader.rs index edaade37c..ff015be4c 100644 --- a/nodedb-wal/src/reader.rs +++ b/nodedb-wal/src/reader.rs @@ -471,7 +471,7 @@ mod tests { vshard_id: 0, payload_len: (crate::record::MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, - reserved: [0; 8], + apply_key: 0, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/record/header.rs b/nodedb-wal/src/record/header.rs index 4dd9bf225..5c950a7cf 100644 --- a/nodedb-wal/src/record/header.rs +++ b/nodedb-wal/src/record/header.rs @@ -34,7 +34,7 @@ pub const MAX_WAL_PAYLOAD_SIZE: usize = 64 * 1024 * 1024; /// | vshard_id(4) | payload_len(4) | database_id(8) | reserved(8) | crc32c(4) /// /// `database_id` occupies bytes 34–41 (previously part of the 16-byte reserved -/// field). `reserved` occupies bytes 42–49. Bytes 34–41 were zero-filled in +/// field). `apply_key` occupies bytes 42–49. Bytes 34–41 were zero-filled in /// prior records, so `database_id == 0` maps to `DatabaseId(0)` (the default /// database), preserving backward compatibility without a format-version bump. pub const HEADER_SIZE: usize = 54; @@ -64,9 +64,12 @@ pub struct RecordHeader { /// /// Occupies bytes 34–41 of the on-disk header (previously part of reserved). pub database_id: u64, - /// Reserved for future use; must be zero on write; ignored on read - /// (but covered by CRC32C). Occupies bytes 42–49. - pub reserved: [u8; 8], + /// The idempotency key of the replicated proposal whose apply appended + /// this record, `0` for a record no proposal apply appended. The record + /// and the key are durable together, so a node recovers which proposals + /// it applied from the records themselves. Covered by CRC32C. Occupies + /// bytes 42–49. + pub apply_key: u64, pub crc32c: u32, } @@ -81,14 +84,12 @@ impl RecordHeader { buf[26..30].copy_from_slice(&self.vshard_id.to_le_bytes()); buf[30..34].copy_from_slice(&self.payload_len.to_le_bytes()); buf[34..42].copy_from_slice(&self.database_id.to_le_bytes()); - buf[42..50].copy_from_slice(&self.reserved); + buf[42..50].copy_from_slice(&self.apply_key.to_le_bytes()); buf[50..54].copy_from_slice(&self.crc32c.to_le_bytes()); buf } pub fn from_bytes(buf: &[u8; HEADER_SIZE]) -> Self { - let mut reserved = [0u8; 8]; - reserved.copy_from_slice(&buf[42..50]); Self { magic: u32::from_le_bytes([buf[0], buf[1], buf[2], buf[3]]), format_version: u16::from_le_bytes([buf[4], buf[5]]), @@ -104,15 +105,17 @@ impl RecordHeader { database_id: u64::from_le_bytes([ buf[34], buf[35], buf[36], buf[37], buf[38], buf[39], buf[40], buf[41], ]), - reserved, + apply_key: u64::from_le_bytes([ + buf[42], buf[43], buf[44], buf[45], buf[46], buf[47], buf[48], buf[49], + ]), crc32c: u32::from_le_bytes([buf[50], buf[51], buf[52], buf[53]]), } } /// CRC32C over header (excluding the crc32c field) + payload. /// - /// The 16 reserved bytes are included in the CRC so they cannot be - /// silently modified without detection. + /// The apply key is included in the CRC so it cannot be silently + /// modified without detection. pub fn compute_checksum(&self, payload: &[u8]) -> u32 { let header_bytes = self.to_bytes(); let mut digest = crc32c::crc32c(&header_bytes[..HEADER_SIZE - 4]); @@ -168,7 +171,7 @@ mod tests { vshard_id, payload_len: 100, database_id: 0, - reserved: [0u8; 8], + apply_key: 0, crc32c: 0xDEAD_BEEF, } } @@ -184,7 +187,7 @@ mod tests { fn header_golden_54_bytes_exact_offsets() { // magic at 0..4, format_version at 4..6, record_type at 6..10, // lsn at 10..18, tenant_id at 18..26, vshard_id at 26..30, - // payload_len at 30..34, database_id at 34..42, reserved at 42..50, + // payload_len at 30..34, database_id at 34..42, apply_key at 42..50, // crc32c at 50..54. let header = RecordHeader { magic: WAL_MAGIC, @@ -195,7 +198,7 @@ mod tests { vshard_id: 0xCAFE_BABE, payload_len: 256, database_id: 0xABCD_0000_1234_5678, - reserved: [0u8; 8], + apply_key: 0, crc32c: 0x1234_5678, }; let b = header.to_bytes(); @@ -216,7 +219,7 @@ mod tests { assert_eq!(&b[30..34], &256u32.to_le_bytes()); // database_id assert_eq!(&b[34..42], &0xABCD_0000_1234_5678u64.to_le_bytes()); - // reserved — all zero + // apply_key — zero assert_eq!(&b[42..50], &[0u8; 8]); // crc32c assert_eq!(&b[50..54], &0x1234_5678u32.to_le_bytes()); @@ -234,7 +237,7 @@ mod tests { vshard_id: 0, payload_len: 0, database_id: 7, - reserved: [0u8; 8], + apply_key: 0, crc32c: 0, }; let bytes = header.to_bytes(); @@ -268,7 +271,7 @@ mod tests { vshard_id: 0, payload_len: 0, database_id: 0, - reserved: [0u8; 8], + apply_key: 0, crc32c: 0, }; let bytes = header.to_bytes(); diff --git a/nodedb-wal/src/record/mod.rs b/nodedb-wal/src/record/mod.rs index 6e7a0e4bf..b050277fd 100644 --- a/nodedb-wal/src/record/mod.rs +++ b/nodedb-wal/src/record/mod.rs @@ -23,4 +23,4 @@ pub use padding::{MIN_PADDING_RECORD_SIZE, padding_record, padding_span}; pub use surrogate::{SURROGATE_PAYLOAD_SIZE, SurrogateAllocPayload, SurrogateBindPayload}; pub use sync_seq::{SYNC_SEQ_ADVANCE_PAYLOAD_SIZE, SyncSeqAdvancePayload}; pub use types::RecordType; -pub use wal_record::{WalRecord, WalRecordArgs}; +pub use wal_record::{RecordTarget, WalRecord, WalRecordArgs}; diff --git a/nodedb-wal/src/record/types.rs b/nodedb-wal/src/record/types.rs index 89a8af891..b3540cd97 100644 --- a/nodedb-wal/src/record/types.rs +++ b/nodedb-wal/src/record/types.rs @@ -345,6 +345,20 @@ pub enum RecordType { /// exists to prevent, so an older binary pointed at a WAL containing one /// must fail to start loudly rather than readmit refused data. WriteAborted = 61 | 0x8000, + + /// Marks one replicated Raft proposal as applied on this node when its + /// apply writes no record of its own (a `wal=false` timeseries ingest). + /// Payload: empty. The proposal's idempotency key is the header's + /// `apply_key`, as on every record a proposal's apply appends. + /// + /// A proposal re-proposed after a leader change can commit at two log + /// indexes. The data-group apply loop recovers every keyed record at boot + /// and skips the second copy, so a write applies once per proposal. Never + /// replayed into any engine. + /// + /// Required: skipping this record re-applies a duplicate proposal, which + /// double-counts every non-idempotent effect (a timeseries append). + ProposalApplied = 62 | 0x8000, } impl RecordType { @@ -399,6 +413,7 @@ impl RecordType { x if x == 59 | 0x8000 => Some(Self::GraphNodeLabelSet), x if x == 60 | 0x8000 => Some(Self::GraphNodeLabelRemove), x if x == 61 | 0x8000 => Some(Self::WriteAborted), + x if x == 62 | 0x8000 => Some(Self::ProposalApplied), _ => None, } } @@ -483,6 +498,7 @@ mod tests { RecordType::GraphNodeLabelSet, RecordType::GraphNodeLabelRemove, RecordType::WriteAborted, + RecordType::ProposalApplied, ] { assert_eq!(RecordType::from_raw(ty as u32), Some(ty)); } diff --git a/nodedb-wal/src/record/wal_record.rs b/nodedb-wal/src/record/wal_record.rs index a88c089fc..85652fedc 100644 --- a/nodedb-wal/src/record/wal_record.rs +++ b/nodedb-wal/src/record/wal_record.rs @@ -15,6 +15,16 @@ pub struct WalRecord { pub payload: Vec, } +/// The header fields of a record its caller decides: its type and scope. +/// The writer assigns the LSN. +#[derive(Debug, Clone, Copy)] +pub struct RecordTarget { + pub record_type: u32, + pub tenant_id: u64, + pub vshard_id: u32, + pub database_id: u64, +} + /// Parameters for [`WalRecord::new`]. pub struct WalRecordArgs<'a> { pub record_type: u32, @@ -43,6 +53,13 @@ impl WalRecord { /// zero-filled). Pre-existing records with zeros decode to `DatabaseId(0)` /// (the default database), preserving backward compatibility. pub fn new(args: WalRecordArgs<'_>) -> Result { + Self::new_keyed(args, 0) + } + + /// [`Self::new`] for a record appended by the apply of the replicated + /// proposal `apply_key`. The key rides the header, inside the CRC and the + /// encryption AAD, so the record and the key are durable together. + pub fn new_keyed(args: WalRecordArgs<'_>, apply_key: u64) -> Result { let WalRecordArgs { record_type, lsn, @@ -70,7 +87,7 @@ impl WalRecord { vshard_id, payload_len: 0, database_id, - reserved: [0u8; 8], + apply_key, crc32c: 0, }; let header_bytes = temp_header.to_bytes(); @@ -98,7 +115,7 @@ impl WalRecord { vshard_id, payload_len: final_payload.len() as u32, database_id, - reserved: [0u8; 8], + apply_key, crc32c: 0, }; @@ -110,6 +127,12 @@ impl WalRecord { }) } + /// The idempotency key of the proposal whose apply appended this record, + /// `0` when no proposal apply appended it. + pub fn apply_key(&self) -> u64 { + self.header.apply_key + } + /// Decrypt the payload if the record is encrypted. /// /// `epoch` must come from the on-disk segment preamble, not from the diff --git a/nodedb-wal/src/segmented.rs b/nodedb-wal/src/segmented.rs index 1ee5bbe4f..74bbfa1c3 100644 --- a/nodedb-wal/src/segmented.rs +++ b/nodedb-wal/src/segmented.rs @@ -23,7 +23,7 @@ use tracing::info; use crate::crypto::KeyRing; use crate::error::{Result, WalError}; -use crate::record::WalRecord; +use crate::record::{RecordTarget, WalRecord}; use crate::segment::{ DEFAULT_SEGMENT_TARGET_SIZE, SegmentContinuity, SegmentMeta, TruncateResult, check_retained_floor, discover_segments, segment_path, truncate_segments, @@ -190,14 +190,33 @@ impl SegmentedWal { vshard_id: u32, database_id: u64, payload: &[u8], + ) -> Result { + self.append_keyed( + RecordTarget { + record_type, + tenant_id, + vshard_id, + database_id, + }, + payload, + 0, + ) + } + + /// [`Self::append`] for a record appended by the apply of the replicated + /// proposal `apply_key` (see [`crate::WalRecord::new_keyed`]). + pub fn append_keyed( + &mut self, + target: RecordTarget, + payload: &[u8], + apply_key: u64, ) -> Result { // Check if we need to roll to a new segment. if self.writer.file_offset() >= self.segment_target_size { self.roll_segment()?; } - self.writer - .append(record_type, tenant_id, vshard_id, database_id, payload) + self.writer.append_keyed(target, payload, apply_key) } /// Flush all buffered records and fsync the active segment. diff --git a/nodedb-wal/src/writer/core.rs b/nodedb-wal/src/writer/core.rs index 9dde0f137..680de4b2f 100644 --- a/nodedb-wal/src/writer/core.rs +++ b/nodedb-wal/src/writer/core.rs @@ -9,7 +9,7 @@ use crate::align::AlignedBuf; use crate::double_write::{DoubleWriteBuffer, DwbProtection}; use crate::error::{Result, WalError}; use crate::preamble::SegmentPreamble; -use crate::record::{HEADER_SIZE, MIN_PADDING_RECORD_SIZE, WalRecord, WalRecordArgs}; +use crate::record::{HEADER_SIZE, MIN_PADDING_RECORD_SIZE, RecordTarget, WalRecord, WalRecordArgs}; use super::config::{WalWriterConfig, open_dwb_for, resume_offset}; use super::durability::{DurabilityState, fsync_and_track}; @@ -254,6 +254,32 @@ impl WalWriter { database_id: u64, payload: &[u8], ) -> Result { + self.append_keyed( + RecordTarget { + record_type, + tenant_id, + vshard_id, + database_id, + }, + payload, + 0, + ) + } + + /// [`Self::append`] for a record appended by the apply of the replicated + /// proposal `apply_key` (see [`WalRecord::new_keyed`]). + pub fn append_keyed( + &mut self, + target: RecordTarget, + payload: &[u8], + apply_key: u64, + ) -> Result { + let RecordTarget { + record_type, + tenant_id, + vshard_id, + database_id, + } = target; if self.sealed { return Err(WalError::Sealed); } @@ -261,16 +287,19 @@ impl WalWriter { let lsn = self.next_lsn.load(Ordering::Relaxed); let preamble_bytes = self.segment_preamble.as_ref().map(|p| p.to_bytes()); - let record = WalRecord::new(WalRecordArgs { - record_type, - lsn, - tenant_id, - vshard_id, - database_id, - payload: payload.to_vec(), - encryption_key: self.encryption_ring.as_ref().map(|r| r.current()), - preamble_bytes: preamble_bytes.as_ref(), - })?; + let record = WalRecord::new_keyed( + WalRecordArgs { + record_type, + lsn, + tenant_id, + vshard_id, + database_id, + payload: payload.to_vec(), + encryption_key: self.encryption_ring.as_ref().map(|r| r.current()), + preamble_bytes: preamble_bytes.as_ref(), + }, + apply_key, + )?; let header_bytes = record.header.to_bytes(); let total_size = HEADER_SIZE + record.payload.len(); diff --git a/nodedb/src/control/array_sync/raft_apply/cell.rs b/nodedb/src/control/array_sync/raft_apply/cell.rs index 83040de39..10ffd5eea 100644 --- a/nodedb/src/control/array_sync/raft_apply/cell.rs +++ b/nodedb/src/control/array_sync/raft_apply/cell.rs @@ -117,6 +117,7 @@ pub(crate) async fn apply_array_cell_write( // exactly as the generic committed-write branch does. event_source: crate::event::EventSource::User, resolved_now_ms, + apply_key: applied_key, op_label: "array cell write", }, ) diff --git a/nodedb/src/control/array_sync/raft_apply/common.rs b/nodedb/src/control/array_sync/raft_apply/common.rs index 5580ffc77..205002273 100644 --- a/nodedb/src/control/array_sync/raft_apply/common.rs +++ b/nodedb/src/control/array_sync/raft_apply/common.rs @@ -43,6 +43,9 @@ pub(super) struct ArrayWriteSubmit { /// the entry carries none. Passed through as the redo record's /// `now_override` so this replica records the value its peers recorded. pub resolved_now_ms: Option, + /// The idempotency key of the committed entry, carried by the redo + /// record's header. + pub apply_key: u64, /// Contextual label for the error surfaced to the propose waiter. pub op_label: &'static str, } @@ -70,6 +73,7 @@ pub(super) async fn submit_array_write( plan, event_source, resolved_now_ms, + apply_key, op_label, } = params; @@ -92,6 +96,7 @@ pub(super) async fn submit_array_write( // durability path than this record's replay. durability: WalDurability::AppendHere { now_override: resolved_now_ms, + apply_key, }, // Raft committed this entry at a fixed log index and every replica // applies it in that order; re-entering the write-admission gate diff --git a/nodedb/src/control/array_sync/raft_apply/op.rs b/nodedb/src/control/array_sync/raft_apply/op.rs index 2dd480b6f..4e553060c 100644 --- a/nodedb/src/control/array_sync/raft_apply/op.rs +++ b/nodedb/src/control/array_sync/raft_apply/op.rs @@ -201,6 +201,7 @@ pub(crate) async fn apply_array_op( // A sync op carries no proposer-resolved instant; only TTL-bearing // KV writes resolve one, and no array op is such a write. resolved_now_ms: None, + apply_key: applied_key, op_label: "array op", }, ) diff --git a/nodedb/src/control/cluster/array_executor/write.rs b/nodedb/src/control/cluster/array_executor/write.rs index 71c9269d9..cbf956460 100644 --- a/nodedb/src/control/cluster/array_executor/write.rs +++ b/nodedb/src/control/cluster/array_executor/write.rs @@ -210,7 +210,10 @@ fn single_node_submit( event_source: crate::event::EventSource::User, txn_id: None, user_id: None, - durability: WalDurability::AppendHere { now_override: None }, + durability: WalDurability::AppendHere { + now_override: None, + apply_key: 0, + }, ordering: WriteOrdering::Gate, change_feed: ChangeFeedOwner::Funnel, } diff --git a/nodedb/src/control/distributed_applier/apply_loop/driver.rs b/nodedb/src/control/distributed_applier/apply_loop/driver.rs index c97f09e52..a82e645b3 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/driver.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/driver.rs @@ -22,7 +22,12 @@ use crate::types::{DatabaseId, TenantId}; use super::array_dispatch::{apply_array_op_entry, apply_array_schema_entry}; use super::bookkeeping::record_durable_apply; use super::calvin_read_result::{CalvinReadResultFields, forward_calvin_read_result}; +use super::proposal_gate::{EntryOutcome, ProposalGate}; +use super::transaction_redo::apply_transaction_redo_entry; use super::write_dispatch::apply_generic_entry; +use crate::control::distributed_applier::proposal_ledger::{ + PROPOSAL_LEDGER_CAPACITY, ProposalLedger, +}; /// Run the background loop that applies committed Raft entries to the local Data Plane. /// @@ -36,6 +41,26 @@ pub async fn run_apply_loop( std::sync::Mutex>>, >, ) { + // Proposals this node already applied, recovered from its WAL before any + // entry is delivered: every record an entry's apply appended carries the + // entry's idempotency key in its header. + let records = match state.wal.replay() { + Ok(records) => records, + Err(error) => { + // Without the keys an entry re-delivered above the durable floor, + // or a second committed copy of a proposal, would apply a second + // time. Refuse to apply anything rather than risk it: the loop + // stops, and every propose waiter surfaces the stall. + tracing::error!( + %error, + "data-group apply loop cannot read its WAL to recover applied proposals; \ + refusing to apply committed entries" + ); + return; + } + }; + let mut ledger = ProposalLedger::from_records(&records, PROPOSAL_LEDGER_CAPACITY); + drop(records); while let Some(batch) = apply_rx.recv().await { // The floor is saved ONCE per batch, after the loop — never per entry. // `save_applied_index` lands a redb transaction, and redb commits at @@ -52,6 +77,10 @@ pub async fn run_apply_loop( // its outcome — `record` for the ones whose success means a durable // redo record, `skip` for the ones that apply no durable state at all. let mut prefix = AppliedPrefix::new(); + let mut gate = ProposalGate { + ledger: &mut ledger, + group_id: batch.group_id, + }; for entry in &batch.entries { // Decode once; reused for both the idempotency key and the // Array/Calvin fast-path match below. Returns 0 for @@ -74,6 +103,14 @@ pub async fn run_apply_loop( .map(|e| DatabaseId::new(e.database_id)) .unwrap_or(DatabaseId::DEFAULT); + // A second committed copy of a proposal this node already applied + // (a re-proposal after a leader change whose first copy also + // committed) resolves its waiter with the first copy's result and + // applies nothing. + if gate.skip_duplicate(&tracker, &mut prefix, entry.index, applied_key) { + continue; + } + // ── Array CRDT variants — handled on the Control Plane, bypass Data Plane ── if let Some(replicated) = replicated_opt { let target_vshard = replicated.vshard_id; @@ -108,7 +145,11 @@ pub async fn run_apply_loop( // submits through `submit_write`, so its redo is fsynced // before it reports success. A failure breaks the // prefix: the entry must stay replayable. - prefix.record(entry.index, applied_ok); + let outcome = EntryOutcome::Applied { + durable: applied_ok, + result: None, + }; + gate.settle(&mut prefix, entry.index, applied_key, outcome); continue; } ReplicatedWrite::ArraySchema { @@ -143,7 +184,20 @@ pub async fn run_apply_loop( // Data-Plane memtables and exists on disk only as the // redo record the funnel appends, which is why they must // route through `submit_write`. - prefix.record(entry.index, applied_ok); + let outcome = EntryOutcome::Applied { + durable: applied_ok, + result: None, + }; + gate.settle(&mut prefix, entry.index, applied_key, outcome); + continue; + } + ReplicatedWrite::TransactionRedo { .. } => { + let outcome = + apply_transaction_redo_entry(&state, &tracker, pos, &replicated).await; + // Advance the durable prefix when the entry's outcome is + // durable: its keyed redo record fsynced, or a final + // refusal cancelled in the WAL. + gate.settle(&mut prefix, entry.index, applied_key, outcome); continue; } ReplicatedWrite::CalvinReadResult { @@ -182,16 +236,16 @@ pub async fn run_apply_loop( } } - apply_generic_entry( + let outcome = apply_generic_entry( &state, &tracker, - &mut prefix, batch.group_id, entry, applied_key, database_id, ) .await; + gate.settle(&mut prefix, entry.index, applied_key, outcome); } // One save + one compaction check per batch, against the contiguous diff --git a/nodedb/src/control/distributed_applier/apply_loop/mod.rs b/nodedb/src/control/distributed_applier/apply_loop/mod.rs index f8ce17289..e89d1024a 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/mod.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/mod.rs @@ -17,6 +17,10 @@ //! - [`calvin_read_result`]: forwards a committed `CalvinReadResult` entry to //! the local Calvin scheduler. //! - [`write_dispatch`]: the generic decode + Data-Plane `submit_write` path. +//! - [`transaction_redo`]: a committed transaction's redo, stamped with its +//! Raft entry and applied through the WAL replay arms. +//! - [`proposal_gate`]: skips a second committed copy of an applied proposal +//! and records each applied proposal in the ledger and the WAL. //! - [`bookkeeping`]: applied-floor persistence + Raft log compaction trigger. //! - [`helpers`]: shared response/result classification helpers. @@ -25,6 +29,8 @@ mod bookkeeping; mod calvin_read_result; mod driver; mod helpers; +mod proposal_gate; +mod transaction_redo; mod write_dispatch; pub use driver::run_apply_loop; diff --git a/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs b/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs new file mode 100644 index 000000000..1d7f4c0a4 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs @@ -0,0 +1,100 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-entry proposal identity gate: skip a second committed copy of a +//! proposal, and record every durably applied proposal in the ledger. See +//! [`crate::control::distributed_applier::proposal_ledger`]. + +use crate::control::distributed_applier::applied_index::AppliedPrefix; +use crate::control::distributed_applier::proposal_ledger::{ + AppliedOutcome, PriorApply, ProposalLedger, +}; +use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; + +/// What applying one committed entry produced. +pub(super) enum EntryOutcome { + /// The entry carries no durable state and no proposal to record: it + /// neither advances nor breaks the applied prefix. + Skipped, + /// The entry was applied. `durable` says its outcome survives a restart. + /// `result` is what its waiter received, when the apply produced one. + Applied { + durable: bool, + result: Option, + }, +} + +/// The outcome a waiter received, in the form the ledger keeps. A refusal is +/// kept as its typed code; an error with no code keeps its message. +pub(super) fn ledger_outcome(result: &crate::Result) -> AppliedOutcome { + match result { + Ok(applied) => Ok(applied.clone()), + Err(crate::Error::DataPlane(code)) => Err(code.clone()), + Err(other) => Err(crate::bridge::envelope::ErrorCode::Internal { + detail: other.to_string(), + }), + } +} + +/// Per-batch state of the proposal gate. +pub(super) struct ProposalGate<'a> { + pub ledger: &'a mut ProposalLedger, + pub group_id: u64, +} + +impl ProposalGate<'_> { + /// When `proposal_key` already applied on this node, resolve the entry's + /// waiter with the first copy's outcome, extend the prefix, and return + /// `true`: the caller skips the entry. The first copy's outcome is + /// durable: the record that carries its key was durable before the floor + /// passed it, or it applied in this process. + pub fn skip_duplicate( + &self, + tracker: &ProposeTracker, + prefix: &mut AppliedPrefix, + log_index: u64, + proposal_key: u64, + ) -> bool { + let Some(prior) = self.ledger.prior(proposal_key) else { + return false; + }; + let result = match prior { + PriorApply::Outcome(Ok(applied)) => Ok(applied.clone()), + PriorApply::Outcome(Err(code)) => Err(crate::Error::DataPlane(code.clone())), + PriorApply::NoOutcome => Ok(AppliedWrite::unversioned(Vec::new())), + }; + tracing::debug!( + group_id = self.group_id, + log_index, + proposal_key, + "skipping a second committed copy of an applied proposal" + ); + tracker.complete(self.group_id, log_index, proposal_key, result); + prefix.record(log_index, true); + true + } + + /// Record `outcome` for the entry at `log_index`: extend or break the + /// prefix, and note a durable apply of a keyed proposal in the ledger. + /// + /// No separate marker is written: the entry's own records carry its key, + /// durable in the same write as its effect. + pub fn settle( + &mut self, + prefix: &mut AppliedPrefix, + log_index: u64, + proposal_key: u64, + outcome: EntryOutcome, + ) { + let (durable, result) = match outcome { + EntryOutcome::Skipped => { + prefix.skip(); + return; + } + EntryOutcome::Applied { durable, result } => (durable, result), + }; + if durable { + self.ledger.note(proposal_key, result); + } + prefix.record(log_index, durable); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs b/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs new file mode 100644 index 000000000..e33b63a1b --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs @@ -0,0 +1,92 @@ +//! Apply path for a committed `ReplicatedWrite::TransactionRedo` entry. +//! +//! Every replica, the proposer included, applies the entry through +//! [`apply_transaction_redo`]: the redo record is appended to this node's WAL, +//! its header carrying the entry's idempotency key, and installed through the +//! WAL replay arms. The key makes the record the entry's applied-marker, so an +//! entry re-delivered after a restart is recognised by the proposal ledger and +//! skipped before it reaches here. +//! +//! A refusal the Data Plane proves applied nothing (a constraint verdict) is +//! final: every replica reaches it at the same log position against the same +//! state, and the funnel cancels the record in the WAL before it returns. The +//! cancelling marker carries the entry's key, so the ledger counts the +//! refusal as the entry's outcome. It advances the durable prefix like a +//! success, because replaying the entry can only refuse it again. + +use std::sync::Arc; + +use crate::bridge::envelope::Status; +use crate::control::array_sync::raft_apply::AppliedPosition; +use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; +use crate::control::server::dispatch_utils::refusal_is_final; +use crate::control::state::SharedState; +use crate::control::wal_replication::ReplicatedEntry; +use crate::control::wal_replication::decode::transaction_redo_payload; +use crate::control::wal_replication::transaction_redo::{RedoTarget, apply_transaction_redo}; +use crate::types::{DatabaseId, TenantId, VShardId}; + +use super::helpers::committed_response_result; +use super::proposal_gate::{EntryOutcome, ledger_outcome}; + +/// Apply one committed `TransactionRedo` entry and resolve its propose +/// waiter. The outcome says whether the entry's effect is durable on this +/// node, which is what the caller's applied prefix records. +pub(super) async fn apply_transaction_redo_entry( + state: &Arc, + tracker: &Arc, + pos: AppliedPosition, + entry: &ReplicatedEntry, +) -> EntryOutcome { + let payload = match transaction_redo_payload(&entry.write) { + Ok(payload) => payload, + Err(error) => { + tracker.complete(pos.group_id, pos.log_index, pos.applied_key, Err(error)); + // The entry's own bytes are malformed; a re-delivery decodes the + // same bytes and fails the same way, so it holds the floor rather + // than skipping a committed transaction. + return EntryOutcome::Applied { + durable: false, + result: None, + }; + } + }; + let target = RedoTarget { + tenant_id: TenantId::new(entry.tenant_id), + database_id: DatabaseId::new(entry.database_id), + vshard_id: VShardId::new(entry.vshard_id), + }; + let submitted = apply_transaction_redo(state, target, &payload, pos.applied_key).await; + + let (result, durable) = match submitted { + Ok(outcome) if outcome.response.status == Status::Ok => { + (Ok(AppliedWrite::from_response(&outcome.response)), true) + } + Ok(outcome) => { + let refused_finally = outcome + .response + .error_code + .as_deref() + .is_some_and(refusal_is_final); + ( + committed_response_result(&outcome.response), + refused_finally, + ) + } + Err(error) => { + tracing::warn!( + group_id = pos.group_id, + index = pos.log_index, + error = %error, + "applying committed transaction redo failed" + ); + (Err(error), false) + } + }; + let applied = ledger_outcome(&result); + tracker.complete(pos.group_id, pos.log_index, pos.applied_key, result); + EntryOutcome::Applied { + durable, + result: Some(applied), + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs index 3ea436d53..a8230426f 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs @@ -15,7 +15,6 @@ use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::array_sync::raft_apply::{ AppliedPosition, ArrayCellTarget, apply_array_cell_write, }; -use crate::control::distributed_applier::applied_index::AppliedPrefix; use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; use crate::control::server::dispatch_utils::{ ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, @@ -25,19 +24,20 @@ use crate::control::wal_replication::from_replicated_entry; use crate::types::{DatabaseId, TraceId}; use super::helpers::{committed_response_result, deterministic_crdt_fence_noop}; +use super::proposal_gate::{EntryOutcome, ledger_outcome}; /// Decode `entry` and apply it: Raft-native array cell writes route through /// the array-open bootstrap and the write funnel; everything else dispatches -/// through the write funnel directly. Records the outcome into `prefix`. +/// through the write funnel directly. Returns the outcome the caller records +/// into its applied prefix. pub(super) async fn apply_generic_entry( state: &Arc, tracker: &Arc, - prefix: &mut AppliedPrefix, group_id: u64, entry: &LogEntry, applied_key: u64, database_id: DatabaseId, -) { +) -> EntryOutcome { let decoded = from_replicated_entry(&entry.data, Some(state.surrogate_assigner.as_ref())); let (tenant_id, vshard_id, plan, resolved_now_ms) = match decoded { Ok(Some(t)) => t, @@ -60,8 +60,7 @@ pub(super) async fn apply_generic_entry( // it buys nothing and costs a double-apply of every later // write in the batch. It applied no state, so it must not // advance the floor either. - prefix.skip(); - return; + return EntryOutcome::Skipped; } Err(e) => { tracing::warn!( @@ -83,8 +82,10 @@ pub(super) async fn apply_generic_entry( // state rather than on its own bytes, so a re-delivery can // legitimately succeed. Holding the floor below it is what // keeps it replayable. - prefix.record(entry.index, false); - return; + return EntryOutcome::Applied { + durable: false, + result: None, + }; } }; @@ -117,8 +118,10 @@ pub(super) async fn apply_generic_entry( plan, ) .await; - prefix.record(entry.index, applied_ok); - return; + return EntryOutcome::Applied { + durable: applied_ok, + result: None, + }; } let submitted = submit_write( @@ -149,6 +152,7 @@ pub(super) async fn apply_generic_entry( // install the byte-identical value every other replica does. durability: WalDurability::AppendHere { now_override: resolved_now_ms, + apply_key: applied_key, }, // Raft committed this entry at a fixed log index; every // replica applies it in that order. Re-entering the @@ -192,6 +196,7 @@ pub(super) async fn apply_generic_entry( }; let applied_ok = result.is_ok() || deterministic_crdt_fence_noop(&result); + let applied = ledger_outcome(&result); tracker.complete(group_id, entry.index, applied_key, result); // Extend the batch's durable prefix. On success `submit_write`'s @@ -202,5 +207,8 @@ pub(super) async fn apply_generic_entry( // neither a safe compaction boundary nor a safe restart floor; // breaking the prefix is what keeps a genuinely failed apply // replayable rather than silently skipped. - prefix.record(entry.index, applied_ok); + EntryOutcome::Applied { + durable: applied_ok, + result: Some(applied), + } } diff --git a/nodedb/src/control/distributed_applier/mod.rs b/nodedb/src/control/distributed_applier/mod.rs index 8b9f1fd9d..8d9ade3a9 100644 --- a/nodedb/src/control/distributed_applier/mod.rs +++ b/nodedb/src/control/distributed_applier/mod.rs @@ -8,9 +8,11 @@ pub mod applied_index; pub mod applier; pub mod apply_loop; +pub mod proposal_ledger; pub mod propose_tracker; pub use applied_index::{AppliedPrefix, save_applied_index}; pub use applier::{ApplyBatch, DistributedApplier, create_distributed_applier}; pub use apply_loop::run_apply_loop; +pub use proposal_ledger::{AppliedOutcome, PROPOSAL_LEDGER_CAPACITY, PriorApply, ProposalLedger}; pub use propose_tracker::{AppliedWrite, ProposeResult, ProposeTracker}; diff --git a/nodedb/src/control/distributed_applier/proposal_ledger.rs b/nodedb/src/control/distributed_applier/proposal_ledger.rs new file mode 100644 index 000000000..eb0e47bbc --- /dev/null +++ b/nodedb/src/control/distributed_applier/proposal_ledger.rs @@ -0,0 +1,260 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which replicated proposals this node already applied, keyed by proposal +//! identity. +//! +//! Every `ReplicatedEntry` carries an `idempotency_key` minted once by its +//! proposer. A re-proposal after `RetryableLeaderChange` sends the same bytes, +//! so both copies carry the same key. When the first copy committed at a log +//! index the proposer was not waiting on, both copies commit, and applying +//! both double-counts every non-idempotent effect (a materialized sum fold, a +//! columnar append). The apply loop checks this ledger before it applies an +//! entry and skips a copy whose key already applied. +//! +//! ## Durability +//! +//! Every WAL record an entry's apply appends carries the entry's key in its +//! header (`RecordHeader::apply_key`). The key is durable in the same write as +//! the effect it names, so no separate marker and no extra fsync is needed. +//! After a restart, [`ProposalLedger::from_records`] recovers the keys from +//! the replayed WAL: +//! +//! - A forward record the WAL cancelled with `WriteAborted` is absent from the +//! replayed records, so a refused entry stays replayable. +//! - A final refusal's `WriteAborted` marker carries the key, so the refusal +//! counts as the entry's outcome. +//! - An apply that writes no record of its own (a `wal=false` timeseries +//! ingest) appends a payload-free `ProposalApplied` record in its place. +//! +//! A duplicate delivered after a restart is recognised as long as the WAL +//! still retains a record of its original. +//! +//! ## Bound +//! +//! The ledger keeps the most recent [`PROPOSAL_LEDGER_CAPACITY`] keys across +//! every group and evicts the oldest past that. Keys are random 64-bit values, +//! so one set serves every group. A re-proposal commits within the proposer's +//! retry budget, which is orders of magnitude fewer entries than the +//! capacity. A checkpoint truncates the WAL, which bounds the recovered set +//! too. +//! +//! The ledger also keeps the outcome of every proposal this process applied, +//! so the waiter of a skipped copy receives what the first copy's waiter +//! received: the same payload and write version, or the same refusal. A key +//! recovered from the WAL has no outcome: no waiter from before the restart +//! survives it. + +use std::collections::{HashMap, VecDeque}; + +use nodedb_wal::WalRecord; + +use super::propose_tracker::AppliedWrite; +use crate::bridge::envelope::ErrorCode; + +/// Keys the ledger keeps before it evicts the oldest. +pub const PROPOSAL_LEDGER_CAPACITY: usize = 1 << 20; + +/// What an applied proposal's waiter received: the applied write, or the +/// typed refusal a durable refusal answered with. +pub type AppliedOutcome = Result; + +/// Applied proposal keys, in apply order. +#[derive(Debug)] +pub struct ProposalLedger { + results: HashMap>, + order: VecDeque, + capacity: usize, +} + +/// A proposal key the ledger already holds. +#[derive(Debug)] +pub enum PriorApply<'a> { + /// Applied in this process: the outcome its first copy produced. + Outcome(&'a AppliedOutcome), + /// Recovered from the WAL, or applied with no outcome to share. + NoOutcome, +} + +impl ProposalLedger { + /// An empty ledger that keeps `capacity` keys. + pub fn new(capacity: usize) -> Self { + Self { + results: HashMap::new(), + order: VecDeque::new(), + capacity: capacity.max(1), + } + } + + /// Recover the ledger from the keyed records in `records`, the node's + /// replayed WAL in WAL order, with cancelled forward records already + /// removed. A record with key `0` belongs to no proposal. + pub fn from_records(records: &[WalRecord], capacity: usize) -> Self { + let mut ledger = Self::new(capacity); + for record in records { + ledger.note(record.apply_key(), None); + } + ledger + } + + /// The prior apply of `proposal_key`, if any. Key `0` is the "no key" + /// sentinel of a legacy or synthetic entry and never matches. + pub fn prior(&self, proposal_key: u64) -> Option> { + if proposal_key == 0 { + return None; + } + Some(match self.results.get(&proposal_key)? { + Some(outcome) => PriorApply::Outcome(outcome), + None => PriorApply::NoOutcome, + }) + } + + /// Record that `proposal_key` applied, with the outcome its waiter + /// received when there is one. Key `0` is ignored. A key already held + /// keeps its place in the eviction order. + pub fn note(&mut self, proposal_key: u64, result: Option) { + if proposal_key == 0 { + return; + } + if self.results.insert(proposal_key, result).is_none() { + self.order.push_back(proposal_key); + } + while self.order.len() > self.capacity { + if let Some(oldest) = self.order.pop_front() { + self.results.remove(&oldest); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + + fn applied(payload: &[u8]) -> AppliedWrite { + AppliedWrite { + payload: payload.to_vec(), + write_version: Lsn::new(9), + } + } + + #[test] + fn a_second_copy_of_a_proposal_finds_the_first_copys_outcome() { + let mut ledger = ProposalLedger::new(8); + assert!(ledger.prior(77).is_none()); + ledger.note(77, Some(Ok(applied(b"first")))); + match ledger.prior(77) { + Some(PriorApply::Outcome(Ok(result))) => assert_eq!(result.payload, b"first"), + other => panic!("expected the first copy's result, got {other:?}"), + } + assert!(ledger.prior(78).is_none()); + } + + #[test] + fn the_no_key_sentinel_never_deduplicates() { + let mut ledger = ProposalLedger::new(8); + ledger.note(0, None); + assert!(ledger.prior(0).is_none()); + } + + #[test] + fn the_oldest_key_is_evicted_past_capacity() { + let mut ledger = ProposalLedger::new(2); + ledger.note(10, None); + ledger.note(11, None); + ledger.note(12, None); + assert!(ledger.prior(10).is_none()); + assert!(ledger.prior(11).is_some()); + assert!(ledger.prior(12).is_some()); + } + + #[test] + fn a_key_noted_twice_holds_one_eviction_slot() { + let mut ledger = ProposalLedger::new(2); + ledger.note(10, None); + ledger.note(10, None); + ledger.note(11, None); + assert!(ledger.prior(10).is_some()); + assert!(ledger.prior(11).is_some()); + } + + fn open_wal(dir: &tempfile::TempDir) -> crate::wal::WalManager { + crate::wal::WalManager::open_for_testing(&dir.path().join("test.wal")).expect("open wal") + } + + #[test] + fn a_redelivered_key_is_skipped_after_the_ledger_rebuilds_from_keyed_records() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + wal.with_apply_key(0xAB, || wal.append_put(tid, vs, db, b"keyed")) + .expect("append keyed put"); + wal.append_put(tid, vs, db, b"unkeyed") + .expect("append unkeyed put"); + wal.sync().expect("sync wal"); + + let ledger = ProposalLedger::from_records( + &wal.replay().expect("replay wal"), + PROPOSAL_LEDGER_CAPACITY, + ); + assert!(matches!(ledger.prior(0xAB), Some(PriorApply::NoOutcome))); + assert!(ledger.prior(0xCD).is_none()); + } + + #[test] + fn a_cancelled_forward_record_leaves_its_proposal_replayable() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + let forward = wal + .with_apply_key(0xAB, || wal.append_put(tid, vs, db, b"refused")) + .expect("append keyed put"); + wal.append_write_aborted(tid, vs, db, forward) + .expect("append unkeyed abort"); + let final_forward = wal + .with_apply_key(0xCD, || wal.append_put(tid, vs, db, b"refused for good")) + .expect("append keyed put"); + wal.with_apply_key(0xCD, || { + wal.append_write_aborted(tid, vs, db, final_forward) + }) + .expect("append keyed abort"); + wal.sync().expect("sync wal"); + + let ledger = ProposalLedger::from_records( + &wal.replay().expect("replay wal"), + PROPOSAL_LEDGER_CAPACITY, + ); + assert!( + ledger.prior(0xAB).is_none(), + "a non-final refusal stays replayable" + ); + assert!( + ledger.prior(0xCD).is_some(), + "a final refusal is the proposal's outcome" + ); + } + + #[test] + fn a_proposal_applied_marker_is_appended_only_inside_an_apply() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + assert!( + wal.append_proposal_applied(tid, vs, db) + .expect("append outside an apply") + .is_none() + ); + assert!( + wal.with_apply_key(0xEF, || wal.append_proposal_applied(tid, vs, db)) + .expect("append inside an apply") + .is_some() + ); + wal.sync().expect("sync wal"); + + let ledger = ProposalLedger::from_records( + &wal.replay().expect("replay wal"), + PROPOSAL_LEDGER_CAPACITY, + ); + assert!(ledger.prior(0xEF).is_some()); + } +} diff --git a/nodedb/src/control/distributed_applier/propose_tracker.rs b/nodedb/src/control/distributed_applier/propose_tracker.rs index 50351f6fb..b4ebb3aa0 100644 --- a/nodedb/src/control/distributed_applier/propose_tracker.rs +++ b/nodedb/src/control/distributed_applier/propose_tracker.rs @@ -23,7 +23,7 @@ use crate::types::Lsn; /// tracker resolves on the very node that applied locally — so the version this /// carries is that node's own, which is exactly what shard-local OCC validates /// against. -#[derive(Debug)] +#[derive(Debug, Clone)] pub struct AppliedWrite { /// The Data Plane's response payload, verbatim. pub payload: Vec, diff --git a/nodedb/src/control/gateway/router.rs b/nodedb/src/control/gateway/router.rs index 59050b599..52998a6a8 100644 --- a/nodedb/src/control/gateway/router.rs +++ b/nodedb/src/control/gateway/router.rs @@ -63,24 +63,12 @@ pub fn route_plan( strategy_fn: impl Fn(&str) -> PartitionStrategy, extractor: &dyn KeyExtractor, ) -> Result> { - // Commit-time meta-ops (ResolveTxn / TransactionBatch) carry no collection - // name, so their vShard cannot be derived here — the primary_vshard - // fallback would silently send them to vShard 0 and durably apply the - // commit batch on the wrong core. They are dispatched with the - // task's pre-classified `vshard_id` (see `dispatch_single_shard`), never - // through the gateway. - { - use nodedb_physical::physical_plan::MetaOp; - if matches!( - &plan, - PhysicalPlan::Meta(MetaOp::ResolveTxn { .. } | MetaOp::TransactionBatch { .. }) - ) { - return Err(crate::Error::Internal { - detail: "commit meta-op cannot be routed by the gateway; \ - dispatch it with the task's explicit vshard_id" - .to_owned(), - }); - } + if is_task_vshard_scoped(&plan) { + return Err(crate::Error::Internal { + detail: "transaction meta-op cannot be routed by the gateway; \ + dispatch it with the task's explicit vshard_id" + .to_owned(), + }); } // In single-node mode every plan runs locally. @@ -283,6 +271,30 @@ fn route_broadcast( routes } +/// Whether `plan` is a transaction meta-op that runs on the core of the +/// task's own `vshard_id`. +/// +/// These ops name no collection, so the router cannot derive their vShard. The +/// `primary_vshard` fallback would send them to vShard 0: a staged write would +/// land in an overlay the commit never reads, and a commit would apply on the +/// wrong core. Callers dispatch them with the task's `vshard_id`, never +/// through the gateway. +pub fn is_task_vshard_scoped(plan: &PhysicalPlan) -> bool { + use nodedb_physical::physical_plan::MetaOp; + matches!( + plan, + PhysicalPlan::Meta( + MetaOp::StageWrite { .. } + | MetaOp::MarkSavepoint { .. } + | MetaOp::RollbackToSavepoint { .. } + | MetaOp::DropTxnOverlay { .. } + | MetaOp::ResolveTxn { .. } + | MetaOp::TransactionBatch { .. } + | MetaOp::ApplyTransactionRedo { .. } + ) + ) +} + /// Determine the primary vShard for a plan by hashing the first collection name. /// /// Falls back to vShard 0 for plans that have no named collection (Meta ops). @@ -464,15 +476,31 @@ mod tests { unreachable!() } - /// Commit-time meta-ops carry no collection name, so the router cannot - /// derive their vShard — silently falling back to vShard 0 durably applies - /// the commit batch on the wrong core. They must be rejected here; - /// callers dispatch them with the task's pre-classified `vshard_id`. + /// Transaction meta-ops carry no collection name, so the router cannot + /// derive their vShard. The vShard 0 fallback stages a write in an overlay + /// the commit never reads, or applies a commit on the wrong core. #[test] - fn commit_meta_ops_are_rejected() { + fn transaction_meta_ops_are_rejected() { use nodedb_physical::physical_plan::MetaOp; + let txn_id = nodedb_types::id::TxnId::new(7); for plan in [ + PhysicalPlan::Meta(MetaOp::StageWrite { + plan: Box::new(PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + key: vec![], + rls_filters: vec![], + surrogate_ceiling: None, + })), + }), + PhysicalPlan::Meta(MetaOp::MarkSavepoint { txn_id }), + PhysicalPlan::Meta(MetaOp::RollbackToSavepoint { + txn_id, + value_marker: 0, + graph_marker: 0, + array_marker: 0, + }), + PhysicalPlan::Meta(MetaOp::DropTxnOverlay { txn_id }), PhysicalPlan::Meta(MetaOp::TransactionBatch { plans: vec![], txn_id: None, @@ -481,6 +509,11 @@ mod tests { txn_id: nodedb_types::id::TxnId::new(7), plans: vec![], }), + PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo: vec![], + collections: vec![], + sum_targets: vec![], + }), ] { for table in [None, Some(single_node_table())] { let result = route_plan( @@ -491,9 +524,10 @@ mod tests { |_| PartitionStrategy::CollectionHomed, &crate::control::gateway::UnwiredKeyExtractor, ); + assert!(is_task_vshard_scoped(&plan), "{plan:?}"); assert!( result.is_err(), - "commit meta-op must not be routable via the gateway: {plan:?}" + "transaction meta-op must not be routable via the gateway: {plan:?}" ); } } diff --git a/nodedb/src/control/planner/rls_injection/meta.rs b/nodedb/src/control/planner/rls_injection/meta.rs index be87bc245..2f5efdbde 100644 --- a/nodedb/src/control/planner/rls_injection/meta.rs +++ b/nodedb/src/control/planner/rls_injection/meta.rs @@ -101,7 +101,8 @@ pub(super) fn inject_meta(ctx: &RlsCtx<'_>, op: &mut MetaOp) -> crate::Result<() | MetaOp::RollbackToSavepoint { .. } | MetaOp::CalvinFlush { .. } | MetaOp::CalvinDrop { .. } - | MetaOp::CalvinResolve { .. } => Ok(()), + | MetaOp::CalvinResolve { .. } + | MetaOp::ApplyTransactionRedo { .. } => Ok(()), } } diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs b/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs index bf3b5f2a3..f56db1f3e 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs @@ -101,7 +101,8 @@ pub(super) fn apply_meta(ctx: &PermCtx<'_>, op: &mut MetaOp) -> crate::Result<() | MetaOp::RollbackToSavepoint { .. } | MetaOp::CalvinFlush { .. } | MetaOp::CalvinDrop { .. } - | MetaOp::CalvinResolve { .. } => Ok(()), + | MetaOp::CalvinResolve { .. } + | MetaOp::ApplyTransactionRedo { .. } => Ok(()), } } diff --git a/nodedb/src/control/security/identity/plan_permission.rs b/nodedb/src/control/security/identity/plan_permission.rs index 03f5c4c6d..ef66cd9ad 100644 --- a/nodedb/src/control/security/identity/plan_permission.rs +++ b/nodedb/src/control/security/identity/plan_permission.rs @@ -272,6 +272,9 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm // Mirrors `ResolveTxn`: Calvin scheduler's commit path, treated as Write though it doesn't mutate base state. PhysicalPlan::Meta(MetaOp::CalvinResolve { .. }) => Permission::Write, + // Installs a committed transaction's post-images into base state. + PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { .. }) => Permission::Write, + // KV engine: read operations. PhysicalPlan::Kv( KvOp::Get { .. } diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index 1100ec601..7636c4d46 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -57,7 +57,10 @@ pub async fn dispatch_authorized_autocommit_write( trace_id, event_source: crate::event::EventSource::User, txn_id: task.txn_id, - durability: WalDurability::AppendHere { now_override: None }, + durability: WalDurability::AppendHere { + now_override: None, + apply_key: 0, + }, }, ) .await @@ -87,7 +90,10 @@ pub(crate) async fn dispatch_authorized_autocommit_write_with_source( trace_id, event_source, txn_id: task.txn_id, - durability: WalDurability::AppendHere { now_override: None }, + durability: WalDurability::AppendHere { + now_override: None, + apply_key: 0, + }, }, ) .await @@ -236,7 +242,10 @@ pub(crate) async fn dispatch_autocommit_write( txn_id, // The funnel appends the WAL record under the admission guard just // before enqueue and stamps the minted LSN onto the `Request`. - durability: WalDurability::AppendHere { now_override: None }, + durability: WalDurability::AppendHere { + now_override: None, + apply_key: 0, + }, }, ) .await diff --git a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs index 8a8de71b1..2f3af9557 100644 --- a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs +++ b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs @@ -26,7 +26,7 @@ use std::sync::atomic::{AtomicU64, Ordering}; use crate::bridge::envelope::PhysicalPlan; use crate::control::server::shared::write_admission::plan_is_write; -use nodedb_physical::physical_plan::GraphOp; +use nodedb_physical::physical_plan::{GraphOp, MetaOp}; /// Count of writes acknowledged with no durable redo record despite belonging /// to an engine whose every write-class op mints one on this path. @@ -56,9 +56,10 @@ pub fn writes_acked_without_durability() -> u64 { /// * `Crdt` — constraint installs are Raft-log-replay durable and /// `RestoreToVersion` only computes a forward delta that a follow-up /// `Apply` logs; -/// * `Meta` — COMMIT's single transaction redo, the procedural batch flush and -/// the Calvin ops each own durability on their own path and arrive with the -/// LSN they minted (or none, by design); +/// * `Meta` — the procedural batch flush and the Calvin ops each own +/// durability on their own path and arrive with the LSN they minted (or +/// none, by design). `ApplyTransactionRedo` is the one `Meta` write this +/// funnel appends a record for, and it is held to the barrier; /// * `ClusterArray` — a coordinator-side routing wrapper: each owning shard's /// apply mints the redo for the cells it actually holds; /// * an empty `EdgePutBatch` / `EdgeDeleteBatch`, which has no edge to make @@ -86,6 +87,8 @@ pub(super) fn funnel_minted_redo_engine(plan: &PhysicalPlan) -> Option<&'static PhysicalPlan::Graph(GraphOp::EdgePutBatch { edges }) if edges.is_empty() => None, PhysicalPlan::Graph(GraphOp::EdgeDeleteBatch { edges }) if edges.is_empty() => None, PhysicalPlan::Graph(_) => Some("graph"), + // The committed-redo apply appends its `TransactionRedo` record here. + PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { .. }) => Some("transaction"), PhysicalPlan::Document(_) | PhysicalPlan::Crdt(_) | PhysicalPlan::Meta(_) @@ -138,6 +141,18 @@ mod tests { assert_eq!(funnel_minted_redo_engine(&kv_put()), Some("kv")); } + /// A committed transaction's redo apply mints its record in the funnel, + /// so an acknowledgement without it must trip the barrier. + #[test] + fn committed_redo_apply_requires_a_funnel_minted_redo() { + let plan = PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo: vec![1], + collections: vec!["c".into()], + sum_targets: Vec::new(), + }); + assert_eq!(funnel_minted_redo_engine(&plan), Some("transaction")); + } + /// A read carries no durability obligation at all. #[test] fn read_requires_no_redo() { diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index 7f72977c5..05a1975a6 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -30,3 +30,4 @@ pub(crate) use submit_write::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; pub(crate) use types::{AutocommitWrite, WriteDispatch}; +pub(crate) use write_abort::refusal_is_final; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index f41c02435..e9ae4c244 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -63,6 +63,25 @@ pub(crate) async fn submit_write( // `Some(collection)` for such a write, else `None`. let post_apply = wal_dispatch::plan_post_apply_redo(&plan); let appends_here = matches!(&durability, WalDurability::AppendHere { .. }); + let apply_key = match &durability { + WalDurability::AppendHere { apply_key, .. } => *apply_key, + WalDurability::CallerSupplied { .. } => 0, + }; + // A transaction redo's refusal is final: every replica reaches it at the + // same log position against the same state. Its abort marker carries the + // entry's key, so the proposal ledger counts the refusal as the entry's + // outcome after a restart. Any other refused write keeps its entry + // replayable, so its abort marker carries no key. + let final_refusal_key = if matches!( + plan, + nodedb_physical::physical_plan::PhysicalPlan::Meta( + nodedb_physical::physical_plan::MetaOp::ApplyTransactionRedo { .. } + ) + ) { + apply_key + } else { + 0 + }; // Durable-at-ack obligation, also computed before `plan` moves. `Some` only // for a write whose redo record THIS funnel is required to mint; a caller @@ -148,6 +167,8 @@ pub(crate) async fn submit_write( vshard_id, wal_lsn, appends_here, + final_refusal_key, + apply_key, post_apply, funnel_redo_engine, change_set, diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs index 12268b818..c4cb0921f 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -36,6 +36,12 @@ pub(super) struct ResponsePhaseInput { pub vshard_id: VShardId, pub wal_lsn: Option, pub appends_here: bool, + /// The idempotency key every record this write appends carries (see + /// `WalDurability::AppendHere`). + pub apply_key: u64, + /// The key a final refusal's abort marker carries, `0` when this write's + /// refusals are not final (see `AbortTarget::final_refusal_key`). + pub final_refusal_key: u64, pub post_apply: Option, pub funnel_redo_engine: Option<&'static str>, pub change_set: Option, @@ -68,6 +74,8 @@ pub(super) async fn collect_classify_and_finish( vshard_id, wal_lsn, appends_here, + apply_key, + final_refusal_key, post_apply, funnel_redo_engine, change_set, @@ -152,6 +160,7 @@ pub(super) async fn collect_classify_and_finish( vshard_id, wal_lsn, appends_here, + final_refusal_key, }, &response, ) @@ -170,14 +179,16 @@ pub(super) async fn collect_classify_and_finish( rollback_on_err( shared, &ddl_transition, - wal_dispatch::append_write_set_redo( - &shared.wal, - tenant_id, - vshard_id, - database_id, - collection, - &response.write_set, - ), + shared.wal.with_apply_key(apply_key, || { + wal_dispatch::append_write_set_redo( + &shared.wal, + tenant_id, + vshard_id, + database_id, + collection, + &response.write_set, + ) + }), )? } else { None diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs index 40a93089f..307baa6e1 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs @@ -75,18 +75,23 @@ pub(super) fn authorize_and_append( )?; let (wal_lsn, resolved_now_ms) = match durability { - WalDurability::AppendHere { now_override } => { + WalDurability::AppendHere { + now_override, + apply_key, + } => { let outcome = rollback_on_err( shared, &ddl_transition, - wal_dispatch::wal_append(WalAppendRequest { - wal: &shared.wal, - tenant_id, - vshard_id, - database_id, - plan: &plan, - credentials: None, - now_override, + shared.wal.with_apply_key(apply_key, || { + wal_dispatch::wal_append(WalAppendRequest { + wal: &shared.wal, + tenant_id, + vshard_id, + database_id, + plan: &plan, + credentials: None, + now_override, + }) }), )?; (outcome.lsn, outcome.resolved_now_ms) diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs index 642e12d5e..05a57d7e0 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs @@ -20,7 +20,15 @@ pub(crate) enum WalDurability { /// is what makes WAL-LSN order equal dispatcher-enqueue order per key; the /// strict-FIFO per-database WFQ then makes apply order follow enqueue /// order, so restart replay (in LSN order) cannot diverge from live state. - AppendHere { now_override: Option }, + /// + /// `apply_key` is the idempotency key of the replicated proposal this + /// write applies, `0` for a write no proposal carries. Every record the + /// funnel appends for the write carries it in its header, so the record + /// names the proposal it applied in the same durable write. + AppendHere { + now_override: Option, + apply_key: u64, + }, /// The caller already recorded this write's durability elsewhere — COMMIT's /// single `Transaction` record, the procedural batch flush, a trigger / /// sync path that owns its own funnel — and supplies the LSN it minted. diff --git a/nodedb/src/control/server/dispatch_utils/types.rs b/nodedb/src/control/server/dispatch_utils/types.rs index 0255da7a9..ae600589a 100644 --- a/nodedb/src/control/server/dispatch_utils/types.rs +++ b/nodedb/src/control/server/dispatch_utils/types.rs @@ -33,11 +33,11 @@ pub(crate) struct WriteDispatch { pub txn_id: Option, pub wal_lsn: Option, /// Wall-clock instant (ms since epoch) the Control Plane resolved at - /// WAL-append time for a TTL-bearing KV write's `expire_at_ms`. Stamped - /// onto the `Request` (same as `wal_lsn`) so the Data Plane installs the - /// SAME instant the durable WAL record carries instead of re-reading the - /// clock at apply time. `None` for reads, non-TTL writes, and writes whose - /// resolved instant is not (yet) threaded. + /// WAL-append time: a TTL-bearing KV write's expiry base, or a timeseries + /// ingest's default row timestamp. Stamped onto the `Request` (same as + /// `wal_lsn`) so the Data Plane installs the SAME instant the durable WAL + /// record carries instead of re-reading the clock at apply time. `None` + /// for reads and other writes. pub resolved_now_ms: Option, } diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index 353092846..dac86af8a 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -40,6 +40,11 @@ pub(crate) struct AbortTarget { /// Whether the write funnel appended the forward record. A caller that /// recorded durability elsewhere owns the undo semantics of its own record. pub appends_here: bool, + /// The idempotency key of the replicated proposal whose refusal is final, + /// `0` otherwise. A final refusal is the proposal's outcome: its abort + /// marker carries the key, and the proposal ledger rebuilt at boot counts + /// the proposal as applied. The cancelled forward record never counts. + pub final_refusal_key: u64, } /// Cancel a forward write record the Data Plane refused. @@ -89,12 +94,19 @@ pub(crate) async fn abort_refused_write( return Ok(()); } - let abort_lsn = shared.wal.append_write_aborted( - target.tenant_id, - target.vshard_id, - target.database_id, - wal_lsn, - )?; + let marker_key = if refusal_is_final(code) { + target.final_refusal_key + } else { + 0 + }; + let abort_lsn = shared.wal.with_apply_key(marker_key, || { + shared.wal.append_write_aborted( + target.tenant_id, + target.vshard_id, + target.database_id, + wal_lsn, + ) + })?; shared.wal.wait_durable(abort_lsn).await?; tracing::debug!( aborted_lsn = wal_lsn.as_u64(), @@ -104,6 +116,29 @@ pub(crate) async fn abort_refused_write( Ok(()) } +/// Whether a replicated proposal refused with `code` is refused for good: a +/// redelivery of the same entry against the same state refuses it again. +/// +/// A verdict that depends on this node's momentary load or on a transient +/// precondition is not final: another replica can apply the same entry, and +/// a redelivery here can too. That covers admission and capacity verdicts, +/// concurrency retries, the staging byte budget, and `RetryableRefusal`, +/// which a committed-redo apply answers with after it rolled a failed +/// install back. +pub(crate) fn refusal_is_final(code: &ErrorCode) -> bool { + write_definitely_not_applied(code) + && !matches!( + code, + ErrorCode::RetryableRefusal { .. } + | ErrorCode::RateExceeded { .. } + | ErrorCode::CollectionDraining { .. } + | ErrorCode::DispatchCapacity { .. } + | ErrorCode::ConflictRetry + | ErrorCode::OllpRetryRequired + | ErrorCode::TxnOverlayMemoryExceeded { .. } + ) +} + /// Whether `code` proves the write was refused without applying anything. /// /// `false` means "not established", not "the write applied". @@ -231,4 +266,20 @@ mod tests { })); assert!(!write_definitely_not_applied(&ErrorCode::DuplicateWrite)); } + + #[test] + fn a_constraint_verdict_is_final_and_a_retryable_one_is_not() { + assert!(refusal_is_final(&ErrorCode::RejectedPrevalidation { + reason: "sub-record does not decode".into(), + })); + assert!(!refusal_is_final(&ErrorCode::RetryableRefusal { + reason: "install rolled back".into(), + })); + assert!(!refusal_is_final(&ErrorCode::DispatchCapacity { + reason: "core 0 queue is full".into(), + })); + assert!(!refusal_is_final(&ErrorCode::Internal { + detail: "io_uring".into(), + })); + } } diff --git a/nodedb/src/control/server/exchange/all_cores/dispatch.rs b/nodedb/src/control/server/exchange/all_cores/dispatch.rs index 223a56739..a80b18578 100644 --- a/nodedb/src/control/server/exchange/all_cores/dispatch.rs +++ b/nodedb/src/control/server/exchange/all_cores/dispatch.rs @@ -168,7 +168,8 @@ pub(crate) async fn execute_plan_all_local_cores( | MetaOp::RollbackToSavepoint { .. } | MetaOp::RecordCalvinWriteVersions { .. } | MetaOp::CalvinFlush { .. } - | MetaOp::CalvinDrop { .. } => { + | MetaOp::CalvinDrop { .. } + | MetaOp::ApplyTransactionRedo { .. } => { generic_gather(state, tenant_id, database_id, plan, trace_id, txn_id).await } }, diff --git a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs index 5ed040d5d..540bb7a5e 100644 --- a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs @@ -7,6 +7,7 @@ use std::sync::Arc; use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext as GatewayQueryContext; +use crate::control::gateway::router::is_task_vshard_scoped; use crate::control::server::shared::clone_write::CloneCheckedOutcome; use crate::types::{Lsn, RequestId, TenantId, TraceId, TxnId, VShardId}; @@ -57,7 +58,14 @@ pub(super) async fn dispatch_authorized_single_task( CloneCheckedOutcome::Handled(resp) => return Ok(resp), CloneCheckedOutcome::Proceed(checked) => checked, }; - match ctx.state.gateway.get() { + // A staged write and the other transaction meta-ops run on the core of + // the task's own vShard. The gateway would route them to vShard 0. + let gateway = ctx + .state + .gateway + .get() + .filter(|_| !is_task_vshard_scoped(checked.plan())); + match gateway { Some(gateway) => { let query = GatewayQueryContext { tenant_id, diff --git a/nodedb/src/control/server/native/dispatch/sql_gateway.rs b/nodedb/src/control/server/native/dispatch/sql_gateway.rs index bf38963d7..94eade89b 100644 --- a/nodedb/src/control/server/native/dispatch/sql_gateway.rs +++ b/nodedb/src/control/server/native/dispatch/sql_gateway.rs @@ -14,6 +14,7 @@ use std::sync::Arc; use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext as GatewayQueryContext; +use crate::control::gateway::router::is_task_vshard_scoped; use crate::control::server::shared::clone_write::CloneCheckedOutcome; use crate::types::{Lsn, RequestId, TraceId}; use nodedb_physical::physical_task::PhysicalTask; @@ -77,7 +78,14 @@ pub(super) async fn dispatch_task_via_gateway( let database_id = checked.database_id(); let txn_id = checked.txn_id(); - match ctx.state.gateway.get() { + // A staged write and the other transaction meta-ops run on the core of + // the task's own vShard. The gateway would route them to vShard 0. + let gateway = ctx + .state + .gateway + .get() + .filter(|_| !is_task_vshard_scoped(checked.plan())); + match gateway { Some(gw) => { let gw_ctx = GatewayQueryContext { tenant_id, diff --git a/nodedb/src/control/server/native/dispatch/transaction.rs b/nodedb/src/control/server/native/dispatch/transaction.rs index 1e21eadb9..6a1a0dc6e 100644 --- a/nodedb/src/control/server/native/dispatch/transaction.rs +++ b/nodedb/src/control/server/native/dispatch/transaction.rs @@ -66,6 +66,10 @@ impl TxnDataPlane for NativeTxnDp<'_> { .await }) } + + fn event_source(&self) -> crate::event::EventSource { + crate::event::EventSource::User + } } pub(crate) fn handle_begin(ctx: &DispatchCtx<'_>, seq: u64) -> NativeResponse { diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs index d8be234d7..cbbff6317 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs @@ -26,7 +26,10 @@ impl NodeDbPgHandler { self.submit_authorized_to_data_plane( checked, user_id, - WalDurability::AppendHere { now_override: None }, + WalDurability::AppendHere { + now_override: None, + apply_key: 0, + }, ) .await } diff --git a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs index 241402b11..4f8a6a6e2 100644 --- a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs +++ b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs @@ -42,6 +42,10 @@ impl TxnDataPlane for PgwireTxnDp<'_> { ) -> Pin> + Send + 'a>> { Box::pin(self.handler.dispatch_task_no_wal(task, None, wal_lsn)) } + + fn event_source(&self) -> crate::event::EventSource { + crate::event::EventSource::User + } } impl NodeDbPgHandler { diff --git a/nodedb/src/control/server/shared/returning/inject.rs b/nodedb/src/control/server/shared/returning/inject.rs index 0b457bd2d..dc6cbb826 100644 --- a/nodedb/src/control/server/shared/returning/inject.rs +++ b/nodedb/src/control/server/shared/returning/inject.rs @@ -412,7 +412,8 @@ pub fn inject_returning_spec(plan: &mut PhysicalPlan, spec: ReturningSpec) { | MetaOp::CalvinFlush { .. } | MetaOp::CalvinDrop { .. } | MetaOp::ResolveTxn { .. } - | MetaOp::CalvinResolve { .. }, + | MetaOp::CalvinResolve { .. } + | MetaOp::ApplyTransactionRedo { .. }, ) | PhysicalPlan::Array( ArrayOp::OpenArray { .. } diff --git a/nodedb/src/control/server/shared/session/commit/single_shard.rs b/nodedb/src/control/server/shared/session/commit/single_shard.rs index a46be659f..fefb94750 100644 --- a/nodedb/src/control/server/shared/session/commit/single_shard.rs +++ b/nodedb/src/control/server/shared/session/commit/single_shard.rs @@ -1,23 +1,31 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Single-shard COMMIT: one `TransactionRedo` WAL record, then one atomic -//! `TransactionBatch` dispatch stamped with that record's LSN. +//! Single-shard COMMIT: resolve the transaction's staged post-images into one +//! redo record, then commit that record through the vShard's apply log. +//! +//! The vShard has one apply log. With Raft it is the data-group log: the +//! record is proposed there and every replica, this node included, appends +//! it to its own WAL and installs it when the entry commits. With no Raft the +//! record goes through the same apply on this node alone. COMMIT returns only +//! once the record is durable and installed here. use nodedb_physical::physical_plan::MetaOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -use crate::bridge::envelope::{PhysicalPlan, Response, Status}; +use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::gateway::RouteDecision; use crate::control::state::SharedState; +use crate::control::wal_replication::encode::transaction_redo_entry; +use crate::control::wal_replication::propose_replicated_entry; +use crate::control::wal_replication::transaction_redo::{ + RedoTarget, TransactionRedoPayload, apply_transaction_redo, +}; use super::super::outcome::{AbortReason, TxnDataPlane}; /// Single-shard commit: resolve the transaction's staged post-images into one -/// replayable `TransactionRedo` WAL record, then dispatch the buffered plans as -/// one atomic `TransactionBatch` stamped with that record's LSN. The redo -/// record restores restart durability for in-transaction writes into in-memory -/// secondary indexes (vector HNSW, FTS) that the base storage engine cannot -/// rebuild on its own. Returns `Some(reason)` on failure. +/// `RedoRecord`, then commit it through the vShard's apply log. Returns +/// `Some(reason)` on failure. pub(super) async fn dispatch_single_shard( state: &SharedState, dp: &impl TxnDataPlane, @@ -74,16 +82,13 @@ pub(super) async fn dispatch_single_shard( } }; - // Re-verify local vShard ownership immediately before the durable WAL - // append. `run_commit` resolved this vShard as `Local`, but a leadership - // handoff can land during the `ResolveTxn` await above. Without this - // re-check the transaction redo would be appended to a WAL this node no - // longer owns, and the batch dispatch below (which re-resolves leadership) - // would then reject the now-non-local commit — leaving an orphaned durable - // redo record behind while the client is told the commit aborted. Aborting - // here, BEFORE any durable write, keeps the failure side-effect-free and - // retryable: the client's retry re-enters `run_commit`, sees the vShard is - // non-local, and routes the commit through Calvin's replicated barrier. + // Re-verify local vShard leadership before anything durable happens. + // `run_commit` resolved this vShard as `Local` and validated the read set + // against this node's write versions, but a leadership handoff can land + // during the `ResolveTxn` await above. The new leader may already have + // applied writes this node's validation never saw, so the commit aborts + // side-effect-free and retryable: the retry sees the vShard is non-local + // and routes through Calvin's replicated barrier. if !matches!( crate::control::server::graph_dispatch::cluster_resolve::resolve_for_vshard( state, @@ -94,75 +99,91 @@ pub(super) async fn dispatch_single_shard( return Some(AbortReason::Serialization); } - // 2. Write-ahead the transaction as ONE replayable `TransactionRedo` record - // (each sub-op keeps its real engine `record_type`). `None` when the txn - // has no durable writes (all reads / CRDT / text). Its LSN stamps the - // batch install so the Data Plane records the committed write version for - // every key in the batch. - let wal_lsn = if redo.ops.is_empty() { - None - } else { - match state - .wal - .append_transaction_redo(tenant_id, vshard_id, database_id, &redo) - { - Ok(lsn) => Some(lsn), - Err(e) => { - return Some(AbortReason::Dispatch(crate::Error::Internal { - detail: format!("single-shard commit: transaction redo WAL append failed: {e}"), - })); - } - } - }; - let batch_task = PhysicalTask { - tenant_id, - vshard_id, + // A transaction with no durable write (all reads) installs nothing. + if redo.ops.is_empty() { + return None; + } + + // 2. Commit the record through the vShard's apply log. The payload carries + // the resolve-time bitemporal stamps inside the redo sub-records, so + // every replica installs each row on the same version key. + let payload = match TransactionRedoPayload::from_commit( + state, database_id, - plan: PhysicalPlan::Meta(MetaOp::TransactionBatch { - plans, - // Reuse the resolve-time bitemporal stamps recorded in this - // transaction's staging overlay so a `bitemporal=true` document put - // installs on the same version key the redo (WAL-appended just - // above) carries — otherwise a normal restart writes a second - // version of the row. - txn_id: Some(txn_id), - }), - post_set_op: PostSetOp::None, - txn_id: None, + tenant_id, + redo, + &plans, + dp.event_source(), + ) { + Ok(payload) => payload, + Err(e) => return Some(AbortReason::Dispatch(e)), }; - classify_batch_dispatch(dispatch_batch(dp, batch_task, wal_lsn).await) + commit_redo( + state, + RedoTarget { + tenant_id, + database_id, + vshard_id, + }, + &payload, + ) + .await } -/// The transaction batch's Data-Plane dispatch, with a fail point ahead of -/// the real call so a test can force this exact synchronous-failure branch -/// (`Option` back to `run_commit`, which compensates a -/// finalized DDL) without a real Data-Plane rejection or touching disk. -/// Compiles to a bare `dp.dispatch_no_wal` call outside the `failpoints` -/// feature. -async fn dispatch_batch( - dp: &impl TxnDataPlane, - batch_task: PhysicalTask, - wal_lsn: Option, -) -> crate::Result { - crate::fail_point_err!("commit::single_shard_batch_dispatch", |detail| { - crate::Error::Internal { detail } - }); - dp.dispatch_no_wal(batch_task, wal_lsn).await -} - -/// Convert a transaction-batch dispatch result into a commit abort reason, if -/// any. `dispatch_no_wal` returns `Ok(Response { status: Error, .. })` for a -/// failed batch rather than a Rust `Err` — the status must be checked -/// explicitly or a failed sub-plan reports as COMMIT success. -fn classify_batch_dispatch(result: crate::Result) -> Option { - match result { - Err(e) => { - tracing::warn!(error = %e, "transaction batch dispatch failed"); - Some(AbortReason::Dispatch(e)) +/// Commit `payload` through the vShard's apply log and wait until it is +/// durable and installed on this node. +/// +/// A fail point ahead of the real commit lets a test force this exact +/// synchronous-failure branch (`Option` back to `run_commit`, +/// which compensates a finalized DDL) without touching disk. +async fn commit_redo( + state: &SharedState, + target: RedoTarget, + payload: &TransactionRedoPayload, +) -> Option { + if let Err(e) = inject_commit_failure() { + return Some(AbortReason::Dispatch(e)); + } + match state.async_raft_proposer() { + // The proposer forwards to the group leader and returns once the entry + // is committed and applied on this node by the apply loop. + Some(proposer) => { + let entry = transaction_redo_entry( + target.tenant_id, + target.database_id, + target.vshard_id, + payload, + ); + match propose_replicated_entry(state, proposer, entry).await { + Ok(_) => None, + Err(crate::Error::DataPlane(code)) => { + Some(AbortReason::BatchRejected { code: Some(code) }) + } + Err(e) => { + tracing::warn!(error = %e, "transaction redo commit failed"); + Some(AbortReason::Dispatch(e)) + } + } } - Ok(resp) if resp.status != Status::Ok => Some(AbortReason::BatchRejected { - code: resp.error_code.as_deref().cloned(), - }), - Ok(_) => None, + // No Raft: this node is the only replica, and its apply is the log. + None => match apply_transaction_redo(state, target, payload, 0).await { + Ok(outcome) if outcome.response.status == Status::Ok => None, + Ok(outcome) => Some(AbortReason::BatchRejected { + code: outcome.response.error_code.as_deref().cloned(), + }), + Err(e) => { + tracing::warn!(error = %e, "transaction redo commit failed"); + Some(AbortReason::Dispatch(e)) + } + }, } } + +/// The `commit::single_shard_redo_commit` fail point. Compiles to `Ok(())` +/// outside the `failpoints` feature. +fn inject_commit_failure() -> crate::Result<()> { + crate::fail_point_err!("commit::single_shard_redo_commit", |detail| { + crate::Error::Internal { detail } + }); + Ok(()) +} diff --git a/nodedb/src/control/server/shared/session/lifecycle.rs b/nodedb/src/control/server/shared/session/lifecycle.rs index b472fb788..207b530a5 100644 --- a/nodedb/src/control/server/shared/session/lifecycle.rs +++ b/nodedb/src/control/server/shared/session/lifecycle.rs @@ -224,6 +224,10 @@ mod tests { }) }) } + + fn event_source(&self) -> crate::event::EventSource { + crate::event::EventSource::User + } } /// A benign staged write task homed on `vshard`. The plan content is irrelevant diff --git a/nodedb/src/control/server/shared/session/outcome.rs b/nodedb/src/control/server/shared/session/outcome.rs index 96641dfe5..18c971646 100644 --- a/nodedb/src/control/server/shared/session/outcome.rs +++ b/nodedb/src/control/server/shared/session/outcome.rs @@ -81,4 +81,8 @@ pub trait TxnDataPlane { task: PhysicalTask, wal_lsn: Option, ) -> Pin> + Send + 'a>>; + + /// The source the transaction's committed writes carry into the Event + /// Plane, on every replica that applies them. + fn event_source(&self) -> crate::event::EventSource; } diff --git a/nodedb/src/control/server/shared/session/savepoint_ops.rs b/nodedb/src/control/server/shared/session/savepoint_ops.rs index 4987f48c4..037fa81be 100644 --- a/nodedb/src/control/server/shared/session/savepoint_ops.rs +++ b/nodedb/src/control/server/shared/session/savepoint_ops.rs @@ -311,6 +311,10 @@ mod tests { }) }) } + + fn event_source(&self) -> crate::event::EventSource { + crate::event::EventSource::User + } } /// A benign staged write task homed on `vshard`. The plan content is irrelevant diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 0d7c7d001..4d5b169c9 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -350,7 +350,8 @@ pub fn plan_requires_txn_buffering(plan: &PhysicalPlan) -> bool { | MetaOp::CalvinFlush { .. } | MetaOp::CalvinDrop { .. } | MetaOp::CalvinResolve { .. } - | MetaOp::ResolveTxn { .. }, + | MetaOp::ResolveTxn { .. } + | MetaOp::ApplyTransactionRedo { .. }, ) => false, // ---- Array reads / DDL / Flush: `to_replicated_entry` returns `None` — matches oracle. diff --git a/nodedb/src/control/server/wal_dispatch/columnar.rs b/nodedb/src/control/server/wal_dispatch/columnar.rs index 984dc5127..a4f74505b 100644 --- a/nodedb/src/control/server/wal_dispatch/columnar.rs +++ b/nodedb/src/control/server/wal_dispatch/columnar.rs @@ -27,8 +27,8 @@ pub(super) fn wal_append_columnar_op( collection, payload, format: _, - intent: _, - on_conflict_updates: _, + intent, + on_conflict_updates, surrogates, schema_bytes: _, provenance, @@ -44,16 +44,23 @@ pub(super) fn wal_append_columnar_op( rls_filters: _, } => { // Encode a map-shaped `ColumnarWalRecord` carrying the per-row - // cross-engine surrogates so replay restores the exact same - // identity after a restart. `surrogates` is index-aligned with the + // cross-engine surrogates and the insert's conflict policy, so + // replay restores the exact same identity and decides each + // existing key the way the live insert did. `surrogates` is index-aligned with the // rows in `payload`. The map shape is distinct from the legacy // 4-tuple array, so old on-disk records still decode via the // replay fallback path. let wal_payload = super::timeseries::encode_columnar_batch_payload( - collection.as_str(), - payload, - provenance.as_ref(), - surrogates, + super::timeseries::ColumnarBatchRecord { + collection: collection.as_str(), + payload, + provenance: provenance.as_ref(), + surrogates, + conflict_policy: &crate::wal::ColumnarConflictPolicy { + intent: *intent, + on_conflict_updates: on_conflict_updates.clone(), + }, + }, )?; Some(wal.append_timeseries_batch(tenant_id, vshard_id, database_id, &wal_payload)?) } diff --git a/nodedb/src/control/server/wal_dispatch/core.rs b/nodedb/src/control/server/wal_dispatch/core.rs index 782074b07..c6566851f 100644 --- a/nodedb/src/control/server/wal_dispatch/core.rs +++ b/nodedb/src/control/server/wal_dispatch/core.rs @@ -11,7 +11,8 @@ use super::super::wal_dispatch_kv; /// Outcome of [`wal_append_if_write`] / [`wal_append_if_write_with_creds`]: /// the allocated WAL LSN (if a durable record was appended) and, for a -/// TTL-bearing KV write, the wall-clock instant resolved at append time. +/// TTL-bearing KV write or a timeseries ingest, the wall-clock instant +/// resolved at append time. /// /// `resolved_now_ms` mirrors `lsn`'s cross-plane contract: the caller stamps /// it onto the dispatched `Request` (via `WriteDispatch` / `DataPlaneDispatch`) @@ -29,8 +30,8 @@ pub struct WalAppendOutcome { /// WAL-bypassed writes. pub lsn: Option, /// Wall-clock instant (ms since epoch) resolved for a TTL-bearing KV - /// write's `expire_at_ms`. `None` for every non-KV plan and every KV write - /// without a TTL. + /// write's `expire_at_ms`, or a timeseries ingest's untimed rows. `None` + /// for every other write. pub resolved_now_ms: Option, } @@ -46,12 +47,13 @@ pub struct WalAppendRequest<'a> { /// Credential store for the timeseries `wal=false` bypass check. pub credentials: Option<&'a CredentialStore>, /// Wall-clock instant (ms since epoch) to resolve a TTL-bearing KV write's - /// `expire_at_ms` against, instead of reading this node's clock. `Some` - /// only when the instant was decided elsewhere and the durable record must - /// carry that exact value: a Raft-committed entry carries the instant the - /// proposing node resolved, and every replica's redo record — like every - /// replica's live apply — must install it verbatim, or a replica's WAL - /// replay resurrects a different `expire_at_ms` than its peers. + /// `expire_at_ms`, or a timeseries ingest's untimed rows, against, instead + /// of reading this node's clock. `Some` only when the instant was decided + /// elsewhere and the durable record must carry that exact value: a + /// Raft-committed entry carries the instant the proposing node resolved, + /// and every replica's redo record — like every replica's live apply — + /// must install it verbatim, or a replica's WAL replay resurrects a + /// different value than its peers. pub now_override: Option, } @@ -133,16 +135,24 @@ pub fn wal_append(req: WalAppendRequest<'_>) -> crate::Result PhysicalPlan::Columnar(op) => { super::columnar::wal_append_columnar_op(wal, tenant_id, vshard_id, database_id, op)? } - PhysicalPlan::Timeseries(op) => super::timeseries::wal_append_timeseries_op( - wal, - tenant_id, - vshard_id, - database_id, - op, - credentials, - )?, - // KV write operations — delegated to wal_dispatch_kv. The only engine - // that resolves a wall-clock instant (TTL `expire_at_ms`), threaded back + // A timeseries ingest resolves the instant its untimed rows take, + // threaded back out via `resolved_now_ms`. + PhysicalPlan::Timeseries(op) => { + let outcome = + super::timeseries::wal_append_timeseries_op(super::timeseries::TimeseriesAppend { + wal, + tenant_id, + vshard_id, + database_id, + op, + credentials, + now_override, + })?; + resolved_now_ms = outcome.resolved_now_ms; + outcome.lsn + } + // KV write operations — delegated to wal_dispatch_kv. A TTL-bearing + // write resolves a wall-clock instant (`expire_at_ms`), threaded back // out via `resolved_now_ms`. PhysicalPlan::Kv(kv_op) => { let outcome = wal_dispatch_kv::wal_append_kv_op( @@ -165,6 +175,12 @@ pub fn wal_append(req: WalAppendRequest<'_>) -> crate::Result PhysicalPlan::Spatial(op) => { super::spatial::wal_append_spatial_op(wal, tenant_id, vshard_id, database_id, op)? } + // A committed transaction's redo record: one `TransactionRedo` WAL + // record per apply, on every replica, in apply order. + PhysicalPlan::Meta(nodedb_physical::physical_plan::MetaOp::ApplyTransactionRedo { + redo, + .. + }) => Some(wal.append_transaction_redo_bytes(tenant_id, vshard_id, database_id, redo)?), // NotAWrite — reads / query ops / control commands. `Meta` durable // writes (WAL append, transaction batch, Calvin apply) are logged on // their own dedicated paths, never through this autocommit oracle; @@ -210,6 +226,39 @@ mod tests { .expect("expected record of this type") } + #[test] + fn committed_redo_apply_appends_its_redo_record_verbatim() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(dir.path()); + let redo = crate::wal::RedoRecord { + version: 1, + ops: Vec::new(), + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let plan = PhysicalPlan::Meta( + nodedb_physical::physical_plan::MetaOp::ApplyTransactionRedo { + redo: redo.clone(), + collections: Vec::new(), + sum_targets: Vec::new(), + }, + ); + + let outcome = wal_append_if_write( + &wal, + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + &plan, + ) + .expect("append"); + assert!(outcome.lsn.is_some(), "the redo apply must mint an LSN"); + + let record = last_record_of_type(&wal, nodedb_wal::record::RecordType::TransactionRedo); + assert_eq!(record.payload, redo); + } + #[test] fn fts_index_doc_appends_and_decodes() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/nodedb/src/control/server/wal_dispatch/crdt.rs b/nodedb/src/control/server/wal_dispatch/crdt.rs index bc935f5ab..73743e2c6 100644 --- a/nodedb/src/control/server/wal_dispatch/crdt.rs +++ b/nodedb/src/control/server/wal_dispatch/crdt.rs @@ -5,17 +5,37 @@ #![deny(clippy::wildcard_enum_match_arm)] use nodedb_physical::physical_plan::CrdtOp; +use nodedb_wal::record::RecordType; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::manager::WalManager; +/// Which CRDT WAL record class a `CrdtOp` write journals as. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum CrdtRecordKind { + /// A Loro delta or snapshot import. + Delta, + /// A block-list mutation intent. + ListOp, + /// A document-row mutation intent. + DocOp, +} + +impl CrdtRecordKind { + /// The WAL record type this class is journalled under. + pub(crate) fn record_type(self) -> RecordType { + match self { + Self::Delta => RecordType::CrdtDelta, + Self::ListOp => RecordType::CrdtListOp, + Self::DocOp => RecordType::CrdtDocOp, + } + } +} + /// Append the WAL record for a single `CrdtOp`, returning the allocated LSN for /// the delta / snapshot-import / block-list write variants (`Some`) or `None` /// for every read / constraint / policy variant that carries no durable /// per-write effect on THIS path. -/// -/// The match over [`CrdtOp`] is **exhaustive** (`wildcard_enum_match_arm` is -/// denied), so a future write variant cannot silently become non-durable. pub(super) fn wal_append_crdt_op( wal: &WalManager, tenant_id: TenantId, @@ -23,6 +43,35 @@ pub(super) fn wal_append_crdt_op( database_id: DatabaseId, op: &CrdtOp, ) -> crate::Result> { + let Some((kind, payload)) = encode_crdt_op_record(op)? else { + return Ok(None); + }; + let lsn = match kind { + CrdtRecordKind::Delta => { + wal.append_crdt_delta(tenant_id, vshard_id, database_id, &payload)? + } + CrdtRecordKind::ListOp => { + wal.append_crdt_list_op(tenant_id, vshard_id, database_id, &payload)? + } + CrdtRecordKind::DocOp => { + wal.append_crdt_doc_op(tenant_id, vshard_id, database_id, &payload)? + } + }; + Ok(Some(lsn)) +} + +/// Encode the WAL record a single `CrdtOp` write journals as: its record class +/// and payload. `None` for every read / constraint / policy variant that +/// carries no durable per-write effect on this path. +/// +/// The match over [`CrdtOp`] is **exhaustive** (`wildcard_enum_match_arm` is +/// denied), so a future write variant cannot silently become non-durable. +/// Shared by the autocommit WAL append and the transaction resolver, so a +/// CRDT write inside a transaction journals the exact record its autocommit +/// form does. +pub(crate) fn encode_crdt_op_record( + op: &CrdtOp, +) -> crate::Result)>> { let appended = match op { CrdtOp::Apply { collection, @@ -47,7 +96,7 @@ pub(super) fn wal_append_crdt_op( format: "msgpack".into(), detail: format!("wal crdt delta: {e}"), })?; - Some(wal.append_crdt_delta(tenant_id, vshard_id, database_id, &crdt_payload)?) + Some((CrdtRecordKind::Delta, crdt_payload)) } CrdtOp::ApplyAuthenticated { collection, @@ -82,7 +131,7 @@ pub(super) fn wal_append_crdt_op( format: "msgpack".into(), detail: format!("wal authenticated crdt delta: {e}"), })?; - Some(wal.append_crdt_delta(tenant_id, vshard_id, database_id, &crdt_payload)?) + Some((CrdtRecordKind::Delta, crdt_payload)) } CrdtOp::ImportSnapshot { collection, bytes, .. @@ -103,7 +152,7 @@ pub(super) fn wal_append_crdt_op( format: "msgpack".into(), detail: format!("wal crdt snapshot import: {e}"), })?; - Some(wal.append_crdt_delta(tenant_id, vshard_id, database_id, &crdt_payload)?) + Some((CrdtRecordKind::Delta, crdt_payload)) } CrdtOp::ListInsert { collection, @@ -125,7 +174,7 @@ pub(super) fn wal_append_crdt_op( fields_json: fields_json.clone(), }; let bytes = encode_crdt_list_op_payload(payload)?; - Some(wal.append_crdt_list_op(tenant_id, vshard_id, database_id, &bytes)?) + Some((CrdtRecordKind::ListOp, bytes)) } CrdtOp::ListDelete { collection, @@ -141,7 +190,7 @@ pub(super) fn wal_append_crdt_op( index: *index as u64, }; let bytes = encode_crdt_list_op_payload(payload)?; - Some(wal.append_crdt_list_op(tenant_id, vshard_id, database_id, &bytes)?) + Some((CrdtRecordKind::ListOp, bytes)) } CrdtOp::ListMove { collection, @@ -159,7 +208,7 @@ pub(super) fn wal_append_crdt_op( to_index: *to_index as u64, }; let bytes = encode_crdt_list_op_payload(payload)?; - Some(wal.append_crdt_list_op(tenant_id, vshard_id, database_id, &bytes)?) + Some((CrdtRecordKind::ListOp, bytes)) } CrdtOp::DocUpsert { collection, @@ -183,7 +232,7 @@ pub(super) fn wal_append_crdt_op( partial: *partial, }; let bytes = encode_crdt_doc_op_payload(payload)?; - Some(wal.append_crdt_doc_op(tenant_id, vshard_id, database_id, &bytes)?) + Some((CrdtRecordKind::DocOp, bytes)) } CrdtOp::DocDelete { collection, @@ -198,7 +247,7 @@ pub(super) fn wal_append_crdt_op( surrogate: surrogate.as_u32(), }; let bytes = encode_crdt_doc_op_payload(payload)?; - Some(wal.append_crdt_doc_op(tenant_id, vshard_id, database_id, &bytes)?) + Some((CrdtRecordKind::DocOp, bytes)) } // NotAWrite — reads / query ops / DDL that produces no engine mutation here CrdtOp::Read { .. } diff --git a/nodedb/src/control/server/wal_dispatch/mod.rs b/nodedb/src/control/server/wal_dispatch/mod.rs index 6c3cf1433..903cee090 100644 --- a/nodedb/src/control/server/wal_dispatch/mod.rs +++ b/nodedb/src/control/server/wal_dispatch/mod.rs @@ -34,11 +34,13 @@ pub use write_set_redo::{append_write_set_redo, mint_dispatch_local_redo, plan_p // Payload encoders shared by the autocommit WAL path and transaction resolve, so // each engine's record shape lives in exactly one place. +pub(crate) use crdt::encode_crdt_op_record; pub(crate) use graph_labels::encode_graph_node_label_payload; +pub(crate) use text::encode_text_op_record; +#[cfg(test)] +pub(crate) use timeseries::{TimeseriesIngestRecord, encode_timeseries_ingest_payload}; pub(crate) use timeseries::{ - encode_columnar_batch_payload, encode_columnar_dml_payload, - encode_columnar_resolved_dml_payload, encode_columnar_truncate_payload, - encode_timeseries_batch_payload_with_format, + encode_columnar_truncate_payload, encode_timeseries_batch_payload_with_format, }; pub(crate) use vector::{ VectorDirectDeleteRecord, VectorDirectTruncateRecord, VectorDirectUpdatePayload, diff --git a/nodedb/src/control/server/wal_dispatch/text.rs b/nodedb/src/control/server/wal_dispatch/text.rs index 20d27c9e2..9af27dc41 100644 --- a/nodedb/src/control/server/wal_dispatch/text.rs +++ b/nodedb/src/control/server/wal_dispatch/text.rs @@ -3,12 +3,11 @@ //! WAL append dispatch for `PhysicalPlan::Text(TextOp)`. use nodedb_physical::physical_plan::TextOp; +use nodedb_wal::record::RecordType; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::manager::WalManager; -use super::super::wal_dispatch_fts_spatial; - /// Append the WAL record for a single `TextOp`, returning the allocated LSN /// for the FTS write variants (`Some`) or `None` for every read/search /// variant, which carries no durable per-write effect. @@ -29,7 +28,24 @@ pub(crate) fn wal_append_text_op( database_id: DatabaseId, op: &TextOp, ) -> crate::Result> { - let appended = match op { + let Some((record_type, payload)) = encode_text_op_record(op)? else { + return Ok(None); + }; + let lsn = if record_type == RecordType::FtsIndex { + wal.append_fts_index(tenant_id, vshard_id, database_id, &payload)? + } else { + wal.append_fts_delete(tenant_id, vshard_id, database_id, &payload)? + }; + Ok(Some(lsn)) +} + +/// Encode the WAL record a single `TextOp` write journals as: its record type +/// (`FtsIndex` or `FtsDelete`) and payload. `None` for every read / search / +/// analyzer-config variant. Shared by the autocommit WAL append and the +/// transaction resolver, so an FTS write inside a transaction journals the +/// exact record its autocommit form does. +pub(crate) fn encode_text_op_record(op: &TextOp) -> crate::Result)>> { + let encoded = match op { TextOp::FtsIndexDoc { collection, surrogate, @@ -41,13 +57,10 @@ pub(crate) fn wal_append_text_op( let prov = provenance.clone().unwrap_or_default(); let payload = nodedb_wal::record::FtsIndexPayload::new(prov, collection.as_str(), &doc_id, text); - Some(wal_dispatch_fts_spatial::wal_append_fts_index( - wal, - tenant_id, - vshard_id, - database_id, - &payload, - )?) + Some(( + RecordType::FtsIndex, + payload.to_bytes().map_err(crate::Error::Wal)?, + )) } TextOp::FtsDeleteDoc { collection, @@ -59,13 +72,10 @@ pub(crate) fn wal_append_text_op( let prov = provenance.clone().unwrap_or_default(); let payload = nodedb_wal::record::FtsDeletePayload::new(prov, collection.as_str(), &doc_id); - Some(wal_dispatch_fts_spatial::wal_append_fts_delete( - wal, - tenant_id, - vshard_id, - database_id, - &payload, - )?) + Some(( + RecordType::FtsDelete, + payload.to_bytes().map_err(crate::Error::Wal)?, + )) } // Reads / scans / analyzer config: no durable effect. TextOp::Search { .. } @@ -75,5 +85,5 @@ pub(crate) fn wal_append_text_op( | TextOp::HybridSearchTriple { .. } | TextOp::SetTextConfig { .. } => None, }; - Ok(appended) + Ok(encoded) } diff --git a/nodedb/src/control/server/wal_dispatch/timeseries.rs b/nodedb/src/control/server/wal_dispatch/timeseries.rs index cd7e3ece3..31401c139 100644 --- a/nodedb/src/control/server/wal_dispatch/timeseries.rs +++ b/nodedb/src/control/server/wal_dispatch/timeseries.rs @@ -11,50 +11,99 @@ use crate::control::security::credential::CredentialStore; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::manager::WalManager; -/// Append the WAL record for a `TimeseriesOp`: LSN for ingest, `None` for -/// `Scan` or a `wal=false` collection. `credentials` is threaded through -/// solely for the per-collection bypass check. +/// Inputs of [`wal_append_timeseries_op`]. +pub(super) struct TimeseriesAppend<'a> { + pub wal: &'a WalManager, + pub tenant_id: TenantId, + pub vshard_id: VShardId, + pub database_id: DatabaseId, + pub op: &'a TimeseriesOp, + /// Threaded through solely for the per-collection `wal=false` bypass. + pub credentials: Option<&'a CredentialStore>, + /// The instant decided elsewhere for the ingest's untimed rows (see + /// `WalAppendRequest::now_override`). + pub now_override: Option, +} + +/// What [`wal_append_timeseries_op`] appended and resolved. +pub(super) struct TimeseriesAppendOutcome { + /// LSN for an ingest or truncate, `None` for a scan. A `wal=false` ingest + /// carries its `ProposalApplied` marker's LSN inside a replicated + /// proposal's apply, and `None` outside one. + pub lsn: Option, + /// The instant an ingest's untimed rows take. `Some` for every ingest, + /// WAL-bypassed or not, so the live apply stores the same rows restart + /// replay does. + pub resolved_now_ms: Option, +} + +/// Append the WAL record for a `TimeseriesOp`. +/// +/// An ingest resolves its default row timestamp here, once: `now_override` +/// when the instant was decided elsewhere, else this node's clock. The record +/// carries it, and the live apply receives it as `resolved_now_ms`. pub(super) fn wal_append_timeseries_op( - wal: &WalManager, - tenant_id: TenantId, - vshard_id: VShardId, - database_id: DatabaseId, - op: &TimeseriesOp, - credentials: Option<&CredentialStore>, -) -> crate::Result> { - let appended = match op { + append: TimeseriesAppend<'_>, +) -> crate::Result { + let TimeseriesAppend { + wal, + tenant_id, + vshard_id, + database_id, + op, + credentials, + now_override, + } = append; + let outcome = match op { TimeseriesOp::Ingest { collection, payload, - format: _, + format, provenance, .. } => { + let now_ms = now_override.unwrap_or_else(crate::engine::kv::current_ms); // WAL bypass: skip WAL if collection has wal=false in timeseries_config. - if let Some(creds) = credentials - && let Ok(Some(coll)) = creds.catalog().get_collection( - database_id, - tenant_id.as_u64(), - collection.as_str(), + let bypassed = credentials.is_some_and(|creds| { + matches!( + creds.catalog().get_collection( + database_id, + tenant_id.as_u64(), + collection.as_str(), + ), + Ok(Some(coll)) if coll + .get_timeseries_config() + .is_some_and(|config| { + config.get("wal").and_then(|v| v.as_str()) == Some("false") + }) ) - && let Some(config) = coll.get_timeseries_config() - && config.get("wal").and_then(|v| v.as_str()) == Some("false") - { - // WAL bypassed — acceptable data loss of last flush interval on crash. - None + }); + let lsn = if bypassed { + // WAL bypassed: the rows since the last flush are lost on a + // crash. A replicated proposal's apply still records its key, + // in a payload-free marker that stands in for the forward + // record, so a second committed copy of the proposal is + // skipped after a restart. Outside a proposal's apply nothing + // is appended. + wal.append_proposal_applied(tenant_id, vshard_id, database_id)? } else { - // Provenance appended last; older 3-element decoders ignore it via arity fallback. - let wal_payload = encode_timeseries_batch_payload( - collection.as_str(), + let wal_payload = encode_timeseries_ingest_payload(TimeseriesIngestRecord { + collection: collection.as_str(), payload, - provenance.as_ref(), - )?; + provenance: provenance.as_ref(), + format, + default_timestamp_ms: i64::try_from(now_ms).unwrap_or(i64::MAX), + })?; Some(wal.append_timeseries_batch( tenant_id, vshard_id, database_id, &wal_payload, )?) + }; + TimeseriesAppendOutcome { + lsn, + resolved_now_ms: Some(now_ms), } } // `restart_identity` is applied by the Control Plane after dispatch, @@ -65,12 +114,23 @@ pub(super) fn wal_append_timeseries_op( restart_identity: _, } => { let wal_payload = encode_columnar_truncate_payload(collection.as_str())?; - Some(wal.append_timeseries_truncate(tenant_id, vshard_id, database_id, &wal_payload)?) + TimeseriesAppendOutcome { + lsn: Some(wal.append_timeseries_truncate( + tenant_id, + vshard_id, + database_id, + &wal_payload, + )?), + resolved_now_ms: None, + } } // Reads / read-only resolve pass — no engine mutation here. - TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_) => None, + TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_) => TimeseriesAppendOutcome { + lsn: None, + resolved_now_ms: None, + }, }; - Ok(appended) + Ok(outcome) } /// Encode the payload of a `ColumnarTruncate` / `TimeseriesTruncate` WAL @@ -85,19 +145,43 @@ pub(crate) fn encode_columnar_truncate_payload(collection: &str) -> crate::Resul }) } -/// Encode the payload of a `TimeseriesBatch` WAL record for a timeseries ingest. -/// Produces the legacy 4-element tuple `("timeseries", collection, payload, -/// provenance)`. New transaction redo must use [`encode_timeseries_batch_payload_with_format`]. -pub(crate) fn encode_timeseries_batch_payload( - collection: &str, - payload: &[u8], - provenance: Option<&nodedb_types::sync::wire::SyncProvenance>, +/// One timeseries ingest's `TimeseriesBatch` WAL record. +pub(crate) struct TimeseriesIngestRecord<'a> { + pub collection: &'a str, + pub payload: &'a [u8], + pub provenance: Option<&'a nodedb_types::sync::wire::SyncProvenance>, + /// The payload's ingest format: payload bytes alone cannot distinguish ILP + /// from row MessagePack. + pub format: &'a str, + /// The timestamp, in epoch milliseconds, of every row that carries none. + pub default_timestamp_ms: i64, +} + +/// Encode an autocommit timeseries ingest's `TimeseriesBatch` WAL record: the +/// six-element tuple `("timeseries", collection, payload, provenance, format, +/// default_timestamp_ms)`. Replay stamps untimed rows with the carried +/// instant, so they store the rows the live apply stored. +pub(crate) fn encode_timeseries_ingest_payload( + record: TimeseriesIngestRecord<'_>, ) -> crate::Result> { - zerompk::to_msgpack_vec(&("timeseries", collection, payload, provenance)).map_err(|e| { - crate::Error::Serialization { - format: "msgpack".into(), - detail: format!("wal timeseries batch: {e}"), - } + let TimeseriesIngestRecord { + collection, + payload, + provenance, + format, + default_timestamp_ms, + } = record; + zerompk::to_msgpack_vec(&( + "timeseries", + collection, + payload, + provenance, + format, + default_timestamp_ms, + )) + .map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("wal timeseries ingest: {e}"), }) } @@ -117,21 +201,39 @@ pub(crate) fn encode_timeseries_batch_payload_with_format( }) } +/// One columnar insert's `TimeseriesBatch` WAL record. +pub(crate) struct ColumnarBatchRecord<'a> { + pub collection: &'a str, + pub payload: &'a [u8], + pub provenance: Option<&'a nodedb_types::sync::wire::SyncProvenance>, + /// Per-row surrogates, index-aligned with the rows in `payload`. + pub surrogates: &'a [nodedb_types::Surrogate], + /// What a row whose primary key already exists does. + pub conflict_policy: &'a crate::wal::ColumnarConflictPolicy, +} + /// Encode the payload of a `TimeseriesBatch` WAL record for a columnar batch. /// Produces the map-shaped `ColumnarWalRecord` (`kind = "columnar"`), distinct -/// from the timeseries tuple so `decode_batch_record` routes correctly. +/// from the timeseries tuple so `decode_batch_record` routes correctly. The +/// record carries the insert's conflict policy, so replay skips, merges or +/// replaces each row exactly as the live insert did. pub(crate) fn encode_columnar_batch_payload( - collection: &str, - payload: &[u8], - provenance: Option<&nodedb_types::sync::wire::SyncProvenance>, - surrogates: &[nodedb_types::Surrogate], + record: ColumnarBatchRecord<'_>, ) -> crate::Result> { + let ColumnarBatchRecord { + collection, + payload, + provenance, + surrogates, + conflict_policy, + } = record; let record = nodedb_types::columnar::ColumnarWalRecord { kind: "columnar".to_string(), collection: collection.to_string(), payload: payload.to_vec(), provenance: provenance.cloned(), surrogates: surrogates.to_vec(), + conflict_policy: conflict_policy.encode()?, }; zerompk::to_msgpack_vec(&record).map_err(|e| crate::Error::Serialization { format: "msgpack".into(), @@ -176,7 +278,13 @@ pub(crate) fn wal_append_timeseries( return Ok(None); } - let wal_payload = encode_timeseries_batch_payload(collection, payload, provenance)?; + let wal_payload = encode_timeseries_ingest_payload(TimeseriesIngestRecord { + collection, + payload, + provenance, + format: "ilp", + default_timestamp_ms: i64::try_from(crate::engine::kv::current_ms()).unwrap_or(i64::MAX), + })?; let lsn = wal.append_timeseries_batch(tenant_id, vshard_id, database_id, &wal_payload)?; Ok(Some(lsn)) } @@ -277,7 +385,14 @@ pub fn wal_append_columnar( provenance, surrogates, } = args; - let wal_payload = encode_columnar_batch_payload(collection, payload, provenance, surrogates)?; + // The sync path applies a plain insert: an existing row is replaced. + let wal_payload = encode_columnar_batch_payload(ColumnarBatchRecord { + collection, + payload, + provenance, + surrogates, + conflict_policy: &crate::wal::ColumnarConflictPolicy::replace(), + })?; let lsn = wal.append_timeseries_batch(tenant_id, vshard_id, database_id, &wal_payload)?; Ok(Some(lsn)) } @@ -330,6 +445,49 @@ mod tests { )); } + #[test] + fn an_ingest_record_carries_the_instant_its_untimed_rows_take() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(dir.path()); + let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), + payload: b"metrics value=1".to_vec(), + format: "ilp".to_string(), + wal_lsn: None, + surrogates: vec![], + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: vec![], + }); + + let outcome = super::super::wal_append(super::super::WalAppendRequest { + wal: &wal, + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + database_id: DatabaseId::DEFAULT, + plan: &plan, + credentials: None, + now_override: Some(1_700_000_000_123), + }) + .expect("append"); + + assert_eq!(outcome.resolved_now_ms, Some(1_700_000_000_123)); + wal.sync().expect("sync wal"); + let record = wal + .replay() + .expect("read wal") + .into_iter() + .find(|r| { + nodedb_wal::record::RecordType::from_raw(r.logical_record_type()) + == Some(nodedb_wal::record::RecordType::TimeseriesBatch) + }) + .expect("ingest record"); + let decoded = crate::wal::decode_batch_record(&record.payload).expect("decode"); + assert_eq!(decoded.default_timestamp_ms, Some(1_700_000_000_123)); + assert_eq!(decoded.format.as_deref(), Some("ilp")); + } + #[test] fn truncate_appends_timeseries_truncate_record() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs index 7381f11be..bb63d05f9 100644 --- a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs +++ b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs @@ -10,7 +10,7 @@ use crate::bridge::envelope::{PhysicalPlan, Response, Status, WriteSetEntry}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::manager::WalManager; -use nodedb_physical::physical_plan::DocumentOp; +use nodedb_physical::physical_plan::{DocumentOp, MetaOp}; use super::document::{encode_document_delete_record, encode_document_put_record}; @@ -46,6 +46,12 @@ pub fn plan_post_apply_redo(plan: &PhysicalPlan) -> Option { // Journals nothing on the pre-dispatch path; without this redo, a WAL-only // restart replays source rows and leaves the total as it stood before. Some(collection.to_string()) + } else if let PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { collections, .. }) = plan { + // A committed transaction's materialized-sum folds write target rows no + // redo sub-record names. Each write-set entry names its own target + // collection, so the transaction's first collection is only the + // fallback an entry without one would use. + collections.first().cloned() } else { None } diff --git a/nodedb/src/control/surrogate/assign/bind_plan/binder.rs b/nodedb/src/control/surrogate/assign/bind_plan/binder.rs index 8bd7f1816..a1a36a40c 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/binder.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/binder.rs @@ -2,10 +2,13 @@ //! The per-identity resolution rule and the top-level plan walk. +use std::cell::RefCell; + use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_types::Surrogate; use super::super::SurrogateAssigner; +use super::carried::CarriedIdentity; use crate::types::{DatabaseId, TenantId}; /// Binds carried identities against one node's catalog under one tenancy scope. @@ -13,6 +16,8 @@ pub struct IdentityBinder<'a> { assigner: &'a SurrogateAssigner, database_id: DatabaseId, tenant_id: TenantId, + /// `Some` when the walk also lists every identity it binds. + recorded: Option>>, } impl<'a> IdentityBinder<'a> { @@ -25,6 +30,35 @@ impl<'a> IdentityBinder<'a> { assigner, database_id, tenant_id, + recorded: None, + } + } + + /// A binder that also lists every identity it binds, with the + /// authoritative surrogate. + pub(super) fn recording( + assigner: &'a SurrogateAssigner, + database_id: DatabaseId, + tenant_id: TenantId, + ) -> Self { + Self { + recorded: Some(RefCell::new(Vec::new())), + ..Self::new(assigner, database_id, tenant_id) + } + } + + /// The identities a recording binder bound, in walk order. + pub(super) fn into_recorded(self) -> Vec { + self.recorded.map(RefCell::into_inner).unwrap_or_default() + } + + fn record(&self, collection: &str, pk_bytes: &[u8], surrogate: Surrogate) { + if let Some(recorded) = &self.recorded { + recorded.borrow_mut().push(CarriedIdentity { + collection: collection.to_string(), + pk_bytes: pk_bytes.to_vec(), + surrogate, + }); } } @@ -42,13 +76,15 @@ impl<'a> IdentityBinder<'a> { carried: Surrogate, ) -> crate::Result { if carried != Surrogate::ZERO { - return self.assigner.bind( + let bound = self.assigner.bind( self.database_id, self.tenant_id, collection, pk_bytes, carried, - ); + )?; + self.record(collection, pk_bytes, bound); + return Ok(bound); } Ok(self .assigner @@ -112,6 +148,7 @@ impl<'a> IdentityBinder<'a> { collection, document_id.as_bytes(), )?; + self.record(collection, document_id.as_bytes(), *slot); Ok(()) } } @@ -125,14 +162,18 @@ pub fn bind_plan_identities( tenant_id: TenantId, plan: &mut PhysicalPlan, ) -> crate::Result<()> { - let binder = IdentityBinder::new(assigner, database_id, tenant_id); + bind_with(&IdentityBinder::new(assigner, database_id, tenant_id), plan) +} + +/// Bind every identity `plan` carries through `binder`. +pub(super) fn bind_with(binder: &IdentityBinder<'_>, plan: &mut PhysicalPlan) -> crate::Result<()> { match plan { - PhysicalPlan::Document(op) => super::document::bind(&binder, op), - PhysicalPlan::Kv(op) => super::kv::bind(&binder, op), - PhysicalPlan::Graph(op) => super::graph::bind(&binder, op), - PhysicalPlan::Vector(op) => super::vector::bind(&binder, op), - PhysicalPlan::Crdt(op) => super::crdt::bind(&binder, op), - PhysicalPlan::Array(op) => super::array::bind(&binder, op), + PhysicalPlan::Document(op) => super::document::bind(binder, op), + PhysicalPlan::Kv(op) => super::kv::bind(binder, op), + PhysicalPlan::Graph(op) => super::graph::bind(binder, op), + PhysicalPlan::Vector(op) => super::vector::bind(binder, op), + PhysicalPlan::Crdt(op) => super::crdt::bind(binder, op), + PhysicalPlan::Array(op) => super::array::bind(binder, op), // Columnar-family rows are keyed by the surrogate alone; text, spatial, // timeseries, query, meta and cluster plans carry no pk binding. PhysicalPlan::Text(_) diff --git a/nodedb/src/control/surrogate/assign/bind_plan/carried.rs b/nodedb/src/control/surrogate/assign/bind_plan/carried.rs new file mode 100644 index 000000000..c0289e22c --- /dev/null +++ b/nodedb/src/control/surrogate/assign/bind_plan/carried.rs @@ -0,0 +1,77 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Identities listed off a set of plans on one node and bound on another. +//! +//! A committed transaction replicates as resolved post-images, not plans, so +//! the replicas never walk the plans that carried its identities. The +//! proposer lists them with the same per-family rules [`bind_plan_identities`] +//! applies, and every replica binds the list before the post-images apply. +//! +//! [`bind_plan_identities`]: super::bind_plan_identities + +use nodedb_physical::physical_plan::PhysicalPlan; +use nodedb_types::Surrogate; + +use super::super::SurrogateAssigner; +use super::binder::{IdentityBinder, bind_with}; +use crate::types::{DatabaseId, TenantId}; + +/// One `(collection, pk_bytes) → surrogate` identity. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CarriedIdentity { + pub collection: String, + pub pk_bytes: Vec, + pub surrogate: Surrogate, +} + +/// Every identity `plans` carry, with the surrogate this node's catalog binds +/// it to. Binding here is first-wins and returns an existing binding +/// unchanged, so on the node that planned the writes it changes nothing. +pub fn collect_plan_identities( + assigner: &SurrogateAssigner, + database_id: DatabaseId, + tenant_id: TenantId, + plans: &[PhysicalPlan], +) -> crate::Result> { + let binder = IdentityBinder::recording(assigner, database_id, tenant_id); + for plan in plans { + let mut plan = plan.clone(); + bind_with(&binder, &mut plan)?; + } + Ok(binder.into_recorded()) +} + +/// Bind every carried identity into this node's catalog. +/// +/// The carried surrogate is the one the rows being applied are stored under. +/// A catalog that already binds the key to a different surrogate has diverged +/// from the proposer's, and applying on top of it would store a row no lookup +/// by key reaches, so that is an error. +pub fn bind_carried_identities( + assigner: &SurrogateAssigner, + database_id: DatabaseId, + tenant_id: TenantId, + identities: &[CarriedIdentity], +) -> crate::Result<()> { + for identity in identities { + let bound = assigner.bind( + database_id, + tenant_id, + &identity.collection, + &identity.pk_bytes, + identity.surrogate, + )?; + if bound != identity.surrogate { + return Err(crate::Error::Internal { + detail: format!( + "surrogate binding diverged on '{}': this node binds the key to {}, the \ + committed transaction stored its row under {}", + identity.collection, + bound.as_u32(), + identity.surrogate.as_u32() + ), + }); + } + } + Ok(()) +} diff --git a/nodedb/src/control/surrogate/assign/bind_plan/mod.rs b/nodedb/src/control/surrogate/assign/bind_plan/mod.rs index 69424ff25..a6c081024 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/mod.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/mod.rs @@ -10,6 +10,7 @@ mod array; mod binder; +mod carried; mod crdt; mod document; mod graph; @@ -17,3 +18,4 @@ mod kv; mod vector; pub use binder::{IdentityBinder, bind_plan_identities}; +pub use carried::{CarriedIdentity, bind_carried_identities, collect_plan_identities}; diff --git a/nodedb/src/control/surrogate/assign/mod.rs b/nodedb/src/control/surrogate/assign/mod.rs index bf39b4a72..80e2541c8 100644 --- a/nodedb/src/control/surrogate/assign/mod.rs +++ b/nodedb/src/control/surrogate/assign/mod.rs @@ -7,5 +7,8 @@ pub mod bind_plan; pub(super) mod cluster_reserve; pub mod core; -pub use bind_plan::{IdentityBinder, bind_plan_identities}; +pub use bind_plan::{ + CarriedIdentity, IdentityBinder, bind_carried_identities, bind_plan_identities, + collect_plan_identities, +}; pub use core::{SurrogateAssigner, SurrogateRegistryHandle}; diff --git a/nodedb/src/control/surrogate/mod.rs b/nodedb/src/control/surrogate/mod.rs index 49057a6fc..dd7caea35 100644 --- a/nodedb/src/control/surrogate/mod.rs +++ b/nodedb/src/control/surrogate/mod.rs @@ -15,7 +15,8 @@ pub mod registry; pub mod wal_appender; pub use assign::{ - IdentityBinder, SurrogateAssigner, SurrogateRegistryHandle, bind_plan_identities, + CarriedIdentity, IdentityBinder, SurrogateAssigner, SurrogateRegistryHandle, + bind_carried_identities, bind_plan_identities, collect_plan_identities, }; pub use bootstrap::bootstrap_registry; pub use persist::{SURROGATE_HWM, SurrogateHwmPersist, SystemCatalogHwm}; diff --git a/nodedb/src/control/system_txn/data_plane.rs b/nodedb/src/control/system_txn/data_plane.rs index e36d9ee03..cb150569e 100644 --- a/nodedb/src/control/system_txn/data_plane.rs +++ b/nodedb/src/control/system_txn/data_plane.rs @@ -55,4 +55,8 @@ impl TxnDataPlane for SystemTxnDataPlane<'_> { .await }) } + + fn event_source(&self) -> crate::event::EventSource { + self.event_source + } } diff --git a/nodedb/src/control/wal_catchup.rs b/nodedb/src/control/wal_catchup.rs index d12545b8e..264e0f538 100644 --- a/nodedb/src/control/wal_catchup.rs +++ b/nodedb/src/control/wal_catchup.rs @@ -161,37 +161,33 @@ async fn run_catchup_cycle(shared: &SharedState) -> CatchupResult { continue; } - // Deserialize WAL payload. Try the new 4-element shape (with kind - // discriminator, collection, payload, and trailing provenance) first, - // then fall back to the legacy 2-element (collection, payload) shape - // written by pre-3a records. Provenance is decoded and discarded here. - let (collection, payload) = if let Ok((disc, coll, p, _provenance)) = - zerompk::from_msgpack::<( - String, - String, - Vec, - Option, - )>(&record.payload) - { - let _ = disc; - (coll, p) - } else if let Ok((coll, p)) = zerompk::from_msgpack::<(String, Vec)>(&record.payload) { - (coll, p) - } else { + // A columnar record shares this record type and is not a timeseries + // ingest; catch-up re-dispatches timeseries ingests only. + let Ok(decoded) = crate::wal::decode_batch_record(&record.payload) else { max_lsn = max_lsn.max(record.header.lsn); continue; }; + if decoded + .kind + .as_deref() + .is_some_and(|kind| kind != "timeseries") + { + max_lsn = max_lsn.max(record.header.lsn); + continue; + } let tenant_id = TenantId::new(record.header.tenant_id); + let database_id = DatabaseId::new(record.header.database_id); let vshard_id = VShardId::new(record.header.vshard_id); + let format = decoded.format.unwrap_or_else(|| "ilp".to_string()); let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: nodedb_types::QualifiedCollection::from_stored(collection), - payload, - format: "ilp".to_string(), + collection: nodedb_types::QualifiedCollection::from_stored(decoded.collection), + payload: decoded.payload, + format, wal_lsn: Some(record.header.lsn), // Re-derived on the engine side during apply (record carries - // raw ILP — row identities are reconstructed from the wire). + // raw rows — row identities are reconstructed from the wire). surrogates: Vec::new(), provenance: None, // No predicate here: catch-up replays an already-committed WAL @@ -202,13 +198,22 @@ async fn run_catchup_cycle(shared: &SharedState) -> CatchupResult { }); // Dispatch to Data Plane — do NOT re-append to WAL (already there). - match crate::control::server::dispatch_utils::dispatch_to_data_plane( + // Untimed rows take the instant the record carries. + match crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( shared, - tenant_id, - DatabaseId::DEFAULT, - vshard_id, - plan, - TraceId::ZERO, + crate::control::server::dispatch_utils::WriteDispatch { + tenant_id, + database_id, + vshard_id, + plan, + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + wal_lsn: None, + resolved_now_ms: decoded + .default_timestamp_ms + .and_then(|ms| u64::try_from(ms).ok()), + }, ) .await { diff --git a/nodedb/src/control/wal_replication/decode/entry.rs b/nodedb/src/control/wal_replication/decode/entry.rs index 8e32696ff..460b52dd9 100644 --- a/nodedb/src/control/wal_replication/decode/entry.rs +++ b/nodedb/src/control/wal_replication/decode/entry.rs @@ -15,8 +15,10 @@ use crate::control::surrogate::{SurrogateAssigner, bind_plan_identities}; use crate::types::{DatabaseId, TenantId, VShardId}; /// Decoded `(tenant, vshard, plan, resolved_now_ms)` for a committed entry. -/// `resolved_now_ms` is `None` except for a TTL-bearing KV write, where it's -/// stamped onto the request so every replica installs the same `expire_at_ms`. +/// `resolved_now_ms` is the instant the proposer read for the write, stamped +/// onto the request so every replica installs the same value: a TTL-bearing +/// KV write's `expire_at_ms`, and a timeseries ingest's untimed rows. `None` +/// for every other write. pub type DecodedEntry = (TenantId, VShardId, PhysicalPlan, Option); /// Returns `None` if the data is not a valid ReplicatedEntry (e.g., ConfChange or no-op). @@ -58,7 +60,7 @@ pub fn from_replicated_entry( } /// Convert a ReplicatedWrite back into a PhysicalPlan, alongside `resolved_now_ms` -/// (see [`DecodedEntry`]) — `None` except for the KV group's TTL-bearing arms. +/// (see [`DecodedEntry`]). fn to_physical_plan( write: &ReplicatedWrite, ctx: &DecodeCtx, @@ -117,7 +119,7 @@ fn to_physical_plan( | ReplicatedWrite::RemoveNodeLabels { .. } | ReplicatedWrite::EdgePutBatch { .. } | ReplicatedWrite::EdgeDeleteBatch { .. } => Ok((entry_graph::decode_arm(write)?, None)), - // KV family — the only group carrying `resolved_now_ms`. + // KV family: a TTL-bearing write carries `resolved_now_ms`. ReplicatedWrite::KvTruncate { .. } | ReplicatedWrite::KvPut { .. } | ReplicatedWrite::KvDelete { .. } @@ -141,9 +143,17 @@ fn to_physical_plan( | ReplicatedWrite::KvResolvedWrite { .. } | ReplicatedWrite::KvPredicateUpdate { .. } | ReplicatedWrite::KvPredicateDelete { .. } => entry_kv::decode_arm(write), + // A timeseries ingest carries the proposer's instant for its untimed + // rows, installed on every replica as `resolved_now_ms`. + ReplicatedWrite::TimeseriesIngest { + default_timestamp_ms, + .. + } => Ok(( + entry_columnar_family::decode_arm(write)?, + u64::try_from(*default_timestamp_ms).ok(), + )), // Columnar-storage family + overlay sync engines. ReplicatedWrite::ColumnarIngest { .. } - | ReplicatedWrite::TimeseriesIngest { .. } | ReplicatedWrite::FtsIndex { .. } | ReplicatedWrite::FtsDelete { .. } | ReplicatedWrite::SpatialInsert { .. } @@ -170,5 +180,11 @@ fn to_physical_plan( detail: "CalvinReadResult reached to_physical_plan (should have been intercepted)" .into(), }), + // The apply loop applies these through `transaction_redo`, which stamps + // the redo with the entry's Raft coordinates. + ReplicatedWrite::TransactionRedo { .. } => Err(crate::Error::Internal { + detail: "TransactionRedo reached to_physical_plan (should have been intercepted)" + .into(), + }), } } diff --git a/nodedb/src/control/wal_replication/decode/entry_columnar_family.rs b/nodedb/src/control/wal_replication/decode/entry_columnar_family.rs index aebdd0d97..77da86f12 100644 --- a/nodedb/src/control/wal_replication/decode/entry_columnar_family.rs +++ b/nodedb/src/control/wal_replication/decode/entry_columnar_family.rs @@ -50,6 +50,8 @@ pub(super) fn decode_arm(write: &ReplicatedWrite) -> crate::Result payload, format, surrogates, + // Carried as the entry's `resolved_now_ms` (see `decode::entry`). + default_timestamp_ms: _, provenance, returning, rls_filters, @@ -119,3 +121,50 @@ pub(super) fn decode_arm(write: &ReplicatedWrite) -> crate::Result }), } } + +#[cfg(test)] +mod tests { + use crate::control::wal_replication::decode; + use crate::types::{DatabaseId, TenantId, VShardId}; + use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp}; + use nodedb_types::QualifiedCollection; + + #[test] + fn a_replicated_timeseries_ingest_carries_the_proposers_instant() { + let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), + payload: b"metrics value=1".to_vec(), + format: "ilp".to_string(), + wal_lsn: None, + surrogates: Vec::new(), + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: Vec::new(), + }); + let before = crate::engine::kv::current_ms(); + let write = crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan) + .expect("decide"); + let entry = crate::control::wal_replication::encode::to_replicated_entry( + TenantId::new(1), + DatabaseId::DEFAULT, + VShardId::new(0), + &write, + ) + .expect("encode") + .expect("a timeseries ingest replicates"); + let after = crate::engine::kv::current_ms(); + + let bytes = entry.to_bytes(); + let (_, _, _, resolved_now_ms) = decode::from_replicated_entry(&bytes, None) + .expect("decode") + .expect("an entry"); + let instant = resolved_now_ms.expect("the ingest carries an instant"); + assert!((before..=after).contains(&instant)); + // Every replica decodes the same bytes, so every replica stamps alike. + let (_, _, _, again) = decode::from_replicated_entry(&bytes, None) + .expect("decode") + .expect("an entry"); + assert_eq!(again, Some(instant)); + } +} diff --git a/nodedb/src/control/wal_replication/decode/mod.rs b/nodedb/src/control/wal_replication/decode/mod.rs index 4383dcc1f..511cf1fe3 100644 --- a/nodedb/src/control/wal_replication/decode/mod.rs +++ b/nodedb/src/control/wal_replication/decode/mod.rs @@ -17,6 +17,7 @@ //! `Timeseries` / `Text` / `Spatial`. //! - [`vector`]: `PhysicalPlan::Vector` (grouped `decode_arm`). //! - [`vector_direct`]: the vector-primary `DELETE` / `UPDATE` decoders. +//! - [`transaction_redo`]: a committed transaction's redo entry. mod columnar; mod crdt; @@ -32,7 +33,9 @@ mod entry_graph; mod entry_kv; mod graph; mod kv; +mod transaction_redo; mod vector; mod vector_direct; pub use entry::from_replicated_entry; +pub use transaction_redo::transaction_redo_payload; diff --git a/nodedb/src/control/wal_replication/decode/transaction_redo.rs b/nodedb/src/control/wal_replication/decode/transaction_redo.rs new file mode 100644 index 000000000..0dfc4ff18 --- /dev/null +++ b/nodedb/src/control/wal_replication/decode/transaction_redo.rs @@ -0,0 +1,132 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Decode a committed `ReplicatedWrite::TransactionRedo` entry back into the +//! payload every replica applies. +//! +//! The apply loop intercepts these entries before the generic decode: the +//! apply stamps the redo with the entry's Raft coordinates, which the generic +//! `from_replicated_entry` result has no place for. + +use nodedb_types::Surrogate; + +use super::super::transaction_redo::TransactionRedoPayload; +use super::super::types::ReplicatedWrite; +use crate::control::surrogate::CarriedIdentity; + +/// The payload `write` carries, or an error when `write` is another variant. +pub fn transaction_redo_payload(write: &ReplicatedWrite) -> crate::Result { + let ReplicatedWrite::TransactionRedo { + redo, + collections, + sum_targets, + identities, + event_source, + } = write + else { + return Err(crate::Error::Internal { + detail: "transaction redo decode called with a different ReplicatedWrite variant" + .into(), + }); + }; + Ok(TransactionRedoPayload { + redo: redo.clone(), + collections: collections.clone(), + sum_targets: sum_targets.clone(), + identities: identities + .iter() + .map(|identity| CarriedIdentity { + collection: identity.collection.clone(), + pk_bytes: identity.pk_bytes.clone(), + surrogate: Surrogate::new(identity.surrogate), + }) + .collect(), + event_source: (*event_source).into(), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::wal_replication::ReplicatedEntry; + use crate::control::wal_replication::encode::transaction_redo_entry; + use crate::event::EventSource; + use crate::types::{DatabaseId, TenantId, VShardId}; + use crate::wal::{CalvinStamp, RedoRecord, RedoSubRecord}; + use nodedb_physical::physical_plan::{RedoSumTargets, ResolvedSumTarget}; + + fn payload() -> TransactionRedoPayload { + TransactionRedoPayload { + redo: RedoRecord { + version: 1, + ops: vec![RedoSubRecord { + record_type: nodedb_wal::record::RecordType::Put as u32, + payload: vec![4, 5, 6], + }], + calvin_stamp: Some(CalvinStamp { + epoch: 11, + position: 2, + vshard_id: 7, + }), + }, + collections: vec!["accounts".into(), "entries".into()], + sum_targets: vec![RedoSumTargets { + collection: "entries".into(), + resolved: vec![ResolvedSumTarget::new("accounts", "a1", Surrogate::new(9))], + deferred: vec!["audit".into()], + }], + identities: vec![CarriedIdentity { + collection: "entries".into(), + pk_bytes: b"e1".to_vec(), + surrogate: Surrogate::new(3), + }], + event_source: EventSource::Trigger, + } + } + + #[test] + fn transaction_redo_entry_round_trips_through_the_raft_bytes() { + let original = payload(); + let entry = transaction_redo_entry( + TenantId::new(1), + DatabaseId::new(5), + VShardId::new(7), + &original, + ); + let bytes = entry.to_bytes(); + let decoded_entry = ReplicatedEntry::from_bytes(&bytes).expect("entry decodes"); + assert_eq!(decoded_entry.tenant_id, 1); + assert_eq!(decoded_entry.database_id, 5); + assert_eq!(decoded_entry.vshard_id, 7); + assert_eq!(decoded_entry.idempotency_key, entry.idempotency_key); + + let decoded = transaction_redo_payload(&decoded_entry.write).expect("payload decodes"); + assert_eq!(decoded.redo, original.redo); + assert_eq!(decoded.collections, original.collections); + assert_eq!(decoded.sum_targets, original.sum_targets); + assert_eq!(decoded.identities, original.identities); + assert_eq!(decoded.event_source, EventSource::Trigger); + } + + #[test] + fn apply_plan_carries_the_committed_redo_bytes_unchanged() { + let original = payload(); + let plan = original.apply_plan().expect("plan builds"); + let nodedb_physical::physical_plan::PhysicalPlan::Meta( + nodedb_physical::physical_plan::MetaOp::ApplyTransactionRedo { redo, .. }, + ) = plan + else { + panic!("apply plan must be ApplyTransactionRedo"); + }; + let redo = RedoRecord::from_bytes(&redo).expect("redo decodes"); + assert_eq!(redo, original.redo); + } + + #[test] + fn a_different_variant_is_refused() { + let write = ReplicatedWrite::KvTruncate { + collection: "kv".into(), + restart_identity: false, + }; + assert!(transaction_redo_payload(&write).is_err()); + } +} diff --git a/nodedb/src/control/wal_replication/encode/columnar.rs b/nodedb/src/control/wal_replication/encode/columnar.rs index ccd036764..553a9d02a 100644 --- a/nodedb/src/control/wal_replication/encode/columnar.rs +++ b/nodedb/src/control/wal_replication/encode/columnar.rs @@ -39,23 +39,28 @@ pub(super) fn columnar_ingest(fields: ColumnarIngestFields<'_>) -> ReplicatedWri } } -pub(super) fn timeseries_ingest( - collection: &str, - payload: &[u8], - format: &str, - surrogates: &[Surrogate], - provenance: Option>, - returning: Option>, - rls_filters: &[u8], -) -> ReplicatedWrite { +/// Fields of a `TimeseriesOp::Ingest` that cross the wire. +pub(super) struct TimeseriesIngestFields<'a> { + pub collection: &'a str, + pub payload: &'a [u8], + pub format: &'a str, + pub surrogates: &'a [Surrogate], + pub default_timestamp_ms: i64, + pub provenance: Option>, + pub returning: Option>, + pub rls_filters: &'a [u8], +} + +pub(super) fn timeseries_ingest(fields: TimeseriesIngestFields<'_>) -> ReplicatedWrite { ReplicatedWrite::TimeseriesIngest { - collection: collection.to_owned(), - payload: payload.to_vec(), - format: format.to_owned(), - surrogates: surrogates.iter().map(|s| s.as_u32()).collect(), - provenance, - returning, - rls_filters: rls_filters.to_vec(), + collection: fields.collection.to_owned(), + payload: fields.payload.to_vec(), + format: fields.format.to_owned(), + surrogates: fields.surrogates.iter().map(|s| s.as_u32()).collect(), + default_timestamp_ms: fields.default_timestamp_ms, + provenance: fields.provenance, + returning: fields.returning, + rls_filters: fields.rls_filters.to_vec(), } } diff --git a/nodedb/src/control/wal_replication/encode/entry_columnar_family.rs b/nodedb/src/control/wal_replication/encode/entry_columnar_family.rs index 13e13b14c..8303689ab 100644 --- a/nodedb/src/control/wal_replication/encode/entry_columnar_family.rs +++ b/nodedb/src/control/wal_replication/encode/entry_columnar_family.rs @@ -153,15 +153,18 @@ pub(super) fn timeseries_write(op: &TimeseriesOp) -> Option { rls_write_check: _, returning, rls_filters, - } => columnar::timeseries_ingest( - collection.as_str(), + } => columnar::timeseries_ingest(columnar::TimeseriesIngestFields { + collection: collection.as_str(), payload, format, surrogates, - encode_provenance(provenance), - encode_returning(returning), + // Read once, here on the proposer: every replica stamps an + // untimed row with this instant instead of its own clock. + default_timestamp_ms: crate::engine::kv::current_ms() as i64, + provenance: encode_provenance(provenance), + returning: encode_returning(returning), rls_filters, - ), + }), TimeseriesOp::Truncate { collection, diff --git a/nodedb/src/control/wal_replication/encode/mod.rs b/nodedb/src/control/wal_replication/encode/mod.rs index 2d8f6124f..5ac14a582 100644 --- a/nodedb/src/control/wal_replication/encode/mod.rs +++ b/nodedb/src/control/wal_replication/encode/mod.rs @@ -5,6 +5,8 @@ //! Split by `PhysicalPlan` family. Each `entry_*` module holds the exhaustive //! per-op write/not-write classification for one engine; its sibling module //! holds the wire encoders it calls into. [`entry`] is the top-level dispatcher. +//! [`transaction_redo`] encodes a committed transaction's redo, which no plan +//! produces. mod columnar; mod crdt; @@ -18,6 +20,8 @@ mod entry_graph; mod entry_kv; mod graph; mod kv; +mod transaction_redo; mod vector; pub use entry::to_replicated_entry; +pub use transaction_redo::transaction_redo_entry; diff --git a/nodedb/src/control/wal_replication/encode/transaction_redo.rs b/nodedb/src/control/wal_replication/encode/transaction_redo.rs new file mode 100644 index 000000000..bc19a4fa8 --- /dev/null +++ b/nodedb/src/control/wal_replication/encode/transaction_redo.rs @@ -0,0 +1,42 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Encode a committed transaction's redo as a `ReplicatedWrite::TransactionRedo` +//! entry for its vShard's data-group log. +//! +//! A redo is not built from a `PhysicalPlan`, so it has no arm in +//! `to_replicated_entry`: the commit that resolved it encodes it here. + +use super::super::transaction_redo::TransactionRedoPayload; +use super::super::types::{ + ReplicatedEntry, ReplicatedEventSource, ReplicatedIdentity, ReplicatedWrite, +}; +use crate::types::{DatabaseId, TenantId, VShardId}; + +/// The Raft entry that carries `payload` to every replica of `vshard_id`. +pub fn transaction_redo_entry( + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + payload: &TransactionRedoPayload, +) -> ReplicatedEntry { + ReplicatedEntry::new( + tenant_id.as_u64(), + database_id.as_u64(), + vshard_id.as_u32(), + ReplicatedWrite::TransactionRedo { + redo: payload.redo.clone(), + collections: payload.collections.clone(), + sum_targets: payload.sum_targets.clone(), + identities: payload + .identities + .iter() + .map(|identity| ReplicatedIdentity { + collection: identity.collection.clone(), + pk_bytes: identity.pk_bytes.clone(), + surrogate: identity.surrogate.as_u32(), + }) + .collect(), + event_source: ReplicatedEventSource::from(payload.event_source), + }, + ) +} diff --git a/nodedb/src/control/wal_replication/mod.rs b/nodedb/src/control/wal_replication/mod.rs index 579e92443..0e52a18e0 100644 --- a/nodedb/src/control/wal_replication/mod.rs +++ b/nodedb/src/control/wal_replication/mod.rs @@ -3,7 +3,9 @@ //! Distributed WAL write path — propose writes through Raft, apply after commit. //! //! [`types`] holds the wire types; [`replicable_write`] the decided-write type -//! [`encode`] consumes; [`decode`] converts wire bytes back to `PhysicalPlan`. +//! [`encode`] consumes; [`decode`] converts wire bytes back to `PhysicalPlan`; +//! [`transaction_redo`] carries committed transactions through the data-group +//! log. pub mod decode; mod decode_sync_engines; @@ -11,6 +13,7 @@ pub mod encode; mod legacy_entry; pub mod propose; pub mod replicable_write; +pub mod transaction_redo; pub mod types; pub use decode::from_replicated_entry; @@ -19,7 +22,8 @@ pub(crate) use propose::propose_replicated_entry; pub use replicable_write::ReplicableWrite; pub use types::{ AsyncRaftProposer, ConstraintChangeOp, RaftAppliedIndexSink, RaftCompactor, RaftProposer, - ReplicatedEntry, ReplicatedSumTarget, ReplicatedWrite, + ReplicatedEntry, ReplicatedEventSource, ReplicatedIdentity, ReplicatedSumTarget, + ReplicatedWrite, }; pub use crate::control::distributed_applier::{ diff --git a/nodedb/src/control/wal_replication/transaction_redo/apply.rs b/nodedb/src/control/wal_replication/transaction_redo/apply.rs new file mode 100644 index 000000000..17103a71c --- /dev/null +++ b/nodedb/src/control/wal_replication/transaction_redo/apply.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply one committed transaction redo on this node. +//! +//! The one apply path for a committed redo record. The data-group apply loop +//! runs it for every committed `TransactionRedo` entry on every replica, the +//! proposer included; a node with no Raft runs it for its own commit. Either +//! way the write funnel appends the record to this node's WAL and dispatches +//! it to the owning core, and the fsync completes before the result returns. + +use crate::control::server::dispatch_utils::{ + ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, +}; +use crate::control::state::SharedState; +use crate::control::surrogate::bind_carried_identities; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; + +use super::payload::TransactionRedoPayload; + +/// Where a committed redo applies. +#[derive(Debug, Clone, Copy)] +pub struct RedoTarget { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard_id: VShardId, +} + +/// Bind the redo's identities, then append and apply it on this node. +/// +/// `apply_key` is the idempotency key of the Raft entry the redo comes from, +/// `0` on a node with no Raft. The redo record's header carries it. The +/// outcome carries the Data Plane's response verbatim, including an error +/// status. +pub(crate) async fn apply_transaction_redo( + state: &SharedState, + target: RedoTarget, + payload: &TransactionRedoPayload, + apply_key: u64, +) -> crate::Result { + bind_carried_identities( + &state.surrogate_assigner, + target.database_id, + target.tenant_id, + &payload.identities, + )?; + let plan = payload.apply_plan()?; + submit_write( + state, + SubmitWrite { + tenant_id: target.tenant_id, + database_id: target.database_id, + vshard_id: target.vshard_id, + plan, + trace_id: TraceId::generate(), + event_source: payload.event_source, + txn_id: None, + // The proposer authorized every write before it resolved them. + user_id: None, + // Every replica appends the record to its own WAL, in apply order. + durability: WalDurability::AppendHere { + now_override: None, + apply_key, + }, + // Raft fixed the order; a node with no Raft already validated the + // commit and holds no gate for it. + ordering: WriteOrdering::AlreadyOrdered, + // A committed transaction publishes no Control-Plane change event; + // its rows reach subscribers through the Data Plane's events. + change_feed: ChangeFeedOwner::Unowned, + }, + ) + .await +} diff --git a/nodedb/src/control/wal_replication/transaction_redo/collections.rs b/nodedb/src/control/wal_replication/transaction_redo/collections.rs new file mode 100644 index 000000000..afdb44fcf --- /dev/null +++ b/nodedb/src/control/wal_replication/transaction_redo/collections.rs @@ -0,0 +1,68 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every collection a transaction's buffered write plans write. +//! +//! The redo apply gives each one a collection-floor write version at the +//! record's LSN, as the transaction batch did for every written plan. A floor +//! on a collection the plans only read is harmless: it makes OCC validation +//! stricter, never looser. + +use std::collections::BTreeSet; + +use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan, SpatialOp, VectorOp}; + +/// The written collections of `plans`, sorted and deduplicated. +pub fn written_collections(plans: &[PhysicalPlan]) -> Vec { + let mut collections: BTreeSet = BTreeSet::new(); + for plan in plans { + collect_plan(plan, &mut collections); + } + collections.into_iter().collect() +} + +fn collect_plan(plan: &PhysicalPlan, out: &mut BTreeSet) { + // Ops whose single target `PhysicalPlan::collection` does not report. + if let PhysicalPlan::Spatial( + SpatialOp::Insert { collection, .. } | SpatialOp::Delete { collection, .. }, + ) + | PhysicalPlan::Graph( + GraphOp::EdgePut { collection, .. } | GraphOp::EdgeDelete { collection, .. }, + ) + | PhysicalPlan::Vector( + VectorOp::DeleteBySurrogate { collection, .. } + | VectorOp::SparseInsert { collection, .. } + | VectorOp::SparseDelete { collection, .. } + | VectorOp::MultiVectorInsert { collection, .. } + | VectorOp::MultiVectorDelete { collection, .. }, + ) = plan + { + out.insert(collection.to_string()); + } else if let PhysicalPlan::Graph( + GraphOp::EdgePutBatch { edges } | GraphOp::EdgeDeleteBatch { edges }, + ) = plan + { + out.extend(edges.iter().map(|edge| edge.collection.to_string())); + } else if let Some(collection) = plan.collection() { + out.insert(collection.to_string()); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{DatabaseId, QualifiedCollection}; + + #[test] + fn spatial_writes_name_their_collection_once() { + let delete = PhysicalPlan::Spatial(SpatialOp::Delete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "places"), + field: "loc".into(), + surrogate: nodedb_types::Surrogate::new(1), + provenance: None, + }); + assert_eq!( + written_collections(&[delete.clone(), delete]), + vec!["places".to_string()] + ); + } +} diff --git a/nodedb/src/control/wal_replication/transaction_redo/mod.rs b/nodedb/src/control/wal_replication/transaction_redo/mod.rs new file mode 100644 index 000000000..4c217a9f2 --- /dev/null +++ b/nodedb/src/control/wal_replication/transaction_redo/mod.rs @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Committed-transaction redo replication: one apply log per vShard. +//! +//! - [`payload`]: the redo a commit hands to the data-group log. +//! - [`apply`]: the one path that appends and applies it on a node. +//! - [`collections`] / [`sum_targets`]: what the payload derives from the +//! commit's buffered plans. + +pub mod apply; +pub mod collections; +pub mod payload; +pub mod sum_targets; + +pub use apply::RedoTarget; +pub(crate) use apply::apply_transaction_redo; +pub use payload::TransactionRedoPayload; diff --git a/nodedb/src/control/wal_replication/transaction_redo/payload.rs b/nodedb/src/control/wal_replication/transaction_redo/payload.rs new file mode 100644 index 000000000..005496d18 --- /dev/null +++ b/nodedb/src/control/wal_replication/transaction_redo/payload.rs @@ -0,0 +1,66 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One committed transaction's redo for one vShard, as a commit hands it to +//! the data-group log and as every replica applies it. + +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan, RedoSumTargets}; + +use crate::control::state::SharedState; +use crate::control::surrogate::{CarriedIdentity, collect_plan_identities}; +use crate::event::EventSource; +use crate::types::{DatabaseId, TenantId}; +use crate::wal::RedoRecord; + +use super::collections::written_collections; +use super::sum_targets::redo_sum_targets; + +/// Everything a replica needs to apply one committed transaction's redo. +#[derive(Debug, Clone)] +pub struct TransactionRedoPayload { + /// The resolved post-images. + pub redo: RedoRecord, + /// Every collection the transaction wrote. + pub collections: Vec, + /// Materialized-sum resolution the document writes fold into, keyed by + /// source collection. + pub sum_targets: Vec, + /// Identities every replica binds before the apply. + pub identities: Vec, + /// Event source every replica stamps on the writes. + pub event_source: EventSource, +} + +impl TransactionRedoPayload { + /// The payload for a commit whose buffered write plans are `plans` and + /// whose resolved redo is `redo`, built on the node that resolved it. + pub fn from_commit( + state: &SharedState, + database_id: DatabaseId, + tenant_id: TenantId, + redo: RedoRecord, + plans: &[PhysicalPlan], + event_source: EventSource, + ) -> crate::Result { + Ok(Self { + redo, + collections: written_collections(plans), + sum_targets: redo_sum_targets(plans), + identities: collect_plan_identities( + &state.surrogate_assigner, + database_id, + tenant_id, + plans, + )?, + event_source, + }) + } + + /// The Data-Plane plan that applies this redo. + pub fn apply_plan(&self) -> crate::Result { + Ok(PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo: self.redo.to_bytes()?, + collections: self.collections.clone(), + sum_targets: self.sum_targets.clone(), + })) + } +} diff --git a/nodedb/src/control/wal_replication/transaction_redo/sum_targets.rs b/nodedb/src/control/wal_replication/transaction_redo/sum_targets.rs new file mode 100644 index 000000000..b2acfba1f --- /dev/null +++ b/nodedb/src/control/wal_replication/transaction_redo/sum_targets.rs @@ -0,0 +1,176 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The materialized-sum resolution a transaction's buffered plans carry, +//! regrouped by source collection for the redo apply. + +#![deny(clippy::wildcard_enum_match_arm)] + +use std::collections::BTreeMap; + +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan, RedoSumTargets, ResolvedSumTarget}; + +/// Every resolved and deferred sum target `plans` carry, one entry per source +/// collection, in collection order. A target resolved by several plans is +/// listed once. +pub fn redo_sum_targets(plans: &[PhysicalPlan]) -> Vec { + let mut by_collection: BTreeMap = BTreeMap::new(); + for plan in plans { + let PhysicalPlan::Document(op) = plan else { + continue; + }; + let Some((collection, resolved, deferred)) = document_sum_targets(op) else { + continue; + }; + if resolved.is_empty() && deferred.is_empty() { + continue; + } + let entry = by_collection + .entry(collection.to_string()) + .or_insert_with(|| RedoSumTargets { + collection: collection.to_string(), + resolved: Vec::new(), + deferred: Vec::new(), + }); + for target in resolved { + if !entry.resolved.contains(target) { + entry.resolved.push(target.clone()); + } + } + for target in deferred { + if !entry.deferred.contains(target) { + entry.deferred.push(target.clone()); + } + } + } + by_collection.into_values().collect() +} + +type SumTargetSlots<'a> = (&'a str, &'a [ResolvedSumTarget], &'a [String]); + +/// The source collection and sum-target slots of one document write. +fn document_sum_targets(op: &DocumentOp) -> Option> { + match op { + DocumentOp::PointInsert { + collection, + resolved_sum_targets, + deferred_sum_targets, + .. + } + | DocumentOp::BatchInsert { + collection, + resolved_sum_targets, + deferred_sum_targets, + .. + } => Some(( + collection.as_str(), + resolved_sum_targets, + deferred_sum_targets, + )), + DocumentOp::PointPut { + collection, + resolved_sum_targets, + .. + } + | DocumentOp::PointDelete { + collection, + resolved_sum_targets, + .. + } + | DocumentOp::PointUpdate { + collection, + resolved_sum_targets, + .. + } + | DocumentOp::Upsert { + collection, + resolved_sum_targets, + .. + } + | DocumentOp::BulkUpdate { + collection, + resolved_sum_targets, + .. + } + | DocumentOp::BulkDelete { + collection, + resolved_sum_targets, + .. + } + | DocumentOp::Truncate { + collection, + resolved_sum_targets, + .. + } => Some((collection.as_str(), resolved_sum_targets, &[])), + DocumentOp::UpdateFromJoin { + target_collection, + resolved_sum_targets, + .. + } + | DocumentOp::Merge { + target_collection, + resolved_sum_targets, + .. + } => Some((target_collection.as_str(), resolved_sum_targets, &[])), + // Reads, index DDL, and the ops that carry no sum resolution of their + // own: a balance delta IS the target write, and a resolved write + // carries its resolution per mutation, never staged in a transaction. + DocumentOp::PointGet { .. } + | DocumentOp::Scan { .. } + | DocumentOp::RangeScan { .. } + | DocumentOp::Register { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::IndexedFetch { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. } + | DocumentOp::EstimateCount { .. } + | DocumentOp::InsertSelect { .. } + | DocumentOp::MaterializeScan { .. } + | DocumentOp::ApplyBalanceDelta { .. } + | DocumentOp::ResolveWrite(_) + | DocumentOp::ResolvedWrite { .. } => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate}; + + fn delete_with(target: ResolvedSumTarget) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointDelete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "entries"), + document_id: "e1".into(), + surrogate: Surrogate::new(1), + pk_bytes: b"e1".to_vec(), + returning: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: vec![target], + }) + } + + #[test] + fn a_target_resolved_by_two_writes_is_listed_once_per_source() { + let target = ResolvedSumTarget::new("accounts", "a1", Surrogate::new(9)); + let plans = vec![delete_with(target.clone()), delete_with(target.clone())]; + let targets = redo_sum_targets(&plans); + assert_eq!(targets.len(), 1); + assert_eq!(targets[0].collection, "entries"); + assert_eq!(targets[0].resolved, vec![target]); + } + + #[test] + fn a_write_with_no_binding_contributes_nothing() { + let plan = PhysicalPlan::Document(DocumentOp::PointDelete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), + document_id: "n1".into(), + surrogate: Surrogate::new(1), + pk_bytes: b"n1".to_vec(), + returning: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + }); + assert!(redo_sum_targets(&[plan]).is_empty()); + } +} diff --git a/nodedb/src/control/wal_replication/types/mod.rs b/nodedb/src/control/wal_replication/types/mod.rs index 7af59d455..a40a938a6 100644 --- a/nodedb/src/control/wal_replication/types/mod.rs +++ b/nodedb/src/control/wal_replication/types/mod.rs @@ -16,15 +16,18 @@ //! - [`wire_shapes`]: small wire-format types embedded inside [`ReplicatedWrite`]. //! - [`replicated_write`]: the [`ReplicatedWrite`] enum itself. //! - [`replicated_entry`]: [`ReplicatedEntry`] (routing envelope) + (de)serialization. +//! - [`transaction_redo_wire`]: wire types of the committed-transaction redo entry. mod aliases; mod replicated_entry; mod replicated_write; +mod transaction_redo_wire; mod wire_shapes; pub use aliases::{AsyncRaftProposer, RaftAppliedIndexSink, RaftCompactor, RaftProposer}; pub use replicated_entry::ReplicatedEntry; pub use replicated_write::ReplicatedWrite; +pub use transaction_redo_wire::{ReplicatedEventSource, ReplicatedIdentity}; pub use wire_shapes::{ BalanceDeltaFields, ColumnarResolvedRow, ConstraintChangeOp, DocumentResolvedMutationWire, KvResolvedMutationWire, ReplicatedBatchEdge, ReplicatedSumTarget, diff --git a/nodedb/src/control/wal_replication/types/replicated_write.rs b/nodedb/src/control/wal_replication/types/replicated_write.rs index 19ac37561..9020f5236 100644 --- a/nodedb/src/control/wal_replication/types/replicated_write.rs +++ b/nodedb/src/control/wal_replication/types/replicated_write.rs @@ -5,6 +5,7 @@ use super::aliases::{ default_columnar_ingest_format, default_columnar_insert_intent, default_ivf_cells, default_ivf_nprobe, default_pq_m, }; +use super::transaction_redo_wire::{ReplicatedEventSource, ReplicatedIdentity}; use super::wire_shapes::{ ColumnarResolvedRow, ConstraintChangeOp, DocumentResolvedMutationWire, KvResolvedMutationWire, ReplicatedBatchEdge, ReplicatedSumTarget, @@ -299,6 +300,10 @@ pub enum ReplicatedWrite { format: String, /// Leader-assigned global surrogates, parallel to the rows in `payload`. surrogates: Vec, + /// The timestamp, in epoch milliseconds, of every row that carries + /// none. The proposer reads its clock once, and every replica stores + /// the same instant. + default_timestamp_ms: i64, /// Sync provenance encoded as zerompk bytes. #[serde(default)] provenance: Option>, @@ -987,6 +992,25 @@ pub enum ReplicatedWrite { /// See `PointUpdate::declared_primary_key`. declared_primary_key: Option, }, + /// One committed transaction's resolved post-images for one vShard. + /// + /// Every replica, the proposer included, appends `redo` to its own WAL + /// and applies it through the WAL replay arms, in Raft log order. The + /// record is resolved once, on the proposer; no replica re-derives it. + /// `redo.calvin_stamp` names the Calvin `(epoch, position)` the record + /// applies, when a Calvin flush produced it. + TransactionRedo { + redo: crate::wal::RedoRecord, + /// Every collection the transaction wrote, for the collection-floor + /// write versions. + collections: Vec, + /// Materialized-sum resolution the document writes fold into their + /// targets, keyed by source collection. + sum_targets: Vec, + /// Identities every replica binds before the apply. + identities: Vec, + event_source: ReplicatedEventSource, + }, } #[cfg(test)] diff --git a/nodedb/src/control/wal_replication/types/transaction_redo_wire.rs b/nodedb/src/control/wal_replication/types/transaction_redo_wire.rs new file mode 100644 index 000000000..3aad56ff5 --- /dev/null +++ b/nodedb/src/control/wal_replication/types/transaction_redo_wire.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Wire shapes carried by [`super::ReplicatedWrite::TransactionRedo`]. + +use crate::event::EventSource; + +/// One `(collection, primary key) → surrogate` identity a committed +/// transaction's writes carry. Every replica binds it into its own catalog +/// before the redo applies, so a later write by primary key on any replica +/// resolves to the row the redo installed. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct ReplicatedIdentity { + pub collection: String, + /// The binding key: a document id's bytes, a KV key, a node id's bytes, an + /// encoded array coordinate, or a surrogate's own big-endian bytes for a + /// row with no user key. + pub pk_bytes: Vec, + pub surrogate: u32, +} + +/// The source a committed transaction's writes carry into the Event Plane. +/// +/// Every replica stamps the same source, so a transaction a trigger issued +/// does not re-fire that trigger on any replica. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum ReplicatedEventSource { + User, + Trigger, + RaftFollower, + CrdtSync, + Deferred, +} + +impl From for ReplicatedEventSource { + fn from(source: EventSource) -> Self { + match source { + EventSource::User => Self::User, + EventSource::Trigger => Self::Trigger, + EventSource::RaftFollower => Self::RaftFollower, + EventSource::CrdtSync => Self::CrdtSync, + EventSource::Deferred => Self::Deferred, + } + } +} + +impl From for EventSource { + fn from(source: ReplicatedEventSource) -> Self { + match source { + ReplicatedEventSource::User => Self::User, + ReplicatedEventSource::Trigger => Self::Trigger, + ReplicatedEventSource::RaftFollower => Self::RaftFollower, + ReplicatedEventSource::CrdtSync => Self::CrdtSync, + ReplicatedEventSource::Deferred => Self::Deferred, + } + } +} diff --git a/nodedb/src/data/executor/array_checkpoint.rs b/nodedb/src/data/executor/array_checkpoint.rs index a3c2345b7..aea6f9060 100644 --- a/nodedb/src/data/executor/array_checkpoint.rs +++ b/nodedb/src/data/executor/array_checkpoint.rs @@ -449,4 +449,73 @@ mod tests { "the flushed segment and the live memtable must read as one array" ); } + + /// A committed record applied at an LSN the array's durable watermark + /// already covers is flushed into a segment: restart replay skips it, so + /// the segment is its only copy. + #[test] + fn a_committed_record_applied_below_the_durable_lsn_is_flushed() { + use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; + use crate::engine::array::wal::{ArrayPutPayload, encode_put_with_version}; + use crate::wal::{RedoRecord, RedoSubRecord}; + use nodedb_wal::record::RecordType; + + let dir = tempfile::tempdir().expect("tempdir"); + let id = aid(); + + let mut before = Core::open_at(dir.path()); + before.open_array(&id); + before.put(&id, 1, 2, 30, 100); + before.core.advance_watermark(Lsn::new(100)); + before + .core + .checkpoint_array_engines() + .expect("flush at lsn 100"); + + let payload = encode_put_with_version(&ArrayPutPayload { + array_id: id.clone(), + cells: vec![ArrayPutCell { + coord: vec![CoordValue::Int64(9), CoordValue::Int64(9)], + attrs: vec![CellValue::Int64(40)], + surrogate: nodedb_types::Surrogate::ZERO, + system_from_ms: 1, + valid_from_ms: 0, + valid_until_ms: i64::MAX, + }], + provenance: None, + }) + .expect("encode put"); + let redo = RedoRecord { + version: 1, + ops: vec![RedoSubRecord { + record_type: RecordType::ArrayPut as u32, + payload, + }], + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let mut task = crate::data::executor::core_loop::tests::make_default_task(); + task.wal_lsn = Some(Lsn::new(50)); + let response = before.core.execute_apply_transaction_redo( + &task, + TID, + CommittedRedo { + redo: &redo, + collections: &[], + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "apply: {response:?}"); + drop(before); + + let mut after = Core::open_at(dir.path()); + after.open_array(&id); + assert_eq!( + after.slice_all(&id), + vec![(1, 2, 30), (9, 9, 40)], + "the cell committed at lsn 50 survives a restart that replays nothing \ + at or below lsn 100" + ); + } } diff --git a/nodedb/src/data/executor/columnar_checkpoint/load.rs b/nodedb/src/data/executor/columnar_checkpoint/load.rs index bf8340bd3..04cc666ac 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/load.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/load.rs @@ -112,6 +112,7 @@ impl CoreLoop { .columnar .set(Lsn::new(manifest.durable_through_lsn)); self.floors.columnar_durable_lsn = Lsn::new(manifest.durable_through_lsn); + self.floors.columnar_published_lsn = Lsn::new(manifest.durable_through_lsn); info!( core = self.core_id, diff --git a/nodedb/src/data/executor/columnar_checkpoint/write.rs b/nodedb/src/data/executor/columnar_checkpoint/write.rs index 61d2afc51..c05ebb079 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/write.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/write.rs @@ -47,7 +47,10 @@ impl CoreLoop { /// fall above the stamp and replay. That is safe and stays safe: it matched /// nothing against the state that the export captured, so re-executing the /// same predicate against that same restored state matches nothing again. - pub(in crate::data::executor) fn checkpoint_columnar_engines(&self) -> crate::Result { + /// + /// Every published generation raises `columnar_published_lsn`, the LSN + /// restart restores columnar from (see `redo_apply::cover`). + pub(in crate::data::executor) fn checkpoint_columnar_engines(&mut self) -> crate::Result { let durable_through = self.watermark; let ckpt_dir = columnar_ckpt_dir(&self.data_dir, self.core_id); @@ -71,6 +74,8 @@ impl CoreLoop { let written = self.write_columnar_generation(&gen_dir)?; self.publish_columnar_generation(&ckpt_dir, generation, durable_through)?; + self.floors.columnar_published_lsn = + self.floors.columnar_published_lsn.max(durable_through); // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs index af71f373f..322d76875 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs @@ -42,6 +42,10 @@ impl CheckpointFloors { // neither — so all three stay at zero until this process's own flush // succeeds, and clamp truncation to zero until it does. vector_durable_lsn: Lsn::ZERO, + // No generation is published until one is loaded or written. + vector_published_lsn: Lsn::ZERO, + kv_published_lsn: Lsn::ZERO, + columnar_published_lsn: Lsn::ZERO, crdt_durable_lsn: Lsn::ZERO, spatial_durable_lsn: Lsn::ZERO, replay_floors: ReplayFloors::default(), diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs index 9a02c020a..482568022 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs @@ -125,6 +125,25 @@ pub(in crate::data::executor) struct CheckpointFloors { /// the watermark — is what the core may report. pub(in crate::data::executor) vector_durable_lsn: Lsn, + /// LSN of the newest KV checkpoint generation on disk, restored at boot + /// from the manifest. Restart replay skips a KV record at or below it, so + /// a committed record applied at or below it must be published again + /// (`redo_apply::cover`). Never a truncation floor. + pub(in crate::data::executor) kv_published_lsn: Lsn, + + /// LSN of the newest columnar checkpoint generation on disk, whichever + /// flush published it, restored at boot from the manifest. Same rule as + /// `kv_published_lsn`. + pub(in crate::data::executor) columnar_published_lsn: Lsn, + + /// LSN of the newest vector checkpoint generation on disk, whichever + /// flush published it, restored at boot from the manifest. Restart + /// replay skips a vector record at or below the LSN its collection was + /// published with, so a committed record applied at or below this one + /// must be published again (`redo_apply::cover`). Never a truncation + /// floor: that is `vector_durable_lsn`. + pub(in crate::data::executor) vector_published_lsn: Lsn, + /// Highest LSN the CRDT engines are known to be durable through OUTSIDE the /// WAL (i.e. in `{data_dir}/crdt-ckpt/`), advanced only by a fully /// successful `checkpoint_crdt_engines`. diff --git a/nodedb/src/data/executor/core_loop/event_emit.rs b/nodedb/src/data/executor/core_loop/event_emit.rs index 608163887..16ca433d6 100644 --- a/nodedb/src/data/executor/core_loop/event_emit.rs +++ b/nodedb/src/data/executor/core_loop/event_emit.rs @@ -274,18 +274,17 @@ impl CoreLoop { new_value: Option<&[u8]>, old_value: Option<&[u8]>, ) { - let producer = match self.event_producer.as_mut() { - Some(p) => p, - None => return, // Event Plane not configured. - }; - - self.event_sequence += 1; + if self.event_producer.is_none() { + return; // Event Plane not configured. + } let (system_time_ms, valid_time_ms) = crate::event::bitemporal_extract::extract_stamps(new_value.or(old_value)); let event = crate::event::WriteEvent { - sequence: self.event_sequence, + // Assigned when the event is sent, so a held event never leaves + // a gap in the sequence. + sequence: 0, collection: Arc::from(collection), op, row_id, @@ -302,6 +301,28 @@ impl CoreLoop { statement_digest: task.request.statement_digest.clone(), }; + // The install pass of a committed-redo apply holds its events until + // the whole record landed. A rolled-back install sends none. + if let Some(scope) = self.redo_apply.scope.as_mut() + && scope.pass + == crate::data::executor::handlers::transaction::redo_apply::RedoApplyPass::Install + { + scope.pending_events.push(event); + return; + } + self.send_write_event(event); + } + + /// Number `event` with the next sequence and hand it to the Event Plane. + pub(in crate::data::executor) fn send_write_event( + &mut self, + mut event: crate::event::WriteEvent, + ) { + let Some(producer) = self.event_producer.as_mut() else { + return; + }; + self.event_sequence += 1; + event.sequence = self.event_sequence; producer.emit(event); } diff --git a/nodedb/src/data/executor/core_loop/maintenance.rs b/nodedb/src/data/executor/core_loop/maintenance.rs index f30487875..6a83fe662 100644 --- a/nodedb/src/data/executor/core_loop/maintenance.rs +++ b/nodedb/src/data/executor/core_loop/maintenance.rs @@ -72,6 +72,14 @@ impl CoreLoop { self.segment_compaction_config = config; } + /// Set the number of Data Plane cores on this node. The committed-redo + /// apply routes its records through the replay arms, which pick a record's + /// core as `vshard_id % num_cores`. Called after open, before the event + /// loop starts. + pub fn set_num_cores(&mut self, num_cores: usize) { + self.redo_apply.num_cores = num_cores; + } + /// Set query execution tuning parameters (called after open, before event loop). /// /// Also resizes the doc cache if `doc_cache_entries` differs from the current size. diff --git a/nodedb/src/data/executor/core_loop/open.rs b/nodedb/src/data/executor/core_loop/open.rs index 379c59e8e..47b013ebb 100644 --- a/nodedb/src/data/executor/core_loop/open.rs +++ b/nodedb/src/data/executor/core_loop/open.rs @@ -210,6 +210,8 @@ impl CoreLoop { active_bitemporal_stamps: HashMap::new(), active_graph_system_from: None, balanced_txn_entries: None, + redo_apply: + crate::data::executor::handlers::transaction::redo_apply::RedoApplyState::new(), }) } } diff --git a/nodedb/src/data/executor/core_loop/state.rs b/nodedb/src/data/executor/core_loop/state.rs index b093eb8ce..59d92273c 100644 --- a/nodedb/src/data/executor/core_loop/state.rs +++ b/nodedb/src/data/executor/core_loop/state.rs @@ -523,12 +523,13 @@ pub struct CoreLoop { pub(in crate::data::executor) write_index: super::write_index::WriteVersionIndex, /// Scratch map (surrogate → resolve-time bitemporal stamp) consulted ONLY - /// by `apply_point_put`. Populated right before a bitemporal document apply - /// scope — from a committing transaction's overlay sidecar (commit-time - /// install) or a decoded 8-tuple redo sub-record (WAL replay) — and cleared - /// right after. When a surrogate has an entry, the put is forced onto the - /// versioned store at the carried stamp rather than deriving a fresh one, - /// so the base install and the redo agree on the version key even when + /// by `apply_point_put` and `apply_point_delete`. Populated right before a + /// bitemporal document apply scope — from a committing transaction's + /// overlay sidecar (commit-time install) or a decoded stamped redo put or + /// delete sub-record (WAL replay, committed-redo apply) — and cleared right + /// after. When a surrogate has an entry, the put or tombstone is forced + /// onto the versioned store at the carried system time rather than a fresh + /// one, so every apply of the record agrees on the version key even when /// `doc_configs` is empty (the real replay-time boot state). pub(in crate::data::executor) active_bitemporal_stamps: HashMap, @@ -577,4 +578,8 @@ pub struct CoreLoop { /// flush's undo log, awaiting the post-apply `RecordCalvinWriteVersions` op. pub(in crate::data::executor) calvin_flush_index_tuples: crate::data::executor::handlers::transaction::index_write_values::StagedCalvinIndexTuples, + + /// Core count and per-record scratch of the committed-redo apply. + pub(in crate::data::executor) redo_apply: + crate::data::executor::handlers::transaction::redo_apply::RedoApplyState, } diff --git a/nodedb/src/data/executor/dispatch/meta.rs b/nodedb/src/data/executor/dispatch/meta.rs index f15708ce1..e47b32d62 100644 --- a/nodedb/src/data/executor/dispatch/meta.rs +++ b/nodedb/src/data/executor/dispatch/meta.rs @@ -298,6 +298,22 @@ impl CoreLoop { self.execute_calvin_resolve(task, *epoch, *position) } + // Install a committed transaction's redo record through the WAL + // replay arms, on every replica, in Raft log order. + MetaOp::ApplyTransactionRedo { + redo, + collections, + sum_targets, + } => self.execute_apply_transaction_redo( + task, + tid, + crate::data::executor::handlers::transaction::redo_apply::CommittedRedo { + redo, + collections, + sum_targets, + }, + ), + MetaOp::StageWrite { plan } => self.execute_stage_write(task, tid, plan), // Release the staging overlay once a transaction resolves (commit @@ -495,6 +511,7 @@ mod txn_created_columnar_engine_tests { payload: &pl, surrogates: &surrogates, schema_bytes: &sb, + intent: nodedb_physical::physical_plan::ColumnarInsertIntent::Insert, on_conflict_updates: &[], rls_write_check: &no_policy, }); @@ -624,6 +641,7 @@ mod txn_created_columnar_engine_tests { payload: &pl, surrogates: &surrogates, schema_bytes: &sb, + intent: nodedb_physical::physical_plan::ColumnarInsertIntent::Insert, on_conflict_updates: &[], rls_write_check: &no_policy, }); diff --git a/nodedb/src/data/executor/handlers/columnar_mutation_apply.rs b/nodedb/src/data/executor/handlers/columnar_mutation_apply.rs index 95533b1e1..6d9998661 100644 --- a/nodedb/src/data/executor/handlers/columnar_mutation_apply.rs +++ b/nodedb/src/data/executor/handlers/columnar_mutation_apply.rs @@ -41,6 +41,25 @@ pub(in crate::data::executor) struct ColumnarDeleteOutcome { pub restored: Vec<(Vec, RowLocation)>, } +/// The surrogate the segment sidecar records for the flushed row at `loc`. +/// Segment ids are 1-based: segment `n` is sidecar index `n - 1`. +pub(in crate::data::executor) fn flushed_row_surrogate( + sidecars: &std::collections::HashMap< + ColumnarEngineKey, + nodedb_columnar::mutation::snapshot::FlushedSurrogateTable, + >, + key: &ColumnarEngineKey, + loc: RowLocation, +) -> Option { + let segment_index = usize::try_from(loc.segment_id.checked_sub(1)?).ok()?; + sidecars + .get(key)? + .get(segment_index)? + .get(loc.row_index as usize) + .copied() + .flatten() +} + impl CoreLoop { /// Apply `(old_pk, post_image)` rows through `MutationEngine::update` /// (delete-old-PK + insert-new-row). For a collection with geometry @@ -100,7 +119,16 @@ impl CoreLoop { } else { None }; - match engine.update(old_pk, new_row) { + // A flushed row's surrogate lives in its segment's sidecar, which + // the engine does not hold; the replacement row keeps it. + let flushed_surrogate = engine + .pk_index() + .get(&old_pk_bytes) + .filter(|loc| loc.segment_id != engine.memtable_segment_id()) + .and_then(|loc| { + flushed_row_surrogate(&self.columnar_flushed_surrogates, key, *loc) + }); + match engine.update(old_pk, new_row, flushed_surrogate) { Ok(_) => {} Err(e) => { warn!(core = self.core_id, %collection, error = %e, "columnar update row failed"); @@ -220,3 +248,69 @@ fn push_removed_spatial_undo(undo_log: &mut Vec, removed: Vec ColumnarSchema { + ColumnarSchema { + columns: vec![ + ColumnDef::required("id", ColumnType::Int64).with_primary_key(), + ColumnDef::required("v", ColumnType::Int64), + ], + version: 1, + } + } + + #[test] + fn an_update_of_a_flushed_row_keeps_the_surrogate_its_segment_records() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let key: ColumnarEngineKey = ( + DatabaseId::DEFAULT, + crate::types::TenantId::new(1), + "m".to_string(), + ); + let mut engine = nodedb_columnar::MutationEngine::new("m".to_string(), schema()); + engine + .insert_with_surrogate(&[Value::Integer(1), Value::Integer(10)], Surrogate::new(42)) + .expect("insert"); + let segment_id = engine.next_segment_id(); + let sidecar = engine.memtable_surrogates().to_vec(); + let _drained = engine.memtable_mut().drain_optimized(); + engine.on_memtable_flushed(segment_id).expect("flush"); + core.columnar_engines.insert(key.clone(), engine); + core.columnar_flushed_surrogates + .insert(key.clone(), vec![sidecar]); + + let outcome = core.apply_columnar_update_rows( + &make_default_task(), + &key, + &schema(), + &[( + Value::Integer(1), + vec![Value::Integer(1), Value::Integer(99)], + )], + None, + ); + + assert_eq!(outcome.affected, 1); + let live: Vec<(Option, Vec)> = core + .columnar_engines + .get(&key) + .expect("engine") + .scan_memtable_rows_with_surrogates() + .collect(); + assert_eq!( + live, + vec![( + Some(Surrogate::new(42)), + vec![Value::Integer(1), Value::Integer(99)] + )] + ); + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs b/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs index 673800c52..f08eb869a 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs @@ -168,10 +168,16 @@ impl CoreLoop { Ok(id) => id, Err(e) => return self.response_error(task, e), }; - for plan in plans { - if let Err(e) = self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan) { - return self.response_error(task, e); - } + // Staging reads the epoch's time anchor, so every replica stages the + // same images. + let prev_epoch_ms = self.epoch_system_ms; + self.epoch_system_ms = Some(epoch_system_ms); + let staged = plans.iter().try_for_each(|plan| { + self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan) + }); + self.epoch_system_ms = prev_epoch_ms; + if let Err(e) = staged { + return self.response_error(task, e); } Response { diff --git a/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs b/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs index bea3184d0..cb73dc6b5 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs @@ -79,7 +79,10 @@ impl CoreLoop { }; // Stage every plan before publishing `PendingCommit`. Panic isolation - // cleans both staging representations after any failure. + // cleans both staging representations after any failure. Staging reads + // the epoch's time anchor, so every replica stages the same images. + let prev_epoch_ms = self.epoch_system_ms; + self.epoch_system_ms = Some(epoch_system_ms); let stage_result = catch_unwind(AssertUnwindSafe(|| { for plan in plans { self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan)?; @@ -88,6 +91,7 @@ impl CoreLoop { } Ok::<(), ErrorCode>(()) })); + self.epoch_system_ms = prev_epoch_ms; match stage_result { Ok(Ok(())) => {} Ok(Err(error)) => { diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs index 596661f33..cdc191c35 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs @@ -13,7 +13,7 @@ //! `MetaOp::ResolveTxn` already does for session transactions //! (`resolve/entry.rs`). -use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan, TimeseriesOp}; +use nodedb_physical::physical_plan::{ColumnarOp, DocumentOp, GraphOp, PhysicalPlan, TimeseriesOp}; use nodedb_types::RowIdentity; use crate::bridge::envelope::{ErrorCode, Response, Status}; @@ -45,9 +45,11 @@ impl CoreLoop { /// than a live predicate rescan — see that module's docs for the /// determinism rationale. `TimeseriesOp::Ingest` is staged through the /// same canonical row decoder as session writes; its per-row tokens are - /// overlay-local and never become base-storage identities. Columnar and - /// spatial predicate writes remain unstaged because they have no - /// deterministic post-image overlay representation. + /// overlay-local and never become base-storage identities. Every columnar + /// write is staged through the session handlers, so COMMIT resolve reads + /// its post-images from the overlay. Staging runs under the epoch's time + /// anchor, so every replica stages the same images. Spatial writes carry + /// their absolute post-image on the plan node and stay unstaged. pub(in crate::data::executor) fn stage_calvin_overlay( &mut self, task: &ExecutionTask, @@ -218,6 +220,22 @@ impl CoreLoop { }); Self::stage_result(&response) } + // Every columnar write stages its post-image the way a session + // write does, so the transaction's redo carries the rows the + // flush installs: an ON CONFLICT merge and a predicate DML's + // matched rows are resolved once, here, against the state this + // position observes. + PhysicalPlan::Columnar( + op @ (ColumnarOp::Insert { .. } + | ColumnarOp::Update { .. } + | ColumnarOp::Delete { .. } + | ColumnarOp::ResolvedUpdate { .. } + | ColumnarOp::ResolvedDelete { .. } + | ColumnarOp::Truncate { .. }), + ) => { + let resp = self.execute_stage_columnar(task, tid, txn_id, op); + Self::stage_result(&resp) + } PhysicalPlan::Graph( op @ (GraphOp::EdgePut { .. } | GraphOp::EdgeDelete { .. } @@ -259,11 +277,8 @@ impl CoreLoop { /// itself failing) is a silent no-op — the same shape as the /// `commit_pending` removal it accompanies. /// - /// Mirrors `MetaOp::DropTxnOverlay`'s gauge accounting exactly: every map - /// was populated (if at all) via the `txn_overlay_mut` / - /// `graph_txn_overlay_mut` / `array_txn_overlay_mut` choke points, which bump `active_txn_overlays` - /// on first creation, so removal here must decrement by the same count - /// or the gauge drifts upward forever on every Calvin-staged transaction. + /// Runs the same teardown as `MetaOp::DropTxnOverlay` + /// (`drop_overlay_entry`), so the `active_txn_overlays` gauge stays exact. pub(in crate::data::executor) fn drop_calvin_synthetic_overlay( &mut self, epoch: u64, @@ -271,15 +286,9 @@ impl CoreLoop { vshard: u32, ) { if let Ok(synthetic_txn_id) = calvin_synthetic_txn_id(epoch, position, vshard) { - let removed = u64::from(self.txn_overlays.remove(&synthetic_txn_id).is_some()) - + u64::from(self.graph_txn_overlays.remove(&synthetic_txn_id).is_some()) - + u64::from(self.array_txn_overlays.remove(&synthetic_txn_id).is_some()); - if removed > 0 - && let Some(m) = &self.metrics - { - m.active_txn_overlays - .fetch_sub(removed, std::sync::atomic::Ordering::Relaxed); - } + // The shared teardown also drops a columnar engine staging + // auto-created that the flush left empty. + self.drop_overlay_entry(synthetic_txn_id); } } } diff --git a/nodedb/src/data/executor/handlers/control/calvin_resolve.rs b/nodedb/src/data/executor/handlers/control/calvin_resolve.rs index f6ac2d30b..fc8a97bbe 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_resolve.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_resolve.rs @@ -19,6 +19,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; use super::calvin_txn_id::calvin_synthetic_txn_id; +use crate::data::executor::handlers::transaction::resolve::StagedWrites; impl CoreLoop { /// Resolve the Calvin transaction staged under `(epoch, position)` on @@ -79,7 +80,10 @@ impl CoreLoop { // stamps back from the overlay, so redo and base install agree. let prev_epoch_ms = self.epoch_system_ms; self.epoch_system_ms = Some(epoch_system_ms); - let resp = self.execute_resolve_txn(task, tid, synthetic_txn_id, &plans); + // Calvin stages no vector-primary direct write, so those resolve from + // their plan nodes. + let resp = + self.execute_resolve_staged(task, tid, synthetic_txn_id, &plans, StagedWrites::Calvin); self.epoch_system_ms = prev_epoch_ms; resp } @@ -439,4 +443,80 @@ mod tests { "resolving an (epoch, position) that was never staged must error" ); } + + #[test] + fn a_calvin_columnar_upsert_resolves_to_the_merged_row_it_installs() { + use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp}; + use nodedb_types::columnar::{ + COLUMNAR_IMAGE_KIND, ColumnDef, ColumnType, ColumnarImageWalRecord, ColumnarSchema, + }; + + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let task = make_task(); + let schema = ColumnarSchema { + columns: vec![ + ColumnDef::required("id", ColumnType::Int64).with_primary_key(), + ColumnDef::required("v", ColumnType::Int64), + ], + version: 1, + }; + let mut engine = nodedb_columnar::MutationEngine::new("m".to_string(), schema); + engine + .insert_with_surrogate(&[Value::Integer(1), Value::Integer(10)], Surrogate::new(5)) + .expect("seed base row"); + core.columnar_engines.insert( + (DatabaseId::DEFAULT, TenantId::new(1), "m".to_string()), + engine, + ); + + let mut submitted = std::collections::HashMap::new(); + submitted.insert("id".to_string(), Value::Integer(1)); + submitted.insert("v".to_string(), Value::Integer(20)); + let upsert = PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "m"), + payload: nodedb_types::value_to_msgpack(&Value::Array(vec![Value::Object(submitted)])) + .expect("encode row"), + format: "msgpack".to_string(), + intent: ColumnarInsertIntent::Put, + on_conflict_updates: vec![( + "v".to_string(), + UpdateValue::Literal( + nodedb_types::value_to_msgpack(&Value::Integer(99)).expect("literal"), + ), + )], + surrogates: vec![Surrogate::new(5)], + schema_bytes: Vec::new(), + provenance: None, + wal_lsn: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }); + stage(&mut core, &task, 6, 0, &[upsert]); + + let resp = core.execute_calvin_resolve(&task, 6, 0); + assert_eq!(resp.status, Status::Ok, "resolve must succeed: {resp:?}"); + let record = RedoRecord::from_bytes(resp.payload.as_bytes()).expect("decode redo"); + let image = record + .ops + .iter() + .find_map(|op| { + zerompk::from_msgpack::(&op.payload) + .ok() + .filter(|image| image.kind == COLUMNAR_IMAGE_KIND) + }) + .expect("the upsert resolves to a columnar image record"); + assert_eq!(image.rows.len(), 1); + let row = + nodedb_types::value_from_msgpack(&image.rows[0].image_msgpack).expect("decode image"); + let Value::Object(row) = row else { + panic!("image is a column-name object: {row:?}"); + }; + assert_eq!( + row.get("v"), + Some(&Value::Integer(99)), + "the redo carries the merged row, not the submitted one" + ); + } } diff --git a/nodedb/src/data/executor/handlers/control/crdt_doc.rs b/nodedb/src/data/executor/handlers/control/crdt_doc.rs index 983094778..2e46b2e92 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_doc.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_doc.rs @@ -16,6 +16,9 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::handlers::returning_doc; use crate::data::executor::handlers::returning_rows; +use crate::data::executor::handlers::transaction::undo::document_outcome::{ + DocumentRow, push_delete_undo, +}; use crate::data::executor::task::ExecutionTask; use crate::engine::document::store::{RowIdentity, StorageKey}; use nodedb_physical::physical_plan::ReturningSpec; @@ -111,7 +114,9 @@ impl CoreLoop { }; let response = if let Some(bytes) = materialized { - self.materialize_document_write( + // The Loro row and its sparse projection answer the same reads, so + // a projection that did not land fails the write. + if let Err(error) = self.materialize_document_write( task, super::crdt_materialize::CrdtMaterializeWrite { tid: tenant_id.as_u64(), @@ -121,7 +126,9 @@ impl CoreLoop { value: &bytes, index_text: true, }, - ); + ) { + return self.response_error(task, error); + } if let Some(spec) = returning { // No strict schema: a CRDT row's stored body is whatever // `encode_crdt_row` materialized from Loro, which is always @@ -256,11 +263,27 @@ impl CoreLoop { ); } self.checkpoint_coordinator.mark_dirty("sparse", 1); + let prior_value = outcome.prior_value.clone(); + if self.recording_redo_undo() { + let mut undo = Vec::new(); + push_delete_undo( + &mut undo, + DocumentRow { + database_id: task.request.database_id.as_u64(), + tid, + collection, + storage_key, + identity: RowIdentity::from_user_key(document_id), + }, + outcome, + ); + self.record_redo_undo(undo); + } // Emit the delete to the Event Plane only when a row was actually // removed, threading the pre-delete bytes through as `old_value` so // CDC/change-stream consumers observe the prior state. - if let Some(prior_bytes) = outcome.prior_value.as_deref() { + if let Some(prior_bytes) = prior_value.as_deref() { let old_converted = self.resolve_event_payload( task.request.database_id.as_u64(), tid, @@ -275,12 +298,12 @@ impl CoreLoop { ); } - // Project the pre-deletion row for RETURNING. `outcome.prior_value` is + // Project the pre-deletion row for RETURNING. `prior_value` is // only borrowed by the CDC emit above (via `.as_deref()`), so it is // still available here; the user-visible `document_id` is injected as // `id` exactly like PointDelete. let response = if let Some(spec) = returning { - if let Some(prior_bytes) = outcome.prior_value.as_deref() { + if let Some(prior_bytes) = prior_value.as_deref() { // No strict schema — see the upsert path: a CRDT row is // materialized as MessagePack in either storage mode. let doc = match returning_doc::from_stored( @@ -318,7 +341,7 @@ impl CoreLoop { } else { // No RETURNING: report what the delete actually removed. A tombstone // written over an already-absent document removes nothing. - self.response_affected(task, u64::from(outcome.prior_value.is_some())) + self.response_affected(task, u64::from(prior_value.is_some())) }; self.checkpoint_coordinator.mark_dirty("crdt", 1); response diff --git a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs index dcd8acf60..e01ae5e4e 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs @@ -29,6 +29,10 @@ use tracing::warn; +use crate::data::executor::handlers::transaction::undo::document_outcome::{ + DocumentRow, push_put_undo, +}; + use nodedb_types::{RowIdentity, Surrogate}; use crate::data::executor::core_loop::CoreLoop; @@ -102,7 +106,8 @@ impl CoreLoop { surrogate: Surrogate, value: &[u8], ) { - self.materialize_document_write( + // A materialization miss is logged and never wedges the sync stream. + if let Err(error) = self.materialize_document_write( task, CrdtMaterializeWrite { tid, @@ -112,7 +117,15 @@ impl CoreLoop { value, index_text: false, }, - ); + ) { + warn!( + core = self.core_id, + %collection, + surrogate = surrogate.as_u32(), + %error, + "crdt sync materialize into sparse document store failed" + ); + } } /// Shared body of the sparse-store materialization. `index_text` gates @@ -125,7 +138,7 @@ impl CoreLoop { &mut self, task: &ExecutionTask, write: CrdtMaterializeWrite<'_>, - ) { + ) -> crate::Result<()> { let CrdtMaterializeWrite { tid, collection, @@ -137,15 +150,8 @@ impl CoreLoop { let database_id = task.request.database_id.as_u64(); let storage_key = StorageKey::for_surrogate(surrogate); - let txn = match self.sparse.begin_write() { - Ok(t) => t, - Err(e) => { - warn!(core = self.core_id, %collection, error = %e, "crdt sync materialize: begin_write failed"); - return; - } - }; - - let prior = match self.apply_point_put( + let txn = self.sparse.begin_write()?; + let outcome = self.apply_point_put( &txn, PointPutParams { database_id, @@ -160,24 +166,11 @@ impl CoreLoop { wal_lsn: task.wal_lsn(), resolved_targets: &[], }, - ) { - Ok(p) => p, - Err(e) => { - warn!( - core = self.core_id, - %collection, - document_id = %storage_key, - error = %e, - "crdt sync materialize into sparse document store failed" - ); - return; - } - }; - - if let Err(e) = txn.commit() { - warn!(core = self.core_id, %collection, error = %e, "crdt sync materialize: commit failed"); - return; - } + )?; + txn.commit().map_err(|e| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("crdt materialize commit: {e}"), + })?; self.checkpoint_coordinator.mark_dirty("sparse", 1); @@ -190,7 +183,24 @@ impl CoreLoop { collection, RowIdentity::from_user_key(document_id), value, - prior.prior_value.as_deref(), + outcome.prior_value.as_deref(), ); + if self.recording_redo_undo() { + let mut undo = Vec::new(); + push_put_undo( + &mut undo, + DocumentRow { + database_id, + tid, + collection, + storage_key, + identity: RowIdentity::from_user_key(document_id), + }, + outcome, + None, + ); + self.record_redo_undo(undo); + } + Ok(()) } } diff --git a/nodedb/src/data/executor/handlers/kv/atomic.rs b/nodedb/src/data/executor/handlers/kv/atomic.rs index 434e6f0ea..50a705349 100644 --- a/nodedb/src/data/executor/handlers/kv/atomic.rs +++ b/nodedb/src/data/executor/handlers/kv/atomic.rs @@ -26,6 +26,26 @@ pub(in crate::data::executor) struct KvAtomicCtx<'a> { pub(in crate::data::executor) rls_write_check: &'a nodedb_types::RlsWriteCheck, } +/// The `ErrorCode` an atomic that computed or stored no value answers with. +pub(in crate::data::executor) fn atomic_error_code( + error: AtomicError, + collection: &str, +) -> ErrorCode { + match error { + AtomicError::TypeMismatch { detail } => ErrorCode::TypeMismatch { + collection: collection.to_string(), + detail, + }, + AtomicError::Overflow => ErrorCode::OverflowError { + collection: collection.to_string(), + }, + AtomicError::Encode { detail } => ErrorCode::Internal { detail }, + // Nothing was written: the engine consults the gate before it + // installs the computed value. + AtomicError::Rejected(error) => (*error).into(), + } +} + impl CoreLoop { pub(in crate::data::executor) fn execute_kv_incr( &mut self, @@ -98,25 +118,7 @@ impl CoreLoop { ), } } - Err(AtomicError::TypeMismatch { detail }) => self.response_error( - task, - ErrorCode::TypeMismatch { - collection: collection.to_string(), - detail, - }, - ), - Err(AtomicError::Overflow) => self.response_error( - task, - ErrorCode::OverflowError { - collection: collection.to_string(), - }, - ), - Err(AtomicError::Encode { detail }) => { - self.response_error(task, ErrorCode::Internal { detail }) - } - // Nothing was written: the engine consults the gate before it - // installs the computed value. - Err(AtomicError::Rejected(error)) => self.response_error(task, *error), + Err(error) => self.response_atomic_error(task, collection, error), } } @@ -186,25 +188,7 @@ impl CoreLoop { ), } } - Err(AtomicError::TypeMismatch { detail }) => self.response_error( - task, - ErrorCode::TypeMismatch { - collection: collection.to_string(), - detail, - }, - ), - Err(AtomicError::Overflow) => self.response_error( - task, - ErrorCode::OverflowError { - collection: collection.to_string(), - }, - ), - Err(AtomicError::Encode { detail }) => { - self.response_error(task, ErrorCode::Internal { detail }) - } - // Nothing was written: the engine consults the gate before it - // installs the computed value. - Err(AtomicError::Rejected(error)) => self.response_error(task, *error), + Err(error) => self.response_atomic_error(task, collection, error), } } @@ -229,18 +213,16 @@ impl CoreLoop { return self.response_error(task, ErrorCode::ResourcesExhausted); } - // `new_value` is caller-supplied, so the row that would exist after a - // successful swap is known before the engine is entered — decided here - // rather than after the fact. - if let Err(e) = super::rls::admit_kv_row(rls_write_check, new_value, key, tid, collection) { - return self.response_error(task, e); - } - let now_ms: u64 = self .epoch_system_ms .map(|ms| ms as u64) .unwrap_or_else(current_ms); - let result = self.kv_engine.cas( + // A swap into a typed row stores the row with one column replaced, not + // `new_value` itself, so the policy decides the image the engine + // computes — see `Incr`. + let admit = + |image: &[u8]| super::rls::admit_kv_row(rls_write_check, image, key, tid, collection); + let result = match self.kv_engine.cas( crate::engine::kv::AtomicKeyCtx { database_id: did, tenant_id: tid, @@ -251,9 +233,13 @@ impl CoreLoop { }, expected, new_value, - ); + &admit, + ) { + Ok(result) => result, + Err(error) => return self.response_atomic_error(task, collection, error), + }; - if result.success { + if let Some(written) = &result.written { if let Some(ref m) = self.metrics { m.record_kv_put(); } @@ -263,7 +249,7 @@ impl CoreLoop { collection, crate::event::WriteOp::Update, crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), - Some(new_value), + Some(written.as_slice()), None, ); self.note_kv_write_lsn(task, did, tid, collection, key); @@ -274,7 +260,7 @@ impl CoreLoop { .as_ref() .map(|v| base64::Engine::encode(&base64::engine::general_purpose::STANDARD, v)); match response_codec::encode_json_as_msgpack(&serde_json::json!({ - "success": result.success, + "success": result.success(), "current_value": current_b64, })) { Ok(payload) => self.response_with_payload(task, payload), @@ -290,7 +276,7 @@ impl CoreLoop { /// `rls_filters` decides the OLD value handed back: `GETSET` is a read as /// much as a write, so a row the read policy hides must come back absent /// rather than being disclosed by the write that replaced it. The write - /// half is a separate decision on `new_value`. + /// half is a separate decision on the image the write stores. pub(in crate::data::executor) fn execute_kv_getset( &mut self, ctx: KvAtomicCtx<'_>, @@ -312,17 +298,15 @@ impl CoreLoop { return self.response_error(task, ErrorCode::ResourcesExhausted); } - // The stored row is replaced wholesale, so the post-image is known - // before the engine call. - if let Err(e) = super::rls::admit_kv_row(rls_write_check, new_value, key, tid, collection) { - return self.response_error(task, e); - } - let now_ms: u64 = self .epoch_system_ms .map(|ms| ms as u64) .unwrap_or_else(current_ms); - let old = self.kv_engine.getset( + // A write into a typed row stores the row with one column replaced, so + // the policy decides the image the engine computes — see `Incr`. + let admit = + |image: &[u8]| super::rls::admit_kv_row(rls_write_check, image, key, tid, collection); + let crate::engine::kv::GetSetResult { old, written } = match self.kv_engine.getset( crate::engine::kv::AtomicKeyCtx { database_id: did, tenant_id: tid, @@ -332,7 +316,11 @@ impl CoreLoop { surrogate, }, new_value, - ); + &admit, + ) { + Ok(result) => result, + Err(error) => return self.response_atomic_error(task, collection, error), + }; if let Some(ref m) = self.metrics { m.record_kv_put(); @@ -343,7 +331,7 @@ impl CoreLoop { collection, crate::event::WriteOp::Update, crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), - Some(new_value), + Some(written.as_slice()), old.as_deref(), ); self.note_kv_write_lsn(task, did, tid, collection, key); @@ -381,4 +369,14 @@ impl CoreLoop { ), } } + + /// The error response for an atomic that computed or stored no value. + pub(in crate::data::executor) fn response_atomic_error( + &self, + task: &ExecutionTask, + collection: &str, + error: AtomicError, + ) -> Response { + self.response_error(task, atomic_error_code(error, collection)) + } } diff --git a/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs b/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs index 31102ca43..885bde5ca 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs @@ -10,7 +10,7 @@ use nodedb_physical::physical_plan::KvResolveOutcome; use super::context::{ResolveResult, ResolvedPut, expiry_from_ttl, one, put_mutation}; use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::handlers::kv::atomic::KvAtomicCtx; +use crate::data::executor::handlers::kv::atomic::{KvAtomicCtx, atomic_error_code}; use crate::data::executor::handlers::kv::rls::admit_kv_row; use crate::data::executor::response_codec; use crate::engine::kv::current_ms; @@ -22,24 +22,6 @@ fn base64_body(body: Option<&[u8]>) -> Option { body.map(|v| base64::Engine::encode(&base64::engine::general_purpose::STANDARD, v)) } -/// Translate an `AtomicError` into the `ErrorCode` the live handler returns -/// for it. -fn atomic_error_code(error: crate::engine::kv::AtomicError, collection: &str) -> ErrorCode { - match error { - crate::engine::kv::AtomicError::TypeMismatch { detail } => ErrorCode::TypeMismatch { - collection: collection.to_string(), - detail, - }, - crate::engine::kv::AtomicError::Overflow => ErrorCode::OverflowError { - collection: collection.to_string(), - }, - crate::engine::kv::AtomicError::Encode { detail } => ErrorCode::Internal { detail }, - // The gate is consulted out here, not inside the engine, so the - // engine's own rejection path is unreachable from a compute call. - crate::engine::kv::AtomicError::Rejected(error) => (*error).into(), - } -} - impl CoreLoop { /// Resolve `INCR`. `ttl_ms > 0` installs a fresh expiry; `ttl_ms == 0` /// preserves the existing one, matching `atomic_put` (not `KvEngine::put`). @@ -146,13 +128,15 @@ impl CoreLoop { if self.kv_engine.is_over_budget() { return Err(ErrorCode::ResourcesExhausted); } - // Decided before the swap, same as `execute_kv_cas`: `new_value` is - // caller-supplied, so the post-swap row is known up front. - admit_kv_row(rls_write_check, new_value, key, tid, collection)?; - let now_ms = self.kv_atomic_now_ms(); let current = self.kv_resolve_read(did, tid, collection, key, now_ms); - let (matches, write_bytes) = compute::cas(current.as_deref(), expected, new_value); + let (matches, write_bytes) = compute::cas(current.as_deref(), expected, new_value) + .map_err(|e| atomic_error_code(e, collection))?; + // Decided on the image the swap stores, same as `execute_kv_cas`: a + // swap into a typed row stores the row, not `new_value` itself. + if matches { + admit_kv_row(rls_write_check, &write_bytes, key, tid, collection)?; + } let response_payload = response_codec::encode_json_as_msgpack(&serde_json::json!({ "success": matches, @@ -199,11 +183,12 @@ impl CoreLoop { if self.kv_engine.is_over_budget() { return Err(ErrorCode::ResourcesExhausted); } - admit_kv_row(rls_write_check, new_value, key, tid, collection)?; - let now_ms = self.kv_atomic_now_ms(); let old = self.kv_resolve_read(did, tid, collection, key, now_ms); - let write_bytes = compute::getset(old.as_deref(), new_value); + let write_bytes = compute::getset(old.as_deref(), new_value) + .map_err(|e| atomic_error_code(e, collection))?; + // Decided on the image the write stores, same as `execute_kv_getset`. + admit_kv_row(rls_write_check, &write_bytes, key, tid, collection)?; let disclosable_old = match &old { Some(bytes) => match self.row_passes_rls(bytes, rls_filters) { diff --git a/nodedb/src/data/executor/handlers/mod.rs b/nodedb/src/data/executor/handlers/mod.rs index 9d785eccd..d87918471 100644 --- a/nodedb/src/data/executor/handlers/mod.rs +++ b/nodedb/src/data/executor/handlers/mod.rs @@ -67,6 +67,7 @@ pub mod timeseries; mod timeseries_gap_fill; pub mod timeseries_wal; pub(super) mod timeseries_wal_decode; +mod timeseries_wal_payload; pub mod transaction; pub mod truncate; pub mod truncate_response; diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index 8e9d1931e..bfb050d9b 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -146,7 +146,17 @@ impl CoreLoop { let _ = user_roles; let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); - let bitemporal = self.is_bitemporal(database_id, tid, collection); + // A stamp in `active_bitemporal_stamps` (a committed redo delete) + // forces the versioned branch at the EXACT resolve-time system time, + // so every replica and every restart tombstones the same version key. + // Absent an override, derive bitemporality from config and mint the + // system time here. + let carried_sys_from = self + .active_bitemporal_stamps + .get(&surrogate.as_u32()) + .map(|stamp| stamp.sys_from_ms); + let bitemporal = + carried_sys_from.is_some() || self.is_bitemporal(database_id, tid, collection); let config_key = ( crate::types::DatabaseId::new(database_id), crate::types::TenantId::new(tid), @@ -177,7 +187,7 @@ impl CoreLoop { resolved_targets, )?; } - let sys_from = self.bitemporal_now_ms(); + let sys_from = carried_sys_from.unwrap_or_else(|| self.bitemporal_now_ms()); bitemporal_sys_from_ms = Some(sys_from); self.sparse.versioned_tombstone_in_txn( txn, @@ -438,7 +448,7 @@ impl CoreLoop { /// (`apply_point_delete`) and transactional (`tx_point_delete`) paths. /// These checks have no persistent side effect, so a violation here /// simply aborts before the write. -fn run_delete_enforcement( +pub(in crate::data::executor) fn run_delete_enforcement( sparse: &crate::engine::sparse::btree::SparseEngine, database_id: u64, tid: u64, diff --git a/nodedb/src/data/executor/handlers/point/apply_put/enforce.rs b/nodedb/src/data/executor/handlers/point/apply_put/enforce.rs index 661755625..31d5934d5 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/enforce.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/enforce.rs @@ -22,22 +22,22 @@ use nodedb_physical::physical_plan::ResolvedSumTarget; use super::types::map_enforcement_error; /// The write being admitted, and the pre-image it is judged against. -pub(in crate::data::executor::handlers::point) struct PutEnforcement<'a> { - pub(in crate::data::executor::handlers::point) config_key: &'a (DatabaseId, TenantId, String), - pub(in crate::data::executor::handlers::point) database_id: u64, - pub(in crate::data::executor::handlers::point) tid: u64, - pub(in crate::data::executor::handlers::point) collection: &'a str, +pub(in crate::data::executor) struct PutEnforcement<'a> { + pub(in crate::data::executor) config_key: &'a (DatabaseId, TenantId, String), + pub(in crate::data::executor) database_id: u64, + pub(in crate::data::executor) tid: u64, + pub(in crate::data::executor) collection: &'a str, /// The incoming body in MessagePack form for both storage modes (a strict /// collection encodes its Binary Tuple separately). - pub(in crate::data::executor::handlers::point) value: &'a [u8], + pub(in crate::data::executor) value: &'a [u8], /// The row as currently stored, when one exists. - pub(in crate::data::executor::handlers::point) old_value: &'a Option>, - pub(in crate::data::executor::handlers::point) user_roles: &'a [String], + pub(in crate::data::executor) old_value: &'a Option>, + pub(in crate::data::executor) user_roles: &'a [String], /// `(target collection, join-key value)` → target row surrogate, resolved /// on the Control Plane at plan time. A period-lock check reads its /// reference row's surrogate off this slice, keyed by /// `(config.ref_table, period value)`. - pub(in crate::data::executor::handlers::point) resolved_targets: &'a [ResolvedSumTarget], + pub(in crate::data::executor) resolved_targets: &'a [ResolvedSumTarget], } impl CoreLoop { @@ -51,7 +51,7 @@ impl CoreLoop { /// CRDT-sync materialization (which passes `enforce == false`): those /// deltas already passed admission on their origin replica at Raft commit /// time. - pub(in crate::data::executor::handlers::point) fn check_stateless_put_enforcement( + pub(in crate::data::executor) fn check_stateless_put_enforcement( &self, enforce: bool, p: PutEnforcement<'_>, diff --git a/nodedb/src/data/executor/handlers/point/apply_put/mod.rs b/nodedb/src/data/executor/handlers/point/apply_put/mod.rs index 47bd78657..eb9ade446 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/mod.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/mod.rs @@ -4,7 +4,7 @@ //! (spatial/vector/sparse), and UNIQUE-constraint check. pub(in crate::data::executor::handlers::point) mod core; -pub(in crate::data::executor::handlers::point) mod enforce; +pub(in crate::data::executor) mod enforce; pub(in crate::data::executor::handlers::point) mod index; pub(in crate::data::executor::handlers::point) mod sparse; pub(in crate::data::executor) mod stored_body; @@ -12,6 +12,7 @@ pub(in crate::data::executor::handlers::point) mod types; pub(in crate::data::executor) mod unique; pub(in crate::data::executor::handlers::point) mod vector; +pub(in crate::data::executor) use enforce::PutEnforcement; pub(in crate::data::executor) use index::SpatialEntryId; pub(in crate::data::executor) use types::{PointPutOutcome, PointPutParams, map_enforcement_error}; pub(in crate::data::executor) use vector::{VectorIndexDelta, VectorIndexPutParams}; diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs index 49af195d9..42577ee8d 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs @@ -22,6 +22,7 @@ pub(in crate::data::executor) struct VectorIndexPutParams<'a> { /// (`collection`, `field`, `doc_id`). Replaces a raw `(index_key, vector_id)` /// tuple so undo can restore/remove the reverse-lookup map symmetrically with /// the R-tree's `SpatialInsert`/`SpatialDelete` undo pattern. +#[derive(Clone)] pub(in crate::data::executor) struct VectorIndexDelta { pub index_key: (nodedb_types::DatabaseId, crate::types::TenantId, String), pub vector_id: u32, diff --git a/nodedb/src/data/executor/handlers/timeseries/ingest.rs b/nodedb/src/data/executor/handlers/timeseries/ingest.rs index 39f825d98..ae9949231 100644 --- a/nodedb/src/data/executor/handlers/timeseries/ingest.rs +++ b/nodedb/src/data/executor/handlers/timeseries/ingest.rs @@ -184,6 +184,18 @@ impl CoreLoop { return self.response_error(task, error); } + if mode == TimeseriesApplyMode::RedoInstall + && let Err(error) = self.prepare_redo_ts_ingest( + task.request.database_id, + tid, + collection, + &lines, + now_ms, + ) + { + return self.response_error(task, error); + } + let bitemporal = self.is_bitemporal(task.request.database_id.as_u64(), tid.as_u64(), collection); let is_new_memtable = !self.columnar_memtables.contains_key(&key); @@ -216,26 +228,8 @@ impl CoreLoop { // The WAL has already committed this record, so the admission gate // resolves every possible mid-record stop before the first row lands. - let governor_pressure = self - .governor - .try_reserve( - task.request.database_id, - tid, - nodedb_mem::EngineId::Timeseries, - 0, - ) - .is_err(); let soft_limit = self.ts_tuning.memtable_budget_bytes; - let hard_limit = self.ts_tuning.memtable_hard_limit_bytes; - let max_tag_cardinality = self.ts_tuning.max_tag_cardinality; - let needs_flush = self.columnar_memtables.get(&key).is_some_and(|mt| { - let resident = mt.memory_bytes(); - resident >= soft_limit - || resident >= hard_limit - || governor_pressure - || !admission::has_tag_headroom(mt, &lines, max_tag_cardinality) - }); - if needs_flush { + if self.ts_ingest_needs_flush(&key, &lines) { if mode == TimeseriesApplyMode::CommitDeferred { return self.response_error( task, @@ -245,6 +239,20 @@ impl CoreLoop { }, ); } + // A redo install flushed before it took its pre-image. A flush + // now would drain rows that pre-image holds, so the install + // fails and rolls back instead. + if mode == TimeseriesApplyMode::RedoInstall { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "'{collection}' still needs a flush after the flush that preceded \ + its committed-redo install" + ), + }, + ); + } if let Err(e) = self.flush_ts_collection(tid, task.request.database_id, collection, now_ms) { diff --git a/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs b/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs index d76e33f00..e97143f9a 100644 --- a/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs +++ b/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs @@ -16,6 +16,12 @@ use crate::data::executor::task::ExecutionTask; pub(in crate::data::executor) enum TimeseriesApplyMode { Immediate, CommitDeferred, + /// The install pass of a committed redo record. The memtable flushes the + /// rows it already holds when it has no room, then the ingest records + /// the pre-image its undo restores. No flush, budget recharge or timer + /// update runs after the rows land: the apply settles those once the + /// whole record installed (`settle_redo_timeseries`). + RedoInstall, } /// Parameters for a timeseries ingest operation on the Data Plane. @@ -148,13 +154,19 @@ impl CoreLoop { }; // A record written before a later truncate describes rows the // truncate removed: the same "nothing to write" answer as a record - // already on disk. + // already on disk. Strictly before: a record at the truncate's own LSN + // is a sibling sub-record of the same transaction redo group, applied + // in the order the transaction wrote it. let already_flushed = already_flushed || wal_lsn.is_some_and(|lsn| { self.ts_truncate_floors .get(&key) - .is_some_and(|floor| lsn <= *floor) + .is_some_and(|floor| lsn < *floor) }); + // Both tests are restart-replay watermarks. A committed-redo apply + // installs its record once, whatever a live flush or truncate stamped + // since the record's LSN was minted. + let already_flushed = self.replay_watermark_skips(already_flushed); if already_flushed { if let Some(prov) = provenance @@ -204,7 +216,12 @@ impl CoreLoop { }; } - let now_ms = self.ingest_now_ms(); + // The instant the write funnel resolved and the WAL record carries, + // so live apply and replay stamp untimed rows alike. + let now_ms = task + .resolved_now_ms() + .and_then(|ms| i64::try_from(ms).ok()) + .unwrap_or_else(|| self.ingest_now_ms()); let ingest_response = match format { "ilp" => self.execute_ilp_ingest(TimeseriesIngestParams { @@ -267,13 +284,13 @@ impl CoreLoop { if let Some(prov) = provenance && ingest_response.status == Status::Ok - && mode == TimeseriesApplyMode::Immediate + && mode != TimeseriesApplyMode::CommitDeferred { self.sync_commit(prov); let applied_seq = self.sync_hwm_value(prov.producer_id, prov.stream_id); return self.sync_ack_response(task, AckStatus::Applied, applied_seq); } - if ingest_response.status == Status::Ok && mode == TimeseriesApplyMode::Immediate { + if ingest_response.status == Status::Ok && mode != TimeseriesApplyMode::CommitDeferred { self.note_collection_write_lsn(task, collection); } ingest_response diff --git a/nodedb/src/data/executor/handlers/timeseries/mod.rs b/nodedb/src/data/executor/handlers/timeseries/mod.rs index 0a97b0020..a109d5e27 100644 --- a/nodedb/src/data/executor/handlers/timeseries/mod.rs +++ b/nodedb/src/data/executor/handlers/timeseries/mod.rs @@ -14,6 +14,7 @@ mod msgpack_decode; mod normalize; pub mod paths; pub mod raw_scan; +mod redo_ingest; mod resolve_ingest; mod rls_gate; mod scan; @@ -22,6 +23,7 @@ mod time_range; pub mod truncate; pub(in crate::data::executor) use ingest_dispatch::{TimeseriesApplyMode, TimeseriesIngestExec}; +pub(in crate::data::executor) use resolve_ingest::StampedIngest; pub(in crate::data::executor) use rls_gate::{admit_ilp_lines, admit_msgpack_rows}; pub(in crate::data::executor) use scan::TimeseriesScanParams; pub(in crate::data::executor) use truncate::{is_truncating_leftover, remove_truncating_leftovers}; diff --git a/nodedb/src/data/executor/handlers/timeseries/redo_ingest.rs b/nodedb/src/data/executor/handlers/timeseries/redo_ingest.rs new file mode 100644 index 000000000..8d7ef7d1a --- /dev/null +++ b/nodedb/src/data/executor/handlers/timeseries/redo_ingest.rs @@ -0,0 +1,64 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The admission test every ILP ingest runs before its rows land, and the +//! preparation a committed-redo install adds to it. +//! +//! An install flushes the rows the memtable already holds when it has no +//! room for the new ones, and only then records its pre-image. The undo +//! therefore restores a memtable that holds nothing the flush moved to disk. + +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::engine::timeseries::ilp; +use crate::types::{DatabaseId, TenantId}; + +use super::admission; + +impl CoreLoop { + /// Whether the memtable of `key` must flush before `lines` land in it: it + /// is at its soft or hard limit, the engine budget is under pressure, or + /// its tag dictionaries have no room for the lines' tags. + pub(super) fn ts_ingest_needs_flush( + &self, + key: &(DatabaseId, TenantId, String), + lines: &[ilp::IlpLine<'_>], + ) -> bool { + let governor_pressure = self + .governor + .try_reserve(key.0, key.1, nodedb_mem::EngineId::Timeseries, 0) + .is_err(); + let soft_limit = self.ts_tuning.memtable_budget_bytes; + let hard_limit = self.ts_tuning.memtable_hard_limit_bytes; + let max_tag_cardinality = self.ts_tuning.max_tag_cardinality; + self.columnar_memtables.get(key).is_some_and(|mt| { + let resident = mt.memory_bytes(); + resident >= soft_limit + || resident >= hard_limit + || governor_pressure + || !admission::has_tag_headroom(mt, lines, max_tag_cardinality) + }) + } + + /// Prepare the install of `lines` into `collection`: flush the memtable + /// when it has no room, then record the pre-image the undo restores. + pub(super) fn prepare_redo_ts_ingest( + &mut self, + database_id: DatabaseId, + tid: TenantId, + collection: &str, + lines: &[ilp::IlpLine<'_>], + now_ms: i64, + ) -> Result<(), ErrorCode> { + let key = (database_id, tid, collection.to_string()); + if self.ts_ingest_needs_flush(&key, lines) { + self.flush_ts_collection(tid, database_id, collection, now_ms) + .map_err(|e| ErrorCode::Internal { + detail: format!("pre-install ts flush of '{collection}' failed: {e}"), + })?; + } + let undo = self.capture_timeseries_ingest_undo(&key); + self.record_redo_undo([UndoEntry::TimeseriesIngest(undo)]); + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/timeseries/resolve_ingest.rs b/nodedb/src/data/executor/handlers/timeseries/resolve_ingest.rs index 60dac3c85..6f56c93b0 100644 --- a/nodedb/src/data/executor/handlers/timeseries/resolve_ingest.rs +++ b/nodedb/src/data/executor/handlers/timeseries/resolve_ingest.rs @@ -15,6 +15,17 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; +/// One ingest payload to normalize and stamp. +pub(in crate::data::executor) struct StampedIngest<'a> { + pub database_id: crate::types::DatabaseId, + pub tid: crate::types::TenantId, + pub collection: &'a str, + pub payload: &'a [u8], + pub format: &'a str, + /// The default timestamp of a row that carries none. + pub now_ms: i64, +} + /// The measurement an ingest writes to, taken from its routing collection. fn measurement_of(collection: &str) -> &str { collection @@ -51,36 +62,18 @@ impl CoreLoop { let time_key = self .declared_ts_time_key(task.request.database_id, tid, collection.as_str()) .map(str::to_string); - let batch = match Self::normalized_ilp_batch( - collection.as_str(), + let now_ms = self.ingest_now_ms(); + let lines = match self.stamped_ingest_lines(StampedIngest { + database_id: task.request.database_id, + tid, + collection: collection.as_str(), payload, format, - time_key.as_deref(), - ) { - Ok(batch) => batch, - Err(error) => return self.response_error(task, error), - }; - - let now_ms = self.ingest_now_ms(); - let lines = match normalize::stamp_timestamps(&batch, now_ms) { + now_ms, + }) { Ok(lines) => lines, - Err(error) => { - return self.response_error( - task, - ErrorCode::RejectedPrevalidation { - reason: format!("timeseries resolve: unparsable line protocol: {error}"), - }, - ); - } + Err(error) => return self.response_error(task, error), }; - if lines.is_empty() { - return self.response_error( - task, - ErrorCode::RejectedPrevalidation { - reason: format!("timeseries resolve: '{collection}' payload holds no rows"), - }, - ); - } // Decide the policy against the stamped lines — the exact images the // proposed ingest will store on every replica. @@ -118,6 +111,37 @@ impl CoreLoop { } } + /// The canonical lines `payload` stores, with every row that carries no + /// timestamp stamped `now_ms`. A transaction's resolve and the governed + /// ingest's resolve pass both stamp here, once, before the write is + /// proposed, so every replica stores identical rows. + pub(in crate::data::executor) fn stamped_ingest_lines( + &self, + ingest: StampedIngest<'_>, + ) -> Result, ErrorCode> { + let StampedIngest { + database_id, + tid, + collection, + payload, + format, + now_ms, + } = ingest; + let time_key = self.declared_ts_time_key(database_id, tid, collection); + let batch = Self::normalized_ilp_batch(collection, payload, format, time_key)?; + let lines = normalize::stamp_timestamps(&batch, now_ms).map_err(|error| { + ErrorCode::RejectedPrevalidation { + reason: format!("timeseries resolve: unparsable line protocol: {error}"), + } + })?; + if lines.is_empty() { + return Err(ErrorCode::RejectedPrevalidation { + reason: format!("timeseries resolve: '{collection}' payload holds no rows"), + }); + } + Ok(lines) + } + /// Rewrite `payload` into line protocol, mirroring the ingest handler's /// format dispatch so the lines returned are what ingest would parse. fn normalized_ilp_batch( diff --git a/nodedb/src/data/executor/handlers/timeseries_wal.rs b/nodedb/src/data/executor/handlers/timeseries_wal.rs index 9f7d5ecea..9d72947cd 100644 --- a/nodedb/src/data/executor/handlers/timeseries_wal.rs +++ b/nodedb/src/data/executor/handlers/timeseries_wal.rs @@ -4,256 +4,17 @@ //! //! On startup, replays `TimeseriesBatch` records into the per-core //! columnar memtable. Only replays records with LSN > `last_flushed_wal_lsn` -//! per partition (not max_ts — safe with out-of-order data). +//! per partition (not max_ts — safe with out-of-order data). A +//! committed-redo apply runs the same arm without that skip +//! (`replay_policy`). -use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::handlers::timeseries::TimeseriesIngestExec; -use crate::data::executor::task::{ExecutionTask, TaskState}; -use crate::engine::timeseries::columnar_memtable::{ - ColumnarMemtable, ColumnarMemtableConfig, ColumnarSchema, -}; use crate::types::DatabaseId; -use crate::types::ReadConsistency; -use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp, TimeseriesOp}; -use nodedb_types::timeseries::MetricSample; -use super::timeseries_wal_decode::{ColumnarReplayArgs, TimeseriesReplayArgs, decode_batch_record}; -impl CoreLoop { - /// Build a synthetic replay `ExecutionTask` embedding `plan`. - /// - /// Shared with `wal_replay_columnar_dml` — every replay handler that - /// re-invokes a live execute_* method needs the same minimal task shape. - pub(in crate::data::executor) fn replay_task( - tenant_id: crate::types::TenantId, - database_id: DatabaseId, - vshard_id: crate::types::VShardId, - plan: PhysicalPlan, - wal_lsn: Option, - ) -> ExecutionTask { - ExecutionTask { - request: Request { - request_id: crate::types::RequestId::new(0), - tenant_id, - database_id, - vshard_id, - plan, - deadline: std::time::Instant::now() - + crate::data::executor::deadline::REPLAY_DEADLINE, - priority: Priority::Normal, - trace_id: crate::types::TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn, - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::AlreadyOrdered, - ), - }, - state: TaskState::Running, - wal_lsn, - resolved_now_ms: None, - } - } - - /// Ensure a timeseries memtable exists for the given collection, creating if needed. - /// - /// Uses the same operator tuning the live ingest path does. A memtable keeps - /// the limits it was built with for its whole life, so seeding replay with - /// hardcoded defaults would leave a restarted node running budgets the - /// operator did not configure until every collection happened to flush. - fn ensure_columnar_memtable( - &mut self, - key: (DatabaseId, crate::types::TenantId, String), - schema: ColumnarSchema, - ) { - let config = ColumnarMemtableConfig::from_tuning(&self.ts_tuning); - self.columnar_memtables - .entry(key) - .or_insert_with(|| ColumnarMemtable::new(schema, config)); - } - - fn replay_timeseries_payload( - &mut self, - tid: crate::types::TenantId, - db_id: DatabaseId, - args: TimeseriesReplayArgs<'_>, - ) -> usize { - let TimeseriesReplayArgs { - collection, - payload, - record_lsn, - provenance, - format, - } = args; - if let Ok(batch) = - zerompk::from_msgpack::(payload) - { - let key = (db_id, tid, collection.to_string()); - self.ensure_columnar_memtable(key.clone(), ColumnarSchema::metric_default()); - - let Some(mt) = self.columnar_memtables.get_mut(&key) else { - return 0; - }; - for (series_id, timestamp_ms, value) in &batch.samples { - mt.ingest_metric( - *series_id, - MetricSample { - timestamp_ms: *timestamp_ms, - value: *value, - }, - ); - } - let sample_count = batch.samples.len(); - // Re-charge the engine memory budget to the memtable's resident - // footprint after replaying these samples. The reservation is - // held until the memtable is drained on flush, so a replay-driven - // flush balances its release instead of over-releasing. - self.recharge_ts_memtable_budget(tid, db_id, collection); - return sample_count; - } - - let format = format.unwrap_or_else(|| { - if std::str::from_utf8(payload).is_ok() { - "ilp" - } else { - "msgpack" - } - }); - let task = Self::replay_task( - tid, - db_id, - crate::types::VShardId::from_collection_in_database(db_id, collection), - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), - payload: payload.to_vec(), - format: format.to_string(), - wal_lsn: Some(record_lsn), - surrogates: Vec::new(), - provenance: provenance.clone(), - rls_write_check: nodedb_types::RlsWriteCheck::already_decided_elsewhere(), - returning: None, - rls_filters: Vec::new(), - }), - Some(crate::types::Lsn::new(record_lsn)), - ); - let response = self.execute_timeseries_ingest(TimeseriesIngestExec { - task: &task, - tid, - collection, - payload, - format, - wal_lsn: Some(record_lsn), - provenance: provenance.as_ref(), - mode: crate::data::executor::handlers::timeseries::TimeseriesApplyMode::Immediate, - // Replay re-applies a record the policy already decided when it was - // written, and the identity that wrote it is not present at boot to - // resolve `$auth.*` against. A refused write never reaches replay: - // its record is cancelled before the refusal is acknowledged. - rls_write_check: &nodedb_types::RlsWriteCheck::already_decided_elsewhere(), - // Replay rebuilds stored state at boot; there is no client waiting - // on a row set, and no identity whose reads would need gating. The - // projection belongs to the originating request, which was answered - // before the process restarted. - returning: None, - rls_filters: &[], - }); - if response.status != crate::bridge::envelope::Status::Ok { - tracing::warn!( - "timeseries WAL replay failed for collection={collection} lsn={record_lsn}: {:?}", - response.error_code - ); - return 0; - } - if format == "ilp-msgpack" { - return zerompk::from_msgpack::>(payload).map_or(0, |rows| rows.len()); - } - match nodedb_types::value_from_msgpack(payload) { - Ok(nodedb_types::Value::Array(rows)) => rows.len(), - Ok(nodedb_types::Value::Object(_)) => 1, - _ => 0, - } - } - - fn replay_columnar_payload( - &mut self, - tid: crate::types::TenantId, - db_id: DatabaseId, - args: ColumnarReplayArgs<'_>, - ) -> usize { - let ColumnarReplayArgs { - collection, - payload, - record_lsn, - provenance, - surrogates, - } = args; - // `execute_columnar_insert` reads only `task.request.{database_id, - // tenant_id, request_id}` — it never inspects the embedded plan. - // Embed empty vecs for the plan-level surrogates/provenance to avoid - // cloning the owned values we need to pass as explicit args below. - let task = Self::replay_task( - tid, - db_id, - crate::types::VShardId::from_collection_in_database(db_id, collection), - PhysicalPlan::Columnar(ColumnarOp::Insert { - collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), - payload: payload.to_vec(), - format: "msgpack".into(), - intent: ColumnarInsertIntent::Insert, - on_conflict_updates: Vec::new(), - surrogates: Vec::new(), - schema_bytes: Vec::new(), - provenance: None, - wal_lsn: Some(record_lsn), - rls_write_check: nodedb_types::RlsWriteCheck::already_decided_elsewhere(), - returning: None, - rls_filters: Vec::new(), - }), - Some(crate::types::Lsn::new(record_lsn)), - ); - // Restore the persisted per-row surrogates so `execute_columnar_insert` - // rebinds the exact same cross-engine identity via - // `insert_with_surrogate`. An empty slice (legacy records / sync path) - // falls back to fresh allocation as before. - let response = self.execute_columnar_insert( - &task, - crate::data::executor::handlers::columnar_write::ColumnarInsertParams { - collection, - payload, - format: "msgpack", - intent: ColumnarInsertIntent::Insert, - on_conflict_updates: &[], - surrogates: &surrogates, - schema_bytes: &[], - provenance: provenance.as_ref(), - rls_write_check: &nodedb_types::RlsWriteCheck::already_decided_elsewhere(), - // WAL replay reconstructs stored state; there is no client - // waiting on a projection, and no identity to gate reads for. - returning: None, - rls_filters: &[], - spatial_undo: None, - }, - ); - if response.status != crate::bridge::envelope::Status::Ok { - tracing::warn!( - "columnar WAL replay failed for collection={collection} lsn={record_lsn}: {:?}", - response.error_code - ); - return 0; - } - match nodedb_types::value_from_msgpack(payload) { - Ok(nodedb_types::Value::Array(rows)) => rows.len(), - Ok(nodedb_types::Value::Object(_)) => 1, - _ => 0, - } - } +use super::timeseries_wal_decode::{ColumnarReplayArgs, TimeseriesReplayArgs}; +use crate::wal::{DecodedBatchRecord, decode_batch_record}; +impl CoreLoop { /// Replay WAL timeseries records to rebuild in-memory memtable state after crash. /// /// Called once during startup, after `open()` but before the event loop. @@ -308,13 +69,10 @@ impl CoreLoop { continue; } - // Predicate DML (`columnar_dml`) rides the same `TimeseriesBatch` - // record type but a disjoint map shape from both `ColumnarWalRecord` - // and the legacy tuples (see `ColumnarDmlWalRecord`'s doc comment), - // so it must be tried BEFORE `decode_batch_record` below — that - // decoder's tuple fallbacks would otherwise mis-classify it as a - // malformed row-payload record and drop it. - if let Some(applied) = self.try_replay_columnar_predicate_dml( + // A transaction's columnar row images (`columnar_image`) are a + // disjoint map shape too, tried first for the same reason as the + // DML shapes below. + if let Some(applied) = self.try_replay_columnar_image( &record.payload, record.header.tenant_id, DatabaseId::new(record.header.database_id), @@ -325,18 +83,44 @@ impl CoreLoop { replayed += applied; continue; } + // A committed redo record carries columnar rows only as images, + // and timeseries rows only as an ingest. Every other shape is + // left unclaimed, so the validate pass refuses the record. + let redo_apply = self.applying_committed_redo(); + + // Predicate DML (`columnar_dml`) rides the same `TimeseriesBatch` + // record type but a disjoint map shape from both `ColumnarWalRecord` + // and the legacy tuples (see `ColumnarDmlWalRecord`'s doc comment), + // so it must be tried BEFORE `decode_batch_record` below — that + // decoder's tuple fallbacks would otherwise mis-classify it as a + // malformed row-payload record and drop it. + if !redo_apply + && let Some(applied) = self.try_replay_columnar_predicate_dml( + &record.payload, + record.header.tenant_id, + DatabaseId::new(record.header.database_id), + record.header.lsn, + tombstones, + &truncate_floors, + ) + { + replayed += applied; + continue; + } // Resolved-row-set DML (`columnar_resolved_dml`) is likewise a // disjoint map shape and must be tried before the generic decode // below for the same reason as the predicate-DML check above. - if let Some(applied) = self.try_replay_columnar_resolved_predicate_dml( - &record.payload, - record.header.tenant_id, - DatabaseId::new(record.header.database_id), - record.header.lsn, - tombstones, - &truncate_floors, - ) { + if !redo_apply + && let Some(applied) = self.try_replay_columnar_resolved_predicate_dml( + &record.payload, + record.header.tenant_id, + DatabaseId::new(record.header.database_id), + record.header.lsn, + tombstones, + &truncate_floors, + ) + { replayed += applied; continue; } @@ -348,23 +132,26 @@ impl CoreLoop { // surrogates. Records iterate in LSN order (guaranteed by the WAL // segment layout), so provenance-aware replay processes seq in // order. - let Ok(( + let Ok(DecodedBatchRecord { kind, - raw_collection, + collection: raw_collection, payload, - record_provenance, - record_format, - record_surrogates, - )) = decode_batch_record(&record.payload) + provenance: record_provenance, + format: record_format, + surrogates: record_surrogates, + conflict_policy, + default_timestamp_ms, + }) = decode_batch_record(&record.payload) else { - crate::data::executor::replay_abort::abort_replay( + self.replay_record_unapplied( "timeseries", "decode_batch", - self.core_id, record.header.lsn, "TimeseriesBatch payload matched none of the columnar / timeseries \ record shapes", ); + skipped += 1; + continue; }; let tenant_id = record.header.tenant_id; @@ -387,7 +174,8 @@ impl CoreLoop { continue; } - // Check if this record was already flushed (LSN-based skip). + // Check if this record was already flushed (LSN-based skip). A + // restart watermark only: see `replay_policy`. if let Some(registry) = self.ts_registries.get(&key) { // Find the max flushed LSN across all partitions. let max_flushed_lsn = registry @@ -395,12 +183,20 @@ impl CoreLoop { .map(|(_, e)| e.meta.last_flushed_wal_lsn) .max() .unwrap_or(0); - if record_lsn <= max_flushed_lsn { + if self.replay_watermark_skips(record_lsn <= max_flushed_lsn) { skipped += 1; continue; } } + if redo_apply && kind.as_deref() == Some("columnar") { + skipped += 1; + continue; + } + if self.claim_for_validation() { + continue; + } + let accepted = match kind.as_deref() { // The columnar floor is consulted HERE and not above the `kind` // match, because it is the columnar engines' floor and this @@ -409,7 +205,11 @@ impl CoreLoop { // does not cover and whose replay it must therefore not gate. // Gating one engine's records on another engine's durability // would drop the writes outright. - Some("columnar") if self.floors.replay_floors.columnar.covers(record_lsn) => { + Some("columnar") + if self.replay_watermark_skips( + self.floors.replay_floors.columnar.covers(record_lsn), + ) => + { // Already folded into the restored generation. Replaying it // would re-insert every row: an upsert masks the duplicate // on a plain collection, but a `bitemporal=true` collection @@ -427,6 +227,7 @@ impl CoreLoop { record_lsn, provenance: record_provenance, surrogates: record_surrogates, + conflict_policy, }, ), Some("timeseries") | None => self.replay_timeseries_payload( @@ -438,14 +239,15 @@ impl CoreLoop { record_lsn, provenance: record_provenance, format: record_format.as_deref(), + default_timestamp_ms, }, ), Some(other) => { - tracing::warn!( - core = self.core_id, - lsn = record_lsn, - kind = other, - "skipping unknown TimeseriesBatch WAL kind" + self.replay_record_rejected( + "timeseries", + record_lsn, + None, + &format!("unknown TimeseriesBatch WAL kind '{other}'"), ); 0 } @@ -488,10 +290,10 @@ impl CoreLoop { #[cfg(test)] mod tests { - use super::super::timeseries_wal_decode::decode_batch_record; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::write_index::CollKey; use crate::types::{DatabaseId, Lsn, TenantId}; + use crate::wal::{DecodedBatchRecord, decode_batch_record}; use nodedb_types::Surrogate; use nodedb_types::columnar::ColumnarWalRecord; use nodedb_types::sync::wire::SyncProvenance; @@ -636,6 +438,7 @@ mod tests { payload: row_payload("name", "alice"), provenance: None, surrogates: Vec::new(), + conflict_policy: Vec::new(), }; let payload = zerompk::to_msgpack_vec(&rec).expect("encode columnar wal record"); WalRecord::new(WalRecordArgs { @@ -651,6 +454,92 @@ mod tests { .expect("wal record") } + /// A replayed `ON CONFLICT DO NOTHING` insert keeps the row its key + /// already holds, as the live insert did. + #[test] + fn a_replayed_do_nothing_insert_keeps_the_existing_row() { + use nodedb_types::Value; + use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; + + let mut h = make_core(); + let schema = ColumnarSchema { + columns: vec![ + ColumnDef::required("id", ColumnType::Int64).with_primary_key(), + ColumnDef::required("v", ColumnType::Int64), + ], + version: 1, + }; + let mut engine = nodedb_columnar::MutationEngine::new("m".to_string(), schema); + engine + .insert(&[Value::Integer(1), Value::Integer(10)]) + .expect("seed row"); + h.core.columnar_engines.insert( + (DatabaseId::new(0), TenantId::new(7), "m".to_string()), + engine, + ); + + let mut row = std::collections::HashMap::new(); + row.insert("id".to_string(), Value::Integer(1)); + row.insert("v".to_string(), Value::Integer(20)); + let policy = crate::wal::ColumnarConflictPolicy { + intent: nodedb_physical::physical_plan::ColumnarInsertIntent::InsertIfAbsent, + on_conflict_updates: Vec::new(), + }; + let rec = ColumnarWalRecord { + kind: "columnar".to_string(), + collection: "m".to_string(), + payload: nodedb_types::value_to_msgpack(&Value::Array(vec![Value::Object(row)])) + .expect("encode row"), + provenance: None, + surrogates: Vec::new(), + conflict_policy: policy.encode().expect("encode policy"), + }; + let record = WalRecord::new(WalRecordArgs { + record_type: RecordType::TimeseriesBatch as u32, + lsn: 40, + tenant_id: 7, + vshard_id: 0, + database_id: 0, + payload: zerompk::to_msgpack_vec(&rec).expect("encode record"), + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record"); + + h.core.replay_timeseries_wal( + std::slice::from_ref(&record), + 1, + &nodedb_wal::TombstoneSet::new(), + ); + + let rows: Vec> = h + .core + .columnar_engines + .get(&(DatabaseId::new(0), TenantId::new(7), "m".to_string())) + .expect("engine") + .scan_memtable_rows() + .collect(); + assert_eq!(rows, vec![vec![Value::Integer(1), Value::Integer(10)]]); + } + + #[test] + fn a_conflict_policy_round_trips_and_a_plain_insert_encodes_empty() { + let plain = crate::wal::ColumnarConflictPolicy::replace(); + assert!(plain.encode().expect("encode").is_empty()); + let upsert = crate::wal::ColumnarConflictPolicy { + intent: nodedb_physical::physical_plan::ColumnarInsertIntent::Put, + on_conflict_updates: vec![( + "v".to_string(), + nodedb_physical::physical_plan::UpdateValue::Literal(vec![0x05]), + )], + }; + let bytes = upsert.encode().expect("encode"); + assert_eq!( + crate::wal::ColumnarConflictPolicy::decode(&bytes).expect("decode"), + upsert + ); + } + /// WAL replay threads the record LSN into `replay_task` so /// `execute_columnar_insert`'s `note_collection_write_lsn(task, ..)` call /// (gated on `task.wal_lsn().is_some()`) fires during WAL replay too, not @@ -864,11 +753,19 @@ mod tests { payload: vec![7, 8, 9], provenance: Some(prov.clone()), surrogates: vec![Surrogate::new(100), Surrogate::new(101)], + conflict_policy: Vec::new(), }; let bytes = zerompk::to_msgpack_vec(&rec).expect("encode map record"); - let (kind, collection, payload, provenance, format, surrogates) = - decode_batch_record(&bytes).expect("decode map record"); + let DecodedBatchRecord { + kind, + collection, + payload, + provenance, + format, + surrogates, + .. + } = decode_batch_record(&bytes).expect("decode map record"); assert_eq!(kind.as_deref(), Some("columnar")); assert_eq!(collection, "events"); assert_eq!(payload, vec![7, 8, 9]); @@ -890,8 +787,15 @@ mod tests { )) .expect("encode legacy columnar tuple"); - let (kind, collection, payload, provenance, format, surrogates) = - decode_batch_record(&bytes).expect("decode legacy tuple"); + let DecodedBatchRecord { + kind, + collection, + payload, + provenance, + format, + surrogates, + .. + } = decode_batch_record(&bytes).expect("decode legacy tuple"); assert_eq!(kind.as_deref(), Some("columnar")); assert_eq!(collection, "events"); assert_eq!(payload, vec![1, 2, 3]); @@ -914,8 +818,14 @@ mod tests { )) .expect("encode timeseries tuple"); - let (kind, collection, payload, _provenance, format, surrogates) = - decode_batch_record(&bytes).expect("decode timeseries tuple"); + let DecodedBatchRecord { + kind, + collection, + payload, + format, + surrogates, + .. + } = decode_batch_record(&bytes).expect("decode timeseries tuple"); assert_eq!(kind.as_deref(), Some("timeseries")); assert_eq!(collection, "metrics"); assert_eq!(payload, vec![4, 5, 6]); @@ -927,8 +837,14 @@ mod tests { fn legacy_untagged_two_tuple_decodes() { let bytes = zerompk::to_msgpack_vec(&("metrics".to_string(), vec![1u8, 2])) .expect("encode 2-tuple"); - let (kind, collection, payload, _, format, surrogates) = - decode_batch_record(&bytes).expect("decode 2-tuple"); + let DecodedBatchRecord { + kind, + collection, + payload, + format, + surrogates, + .. + } = decode_batch_record(&bytes).expect("decode 2-tuple"); assert_eq!(kind, None); assert_eq!(collection, "metrics"); assert_eq!(payload, vec![1, 2]); @@ -948,11 +864,43 @@ mod tests { "ilp-msgpack".to_string(), )) .expect("encode format-preserving tuple"); - let (kind, collection, _payload, _provenance, format, surrogates) = - decode_batch_record(&bytes).expect("decode format-preserving tuple"); + let DecodedBatchRecord { + kind, + collection, + format, + surrogates, + default_timestamp_ms, + .. + } = decode_batch_record(&bytes).expect("decode format-preserving tuple"); assert_eq!(kind.as_deref(), Some("timeseries")); assert_eq!(collection, "cpu"); assert_eq!(format.as_deref(), Some("ilp-msgpack")); assert!(surrogates.is_empty()); + assert_eq!(default_timestamp_ms, None); + } + + #[test] + fn an_autocommit_ingest_record_carries_its_default_timestamp() { + let bytes = crate::control::server::wal_dispatch::encode_timeseries_ingest_payload( + crate::control::server::wal_dispatch::TimeseriesIngestRecord { + collection: "cpu", + payload: b"cpu value=1", + provenance: None, + format: "ilp", + default_timestamp_ms: 1_700_000_000_123, + }, + ) + .expect("encode ingest record"); + let DecodedBatchRecord { + kind, + collection, + format, + default_timestamp_ms, + .. + } = decode_batch_record(&bytes).expect("decode ingest record"); + assert_eq!(kind.as_deref(), Some("timeseries")); + assert_eq!(collection, "cpu"); + assert_eq!(format.as_deref(), Some("ilp")); + assert_eq!(default_timestamp_ms, Some(1_700_000_000_123)); } } diff --git a/nodedb/src/data/executor/handlers/timeseries_wal_decode.rs b/nodedb/src/data/executor/handlers/timeseries_wal_decode.rs index 7289c8156..c77fc7554 100644 --- a/nodedb/src/data/executor/handlers/timeseries_wal_decode.rs +++ b/nodedb/src/data/executor/handlers/timeseries_wal_decode.rs @@ -1,83 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Decoding compatibility for `TimeseriesBatch` WAL records. - -/// Decoded fields of a `TimeseriesBatch` WAL record. -/// -/// `kind` is `Some("columnar")` / `Some("timeseries")` for tagged records and -/// `None` for the legacy 2-tuple shape. `format` is present only in the new -/// five-element timeseries tuple; absent records use the legacy UTF-8 heuristic. -/// `surrogates` is only non-empty for map-shaped columnar records. -pub(super) type DecodedBatchRecord = ( - Option, - String, - Vec, - Option, - Option, - Vec, -); - -/// Decode a `TimeseriesBatch` WAL payload into its logical fields. -/// -/// Tries the newest map and format-preserving tuple forms before all legacy -/// tuple forms. The map form is unambiguous from the tuple forms. -pub(super) fn decode_batch_record(payload: &[u8]) -> Result { - if let Ok(rec) = zerompk::from_msgpack::(payload) { - return Ok(( - Some(rec.kind), - rec.collection, - rec.payload, - rec.provenance, - None, - rec.surrogates, - )); - } - zerompk::from_msgpack::<( - String, - String, - Vec, - Option, - String, - )>(payload) - .map(|(kind, collection, payload, provenance, format)| { - ( - Some(kind), - collection, - payload, - provenance, - Some(format), - Vec::new(), - ) - }) - .or_else(|_| { - zerompk::from_msgpack::<( - String, - String, - Vec, - Option, - )>(payload) - .map(|(kind, collection, payload, provenance)| { - ( - Some(kind), - collection, - payload, - provenance, - None, - Vec::new(), - ) - }) - }) - .or_else(|_| { - zerompk::from_msgpack::<(String, String, Vec)>(payload).map( - |(kind, collection, payload)| (Some(kind), collection, payload, None, None, Vec::new()), - ) - }) - .or_else(|_| { - zerompk::from_msgpack::<(String, Vec)>(payload) - .map(|(collection, payload)| (None, collection, payload, None, None, Vec::new())) - }) - .map_err(|_| ()) -} +//! Record-level arguments for replaying `TimeseriesBatch` WAL records. /// Record-level fields for replaying a single columnar WAL batch. pub(super) struct ColumnarReplayArgs<'a> { @@ -88,6 +11,8 @@ pub(super) struct ColumnarReplayArgs<'a> { /// Per-row surrogates index-aligned with `payload` rows. An empty `Vec` /// falls back to fresh surrogate allocation for legacy records. pub surrogates: Vec, + /// The insert's encoded conflict policy (`crate::wal::ColumnarConflictPolicy`). + pub conflict_policy: Vec, } /// Record-level fields for replaying one timeseries WAL batch. @@ -97,4 +22,6 @@ pub(super) struct TimeseriesReplayArgs<'a> { pub record_lsn: u64, pub provenance: Option, pub format: Option<&'a str>, + /// The timestamp every untimed row takes, when the record carries one. + pub default_timestamp_ms: Option, } diff --git a/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs b/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs new file mode 100644 index 000000000..0756ea8b8 --- /dev/null +++ b/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs @@ -0,0 +1,258 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply one decoded `TimeseriesBatch` record: a timeseries ingest into the +//! collection's memtable, or a plain columnar insert. + +use crate::bridge::envelope::PhysicalPlan; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::timeseries::{TimeseriesApplyMode, TimeseriesIngestExec}; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::engine::timeseries::columnar_memtable::{ + ColumnarMemtable, ColumnarMemtableConfig, ColumnarSchema, +}; +use crate::types::DatabaseId; +use nodedb_physical::physical_plan::{ColumnarOp, TimeseriesOp}; +use nodedb_types::timeseries::MetricSample; + +use super::timeseries_wal_decode::{ColumnarReplayArgs, TimeseriesReplayArgs}; + +impl CoreLoop { + /// Ensure a timeseries memtable exists for the given collection, creating if needed. + /// + /// Uses the same operator tuning the live ingest path does. A memtable keeps + /// the limits it was built with for its whole life, so seeding replay with + /// hardcoded defaults would leave a restarted node running budgets the + /// operator did not configure until every collection happened to flush. + fn ensure_columnar_memtable( + &mut self, + key: (DatabaseId, crate::types::TenantId, String), + schema: ColumnarSchema, + ) { + let config = ColumnarMemtableConfig::from_tuning(&self.ts_tuning); + self.columnar_memtables + .entry(key) + .or_insert_with(|| ColumnarMemtable::new(schema, config)); + } + + pub(super) fn replay_timeseries_payload( + &mut self, + tid: crate::types::TenantId, + db_id: DatabaseId, + args: TimeseriesReplayArgs<'_>, + ) -> usize { + let TimeseriesReplayArgs { + collection, + payload, + record_lsn, + provenance, + format, + default_timestamp_ms, + } = args; + if let Ok(batch) = + zerompk::from_msgpack::(payload) + { + let key = (db_id, tid, collection.to_string()); + if self.recording_redo_undo() { + let undo = UndoEntry::TimeseriesIngest(self.capture_timeseries_ingest_undo(&key)); + self.record_redo_undo([undo]); + } + self.ensure_columnar_memtable(key.clone(), ColumnarSchema::metric_default()); + + let Some(mt) = self.columnar_memtables.get_mut(&key) else { + return 0; + }; + for (series_id, timestamp_ms, value) in &batch.samples { + mt.ingest_metric( + *series_id, + MetricSample { + timestamp_ms: *timestamp_ms, + value: *value, + }, + ); + } + let sample_count = batch.samples.len(); + if self.recording_redo_undo() { + // The install settles the budget once the whole record landed. + self.note_redo_timeseries_written(key, record_lsn); + } else { + // Re-charge the engine memory budget to the memtable's + // resident footprint after replaying these samples. The + // reservation is held until the memtable is drained on flush, + // so a replay-driven flush balances its release instead of + // over-releasing. + self.recharge_ts_memtable_budget(tid, db_id, collection); + } + return sample_count; + } + + let format = format.unwrap_or_else(|| { + if std::str::from_utf8(payload).is_ok() { + "ilp" + } else { + "msgpack" + } + }); + let mut task = Self::replay_task( + tid, + db_id, + crate::types::VShardId::from_collection_in_database(db_id, collection), + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), + payload: payload.to_vec(), + format: format.to_string(), + wal_lsn: Some(record_lsn), + surrogates: Vec::new(), + provenance: provenance.clone(), + rls_write_check: nodedb_types::RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: Vec::new(), + }), + Some(crate::types::Lsn::new(record_lsn)), + ); + // Untimed rows take the instant the record carries, the one the live + // apply stored them with. + task.resolved_now_ms = default_timestamp_ms.and_then(|ms| u64::try_from(ms).ok()); + let installing = self.recording_redo_undo(); + if installing { + let hwm = self.capture_sync_hwm_undo(provenance.as_ref()); + self.record_redo_undo(hwm); + } + let mode = if installing { + TimeseriesApplyMode::RedoInstall + } else { + TimeseriesApplyMode::Immediate + }; + let response = self.execute_timeseries_ingest(TimeseriesIngestExec { + task: &task, + tid, + collection, + payload, + format, + wal_lsn: Some(record_lsn), + provenance: provenance.as_ref(), + mode, + // Replay re-applies a record the policy already decided when it was + // written, and the identity that wrote it is not present at boot to + // resolve `$auth.*` against. A refused write never reaches replay: + // its record is cancelled before the refusal is acknowledged. + rls_write_check: &nodedb_types::RlsWriteCheck::already_decided_elsewhere(), + // Replay rebuilds stored state at boot; there is no client waiting + // on a row set, and no identity whose reads would need gating. The + // projection belongs to the originating request, which was answered + // before the process restarted. + returning: None, + rls_filters: &[], + }); + if response.status != crate::bridge::envelope::Status::Ok { + self.replay_record_rejected( + "timeseries", + record_lsn, + response.error_code, + &format!("timeseries ingest into '{collection}' failed"), + ); + return 0; + } + if installing { + self.note_redo_timeseries_written((db_id, tid, collection.to_string()), record_lsn); + } + if format == "ilp-msgpack" { + return zerompk::from_msgpack::>(payload).map_or(0, |rows| rows.len()); + } + match nodedb_types::value_from_msgpack(payload) { + Ok(nodedb_types::Value::Array(rows)) => rows.len(), + Ok(nodedb_types::Value::Object(_)) => 1, + _ => 0, + } + } + + pub(super) fn replay_columnar_payload( + &mut self, + tid: crate::types::TenantId, + db_id: DatabaseId, + args: ColumnarReplayArgs<'_>, + ) -> usize { + let ColumnarReplayArgs { + collection, + payload, + record_lsn, + provenance, + surrogates, + conflict_policy, + } = args; + // Each row whose key already exists is skipped, merged or replaced the + // way the live insert decided it. + let conflict_policy = match crate::wal::ColumnarConflictPolicy::decode(&conflict_policy) { + Ok(policy) => policy, + Err(error) => { + self.replay_record_unapplied( + "columnar", + "conflict_policy", + record_lsn, + &format!("columnar insert into '{collection}': {error}"), + ); + return 0; + } + }; + // `execute_columnar_insert` reads only `task.request.{database_id, + // tenant_id, request_id}` — it never inspects the embedded plan. + // Embed empty vecs for the plan-level surrogates/provenance to avoid + // cloning the owned values we need to pass as explicit args below. + let task = Self::replay_task( + tid, + db_id, + crate::types::VShardId::from_collection_in_database(db_id, collection), + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), + payload: payload.to_vec(), + format: "msgpack".into(), + intent: conflict_policy.intent, + on_conflict_updates: conflict_policy.on_conflict_updates.clone(), + surrogates: Vec::new(), + schema_bytes: Vec::new(), + provenance: None, + wal_lsn: Some(record_lsn), + rls_write_check: nodedb_types::RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: Vec::new(), + }), + Some(crate::types::Lsn::new(record_lsn)), + ); + // Restore the persisted per-row surrogates so `execute_columnar_insert` + // rebinds the exact same cross-engine identity via + // `insert_with_surrogate`. An empty slice (legacy records / sync path) + // falls back to fresh allocation as before. + let response = self.execute_columnar_insert( + &task, + crate::data::executor::handlers::columnar_write::ColumnarInsertParams { + collection, + payload, + format: "msgpack", + intent: conflict_policy.intent, + on_conflict_updates: &conflict_policy.on_conflict_updates, + surrogates: &surrogates, + schema_bytes: &[], + provenance: provenance.as_ref(), + rls_write_check: &nodedb_types::RlsWriteCheck::already_decided_elsewhere(), + // WAL replay reconstructs stored state; there is no client + // waiting on a projection, and no identity to gate reads for. + returning: None, + rls_filters: &[], + spatial_undo: None, + }, + ); + if response.status != crate::bridge::envelope::Status::Ok { + self.replay_record_rejected( + "columnar", + record_lsn, + response.error_code, + &format!("columnar insert into '{collection}' failed"), + ); + return 0; + } + match nodedb_types::value_from_msgpack(payload) { + Ok(nodedb_types::Value::Array(rows)) => rows.len(), + Ok(nodedb_types::Value::Object(_)) => 1, + _ => 0, + } + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs index 9afa73572..14d2dc94e 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch.rs @@ -61,9 +61,10 @@ impl CoreLoop { // recorded into per-core apply scratch, so `apply_point_put` installs // each bitemporal document put on the versioned store at the SAME stamp // the redo carries. Calvin threads its stamps in before the call - // (`txn_id = None` here); the session single-shard commit passes its - // `txn_id`. The scratch is consulted ONLY by the forward apply below, so - // it is cleared the moment `run_sub_plans` returns (any path). + // (`txn_id = None` here); a caller applying a session transaction's + // plans passes its `txn_id`. The scratch is consulted ONLY by the + // forward apply below, so it is cleared the moment `run_sub_plans` + // returns (any path). if let Some(txn_id) = txn_id { self.load_bitemporal_stamps_for_txn(txn_id); self.active_graph_system_from = self diff --git a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs index 22cae2335..aa9750140 100644 --- a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs +++ b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs @@ -68,12 +68,23 @@ fn entry_index_tuples(entry: &UndoEntry) -> Option<(String, Vec<(String, String) | UndoEntry::KvBatchPut { .. } | UndoEntry::KvTransfer { .. } | UndoEntry::KvTransferItem { .. } + | UndoEntry::KvTruncate { .. } | UndoEntry::KvTtl { .. } | UndoEntry::SortedIndexDdl { .. } | UndoEntry::MarkNodeDeleted { .. } + | UndoEntry::NodeLabels { .. } + | UndoEntry::SpatialRow(_) + | UndoEntry::VectorWrite(_) + | UndoEntry::CrdtCollection(_) + | UndoEntry::ArrayTiles { .. } + | UndoEntry::SparseDoc { .. } + | UndoEntry::VectorTruncate(_) + | UndoEntry::FtsDocument(_) + | UndoEntry::SyncHwm { .. } | UndoEntry::ColumnarInsert { .. } | UndoEntry::ColumnarUpdate { .. } | UndoEntry::ColumnarDelete { .. } + | UndoEntry::ColumnarEngineCreated { .. } | UndoEntry::TimeseriesIngest(_) | UndoEntry::ColumnarTruncate(_) | UndoEntry::TimeseriesTruncate(_) diff --git a/nodedb/src/data/executor/handlers/transaction/mod.rs b/nodedb/src/data/executor/handlers/transaction/mod.rs index 6666aa559..b95f108f1 100644 --- a/nodedb/src/data/executor/handlers/transaction/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/mod.rs @@ -6,7 +6,8 @@ pub(in crate::data::executor) mod index_write_values; pub mod overlay; mod overlay_gauge; pub(in crate::data::executor) mod overlay_reap; -mod resolve; +pub(in crate::data::executor) mod redo_apply; +pub(in crate::data::executor) mod resolve; pub(in crate::data::executor) mod stage_write; mod sub_plan; mod sub_plan_columnar; @@ -18,6 +19,6 @@ mod sub_plan_kv_ttl_sorted; mod sub_plan_kv_writes; mod sub_plan_write; mod sub_request; -pub(in crate::data::executor::handlers) mod undo; +pub(in crate::data::executor) mod undo; mod write_version; mod write_version_kv; diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs b/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs index a68232adb..9a013f828 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs @@ -54,6 +54,15 @@ pub struct CollectionOverlay { /// by the commit-time base install so redo and install share one stamp. /// See [`BitemporalStamp`]. Never consulted by non-bitemporal collections. pub(super) bitemporal_by_surrogate: HashMap, + /// Primary key (MessagePack) of the base row a staged columnar write + /// displaced, per surrogate. Recorded when a statement first stages a + /// base row, read by COMMIT resolve so the redo names the row it removes. + /// Never consulted by non-columnar collections. + pub(super) base_pk_by_surrogate: HashMap>, + /// The instant a staged timeseries ingest read as its default row + /// timestamp, keyed by the batch's first surrogate. COMMIT resolve stamps + /// the batch's untimed rows with it. Never consulted by other engines. + pub(super) ingest_now_by_surrogate: HashMap, } impl CollectionOverlay { @@ -63,6 +72,8 @@ impl CollectionOverlay { && self.doc_id_to_surrogate.is_empty() && self.ttl_by_surrogate.is_empty() && self.bitemporal_by_surrogate.is_empty() + && self.base_pk_by_surrogate.is_empty() + && self.ingest_now_by_surrogate.is_empty() } } diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/staged_sidecar.rs b/nodedb/src/data/executor/handlers/transaction/overlay/staged_sidecar.rs index 5524407fb..a23972844 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/staged_sidecar.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/staged_sidecar.rs @@ -1,8 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Per-row sidecars of the staging overlay: the KV TTL delta and the -//! bitemporal stamp. Both live beside [`Staged`](super::Staged) rather than -//! inside it because only one engine reads each. +//! Per-row sidecars of the staging overlay: the KV TTL delta, the +//! bitemporal stamp, the displaced columnar base key and a timeseries +//! batch's default timestamp. Each lives beside +//! [`Staged`](super::Staged) rather than inside it because only one engine +//! reads it. use nodedb_types::RowIdentity; @@ -119,6 +121,71 @@ impl TxnOverlay { .copied() } + /// Record the primary key of the base row `surrogate` names, the first + /// time a statement stages that base row. A base row's key does not change + /// inside the transaction, so a later record for the same surrogate is + /// dropped and no undo journalling is needed: after a savepoint rollback + /// the recorded key still names the same base row. + pub fn note_base_pk( + &mut self, + coll_key: &(DatabaseId, TenantId, String), + surrogate: u32, + pk_msgpack: Vec, + ) { + self.collections + .entry(coll_key.clone()) + .or_default() + .base_pk_by_surrogate + .entry(surrogate) + .or_insert(pk_msgpack); + } + + /// The primary key (MessagePack) of the base row a staged columnar write + /// displaced for `surrogate`. `None` when the transaction staged no base + /// row under that surrogate. + pub fn base_pk( + &self, + coll_key: &(DatabaseId, TenantId, String), + surrogate: u32, + ) -> Option<&[u8]> { + self.collections + .get(coll_key)? + .base_pk_by_surrogate + .get(&surrogate) + .map(Vec::as_slice) + } + + /// Record the instant a staged timeseries ingest read as its default row + /// timestamp, under the batch's first surrogate. Surrogates are fresh per + /// staged row, so no later statement records the same key and no undo + /// journalling is needed. + pub fn note_ingest_now( + &mut self, + coll_key: &(DatabaseId, TenantId, String), + first_surrogate: u32, + now_ms: i64, + ) { + self.collections + .entry(coll_key.clone()) + .or_default() + .ingest_now_by_surrogate + .insert(first_surrogate, now_ms); + } + + /// The default row timestamp the staged ingest whose first row is + /// `first_surrogate` read at its statement. + pub fn ingest_now( + &self, + coll_key: &(DatabaseId, TenantId, String), + first_surrogate: u32, + ) -> Option { + self.collections + .get(coll_key)? + .ingest_now_by_surrogate + .get(&first_surrogate) + .copied() + } + /// Iterate every `(surrogate, BitemporalStamp)` staged across all /// collections in this overlay. Surrogates are globally unique, so the /// commit-time install flattens these into one per-core scratch map. diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs new file mode 100644 index 000000000..918fe71b1 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs @@ -0,0 +1,379 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Keep every published engine watermark true after a committed record +//! applies below it. +//! +//! Restart replay skips a record at or below an engine's published +//! watermark: the vector checkpoint's per-collection LSN, the KV and +//! columnar checkpoint floors, an array manifest's durable LSN. The watermark +//! claims that the published artifact holds every record at or below it. +//! Records apply in the order Raft +//! commits them, not in LSN order, so a record can apply after an artifact +//! was published at a higher LSN. That record lives only in memory, and the +//! claim is false for it. Publishing the artifact again, now holding the +//! record, makes the claim true. The install settles a timeseries partition +//! stamp the same way (`settle`). + +use nodedb_types::columnar::{COLUMNAR_IMAGE_KIND, ColumnarImageWalRecord}; +use nodedb_wal::record::RecordType; + +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::types::Lsn; +use crate::wal::{RedoRecord, RedoSubRecord}; + +use super::sub_ops::kv_ops; + +/// The engine-wide checkpoints a redo record writes into. +pub(super) struct WrittenEngines { + /// A vector index the vector checkpoint publishes. + pub vectors: bool, + /// A KV collection the KV checkpoint publishes. + pub kv: bool, + /// A columnar collection the columnar checkpoint publishes. + pub columnar: bool, +} + +impl WrittenEngines { + pub(super) fn of(redo: &RedoRecord) -> Self { + Self { + vectors: redo.ops.iter().any(writes_vector_index), + kv: !kv_ops(&redo.ops).is_empty() || redo.ops.iter().any(is_kv_truncate), + columnar: redo.ops.iter().any(writes_columnar), + } + } +} + +fn writes_vector_index(op: &RedoSubRecord) -> bool { + matches!( + RecordType::from_raw(op.record_type), + Some( + RecordType::VectorPut + | RecordType::VectorDelete + | RecordType::VectorDirectUpsert + | RecordType::VectorDirectUpdate + | RecordType::VectorDirectDelete + | RecordType::VectorDirectTruncate + | RecordType::VectorResolvedDirectWrite + | RecordType::MultiVectorPut + | RecordType::MultiVectorDelete + ) + ) +} + +fn is_kv_truncate(op: &RedoSubRecord) -> bool { + RecordType::from_raw(op.record_type) == Some(RecordType::Delete) + && zerompk::from_msgpack::<(String, String)>(&op.payload) + .is_ok_and(|(disc, _)| disc == "kv_truncate") +} + +fn writes_columnar(op: &RedoSubRecord) -> bool { + match RecordType::from_raw(op.record_type) { + Some(RecordType::ColumnarTruncate) => true, + Some(RecordType::TimeseriesBatch) => { + zerompk::from_msgpack::(&op.payload) + .is_ok_and(|record| record.kind == COLUMNAR_IMAGE_KIND) + } + _ => false, + } +} + +impl CoreLoop { + /// Record that the open redo-apply scope wrote cells to `array_id`. + pub(in crate::data::executor) fn note_redo_array_written( + &mut self, + array_id: &nodedb_array::types::ArrayId, + ) { + if let Some(scope) = self.redo_apply.scope.as_mut() + && !scope.arrays_written.contains(array_id) + { + scope.arrays_written.push(array_id.clone()); + } + } + + /// Publish again every artifact whose watermark covers `lsn` but not the + /// record just applied at it. + pub(super) fn cover_applied_record( + &mut self, + lsn: Lsn, + engines: &WrittenEngines, + arrays: &[nodedb_array::types::ArrayId], + ) -> Result<(), ErrorCode> { + if engines.vectors && lsn <= self.floors.vector_published_lsn { + let published = self.floors.vector_published_lsn; + self.checkpoint_vector_indexes() + .map_err(|error| republish_error("vector checkpoint", lsn, published, &error))?; + } + if engines.kv && lsn <= self.floors.kv_published_lsn { + let published = self.floors.kv_published_lsn; + self.checkpoint_kv_engines() + .map_err(|error| republish_error("KV checkpoint", lsn, published, &error))?; + } + if engines.columnar && lsn <= self.floors.columnar_published_lsn { + let published = self.floors.columnar_published_lsn; + self.checkpoint_columnar_engines() + .map_err(|error| republish_error("columnar checkpoint", lsn, published, &error))?; + } + for array_id in arrays { + let durable = self.array_durable_lsn(array_id); + if lsn.as_u64() > durable { + continue; + } + self.array_engine + .flush(array_id, durable) + .map_err(|error| ErrorCode::Internal { + detail: format!( + "a record applied at lsn {} below array '{}' durable lsn {durable} \ + could not be flushed: {error}", + lsn.as_u64(), + array_id.name + ), + })?; + } + Ok(()) + } +} + +fn republish_error(artifact: &str, lsn: Lsn, published: Lsn, error: &crate::Error) -> ErrorCode { + ErrorCode::Internal { + detail: format!( + "a record applied at lsn {} below the {artifact} at lsn {} could not be \ + published again: {error}", + lsn.as_u64(), + published.as_u64() + ), + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::sync::wire::SyncProvenance; + use nodedb_types::{DatabaseId, Surrogate}; + + use super::*; + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; + use crate::engine::vector::collection::VectorCollection; + use crate::engine::vector::hnsw::HnswParams; + use crate::wal::RedoSubRecord; + + const TID: u64 = 1; + + fn vector_put(collection: &str, surrogate: u32) -> RedoSubRecord { + RedoSubRecord { + record_type: RecordType::VectorPut as u32, + payload: zerompk::to_msgpack_vec(&( + collection, + vec![0.5f32, 0.5], + 2usize, + "", + None::, + surrogate, + None::, + )) + .expect("encode vector put"), + } + } + + #[test] + fn a_vector_record_applied_below_a_published_checkpoint_is_published_again() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let key = CoreLoop::vector_index_key(DatabaseId::DEFAULT.as_u64(), TID, "docs", ""); + let mut collection = VectorCollection::new(2, HnswParams::default()); + collection.insert_with_surrogate(vec![0.1, 0.9], Surrogate::new(1)); + collection.note_checkpoint_lsn(100); + core.vector_collections.insert(key.clone(), collection); + core.advance_watermark(Lsn::new(100)); + core.checkpoint_vector_indexes() + .expect("publish at lsn 100"); + + // Committed at lsn 50, applied after the generation at lsn 100. + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(50)); + let redo = RedoRecord { + version: 1, + ops: vec![vector_put("docs", 2)], + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let response = core.execute_apply_transaction_redo( + &task, + TID, + CommittedRedo { + redo: &redo, + collections: &["docs".to_string()], + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "apply: {response:?}"); + drop(core); + + // Restart replay skips lsn 50 against the restored collection, so the + // published generation is the only copy of the second vector. + let dir_path = dir.path().to_path_buf(); + let (mut restored, _tx2, _rx2) = make_core_with_dir(&dir_path); + restored.load_vector_checkpoints().expect("load"); + assert_eq!( + restored.vector_collections.get(&key).map(|c| c.len()), + Some(2), + "the vector applied below the checkpoint is in the newest generation" + ); + } + + fn kv_put(collection: &str, key: &[u8], value: &[u8], surrogate: u32) -> RedoSubRecord { + RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(&( + "kv_put", + collection, + key.to_vec(), + value.to_vec(), + 0u64, + None::, + surrogate, + )) + .expect("encode kv put"), + } + } + + #[test] + fn a_kv_record_applied_below_a_published_checkpoint_is_published_again() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let _prior = core.kv_engine.put(crate::engine::kv::KvPutParams { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: TID, + collection: "cache", + key: b"a", + value: b"1", + ttl_ms: 0, + now_ms: 0, + surrogate: Surrogate::new(1), + }); + core.advance_watermark(Lsn::new(100)); + core.checkpoint_kv_engines().expect("publish at lsn 100"); + + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(50)); + let redo = RedoRecord { + version: 1, + ops: vec![kv_put("cache", b"b", b"2", 2)], + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let response = core.execute_apply_transaction_redo( + &task, + TID, + CommittedRedo { + redo: &redo, + collections: &["cache".to_string()], + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "apply: {response:?}"); + drop(core); + + let dir_path = dir.path().to_path_buf(); + let (mut restored, _tx2, _rx2) = make_core_with_dir(&dir_path); + restored.load_kv_checkpoints().expect("load"); + let now = crate::engine::kv::current_ms(); + assert_eq!( + restored + .kv_engine + .get(DatabaseId::DEFAULT.as_u64(), TID, "cache", b"b", now), + Some(b"2".to_vec()), + "the KV row applied below the checkpoint is in the newest generation" + ); + } + + #[test] + fn a_columnar_record_applied_below_a_published_checkpoint_is_published_again() { + use nodedb_types::Value; + use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarImageWalRow, ColumnarSchema}; + + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let key = ( + DatabaseId::DEFAULT, + crate::types::TenantId::new(TID), + "m".to_string(), + ); + let schema = ColumnarSchema { + columns: vec![ + ColumnDef::required("id", ColumnType::Int64).with_primary_key(), + ColumnDef::required("v", ColumnType::Int64), + ], + version: 1, + }; + let mut engine = nodedb_columnar::MutationEngine::new("m".to_string(), schema); + engine + .insert_with_surrogate(&[Value::Integer(1), Value::Integer(10)], Surrogate::new(1)) + .expect("seed row"); + core.columnar_engines.insert(key.clone(), engine); + core.advance_watermark(Lsn::new(100)); + core.checkpoint_columnar_engines() + .expect("publish at lsn 100"); + + let mut image = std::collections::HashMap::new(); + image.insert("id".to_string(), Value::Integer(2)); + image.insert("v".to_string(), Value::Integer(20)); + let record = ColumnarImageWalRecord { + kind: COLUMNAR_IMAGE_KIND.to_string(), + collection: "m".to_string(), + schema_bytes: Vec::new(), + rows: vec![ColumnarImageWalRow { + surrogate: 2, + prior_pk_msgpack: Vec::new(), + image_msgpack: nodedb_types::value_to_msgpack(&Value::Object(image)) + .expect("encode image"), + }], + }; + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(50)); + let redo = RedoRecord { + version: 1, + ops: vec![RedoSubRecord { + record_type: RecordType::TimeseriesBatch as u32, + payload: zerompk::to_msgpack_vec(&record).expect("encode image record"), + }], + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let response = core.execute_apply_transaction_redo( + &task, + TID, + CommittedRedo { + redo: &redo, + collections: &["m".to_string()], + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "apply: {response:?}"); + drop(core); + + let dir_path = dir.path().to_path_buf(); + let (mut restored, _tx2, _rx2) = make_core_with_dir(&dir_path); + restored.load_columnar_checkpoints().expect("load"); + let mut ids: Vec = restored + .columnar_engines + .get(&key) + .expect("engine restored") + .scan_memtable_rows() + .map(|row| row[0].clone()) + .collect(); + ids.sort_by_key(|value| match value { + Value::Integer(i) => *i, + _ => i64::MAX, + }); + assert_eq!( + ids, + vec![Value::Integer(1), Value::Integer(2)], + "the columnar row applied below the checkpoint is in the newest generation" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs new file mode 100644 index 000000000..06a576a06 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs @@ -0,0 +1,341 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Document writes of a committed redo record while its apply scope is open. +//! +//! The document redo arm hands each decoded row here instead of its restart +//! path. A replica applying a committed record re-executes the write the way +//! the transaction batch did: it links the hash chain on an insert and folds +//! the row into its materialized-sum targets inside the row's own write +//! transaction. Restart replay does neither, because the chained row and the +//! target rows are already durable by then. +//! +//! Constraint checks ran before the first write (see `validate`), so the +//! writes run with `enforce = false`. In the install pass every written row +//! records the undo entries that reverse it and the target rows it folded +//! into. + +use nodedb_types::Surrogate; +use redb::WriteTransaction; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::enforcement::chain_guard::{ChainGuard, abort_after_apply}; +use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, WriteImages}; +use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; +use crate::data::executor::handlers::point::apply_put::PointPutParams; +use crate::data::executor::handlers::transaction::undo::document_outcome::{ + DocumentRow, push_delete_undo, push_put_undo, push_target_undo, +}; +use crate::engine::document::store::{RowIdentity, StorageKey}; +use crate::event::WriteOp; +use crate::types::Lsn; + +use super::state::{AppliedDocWrite, target_doc_write}; + +/// One document row a committed redo record writes. +pub(in crate::data::executor) struct CommittedDocWrite<'a> { + pub database_id: u64, + pub tenant_id: u64, + pub collection: &'a str, + /// The row's client identity text. + pub document_id: &'a str, + pub surrogate: u32, + pub record_lsn: u64, +} + +impl CoreLoop { + /// Write one document put of a committed record. Returns whether the row + /// was written; an error is kept on the open scope for the apply to report. + pub(in crate::data::executor) fn apply_committed_document_put( + &mut self, + row: CommittedDocWrite<'_>, + value: &[u8], + ) -> bool { + let result = self.committed_document_put(&row, value); + self.settle_committed_write(result.map(|()| true)) + } + + /// Remove one document row of a committed record. Returns whether a row + /// was removed. + pub(in crate::data::executor) fn apply_committed_document_delete( + &mut self, + row: CommittedDocWrite<'_>, + ) -> bool { + let result = self.committed_document_delete(&row); + self.settle_committed_write(result) + } + + fn settle_committed_write(&mut self, result: crate::Result) -> bool { + match result { + Ok(applied) => applied, + Err(error) => { + if let Some(scope) = self.redo_apply.scope.as_mut() { + scope.record_error(error); + } + false + } + } + } + + fn committed_document_put( + &mut self, + row: &CommittedDocWrite<'_>, + value: &[u8], + ) -> crate::Result<()> { + let (resolved, deferred) = self.committed_sum_targets(row.collection); + let surrogate = Surrogate::new(row.surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let wal_lsn = (row.record_lsn != 0).then(|| Lsn::new(row.record_lsn)); + let hook_ctx = HookCtx { + database_id: row.database_id, + tid: row.tenant_id, + collection: row.collection, + resolved_targets: &resolved, + deferred_sum_targets: &deferred, + wal_lsn, + }; + + // The pre-image decides insert-vs-update for the hash chain and is the + // image the materialized-sum fold subtracts. Read only when one of the + // two needs it, as the transaction batch does. + let mut chain = ChainGuard::begin(self, row.database_id, row.tenant_id, row.collection); + let prior_bytes = if chain.enabled() || write_hook::folds_images(self, &hook_ctx) { + self.sparse + .get(row.database_id, row.tenant_id, row.collection, &storage_key)? + } else { + None + }; + let chained = if prior_bytes.is_none() { + chain.chain_insert(self, row.database_id, row.tenant_id, row.document_id, value)? + } else { + None + }; + let stored_value: &[u8] = chained.as_deref().unwrap_or(value); + + let txn = match self.sparse.begin_write() { + Ok(txn) => txn, + Err(error) => { + chain.restore(self); + return Err(error); + } + }; + let outcome = match self.apply_point_put( + &txn, + PointPutParams { + database_id: row.database_id, + tid: row.tenant_id, + collection: row.collection, + storage_key, + surrogate, + value: stored_value, + index_text: true, + user_roles: &[], + enforce: false, + resolved_targets: &resolved, + wal_lsn, + }, + ) { + Ok(outcome) => outcome, + Err(error) => { + self.abort_committed_write(&chain, row, &storage_key); + return Err(error); + } + }; + if let Err(error) = chain.persist_head(self, &txn) { + self.abort_committed_write(&chain, row, &storage_key); + return Err(error); + } + + // The fold reads the SUBMITTED body: `_chain_hash` wraps the row and no + // binding is declared over it. + let images = match prior_bytes { + Some(ref old) => WriteImages::Update { + old: ImageBody::Stored(old), + new: ImageBody::Submitted(value), + }, + None => WriteImages::Insert { + new: ImageBody::Submitted(value), + }, + }; + let target_writes = match write_hook::run(self, &txn, &hook_ctx, images) { + Ok(enforcement) => enforcement.target_writes, + Err(error) => { + self.abort_committed_write(&chain, row, &storage_key); + return Err(error); + } + }; + commit_row(txn)?; + self.checkpoint_coordinator.mark_dirty("sparse", 1); + + let mut index_tuples = outcome.secondary_index_added.clone(); + index_tuples.extend(outcome.secondary_index_removed.iter().cloned()); + index_tuples.extend(outcome.bitemporal_index_tuples.iter().cloned()); + self.record_committed_doc_write(AppliedDocWrite { + collection: row.collection.to_string(), + identity: RowIdentity::from_user_key(row.document_id), + op: if outcome.prior_value.is_some() { + WriteOp::Update + } else { + WriteOp::Insert + }, + old_value: outcome.prior_value.clone(), + index_tuples, + }); + if self.recording_redo_undo() { + let mut undo = Vec::new(); + push_target_undo(&mut undo, &target_writes); + push_put_undo( + &mut undo, + DocumentRow { + database_id: row.database_id, + tid: row.tenant_id, + collection: row.collection, + storage_key, + identity: RowIdentity::from_user_key(row.document_id), + }, + outcome, + chain.prior(), + ); + self.record_redo_undo(undo); + } + self.record_committed_targets(target_writes); + Ok(()) + } + + fn committed_document_delete(&mut self, row: &CommittedDocWrite<'_>) -> crate::Result { + let (resolved, _) = self.committed_sum_targets(row.collection); + let surrogate = Surrogate::new(row.surrogate); + let storage_key = StorageKey::for_surrogate(surrogate); + let row_key = storage_key.to_string(); + let hook_ctx = HookCtx { + database_id: row.database_id, + tid: row.tenant_id, + collection: row.collection, + resolved_targets: &resolved, + // A delete is deferred by omission from `resolved`, never by list. + deferred_sum_targets: &[], + wal_lsn: (row.record_lsn != 0).then(|| Lsn::new(row.record_lsn)), + }; + + let txn = self.sparse.begin_write()?; + let outcome = self.apply_point_delete( + &txn, + PointDeleteParams { + database_id: row.database_id, + tid: row.tenant_id, + collection: row.collection, + document_id: row_key.as_str(), + surrogate, + user_roles: &[], + enforce: false, + resolved_targets: &resolved, + }, + )?; + let target_writes = match outcome.prior_value { + Some(ref old) => { + write_hook::run( + self, + &txn, + &hook_ctx, + WriteImages::Delete { + old: ImageBody::Stored(old), + }, + )? + .target_writes + } + None => Vec::new(), + }; + commit_row(txn)?; + self.checkpoint_coordinator.mark_dirty("sparse", 1); + + let removed = outcome.prior_value.clone(); + let mut index_tuples = outcome.secondary_index_tuples.clone(); + index_tuples.extend(outcome.bitemporal_index_tuples.iter().cloned()); + // A delete that removed no row can still have cascaded side effects to + // reverse, so its undo is recorded either way. + if self.recording_redo_undo() { + let mut undo = Vec::new(); + push_target_undo(&mut undo, &target_writes); + push_delete_undo( + &mut undo, + DocumentRow { + database_id: row.database_id, + tid: row.tenant_id, + collection: row.collection, + storage_key, + identity: RowIdentity::from_user_key(row.document_id), + }, + outcome, + ); + self.record_redo_undo(undo); + } + let Some(old_value) = removed else { + return Ok(false); + }; + self.record_committed_doc_write(AppliedDocWrite { + collection: row.collection.to_string(), + identity: RowIdentity::from_user_key(row.document_id), + op: WriteOp::Delete, + old_value: Some(old_value), + index_tuples, + }); + self.record_committed_targets(target_writes); + Ok(true) + } + + /// Reverse the in-memory effects of a put abandoned after + /// `apply_point_put` ran; the caller drops its transaction uncommitted. + fn abort_committed_write( + &mut self, + chain: &ChainGuard, + row: &CommittedDocWrite<'_>, + storage_key: &StorageKey, + ) { + abort_after_apply( + self, + chain, + row.database_id, + row.tenant_id, + row.collection, + storage_key, + ); + } + + fn committed_sum_targets( + &self, + collection: &str, + ) -> ( + Vec, + Vec, + ) { + self.redo_apply + .scope + .as_ref() + .map(|scope| scope.sum_targets_for(collection)) + .unwrap_or_default() + } + + fn record_committed_doc_write(&mut self, write: AppliedDocWrite) { + if let Some(scope) = self.redo_apply.scope.as_mut() { + scope.doc_writes.push(write); + } + } + + fn record_committed_targets( + &mut self, + targets: Vec, + ) { + if let Some(scope) = self.redo_apply.scope.as_mut() { + for target in targets { + scope.doc_writes.push(target_doc_write(&target)); + scope.target_writes.push(target); + } + } + } +} + +fn commit_row(txn: WriteTransaction) -> crate::Result<()> { + txn.commit().map_err(|e| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("committed redo document commit: {e}"), + }) +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs new file mode 100644 index 000000000..5138662e4 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs @@ -0,0 +1,580 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `MetaOp::ApplyTransactionRedo`: install one committed transaction's redo +//! record on the core that owns its vShard. +//! +//! The record is applied through the SAME per-engine replay arms WAL replay +//! uses (`CoreLoop::replay_engines_in_lsn_order`), handed a one-record slice at +//! the LSN the write funnel minted for it on this node. What restart replay +//! reproduces from the WAL is therefore exactly what the live apply installed. +//! +//! The open apply scope switches the arms to their committed-redo policy +//! (`replay_policy`): no restart watermark skips a sub-record, and every +//! sub-record an arm cannot apply fails this response instead of being logged +//! and skipped or stopping the process. +//! +//! The record applies all or nothing (see `passes`): a validate pass checks +//! every sub-record before any arm writes, and the install pass rolls every +//! write back when one fails. +//! +//! Around the arms the apply adds what a live commit owes and restart replay +//! does not: +//! +//! 1. commit-boundary checks and the validate pass, before any write (see +//! `validate` and `passes`); +//! 2. materialized-sum folds and hash-chain links on document rows, through +//! the open [`RedoApplyScope`]; +//! 3. a collection-floor write version for every collection written, and the +//! index-value versions of every document row; +//! 4. the record's events (see `events`); +//! 5. the fold target rows in `Response::write_set`, so the funnel journals +//! them. + +use nodedb_physical::physical_plan::RedoSumTargets; +use nodedb_wal::WalRecord; +use nodedb_wal::record::{RecordType, WalRecordArgs}; + +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::enforcement::write_hook::target_write_set; +use crate::data::executor::task::ExecutionTask; +use crate::types::TenantId; +use crate::wal::RedoRecord; + +use super::cover::WrittenEngines; +use super::passes::{RedoTarget, final_refusal}; +use super::state::{RedoApplyPass, RedoApplyScope}; +use super::sub_ops::{document_ops, kv_ops, label_ops}; + +/// The `MetaOp::ApplyTransactionRedo` fields the handler reads. +pub(in crate::data::executor) struct CommittedRedo<'a> { + pub redo: &'a [u8], + pub collections: &'a [String], + pub sum_targets: &'a [RedoSumTargets], +} + +impl CoreLoop { + /// Apply one committed redo record. The request carries the LSN of the + /// `TransactionRedo` WAL record the funnel appended for it. + pub(in crate::data::executor) fn execute_apply_transaction_redo( + &mut self, + task: &ExecutionTask, + tid: u64, + committed: CommittedRedo<'_>, + ) -> Response { + let Some(lsn) = task.wal_lsn() else { + return self.response_error( + task, + ErrorCode::Internal { + detail: "committed transaction redo reached the Data Plane without the LSN \ + of its WAL record" + .into(), + }, + ); + }; + let redo = match RedoRecord::from_bytes(committed.redo) { + Ok(redo) => redo, + Err(error) => return self.response_error(task, final_refusal(error.into())), + }; + let database_id = task.request.database_id.as_u64(); + + let doc_ops = document_ops(&redo.ops); + // Nothing is written when a check refuses the record, so the refusal + // is final on every replica. + let check_scope = + RedoApplyScope::new(RedoApplyPass::Validate, committed.sum_targets.to_vec()); + if let Err(error) = + self.validate_redo_document_ops(database_id, tid, &doc_ops, &check_scope) + { + return self.response_error(task, error); + } + let kv_images = self.kv_prior_images(database_id, tid, kv_ops(&redo.ops)); + let labels = label_ops(&redo.ops); + + let record = match WalRecord::new(WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn: lsn.as_u64(), + tenant_id: task.request.tenant_id.as_u64(), + vshard_id: task.request.vshard_id.as_u32(), + database_id, + payload: committed.redo.to_vec(), + encryption_key: None, + preamble_bytes: None, + }) { + Ok(record) => record, + Err(error) => { + return self.response_error(task, final_refusal(crate::Error::from(error).into())); + } + }; + let target = RedoTarget { + record: &record, + sub_records: redo.ops.len(), + database_id, + tid, + vshard_id: task.request.vshard_id, + }; + if let Err(refusal) = self.validate_redo_pass(&target, committed.sum_targets) { + return self.response_error(task, refusal.into_code()); + } + let mut scope = match self.install_redo_pass(&target, committed.sum_targets) { + Ok(scope) => scope, + Err(refusal) => return self.response_error(task, refusal.into_code()), + }; + if let Err(error) = self.settle_redo_install(task, &mut scope) { + return self.response_error(task, error); + } + + // A record applied below a published engine watermark is published + // again, so restart replay does not skip it. + if let Err(error) = + self.cover_applied_record(lsn, &WrittenEngines::of(&redo), &scope.arrays_written) + { + return self.response_error(task, error); + } + + let tenant = TenantId::new(tid); + for collection in committed.collections { + self.note_write_lsn(task.request.database_id, tenant, collection, None, lsn); + } + for write in &scope.doc_writes { + self.note_index_write_values( + task.request.database_id, + tenant, + &write.collection, + &write.index_tuples, + lsn, + ); + } + let write_set = target_write_set(&scope.target_writes); + self.emit_committed_redo_events(task, scope.doc_writes, kv_images, labels); + + let mut response = self.response_ok(task); + response.write_set = write_set; + response + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::types::Lsn; + use crate::wal::RedoSubRecord; + use nodedb_types::sync::wire::SyncProvenance; + use nodedb_types::{StorageKey, Surrogate}; + + const TID: u64 = 1; + + fn doc_body(name: &str) -> Vec { + let mut obj = std::collections::HashMap::new(); + obj.insert( + "name".to_string(), + nodedb_types::Value::String(name.to_string()), + ); + zerompk::to_msgpack_vec(&nodedb_types::Value::Object(obj)).expect("encode body") + } + + fn redo_bytes(ops: Vec) -> Vec { + RedoRecord { + version: 1, + ops, + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo") + } + + fn doc_put(collection: &str, document_id: &str, surrogate: u32) -> RedoSubRecord { + RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(&( + collection, + document_id, + doc_body(document_id), + None::, + surrogate, + )) + .expect("encode doc put"), + } + } + + fn kv_put(collection: &str, key: &[u8], value: &[u8], surrogate: u32) -> RedoSubRecord { + RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(&( + "kv_put", + collection, + key.to_vec(), + value.to_vec(), + 0u64, + None::, + surrogate, + )) + .expect("encode kv put"), + } + } + + fn task_at(lsn: Option) -> ExecutionTask { + let mut task = make_default_task(); + task.wal_lsn = lsn.map(Lsn::new); + task + } + + #[test] + fn committed_redo_installs_its_document_and_kv_rows() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let redo = redo_bytes(vec![ + doc_put("notes", "n1", 11), + kv_put("cache", b"k1", b"v1", 12), + ]); + let collections = vec!["cache".to_string(), "notes".to_string()]; + + let response = core.execute_apply_transaction_redo( + &task_at(Some(50)), + TID, + CommittedRedo { + redo: &redo, + collections: &collections, + sum_targets: &[], + }, + ); + + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + let stored = core + .sparse + .get( + 0, + TID, + "notes", + &StorageKey::for_surrogate(Surrogate::new(11)), + ) + .expect("read document"); + assert!(stored.is_some(), "the document post-image is installed"); + let now_ms = crate::engine::kv::current_ms(); + assert_eq!( + core.kv_engine + .get(0, TID, "cache", b"k1", now_ms) + .as_deref(), + Some(b"v1".as_slice()), + "the KV post-image is installed" + ); + assert!( + core.redo_apply.scope.is_none(), + "the apply scope closes with the apply" + ); + } + + #[test] + fn committed_redo_without_its_wal_lsn_writes_nothing() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let redo = redo_bytes(vec![kv_put("cache", b"k1", b"v1", 12)]); + + let response = core.execute_apply_transaction_redo( + &task_at(None), + TID, + CommittedRedo { + redo: &redo, + collections: &[], + sum_targets: &[], + }, + ); + + assert_eq!(response.status, Status::Error); + let now_ms = crate::engine::kv::current_ms(); + assert!(core.kv_engine.get(0, TID, "cache", b"k1", now_ms).is_none()); + } + + fn metric_samples_sub(collection: &str, timestamp_ms: i64, value: f64) -> RedoSubRecord { + let batch = nodedb_types::timeseries::TimeseriesWalBatch { + collection: collection.to_string(), + samples: vec![(1u64, timestamp_ms, value)], + provenance: None, + }; + let batch_bytes = zerompk::to_msgpack_vec(&batch).expect("encode samples"); + RedoSubRecord { + record_type: RecordType::TimeseriesBatch as u32, + payload: + crate::control::server::wal_dispatch::encode_timeseries_batch_payload_with_format( + collection, + &batch_bytes, + None, + "samples", + ) + .expect("encode timeseries sub-record"), + } + } + + /// Restart replay skips a timeseries record at or below the highest + /// flushed partition stamp. A committed redo applied online after a live + /// flush stamped past its LSN must still install, and must flush so the + /// stamp's claim holds for it on the next restart. + #[test] + fn an_online_redo_apply_below_a_flushed_partition_stamp_still_installs() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let tenant = TenantId::new(TID); + let key = ( + crate::types::DatabaseId::DEFAULT, + tenant, + "metrics".to_string(), + ); + + // A write at LSN 100 lands and flushes: the partition stamps LSN 100. + let earlier = RedoRecord { + version: 1, + ops: vec![metric_samples_sub("metrics", 1_700_000_000_000, 1.0)], + calvin_stamp: None, + }; + let record = WalRecord::new(WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn: 100, + tenant_id: TID, + vshard_id: 0, + database_id: 0, + payload: earlier.to_bytes().expect("encode redo"), + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record"); + core.replay_transaction_redo_wal( + std::slice::from_ref(&record), + 1, + &nodedb_wal::TombstoneSet::new(), + ) + .expect("seed replay"); + core.flush_ts_collection(tenant, crate::types::DatabaseId::DEFAULT, "metrics", 0) + .expect("flush the seeded partition"); + + // A committed redo minted at LSN 50 applies afterwards. + let redo = redo_bytes(vec![metric_samples_sub("metrics", 1_700_000_000_001, 2.0)]); + let collections = vec!["metrics".to_string()]; + let response = core.execute_apply_transaction_redo( + &task_at(Some(50)), + TID, + CommittedRedo { + redo: &redo, + collections: &collections, + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + + let memtable_rows = core + .columnar_memtables + .get(&key) + .map_or(0, |m| m.row_count()); + let flushed_rows = core + .ts_registries + .get(&key) + .map_or(0, |registry| registry.total_row_count()); + assert_eq!( + memtable_rows + flushed_rows, + 2, + "the online apply installs its sample despite the higher partition stamp" + ); + assert_eq!( + memtable_rows, 0, + "the sample applied below the stamp is flushed, so a restart that skips its \ + record still finds it on disk" + ); + } + + /// A sub-record no replay arm owns is refused by the validate pass, so + /// the sub-records beside it write nothing. + #[test] + fn a_record_with_a_sub_record_no_arm_claims_writes_nothing() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let redo = redo_bytes(vec![ + kv_put("cache", b"k1", b"v1", 12), + RedoSubRecord { + record_type: RecordType::VectorParams as u32, + payload: vec![0x90], + }, + ]); + + let response = core.execute_apply_transaction_redo( + &task_at(Some(70)), + TID, + CommittedRedo { + redo: &redo, + collections: &[], + sum_targets: &[], + }, + ); + + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RejectedPrevalidation { .. }) + ), + "{:?}", + response.error_code + ); + let now_ms = crate::engine::kv::current_ms(); + assert!(core.kv_engine.get(0, TID, "cache", b"k1", now_ms).is_none()); + } + + /// A sub-record that passes validation and fails while it installs rolls + /// back every write of the record, those before it and those after it: + /// the overwritten KV key keeps its value, expiry and surrogate, and the + /// document row is absent. + #[test] + fn an_install_failure_rolls_back_every_write_of_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let now_ms = crate::engine::kv::current_ms(); + let expire_at_ms = now_ms + 3_600_000; + core.kv_engine.put_with_absolute_expiry( + crate::engine::kv::KvPutParams { + database_id: 0, + tenant_id: TID, + collection: "cache", + key: b"k1", + value: b"old", + ttl_ms: 0, + now_ms, + surrogate: Surrogate::new(5), + }, + expire_at_ms, + ); + let before = core + .kv_engine + .entry_image(0, TID, "cache", b"k1", now_ms) + .expect("seeded key"); + let redo = redo_bytes(vec![ + kv_put("cache", b"k1", b"new", 12), + mismatched_ingest(), + doc_put("notes", "n1", 11), + ]); + let collections = vec!["cache".to_string(), "notes".to_string()]; + + let response = core.execute_apply_transaction_redo( + &task_at(Some(80)), + TID, + CommittedRedo { + redo: &redo, + collections: &collections, + sum_targets: &[], + }, + ); + + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ), + "{:?}", + response.error_code + ); + assert_eq!( + core.kv_engine.entry_image(0, TID, "cache", b"k1", now_ms), + Some(before), + "the KV key keeps its value, its expiry instant and its surrogate" + ); + let stored = core + .sparse + .get( + 0, + TID, + "notes", + &StorageKey::for_surrogate(Surrogate::new(11)), + ) + .expect("read document"); + assert!( + stored.is_none(), + "the document row the record wrote is gone" + ); + assert!(core.redo_apply.scope.is_none()); + } + + fn mismatched_ingest() -> RedoSubRecord { + let lines = zerompk::to_msgpack_vec(&vec!["other,host=a value=1 1".to_string()]) + .expect("encode lines"); + RedoSubRecord { + record_type: RecordType::TimeseriesBatch as u32, + payload: + crate::control::server::wal_dispatch::encode_timeseries_batch_payload_with_format( + "metrics", + &lines, + None, + "ilp-msgpack", + ) + .expect("encode ingest"), + } + } + + /// The vector node an install inserted is withdrawn with the rest of the + /// record, and the index it created is gone. + #[test] + fn an_install_failure_withdraws_the_vector_node_it_inserted() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let vector_put = RedoSubRecord { + record_type: RecordType::VectorPut as u32, + payload: zerompk::to_msgpack_vec(&( + "docs", + vec![0.5f32, 0.5], + 2usize, + "", + None::, + 21u32, + None::, + )) + .expect("encode vector put"), + }; + let redo = redo_bytes(vec![vector_put, mismatched_ingest()]); + + let response = core.execute_apply_transaction_redo( + &task_at(Some(90)), + TID, + CommittedRedo { + redo: &redo, + collections: &[], + sum_targets: &[], + }, + ); + + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ), + "{:?}", + response.error_code + ); + let key = CoreLoop::vector_index_key(0, TID, "docs", ""); + assert!(!core.vector_collections.contains_key(&key)); + } + + /// A non-document sub-record that cannot be applied fails the online + /// apply response instead of being logged and skipped. + #[test] + fn a_failing_non_document_sub_record_fails_the_apply() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let redo = redo_bytes(vec![RedoSubRecord { + record_type: RecordType::SpatialPut as u32, + payload: vec![0xff, 0x00], + }]); + + let response = core.execute_apply_transaction_redo( + &task_at(Some(60)), + TID, + CommittedRedo { + redo: &redo, + collections: &[], + sum_targets: &[], + }, + ); + + assert_eq!(response.status, Status::Error); + assert!( + core.redo_apply.scope.is_none(), + "the apply scope closes with the failed apply" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs new file mode 100644 index 000000000..0f7b1849e --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs @@ -0,0 +1,159 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Event Plane output of a committed redo apply. +//! +//! The transaction batch emitted three kinds of events, and the apply emits +//! the same set: +//! +//! * one `Deferred` event per document row written (source rows and +//! materialized-sum targets), carrying the pre-image, for DEFERRED triggers; +//! * one event per KV key written, from the KV write handlers; +//! * one event per graph node-label delta, on the label stream. +//! +//! Graph edges and CRDT rows emit from the handlers their replay arms call, +//! and the index engines (vector, FTS, spatial) and the columnar family emit +//! no Data-Plane events on either path. + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::deferred::DeferredWrite; +use crate::data::executor::task::ExecutionTask; +use crate::engine::document::store::RowIdentity; +use crate::event::WriteOp; + +use super::state::AppliedDocWrite; +use super::sub_ops::{RedoKvOp, RedoLabelOp}; + +/// One KV key a redo record writes, with the value it held before. +pub(super) struct KvPriorImage { + collection: String, + key: Vec, + /// `Some` for a put, `None` for a delete. + new_value: Option>, + prior: Option>, +} + +impl CoreLoop { + /// Read the current value of every KV key `ops` writes, before the KV arm + /// overwrites it. + pub(super) fn kv_prior_images( + &self, + database_id: u64, + tid: u64, + ops: Vec, + ) -> Vec { + let now_ms = crate::engine::kv::current_ms(); + let mut images = Vec::new(); + for op in ops { + match op { + RedoKvOp::Put { + collection, + key, + value, + } => { + let prior = self + .kv_engine + .get(database_id, tid, &collection, &key, now_ms); + images.push(KvPriorImage { + collection, + key, + new_value: Some(value), + prior, + }); + } + RedoKvOp::Delete { collection, keys } => { + for key in keys { + let prior = self + .kv_engine + .get(database_id, tid, &collection, &key, now_ms); + images.push(KvPriorImage { + collection: collection.clone(), + key, + new_value: None, + prior, + }); + } + } + } + } + images + } + + /// Emit every event the committed record's writes produce. + pub(super) fn emit_committed_redo_events( + &mut self, + task: &ExecutionTask, + doc_writes: Vec, + kv_images: Vec, + labels: Vec, + ) { + // Every event names the record's LSN; the arms advanced the watermark + // to it for the rows they versioned, and this covers a record whose + // writes versioned none. + if let Some(lsn) = task.wal_lsn() + && lsn > self.watermark + { + self.watermark = lsn; + } + + for image in kv_images { + let identity = RowIdentity::from_user_key(String::from_utf8_lossy(&image.key).as_ref()); + match image.new_value { + Some(new_value) => { + let op = if image.prior.is_some() { + WriteOp::Update + } else { + WriteOp::Insert + }; + self.emit_write_event( + task, + &image.collection, + op, + identity, + Some(new_value.as_slice()), + image.prior.as_deref(), + ); + } + // A delete of an absent key removed nothing. + None if image.prior.is_some() => { + self.emit_write_event( + task, + &image.collection, + WriteOp::Delete, + identity, + None, + None, + ); + } + None => {} + } + } + + for label in labels { + let op = if label.is_set { + WriteOp::Insert + } else { + WriteOp::Delete + }; + self.emit_graph_label_event(task, &label.node_id, &label.labels, op); + } + + let deferred: Vec = doc_writes + .into_iter() + .map(|write| DeferredWrite { + collection: write.collection, + op: write.op, + identity: write.identity, + new_value: None, + old_value: write.old_value, + }) + .collect(); + if !deferred.is_empty() { + self.emit_deferred_events( + deferred, + task.request.database_id, + task.request.tenant_id, + task.request.vshard_id, + ); + } + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs new file mode 100644 index 000000000..1e25c3c66 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs @@ -0,0 +1,28 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Committed-redo apply: install one committed transaction's redo record on +//! the core that owns its vShard, through the WAL replay arms. +//! +//! - [`entry`]: the `MetaOp::ApplyTransactionRedo` handler. +//! - [`cover`]: republishing an engine artifact a record applied below. +//! - [`validate`]: commit-boundary checks that run before any write. +//! - [`passes`]: the validate and install passes over the replay arms. +//! - [`settle`]: the work an install defers until every sub-record landed. +//! - [`document`]: document rows written while the apply scope is open. +//! - [`events`]: the record's Event Plane output. +//! - [`sub_ops`]: typed views of the redo sub-records. +//! - [`state`]: the per-core state and per-record scope. + +mod cover; +mod document; +mod entry; +mod events; +mod passes; +mod settle; +mod state; +mod sub_ops; +mod validate; + +pub(in crate::data::executor) use document::CommittedDocWrite; +pub(in crate::data::executor) use entry::CommittedRedo; +pub(in crate::data::executor) use state::{RedoApplyPass, RedoApplyState}; diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs new file mode 100644 index 000000000..1e3bd4570 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs @@ -0,0 +1,151 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The two passes of a committed redo record's apply over the replay arms. +//! +//! * The validate pass decodes, routes and checks every sub-record and writes +//! nothing. Each arm claims every sub-record it owns. A failure, or a +//! sub-record no arm claimed, refuses the record. The pass reads only state +//! every replica holds at the same log position, so the refusal is final. +//! * The install pass writes every sub-record and records the undo entries +//! that reverse each write. A failure rolls every write back, so the core +//! holds none of the record, which is what restart replay reproduces once +//! the funnel cancels the record in the WAL. The refusal is retryable: the +//! failure did not come from the record. + +use nodedb_physical::physical_plan::RedoSumTargets; +use nodedb_wal::WalRecord; + +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::types::VShardId; + +use super::state::{RedoApplyPass, RedoApplyScope}; + +/// Why a pass refused the record. +pub(super) enum PassRefusal { + /// The validate pass refused it. Nothing was written. + Invalid(ErrorCode), + /// The install pass failed. Every write was rolled back. + RolledBack(ErrorCode), + /// The install pass failed and its rollback failed too. The core's state + /// is unknown. + RollbackFailed(ErrorCode), +} + +impl PassRefusal { + /// The code the apply responds with. + pub(super) fn into_code(self) -> ErrorCode { + match self { + Self::Invalid(code) | Self::RollbackFailed(code) => code, + Self::RolledBack(cause) => ErrorCode::RetryableRefusal { + reason: format!( + "the committed redo record failed part way through its install and every \ + write of it was rolled back: {cause:?}" + ), + }, + } + } +} + +/// Where the record applies. +pub(super) struct RedoTarget<'a> { + pub record: &'a WalRecord, + /// Sub-records in the record. Every one belongs to exactly one arm. + pub sub_records: usize, + pub database_id: u64, + pub tid: u64, + pub vshard_id: VShardId, +} + +impl CoreLoop { + /// Run the validate pass: every arm checks and claims its sub-records and + /// writes nothing. + pub(super) fn validate_redo_pass( + &mut self, + target: &RedoTarget<'_>, + sum_targets: &[RedoSumTargets], + ) -> Result<(), PassRefusal> { + let (applied, scope) = self.run_redo_arms( + target.record, + RedoApplyScope::new(RedoApplyPass::Validate, sum_targets.to_vec()), + )?; + if let Some(code) = applied.err().or(scope.error) { + return Err(PassRefusal::Invalid(final_refusal(code))); + } + if scope.claimed != target.sub_records { + return Err(PassRefusal::Invalid(ErrorCode::RejectedPrevalidation { + reason: format!( + "the replay arms claimed {} of the {} sub-records of the committed redo \ + record: a sub-record matches no engine's shape", + scope.claimed, target.sub_records + ), + })); + } + Ok(()) + } + + /// Run the install pass. On success the scope carries what the arms + /// wrote. On a failure every write is rolled back first. + pub(super) fn install_redo_pass( + &mut self, + target: &RedoTarget<'_>, + sum_targets: &[RedoSumTargets], + ) -> Result { + let (applied, mut scope) = self.run_redo_arms( + target.record, + RedoApplyScope::new(RedoApplyPass::Install, sum_targets.to_vec()), + )?; + let Some(cause) = applied.err().or(scope.error.take()) else { + return Ok(scope); + }; + let undo = std::mem::take(&mut scope.undo); + match self.rollback_undo_log_at(target.database_id, target.tid, target.vshard_id, undo) { + Ok(()) => Err(PassRefusal::RolledBack(cause)), + Err((entry_index, detail)) => { + Err(PassRefusal::RollbackFailed(ErrorCode::RollbackFailed { + entry_index, + detail: format!( + "rolling back a committed redo install that failed with {cause:?}: \ + {detail}" + ), + })) + } + } + } + + /// Drive every replay arm over `record` with `scope` open. Returns the + /// arms' own result and the scope they filled. + fn run_redo_arms( + &mut self, + record: &WalRecord, + scope: RedoApplyScope, + ) -> Result<(Result<(), ErrorCode>, RedoApplyScope), PassRefusal> { + // The arms route a record to `vshard_id % num_cores`. A committed + // record carries no collection tombstone of its own: a collection + // dropped before this entry committed refused the commit instead. + self.redo_apply.scope = Some(scope); + let applied = self + .replay_engines_in_lsn_order( + std::slice::from_ref(record), + self.redo_apply.num_cores, + &nodedb_wal::TombstoneSet::new(), + ) + .map_err(ErrorCode::from); + match self.redo_apply.scope.take() { + Some(scope) => Ok((applied, scope)), + None => Err(PassRefusal::RollbackFailed(ErrorCode::Internal { + detail: "committed transaction redo lost its apply scope".into(), + })), + } + } +} + +/// A validate-pass failure as a final refusal. The pass wrote nothing and +/// read only state every replica holds at the same log position, so a +/// failure the arms report as internal is a verdict on the record's bytes. +pub(super) fn final_refusal(code: ErrorCode) -> ErrorCode { + match code { + ErrorCode::Internal { detail } => ErrorCode::RejectedPrevalidation { reason: detail }, + other => other, + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs new file mode 100644 index 000000000..c1e97c29a --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs @@ -0,0 +1,148 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The work a committed-redo install defers until every sub-record landed. +//! +//! A write the undo cannot reverse waits here: a memtable flush drains rows +//! the undo would restore in memory, a vector seal moves inserted nodes out of +//! the growing segment, a truncate that removes files cannot be renamed back, +//! and an event the Event Plane consumed cannot be withdrawn. Once the install +//! succeeded, this runs them in one step. +//! +//! A failure here comes after every sub-record landed, so it is not rolled +//! back. It fails the response as an ambiguous error, and the funnel keeps +//! the record for restart replay. + +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; + +use super::state::{CollectionKey, RedoApplyPass, RedoApplyScope}; + +impl CoreLoop { + /// Record that the install wrote rows into the columnar collection `key`. + pub(in crate::data::executor) fn note_redo_columnar_written(&mut self, key: CollectionKey) { + if let Some(scope) = self.redo_apply.scope.as_mut() + && scope.pass == RedoApplyPass::Install + && !scope.columnar_written.contains(&key) + { + scope.columnar_written.push(key); + } + } + + /// Record that the install ingested rows into the timeseries collection + /// `key` at `lsn`. + pub(in crate::data::executor) fn note_redo_timeseries_written( + &mut self, + key: CollectionKey, + lsn: u64, + ) { + if let Some(scope) = self.redo_apply.scope.as_mut() + && scope.pass == RedoApplyPass::Install + && !scope + .timeseries_written + .iter() + .any(|(seen, _)| *seen == key) + { + scope.timeseries_written.push((key, lsn)); + } + } + + /// Run everything the successful install deferred. + pub(super) fn settle_redo_install( + &mut self, + task: &ExecutionTask, + scope: &mut RedoApplyScope, + ) -> Result<(), ErrorCode> { + for event in std::mem::take(&mut scope.pending_events) { + self.send_write_event(event); + } + self.finalize_timeseries_truncates(&scope.undo); + self.finalize_vector_truncates(&mut scope.undo); + self.seal_full_vector_collections(); + for key in std::mem::take(&mut scope.columnar_written) { + let collection = key.2.clone(); + self.flush_columnar_memtable_if_needed(task, &key, &collection) + .map_err(|response| { + response.error_code.map_or_else( + || ErrorCode::Internal { + detail: format!("columnar flush of '{collection}' failed"), + }, + |error| *error, + ) + })?; + } + for (key, lsn) in std::mem::take(&mut scope.timeseries_written) { + self.settle_redo_timeseries(key, lsn)?; + } + for array_id in &scope.arrays_written { + self.array_engine + .flush_if_full(array_id) + .map_err(|e| ErrorCode::Internal { + detail: format!("array '{}' threshold flush failed: {e}", array_id.name), + })?; + } + Ok(()) + } + + /// Seal every vector collection whose growing segment filled while the + /// install held its seals back. + fn seal_full_vector_collections(&mut self) { + let full: Vec<_> = self + .vector_collections + .iter() + .filter(|(_, coll)| coll.needs_seal()) + .map(|(key, _)| key.clone()) + .collect(); + for key in full { + let seal_key = CoreLoop::vector_build_key(&key); + if let Some(coll) = self.vector_collections.get_mut(&key) + && let Some(req) = coll.seal(&seal_key) + && let Some(tx) = &self.build_tx + && let Err(e) = tx.send(req) + { + tracing::warn!( + core = self.core_id, + error = %e, + "failed to send HNSW build request" + ); + } + } + } + + /// Settle one timeseries collection the install ingested into: charge + /// the memory budget for its memtable, and flush it when it is over its + /// soft limit or when a partition already claims the record's LSN. + /// + /// Restart replay skips every record at or below the highest partition + /// stamp. A record applied after a flush stamped past its LSN sits only + /// in the memtable, so the claim is false for it until the memtable + /// flushes too. + fn settle_redo_timeseries(&mut self, key: CollectionKey, lsn: u64) -> Result<(), ErrorCode> { + let (database_id, tid, collection) = key; + self.recharge_ts_memtable_budget(tid, database_id, &collection); + self.checkpoint_coordinator.mark_dirty("timeseries", 1); + // no-determinism: Instant::now feeds only the idle-flush timer. + self.last_ts_ingest = Some(std::time::Instant::now()); + let memtable_key = (database_id, tid, collection.clone()); + let over_budget = self + .columnar_memtables + .get(&memtable_key) + .is_some_and(|mt| mt.memory_bytes() >= self.ts_tuning.memtable_budget_bytes); + let below_stamp = self + .ts_registries + .get(&memtable_key) + .is_some_and(|registry| { + registry + .iter() + .any(|(_, entry)| entry.meta.last_flushed_wal_lsn >= lsn) + }); + if !over_budget && !below_stamp { + return Ok(()); + } + let now_ms = self.ingest_now_ms(); + self.flush_ts_collection(tid, database_id, &collection, now_ms) + .map_err(|e| ErrorCode::Internal { + detail: format!("flushing '{collection}' after a committed-redo install: {e}"), + }) + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs new file mode 100644 index 000000000..733597bf0 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Core state the committed-redo apply shares with the document redo arm. +//! +//! WAL replay and the committed-redo apply drive the same per-engine replay +//! arms. The apply opens a [`RedoApplyScope`] for each pass over one record: +//! while it is open, the document arm folds materialized sums, links hash +//! chains, and reports every row it wrote back through the scope. With no +//! scope open the arms are plain restart replay. +//! +//! The apply runs the arms twice. The [`RedoApplyPass::Validate`] pass +//! decodes, routes and checks every sub-record and writes nothing. The +//! [`RedoApplyPass::Install`] pass writes, and records the undo entries that +//! reverse each write, so a failure part way through rolls the record back. + +use std::collections::HashMap; + +use nodedb_physical::physical_plan::{RedoSumTargets, ResolvedSumTarget}; +use nodedb_types::RowIdentity; + +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::enforcement::materialized_sum::apply::TargetWrite; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::event::WriteOp; + +/// Committed-redo apply state owned by one core. +pub(in crate::data::executor) struct RedoApplyState { + /// Number of Data Plane cores on this node. The replay arms route a record + /// to core `vshard_id % num_cores`, so the apply hands them this count. + pub(in crate::data::executor) num_cores: usize, + /// `Some` only while one committed redo record applies on this core. + pub(in crate::data::executor) scope: Option, +} + +impl RedoApplyState { + /// A single-core default. Every multi-core runtime sets the real count + /// through `CoreLoop::set_num_cores` before the core serves requests. + pub(in crate::data::executor) fn new() -> Self { + Self { + num_cores: 1, + scope: None, + } + } +} + +/// One document row the committed-redo apply wrote. +pub(in crate::data::executor) struct AppliedDocWrite { + pub collection: String, + /// The row's client identity. + pub identity: RowIdentity, + pub op: WriteOp, + /// The row as stored before this write, when the write read it. + pub old_value: Option>, + /// Every `(field, value)` index entry the write added, removed, or + /// versioned. + pub index_tuples: Vec<(String, String)>, +} + +/// Which pass over a committed redo record drives the arms. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::data::executor) enum RedoApplyPass { + /// Every arm decodes, routes and checks each sub-record it owns, claims + /// it, and writes nothing. + Validate, + /// Every arm writes each sub-record it owns and records the undo entries + /// that reverse the write. + Install, +} + +/// Scratch for one pass over a committed redo record. +pub(in crate::data::executor) struct RedoApplyScope { + pub(in crate::data::executor) pass: RedoApplyPass, + /// Sub-records the arms claimed in the validate pass. Every sub-record + /// belongs to exactly one arm, so the count equals the record's. + pub(in crate::data::executor) claimed: usize, + /// The entries that reverse every write of the install pass, in write + /// order. + pub(in crate::data::executor) undo: Vec, + /// Materialized-sum resolution keyed by SOURCE collection. + sum_targets: HashMap, + /// Document rows written, source rows and fold targets alike, in order. + pub(in crate::data::executor) doc_writes: Vec, + /// Materialized-sum target rows the folds wrote. + pub(in crate::data::executor) target_writes: Vec, + /// First error any arm hit on a sub-record. The apply reports it once + /// every arm has run. + pub(in crate::data::executor) error: Option, + /// Every array the record wrote cells to. + pub(in crate::data::executor) arrays_written: Vec, + /// Columnar collections the install wrote, flushed once it succeeded. + pub(in crate::data::executor) columnar_written: Vec, + /// Timeseries collections the install ingested into, with the record's + /// LSN, settled once it succeeded. + pub(in crate::data::executor) timeseries_written: Vec<(CollectionKey, u64)>, + /// Events the install's writes raised, sent once it succeeded. + pub(in crate::data::executor) pending_events: Vec, +} + +/// `(database, tenant, collection)`. +pub(in crate::data::executor) type CollectionKey = + (crate::types::DatabaseId, crate::types::TenantId, String); + +impl RedoApplyScope { + pub(in crate::data::executor) fn new( + pass: RedoApplyPass, + sum_targets: Vec, + ) -> Self { + Self { + pass, + claimed: 0, + undo: Vec::new(), + sum_targets: sum_targets + .into_iter() + .map(|targets| (targets.collection.clone(), targets)) + .collect(), + doc_writes: Vec::new(), + target_writes: Vec::new(), + error: None, + arrays_written: Vec::new(), + columnar_written: Vec::new(), + timeseries_written: Vec::new(), + pending_events: Vec::new(), + } + } + + /// The resolved and deferred sum targets for writes to `collection`. + pub(in crate::data::executor) fn sum_targets_for( + &self, + collection: &str, + ) -> (Vec, Vec) { + match self.sum_targets.get(collection) { + Some(targets) => (targets.resolved.clone(), targets.deferred.clone()), + None => (Vec::new(), Vec::new()), + } + } + + /// Keep the first error; a later one is a consequence of it. + pub(in crate::data::executor) fn record_error(&mut self, error: impl Into) { + if self.error.is_none() { + self.error = Some(error.into()); + } + } +} + +/// The applied-write record of one materialized-sum target row. +pub(in crate::data::executor) fn target_doc_write(target: &TargetWrite) -> AppliedDocWrite { + let outcome = &target.outcome; + let mut index_tuples = Vec::with_capacity( + outcome.secondary_index_added.len() + + outcome.secondary_index_removed.len() + + outcome.bitemporal_index_tuples.len(), + ); + index_tuples.extend_from_slice(&outcome.secondary_index_added); + index_tuples.extend_from_slice(&outcome.secondary_index_removed); + index_tuples.extend_from_slice(&outcome.bitemporal_index_tuples); + AppliedDocWrite { + collection: target.collection.clone(), + identity: target.identity.clone(), + op: if outcome.prior_value.is_some() { + WriteOp::Update + } else { + WriteOp::Insert + }, + old_value: outcome.prior_value.clone(), + index_tuples, + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/sub_ops.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/sub_ops.rs new file mode 100644 index 000000000..80010c26e --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/sub_ops.rs @@ -0,0 +1,239 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Typed views of the redo sub-records the committed-redo apply inspects +//! before and after the replay arms run. +//! +//! The decoders accept exactly the payload shapes the transaction resolver +//! emits and the replay arms decode: a document `Put` is the 5-tuple or the +//! bitemporal 8-tuple, a document `Delete` the 4-tuple or the bitemporal +//! 5-tuple, a KV write carries its leading `"kv_*"` discriminator. KV and +//! graph `Put` / `Delete` payloads never decode as a document tuple, and +//! document payloads never decode as KV. + +use nodedb_types::sync::wire::SyncProvenance; +use nodedb_wal::record::RecordType; + +use crate::wal::RedoSubRecord; + +/// A document write inside a redo record. +pub(super) enum RedoDocOp { + Put { + collection: String, + /// The post-image as MessagePack. + value: Vec, + surrogate: u32, + }, + Delete { + collection: String, + surrogate: u32, + }, +} + +impl RedoDocOp { + pub(super) fn collection(&self) -> &str { + match self { + Self::Put { collection, .. } | Self::Delete { collection, .. } => collection, + } + } + + pub(super) fn surrogate(&self) -> u32 { + match self { + Self::Put { surrogate, .. } | Self::Delete { surrogate, .. } => *surrogate, + } + } +} + +type BitemporalPut = ( + String, + String, + Vec, + Option, + u32, + i64, + i64, + i64, +); +type PlainPut = (String, String, Vec, Option, u32); +type BitemporalDelete = (String, String, Option, u32, i64); +type PlainDelete = (String, String, Option, u32); + +/// The document writes in `ops`, in record order. +pub(super) fn document_ops(ops: &[RedoSubRecord]) -> Vec { + ops.iter().filter_map(document_op).collect() +} + +fn document_op(sub: &RedoSubRecord) -> Option { + match RecordType::from_raw(sub.record_type) { + Some(RecordType::Put) => zerompk::from_msgpack::(&sub.payload) + .map(|(collection, _, value, _, surrogate, _, _, _)| (collection, value, surrogate)) + .or_else(|_| { + zerompk::from_msgpack::(&sub.payload) + .map(|(collection, _, value, _, surrogate)| (collection, value, surrogate)) + }) + .ok() + .map(|(collection, value, surrogate)| RedoDocOp::Put { + collection, + value, + surrogate, + }), + Some(RecordType::Delete) => zerompk::from_msgpack::(&sub.payload) + .map(|(collection, _, _, surrogate, _)| (collection, surrogate)) + .or_else(|_| { + zerompk::from_msgpack::(&sub.payload) + .map(|(collection, _, _, surrogate)| (collection, surrogate)) + }) + .ok() + .map(|(collection, surrogate)| RedoDocOp::Delete { + collection, + surrogate, + }), + _ => None, + } +} + +/// A KV write inside a redo record. +pub(super) enum RedoKvOp { + Put { + collection: String, + key: Vec, + value: Vec, + }, + Delete { + collection: String, + keys: Vec>, + }, +} + +type KvPut = (String, String, Vec, Vec, u64, Option, u32); +type KvDelete = (String, String, Vec>); + +/// The KV point writes in `ops`, in record order. A `kv_truncate` carries no +/// per-key identity and is not listed. +pub(super) fn kv_ops(ops: &[RedoSubRecord]) -> Vec { + ops.iter().filter_map(kv_op).collect() +} + +fn kv_op(sub: &RedoSubRecord) -> Option { + match RecordType::from_raw(sub.record_type) { + Some(RecordType::Put) => zerompk::from_msgpack::(&sub.payload) + .ok() + .filter(|(disc, ..)| disc == "kv_put") + .map(|(_, collection, key, value, _, _, _)| RedoKvOp::Put { + collection, + key, + value, + }), + Some(RecordType::Delete) => zerompk::from_msgpack::(&sub.payload) + .ok() + .filter(|(disc, ..)| disc == "kv_delete") + .map(|(_, collection, keys)| RedoKvOp::Delete { collection, keys }), + _ => None, + } +} + +/// A graph node-label delta inside a redo record. +pub(super) struct RedoLabelOp { + pub node_id: String, + pub labels: Vec, + /// `true` for a label set, `false` for a removal. + pub is_set: bool, +} + +/// The node-label deltas in `ops`, in record order. +pub(super) fn label_ops(ops: &[RedoSubRecord]) -> Vec { + ops.iter() + .filter_map(|sub| { + let is_set = match RecordType::from_raw(sub.record_type) { + Some(RecordType::GraphNodeLabelSet) => true, + Some(RecordType::GraphNodeLabelRemove) => false, + _ => return None, + }; + let (node_id, labels) = + zerompk::from_msgpack::<(String, Vec)>(&sub.payload).ok()?; + Some(RedoLabelOp { + node_id, + labels, + is_set, + }) + }) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn sub(record_type: RecordType, payload: Vec) -> RedoSubRecord { + RedoSubRecord { + record_type: record_type as u32, + payload, + } + } + + #[test] + fn document_and_kv_shapes_decode_to_their_own_kind_only() { + let doc_put = + zerompk::to_msgpack_vec(&("docs", "d1", vec![1u8, 2], None::, 7u32)) + .expect("encode doc put"); + let doc_delete = zerompk::to_msgpack_vec(&("docs", "d2", None::, 8u32)) + .expect("encode doc delete"); + let kv_put = zerompk::to_msgpack_vec(&( + "kv_put", + "kvs", + b"k".to_vec(), + b"v".to_vec(), + 0u64, + None::, + 9u32, + )) + .expect("encode kv put"); + let kv_delete = zerompk::to_msgpack_vec(&("kv_delete", "kvs", vec![b"k".to_vec()])) + .expect("encode kv delete"); + let ops = vec![ + sub(RecordType::Put, doc_put), + sub(RecordType::Delete, doc_delete), + sub(RecordType::Put, kv_put), + sub(RecordType::Delete, kv_delete), + ]; + + let docs = document_ops(&ops); + assert_eq!(docs.len(), 2); + assert_eq!(docs[0].surrogate(), 7); + assert!(matches!(docs[1], RedoDocOp::Delete { surrogate: 8, .. })); + + let kvs = kv_ops(&ops); + assert_eq!(kvs.len(), 2); + assert!(matches!(&kvs[0], RedoKvOp::Put { key, .. } if key == b"k")); + assert!(matches!(&kvs[1], RedoKvOp::Delete { keys, .. } if keys.len() == 1)); + } + + #[test] + fn bitemporal_document_put_decodes_with_its_value() { + let prov: Option = None; + let payload = zerompk::to_msgpack_vec(&( + "docs", + "d1", + vec![5u8], + prov, + 3u32, + 10i64, + i64::MIN, + i64::MAX, + )) + .expect("encode bitemporal put"); + let docs = document_ops(&[sub(RecordType::Put, payload)]); + assert!(matches!( + &docs[0], + RedoDocOp::Put { value, surrogate: 3, .. } if value == &vec![5u8] + )); + } + + #[test] + fn bitemporal_document_delete_decodes_as_a_delete() { + let prov: Option = None; + let payload = zerompk::to_msgpack_vec(&("docs", "d1", prov, 4u32, 77i64)) + .expect("encode bitemporal delete"); + let docs = document_ops(&[sub(RecordType::Delete, payload)]); + assert!(matches!(docs[0], RedoDocOp::Delete { surrogate: 4, .. })); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/validate.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/validate.rs new file mode 100644 index 000000000..770e743a2 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/validate.rs @@ -0,0 +1,263 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Commit-boundary checks a committed redo record passes before anything of it +//! is written. +//! +//! The transaction batch judged these at its commit and rolled the whole +//! transaction back on a refusal. A redo record applies as absolute +//! post-images with no undo log, so every check runs first, against this +//! core's current state, and a refusal writes nothing. Every replica applies +//! the record at the same log position against the same state, so each one +//! reaches the same verdict. +//! +//! * BALANCED — the signed entries of every document write, judged per +//! collection across the whole record. +//! * UNIQUE — the record's post-state: a unique value may have one owner among +//! the rows the record leaves untouched and the rows it writes. +//! * Stateless PUT / DELETE enforcement — append-only, period lock, state +//! transitions, transition checks, retention and legal hold, each against +//! the stored pre-image. + +use std::collections::{BTreeMap, HashMap, HashSet}; + +use nodedb_types::Surrogate; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::doc_format; +use crate::data::executor::enforcement::balanced::{self, BalancedEntry}; +use crate::data::executor::enforcement::images::RowImages; +use crate::data::executor::handlers::point::apply_delete::run_delete_enforcement; +use crate::data::executor::handlers::point::apply_put::PutEnforcement; +use crate::engine::document::store::{ + CollectionConfig, DocumentEngine, StorageKey, extract_index_values, +}; +use crate::types::{DatabaseId, TenantId}; + +use super::sub_ops::RedoDocOp; + +impl CoreLoop { + /// Refuse `ops` when applying them would break a constraint the + /// transaction batch checks at commit. Reads only. + pub(super) fn validate_redo_document_ops( + &self, + database_id: u64, + tid: u64, + ops: &[RedoDocOp], + apply_scope: &super::state::RedoApplyScope, + ) -> crate::Result<()> { + let mut by_collection: BTreeMap<&str, Vec<&RedoDocOp>> = BTreeMap::new(); + for op in ops { + by_collection.entry(op.collection()).or_default().push(op); + } + for (collection, collection_ops) in by_collection { + let config_key = ( + DatabaseId::new(database_id), + TenantId::new(tid), + collection.to_string(), + ); + // An unregistered collection declares no constraint to check. + let Some(config) = self.doc_configs.get(&config_key) else { + continue; + }; + let bitemporal = self.is_bitemporal(database_id, tid, collection); + let (resolved, _) = apply_scope.sum_targets_for(collection); + let scope = CollectionScope { + database_id, + tid, + collection, + config_key: &config_key, + config, + bitemporal, + resolved: &resolved, + }; + self.check_collection_ops(&scope, &collection_ops)?; + self.check_unique_post_state(&scope, &collection_ops)?; + } + Ok(()) + } + + /// Stateless enforcement per write, then BALANCED across the collection. + fn check_collection_ops( + &self, + scope: &CollectionScope<'_>, + ops: &[&RedoDocOp], + ) -> crate::Result<()> { + let enforcement = &scope.config.enforcement; + let needs_prior = enforcement.has_put_checks() + || enforcement.balanced.is_some() + || enforcement.retention.is_some() + || enforcement.has_legal_hold; + let mut balanced_entries: Vec = Vec::new(); + for op in ops { + let prior = if needs_prior { + self.stored_prior(scope, op.surrogate())? + } else { + None + }; + match op { + RedoDocOp::Put { value, .. } => { + self.check_stateless_put_enforcement( + true, + PutEnforcement { + config_key: scope.config_key, + database_id: scope.database_id, + tid: scope.tid, + collection: scope.collection, + value, + old_value: &prior, + user_roles: &[], + resolved_targets: scope.resolved, + }, + )?; + if let Some(def) = &enforcement.balanced { + let new_doc = doc_format::decode_document(value).ok(); + let old_doc = match &prior { + Some(bytes) => Some(self.decode_stored_document(scope.config, bytes)?), + None => None, + }; + let images = match (old_doc.as_ref(), new_doc.as_ref()) { + (Some(old_doc), Some(new_doc)) => { + Some(RowImages::Update { old_doc, new_doc }) + } + (None, Some(new_doc)) => Some(RowImages::Insert { new_doc }), + // A post-image with no readable document carries no + // column the definition can read. + (_, None) => None, + }; + if let Some(images) = images { + balanced_entries.extend(balanced::entries_for(def, &images)); + } + } + } + RedoDocOp::Delete { .. } => { + // A bitemporal delete judges only a row that exists; a plain + // delete judges whatever is stored, as the delete path does. + if !scope.bitemporal || prior.is_some() { + run_delete_enforcement( + &self.sparse, + scope.database_id, + scope.tid, + scope.collection, + scope.config, + prior.as_deref(), + scope.resolved, + )?; + } + if let (Some(def), Some(bytes)) = (&enforcement.balanced, &prior) { + let old_doc = self.decode_stored_document(scope.config, bytes)?; + balanced_entries.extend(balanced::entries_for( + def, + &RowImages::Delete { old_doc: &old_doc }, + )); + } + } + } + } + if let Some(def) = &enforcement.balanced { + balanced::check_balanced(scope.collection, def, &balanced_entries)?; + } + Ok(()) + } + + /// Every unique index value the record's post-state holds has one owner. + fn check_unique_post_state( + &self, + scope: &CollectionScope<'_>, + ops: &[&RedoDocOp], + ) -> crate::Result<()> { + let unique_paths: Vec<_> = scope + .config + .index_paths + .iter() + .filter(|path| path.unique) + .collect(); + if unique_paths.is_empty() { + return Ok(()); + } + // Rows this record rewrites or removes: their stored values do not + // count, whatever they are. + let touched: HashSet = ops.iter().map(|op| op.surrogate()).collect(); + let doc_engine = DocumentEngine::new(&self.sparse, scope.database_id, scope.tid); + for path in unique_paths { + // Needle → the surrogate of the record's own row holding it. + let mut claimed: HashMap = HashMap::new(); + for op in ops { + let RedoDocOp::Put { + value, surrogate, .. + } = op + else { + continue; + }; + let Ok(doc) = doc_format::decode_document(value) else { + continue; + }; + if let Some(predicate) = &path.predicate + && !predicate.evaluate_json(&doc) + { + continue; + } + for raw in extract_index_values(&doc, &path.path, path.is_array) { + let needle = if path.case_insensitive { + raw.to_lowercase() + } else { + raw + }; + let violation = || crate::Error::RejectedConstraint { + collection: scope.collection.to_string(), + constraint: "unique".to_string(), + detail: format!( + "unique index '{}' violation on field '{}' (value '{}')", + path.name, path.path, needle + ), + }; + if let Some(owner) = claimed.insert(needle.clone(), *surrogate) + && owner != *surrogate + { + return Err(violation()); + } + let stored_owners = doc_engine.index_lookup( + scope.collection, + &path.path, + &needle, + scope.bitemporal, + )?; + let foreign_owner = stored_owners.iter().any(|key: &StorageKey| { + let owner = key.surrogate().as_u32(); + owner != *surrogate && !touched.contains(&owner) + }); + if foreign_owner { + return Err(violation()); + } + } + } + } + Ok(()) + } + + /// The row as stored now, in the store this collection writes. + fn stored_prior( + &self, + scope: &CollectionScope<'_>, + surrogate: u32, + ) -> crate::Result>> { + let key = StorageKey::for_surrogate(Surrogate::new(surrogate)); + if scope.bitemporal { + self.sparse + .versioned_get_current(scope.database_id, scope.tid, scope.collection, &key) + } else { + self.sparse + .get(scope.database_id, scope.tid, scope.collection, &key) + } + } +} + +/// One collection's slice of a record, with what its checks read. +struct CollectionScope<'a> { + database_id: u64, + tid: u64, + collection: &'a str, + config_key: &'a (DatabaseId, TenantId, String), + config: &'a CollectionConfig, + bitemporal: bool, + resolved: &'a [nodedb_physical::physical_plan::ResolvedSumTarget], +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/classify.rs b/nodedb/src/data/executor/handlers/transaction/resolve/classify.rs new file mode 100644 index 000000000..71c782871 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/resolve/classify.rs @@ -0,0 +1,171 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Plan classification for transaction resolve: which KV and document +//! collections a transaction wrote. Those two engines serialize from the +//! overlay, so the plan walk only collects collections and rejects the ops +//! that leave no staged post-image. + +use std::collections::BTreeSet; + +use nodedb_physical::physical_plan::{DocumentOp, KvOp}; + +/// Classify a KV op for transaction resolve: collect the collection of a +/// row-level write into `collections`, skip read-only ops, and reject the ops +/// that have no row-level redo representation. +pub(super) fn classify_kv_op(op: &KvOp, collections: &mut BTreeSet) -> crate::Result<()> { + match op { + // Row-level writes: the resolved post-image (value or tombstone) is in + // the overlay, keyed by collection. + KvOp::Put { collection, .. } + | KvOp::Insert { collection, .. } + | KvOp::InsertIfAbsent { collection, .. } + | KvOp::InsertOnConflictUpdate { collection, .. } + | KvOp::Delete { collection, .. } + | KvOp::BatchPut { collection, .. } + | KvOp::Incr { collection, .. } + | KvOp::IncrFloat { collection, .. } + | KvOp::Cas { collection, .. } + | KvOp::GetSet { collection, .. } + | KvOp::FieldSet { collection, .. } + | KvOp::Transfer { collection, .. } + // Predicate DML stages each matched row's post-image or tombstone + // at statement time, so the overlay carries it like a keyed write. + | KvOp::PredicateUpdate { collection, .. } + | KvOp::PredicateDelete { collection, .. } => { + collections.insert(collection.to_string()); + Ok(()) + } + // `TransferItem` moves a row across collections: the source holds a + // staged tombstone and the destination a staged value. + KvOp::TransferItem { + source_collection, + dest_collection, + .. + } => { + collections.insert(source_collection.to_string()); + collections.insert(dest_collection.to_string()); + Ok(()) + } + + // Read-only: nothing staged, nothing to persist. + KvOp::Get { .. } + | KvOp::BatchGet { .. } + | KvOp::Scan { .. } + | KvOp::FieldGet { .. } + | KvOp::GetTtl { .. } + | KvOp::MaterializeScan { .. } + | KvOp::SortedIndexRank { .. } + | KvOp::SortedIndexTopK { .. } + | KvOp::SortedIndexRange { .. } + | KvOp::SortedIndexCount { .. } + | KvOp::SortedIndexScore { .. } + // Read-only: reports what a governed write would apply, stages + // nothing. + | KvOp::ResolveWrite(_) => Ok(()), + + // Resolve-before-propose is an autocommit path: never staged into an + // overlay, so no row-level redo shape carries it. + KvOp::ResolvedWrite { .. } => Err(crate::Error::PlanError { + detail: "kv resolved write is not supported in transaction resolve".to_string(), + }), + + // A standalone TTL delta has no value post-image, and KV redo carries + // TTL only as part of a value put, so rejecting avoids a silent drop. + KvOp::Expire { .. } | KvOp::Persist { .. } => Err(crate::Error::PlanError { + detail: "kv EXPIRE/PERSIST is not supported in transaction resolve".to_string(), + }), + + // Truncate: staged as an overlay marker; the serializer emits the + // `kv_truncate` redo ahead of the collection's row entries. + KvOp::Truncate { collection, .. } => { + collections.insert(collection.to_string()); + Ok(()) + } + + // Index / DDL: never stageable into the overlay, so no row-level + // redo shape carries them. + KvOp::RegisterIndex { .. } + | KvOp::DropIndex { .. } + | KvOp::RegisterSortedIndex { .. } + | KvOp::DropSortedIndex { .. } => Err(crate::Error::PlanError { + detail: "kv index/DDL op is not supported in transaction resolve".to_string(), + }), + } +} + +/// Classify a Document op for transaction resolve: collect the collection of a +/// staged point/bulk write into `collections`, skip read-only ops, and reject +/// the writes that leave no overlay post-image. +pub(super) fn classify_document_op( + op: &DocumentOp, + collections: &mut BTreeSet, +) -> crate::Result<()> { + match op { + // Staged writes: the resolved post-image is in the overlay, keyed by the + // user primary key. RETURNING doesn't affect staging, so these serialize + // from the overlay like any other point/bulk write. + DocumentOp::PointPut { collection, .. } + | DocumentOp::PointInsert { collection, .. } + | DocumentOp::Upsert { collection, .. } + | DocumentOp::PointDelete { collection, .. } + | DocumentOp::PointUpdate { collection, .. } + | DocumentOp::BulkUpdate { collection, .. } + | DocumentOp::BulkDelete { collection, .. } + // A balance write stages like any other point write: one target row, + // one absolute post-image, keyed by the row's own surrogate. + | DocumentOp::ApplyBalanceDelta { collection, .. } => { + collections.insert(collection.to_string()); + Ok(()) + } + // `INSERT ... SELECT` stages the copied rows into the target collection. + DocumentOp::InsertSelect { + target_collection, .. + } => { + collections.insert(target_collection.to_string()); + Ok(()) + } + + // Read-only families: scans, lookups, point-gets, and estimates carry + // no persisted post-image. + DocumentOp::ResolveWrite(_) + | DocumentOp::PointGet { .. } + | DocumentOp::Scan { .. } + | DocumentOp::RangeScan { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::IndexedFetch { .. } + | DocumentOp::EstimateCount { .. } + | DocumentOp::MaterializeScan { .. } => Ok(()), + + // Resolve-before-propose is an autocommit path: never staged into an + // overlay, so no row-level redo shape carries it. + DocumentOp::ResolvedWrite { .. } => Err(crate::Error::PlanError { + detail: "document resolved write is not supported in transaction resolve".to_string(), + }), + + // Join/merge have no per-surrogate post-image; `BatchInsert` rides the + // buffered-plan path. None is staged, so rejecting avoids a lossy redo. + DocumentOp::UpdateFromJoin { .. } + | DocumentOp::Merge { .. } + | DocumentOp::BatchInsert { .. } => Err(crate::Error::PlanError { + detail: "document join/merge/batch DML has no staged post-image and is not \ + supported in transaction resolve" + .to_string(), + }), + + // Truncate: staged as an overlay marker; the serializer emits a + // `Delete` per removed base row ahead of the collection's overlay + // entries. + DocumentOp::Truncate { collection, .. } => { + collections.insert(collection.to_string()); + Ok(()) + } + + // Index / DDL: never stageable into the overlay, so no row-level + // redo shape carries them. + DocumentOp::Register { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. } => Err(crate::Error::PlanError { + detail: "document index/DDL op is not supported in transaction resolve".to_string(), + }), + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/columnar.rs b/nodedb/src/data/executor/handlers/transaction/resolve/columnar.rs deleted file mode 100644 index 5fc2c986b..000000000 --- a/nodedb/src/data/executor/handlers/transaction/resolve/columnar.rs +++ /dev/null @@ -1,195 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Columnar + timeseries serializer for transaction resolve. Plan-driven, -//! like the vector serializer: these ops ride the buffered-plan path, not a -//! per-surrogate overlay, so this reads the plan node directly and reuses the -//! autocommit path's `RecordType::TimeseriesBatch` encoders -//! (`control::server::wal_dispatch`). Columnar and timeseries share that one -//! record type, disambiguated on replay by payload shape: a map (`kind = -//! "columnar"`/`"columnar_dml"`) vs. the timeseries 5-tuple. Predicate DML -//! (`Update`/`Delete`) uses the same `ColumnarDmlWalRecord` the autocommit -//! path appends, so an in-tx UPDATE/DELETE is restart-durable identically. -//! Emission is in plan order, already deterministic. - -use nodedb_physical::physical_plan::{ColumnarOp, TimeseriesOp}; -use nodedb_wal::record::RecordType; - -use crate::control::server::wal_dispatch::{ - encode_columnar_batch_payload, encode_columnar_dml_payload, - encode_columnar_resolved_dml_payload, encode_columnar_truncate_payload, - encode_timeseries_batch_payload_with_format, -}; -use crate::wal::RedoSubRecord; - -/// Append the redo sub-record for a single columnar plan op to `ops`. -/// `Insert` tags `"columnar"`; predicate DML tags `"columnar_dml"`; reads -/// emit nothing (see module docs). -pub(super) fn serialize_columnar_op( - op: &ColumnarOp, - ops: &mut Vec, -) -> crate::Result<()> { - match op { - ColumnarOp::Insert { - collection, - payload, - format: _, - intent: _, - on_conflict_updates: _, - surrogates, - schema_bytes: _, - provenance, - wal_lsn: _, - rls_write_check: _, - // The redo record carries the row image, not the response shape a - // projection and its read gate would have produced for one caller. - returning: _, - rls_filters: _, - } => { - let sub_payload = encode_columnar_batch_payload( - collection.as_str(), - payload, - provenance.as_ref(), - surrogates, - )?; - ops.push(RedoSubRecord { - record_type: RecordType::TimeseriesBatch as u32, - payload: sub_payload, - }); - Ok(()) - } - - // Read families: no persisted post-image. `ResolveDml` mutates - // nothing, so it emits no redo sub-record either. - ColumnarOp::Scan { .. } - | ColumnarOp::MaterializeScan { .. } - | ColumnarOp::ResolveDml { .. } => Ok(()), - - // Same `ColumnarDmlWalRecord` the autocommit path appends; replay - // re-executes the predicate through the live handler, so an in-tx - // UPDATE/DELETE is restart-durable exactly like its autocommit twin. - ColumnarOp::Update { - collection, - filters, - updates, - rls_write_check: _, - } => { - let sub_payload = - encode_columnar_dml_payload(collection.as_str(), true, filters, updates)?; - ops.push(RedoSubRecord { - record_type: RecordType::TimeseriesBatch as u32, - payload: sub_payload, - }); - Ok(()) - } - ColumnarOp::Delete { - collection, - filters, - rls_write_check: _, - } => { - let sub_payload = - encode_columnar_dml_payload(collection.as_str(), false, filters, &[])?; - ops.push(RedoSubRecord { - record_type: RecordType::TimeseriesBatch as u32, - payload: sub_payload, - }); - Ok(()) - } - - // Control Plane already resolved these rows, so the redo carries the - // exact images, never a predicate — same as the autocommit encoder. - ColumnarOp::ResolvedUpdate { - collection, - rows, - rls_write_check: _, - } => { - let sub_payload = - encode_columnar_resolved_dml_payload(collection.as_str(), true, rows, &[])?; - ops.push(RedoSubRecord { - record_type: RecordType::TimeseriesBatch as u32, - payload: sub_payload, - }); - Ok(()) - } - ColumnarOp::ResolvedDelete { - collection, - pks, - rls_write_check: _, - } => { - let sub_payload = - encode_columnar_resolved_dml_payload(collection.as_str(), false, &[], pks)?; - ops.push(RedoSubRecord { - record_type: RecordType::TimeseriesBatch as u32, - payload: sub_payload, - }); - Ok(()) - } - // Same record the autocommit path appends (`RecordType::ColumnarTruncate`), - // replayed via `replay_columnar_truncate`. `restart_identity` is a - // Control-Plane sequence concern and never enters the redo record. - ColumnarOp::Truncate { - collection, - restart_identity: _, - } => { - let sub_payload = encode_columnar_truncate_payload(collection.as_str())?; - ops.push(RedoSubRecord { - record_type: RecordType::ColumnarTruncate as u32, - payload: sub_payload, - }); - Ok(()) - } - } -} - -/// Append the redo sub-record for a single timeseries plan op to `ops`. -/// `Ingest` tags `"timeseries"`; the scan op emits nothing. -pub(super) fn serialize_timeseries_op( - op: &TimeseriesOp, - ops: &mut Vec, -) -> crate::Result<()> { - match op { - TimeseriesOp::Ingest { - collection, - payload, - format, - wal_lsn: _, - surrogates: _, - provenance, - rls_write_check: _, - // Redo carries the ingested payload, not one caller's projected - // response shape — replay reconstructs state, nothing else. - returning: _, - rls_filters: _, - } => { - let sub_payload = encode_timeseries_batch_payload_with_format( - collection.as_str(), - payload, - provenance.as_ref(), - format, - )?; - ops.push(RedoSubRecord { - record_type: RecordType::TimeseriesBatch as u32, - payload: sub_payload, - }); - Ok(()) - } - - // Same record the autocommit path appends - // (`RecordType::TimeseriesTruncate`), replayed via - // `replay_timeseries_truncate`. - TimeseriesOp::Truncate { - collection, - restart_identity: _, - } => { - let sub_payload = encode_columnar_truncate_payload(collection.as_str())?; - ops.push(RedoSubRecord { - record_type: RecordType::TimeseriesTruncate as u32, - payload: sub_payload, - }); - Ok(()) - } - - // Read family: no persisted post-image. The resolve pass is read-only - // too — the ingest it reports is proposed as its own plan. - TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_) => Ok(()), - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/columnar_image.rs b/nodedb/src/data/executor/handlers/transaction/resolve/columnar_image.rs new file mode 100644 index 000000000..d378f0ae8 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/resolve/columnar_image.rs @@ -0,0 +1,317 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Columnar serializer for transaction resolve. Overlay-driven. +//! +//! Every columnar write a transaction runs is staged per surrogate: a plain +//! INSERT and an `ON CONFLICT DO UPDATE` stage the row as it will exist, an +//! UPDATE stages its post-image, and a DELETE stages a tombstone. That staged +//! image is what the transaction's own reads showed. The redo record carries +//! it verbatim as one `columnar_image` record per collection +//! (`ColumnarImageWalRecord`). Replay installs the image and never re-runs a +//! merge, a SET list or a predicate against the replaying node's state. +//! +//! * A staged put → a row with its image. When the statement that first +//! staged the surrogate matched a base row (an UPDATE), the row also names +//! that base row's primary key, and replay removes the base row before it +//! installs the image. A key-changing UPDATE therefore leaves no row behind +//! under the old key. +//! * A staged tombstone → a row with the base key and no image. A tombstone +//! for a row the transaction itself inserted names no base row and emits +//! nothing. +//! * A staged TRUNCATE → the `ColumnarTruncate` record ahead of the rows. +//! The truncate removes every base row, so rows after it carry no base key. +//! +//! Rows are emitted in surrogate order, so two resolves of one transaction +//! produce byte-identical records. +//! +//! Session and Calvin transactions both stage every columnar write, so both +//! resolve here. + +use std::collections::BTreeMap; + +use nodedb_physical::physical_plan::ColumnarOp; +use nodedb_types::columnar::{ + COLUMNAR_IMAGE_KIND, ColumnarImageWalRecord, ColumnarImageWalRow, ColumnarSchema, +}; +use nodedb_types::value::Value; +use nodedb_wal::record::RecordType; + +use crate::control::server::wal_dispatch::encode_columnar_truncate_payload; +use crate::data::executor::handlers::columnar_write::row_values_to_object; +use crate::data::executor::handlers::transaction::overlay::{ + Staged, TxnOverlay, decode_staged_row, +}; +use crate::types::{DatabaseId, TenantId}; +use crate::wal::RedoSubRecord; + +/// The columnar collections a transaction wrote, each with the catalog schema +/// its INSERT plans carried (empty when none carried one). +pub(super) type ColumnarCollections = BTreeMap>; + +/// Collect the collection of every columnar write in `op`. Reads contribute +/// nothing. A sync-provenance INSERT is refused: the sync ingest path applies +/// it outside any transaction, and a transaction's image record carries no +/// sync high-water mark. +pub(super) fn classify_columnar_op( + op: &ColumnarOp, + collections: &mut ColumnarCollections, +) -> crate::Result<()> { + match op { + ColumnarOp::Insert { + collection, + schema_bytes, + provenance, + .. + } => { + if provenance.is_some() { + return Err(crate::Error::PlanError { + detail: format!( + "columnar insert into '{collection}' carries sync provenance, which \ + the sync ingest path applies outside any transaction" + ), + }); + } + let entry = collections.entry(collection.to_string()).or_default(); + if entry.is_empty() { + entry.clone_from(schema_bytes); + } + Ok(()) + } + ColumnarOp::Update { collection, .. } + | ColumnarOp::Delete { collection, .. } + | ColumnarOp::ResolvedUpdate { collection, .. } + | ColumnarOp::ResolvedDelete { collection, .. } + | ColumnarOp::Truncate { collection, .. } => { + collections.entry(collection.to_string()).or_default(); + Ok(()) + } + ColumnarOp::Scan { .. } + | ColumnarOp::MaterializeScan { .. } + | ColumnarOp::ResolveDml { .. } => Ok(()), + } +} + +/// What [`serialize_columnar_collection`] reads for one collection. +pub(super) struct ColumnarCollectionImages<'a> { + pub overlay: &'a TxnOverlay, + pub coll_key: &'a (DatabaseId, TenantId, String), + /// The collection's engine schema. `None` when no engine exists on this + /// core, which is legal only when the transaction staged no put. + pub schema: Option<&'a ColumnarSchema>, + pub schema_bytes: &'a [u8], +} + +/// Append the truncate record (when the transaction truncated the +/// collection) and the collection's `columnar_image` record to `ops`. +pub(super) fn serialize_columnar_collection( + params: ColumnarCollectionImages<'_>, + ops: &mut Vec, +) -> crate::Result<()> { + let ColumnarCollectionImages { + overlay, + coll_key, + schema, + schema_bytes, + } = params; + let collection = coll_key.2.as_str(); + let truncated = overlay.is_truncated(coll_key); + if truncated { + ops.push(RedoSubRecord { + record_type: RecordType::ColumnarTruncate as u32, + payload: encode_columnar_truncate_payload(collection)?, + }); + } + + let entries: BTreeMap = overlay.iter_for_collection(coll_key).collect(); + let mut rows = Vec::with_capacity(entries.len()); + for (surrogate, staged) in entries { + let prior_pk_msgpack = if truncated { + Vec::new() + } else { + overlay + .base_pk(coll_key, surrogate) + .map(<[u8]>::to_vec) + .unwrap_or_default() + }; + let image_msgpack = match staged { + Staged::Put(body) => staged_image(collection, schema, surrogate, body)?, + Staged::Tombstone if prior_pk_msgpack.is_empty() => continue, + Staged::Tombstone => Vec::new(), + }; + rows.push(ColumnarImageWalRow { + surrogate, + prior_pk_msgpack, + image_msgpack, + }); + } + if rows.is_empty() { + return Ok(()); + } + + let record = ColumnarImageWalRecord { + kind: COLUMNAR_IMAGE_KIND.to_string(), + collection: collection.to_string(), + schema_bytes: schema_bytes.to_vec(), + rows, + }; + let payload = zerompk::to_msgpack_vec(&record).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("columnar image record for '{collection}': {e}"), + })?; + ops.push(RedoSubRecord { + record_type: RecordType::TimeseriesBatch as u32, + payload, + }); + Ok(()) +} + +/// The staged row body as a column-name object, MessagePack-encoded. +fn staged_image( + collection: &str, + schema: Option<&ColumnarSchema>, + surrogate: u32, + body: &[u8], +) -> crate::Result> { + let schema = schema.ok_or_else(|| crate::Error::Internal { + detail: format!( + "columnar resolve: '{collection}' has a staged row (surrogate {surrogate}) but no \ + engine schema on this core" + ), + })?; + let values = decode_staged_row(body).ok_or_else(|| crate::Error::Internal { + detail: format!( + "columnar resolve: staged row of '{collection}' (surrogate {surrogate}) does not \ + decode" + ), + })?; + if values.len() != schema.columns.len() { + return Err(crate::Error::Internal { + detail: format!( + "columnar resolve: staged row of '{collection}' (surrogate {surrogate}) has {} \ + values for {} columns", + values.len(), + schema.columns.len() + ), + }); + } + let image: Value = row_values_to_object(schema, &values); + nodedb_types::value_to_msgpack(&image).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("columnar resolve image of '{collection}': {e}"), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::RowIdentity; + use nodedb_types::columnar::{ColumnDef, ColumnType}; + + fn coll_key() -> (DatabaseId, TenantId, String) { + (DatabaseId::DEFAULT, TenantId::new(1), "m".to_string()) + } + + fn schema() -> ColumnarSchema { + ColumnarSchema::new(vec![ + ColumnDef::required("id", ColumnType::String).with_primary_key(), + ColumnDef::nullable("v", ColumnType::Int64), + ]) + .expect("valid schema") + } + + fn body(id: &str, v: i64) -> Vec { + nodedb_types::value_to_msgpack(&Value::Array(vec![ + Value::String(id.into()), + Value::Integer(v), + ])) + .expect("encode staged row") + } + + fn decode(op: &RedoSubRecord) -> ColumnarImageWalRecord { + zerompk::from_msgpack(&op.payload).expect("decode image record") + } + + fn serialize(overlay: &TxnOverlay) -> Vec { + let schema = schema(); + let mut ops = Vec::new(); + serialize_columnar_collection( + ColumnarCollectionImages { + overlay, + coll_key: &coll_key(), + schema: Some(&schema), + schema_bytes: &[], + }, + &mut ops, + ) + .expect("serialize"); + ops + } + + #[test] + fn a_staged_put_carries_the_image_the_transaction_was_shown() { + let mut overlay = TxnOverlay::new(); + overlay.insert_put( + coll_key(), + 5, + &RowIdentity::from_user_key("a"), + body("a", 7), + ); + let ops = serialize(&overlay); + assert_eq!(ops.len(), 1); + let record = decode(&ops[0]); + assert_eq!(record.kind, COLUMNAR_IMAGE_KIND); + assert_eq!(record.rows.len(), 1); + assert!(record.rows[0].prior_pk_msgpack.is_empty()); + let image = + nodedb_types::value_from_msgpack(&record.rows[0].image_msgpack).expect("image decodes"); + let Value::Object(map) = image else { + panic!("image must be an object"); + }; + assert_eq!(map.get("v"), Some(&Value::Integer(7))); + assert_eq!(map.get("id"), Some(&Value::String("a".into()))); + } + + #[test] + fn a_tombstone_of_a_base_row_names_its_key_and_an_own_insert_emits_nothing() { + let mut overlay = TxnOverlay::new(); + let key = nodedb_types::value_to_msgpack(&Value::String("b".into())).expect("key"); + overlay.note_base_pk(&coll_key(), 6, key.clone()); + overlay.insert_tombstone(coll_key(), 6, &RowIdentity::from_user_key("b")); + overlay.insert_tombstone(coll_key(), 9, &RowIdentity::from_user_key("own")); + let record = decode(&serialize(&overlay)[0]); + assert_eq!(record.rows.len(), 1); + assert_eq!(record.rows[0].surrogate, 6); + assert_eq!(record.rows[0].prior_pk_msgpack, key); + assert!(record.rows[0].image_msgpack.is_empty()); + } + + #[test] + fn a_truncate_precedes_the_rows_and_drops_their_base_keys() { + let mut overlay = TxnOverlay::new(); + let key = nodedb_types::value_to_msgpack(&Value::String("c".into())).expect("key"); + overlay.note_base_pk(&coll_key(), 4, key); + overlay.mark_truncated(coll_key()); + overlay.insert_put( + coll_key(), + 4, + &RowIdentity::from_user_key("c"), + body("c", 1), + ); + let ops = serialize(&overlay); + assert_eq!(ops.len(), 2); + assert_eq!(ops[0].record_type, RecordType::ColumnarTruncate as u32); + let record = decode(&ops[1]); + assert!(record.rows[0].prior_pk_msgpack.is_empty()); + } + + #[test] + fn rows_emit_in_surrogate_order() { + let mut overlay = TxnOverlay::new(); + for (s, id) in [(30, "c"), (10, "a"), (20, "b")] { + overlay.insert_put(coll_key(), s, &RowIdentity::from_user_key(id), body(id, 1)); + } + let record = decode(&serialize(&overlay)[0]); + let order: Vec = record.rows.iter().map(|r| r.surrogate).collect(); + assert_eq!(order, vec![10, 20, 30]); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/crdt.rs b/nodedb/src/data/executor/handlers/transaction/resolve/crdt.rs new file mode 100644 index 000000000..b9e785afc --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/resolve/crdt.rs @@ -0,0 +1,80 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! CRDT serializer for transaction resolve. +//! +//! **Plan-driven.** A CRDT write buffered inside a transaction is an intent — +//! a document-row upsert or delete, a block-list mutation, a snapshot import — +//! that the live handler re-executes against the collection's Loro state. The +//! sub-record is the exact record the autocommit path journals for the same op +//! (`wal_dispatch::encode_crdt_op_record`), so the CRDT replay arm re-executes +//! it on restart and on every replica applying the committed record. +//! +//! A raw `Apply` / `ApplyAuthenticated` delta is refused inside a transaction +//! before it is buffered; reaching resolve is a typed error, never a silent +//! drop. Reads, constraint installs, policy changes and history compaction +//! emit nothing. + +use nodedb_physical::physical_plan::CrdtOp; + +use crate::control::server::wal_dispatch::encode_crdt_op_record; +use crate::wal::RedoSubRecord; + +/// Append the redo sub-record for a single CRDT plan op to `ops`. +pub(super) fn serialize_crdt_op(op: &CrdtOp, ops: &mut Vec) -> crate::Result<()> { + if matches!(op, CrdtOp::Apply { .. } | CrdtOp::ApplyAuthenticated { .. }) { + return Err(crate::Error::PlanError { + detail: "CRDT Apply is not supported inside a transaction".to_string(), + }); + } + if let Some((kind, payload)) = encode_crdt_op_record(op)? { + ops.push(RedoSubRecord { + record_type: kind.record_type() as u32, + payload, + }); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_physical::physical_plan::CrdtWriteVerb; + use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate}; + use nodedb_wal::record::RecordType; + + #[test] + fn doc_upsert_resolves_to_a_crdt_doc_op_sub_record() { + let op = CrdtOp::DocUpsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), + document_id: "n1".to_string(), + fields_json: r#"{"title":"a"}"#.to_string(), + surrogate: Surrogate::new(5), + partial: false, + verb: CrdtWriteVerb::Insert, + returning: None, + rls_filters: Vec::new(), + }; + let mut ops = Vec::new(); + serialize_crdt_op(&op, &mut ops).expect("resolve crdt doc upsert"); + assert_eq!(ops.len(), 1); + assert_eq!(ops[0].record_type, RecordType::CrdtDocOp as u32); + } + + #[test] + fn raw_delta_apply_is_refused() { + let op = CrdtOp::Apply { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), + document_id: "n1".to_string(), + delta: vec![1], + peer_id: 1, + mutation_id: 1, + surrogate: Surrogate::new(5), + provenance: None, + constraint_version_required: 0, + expected_frontier_digest: None, + }; + let mut ops = Vec::new(); + assert!(serialize_crdt_op(&op, &mut ops).is_err()); + assert!(ops.is_empty()); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/document.rs b/nodedb/src/data/executor/handlers/transaction/resolve/document.rs index dd20c6cb0..f2e4caa71 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/document.rs @@ -16,7 +16,11 @@ //! * A staged tombstone ([`Staged::Tombstone`]) → `RecordType::Delete`, //! `(collection, document_id, Option, surrogate)`. The redo //! delete shape carries the surrogate (unlike the autocommit delete shape) -//! because replay keys redb by `StorageKey::for_surrogate(surrogate)`. +//! because replay keys redb by `StorageKey::for_surrogate(surrogate)`. For a +//! `bitemporal=true` collection it becomes a 5-tuple that appends the +//! resolve-time system time, so every apply of the record writes its +//! tombstone at the same version key. The replay decoder distinguishes the +//! two forms by arity. //! //! ## Stored form vs replay input //! @@ -41,11 +45,10 @@ //! //! ## Staged TRUNCATE //! -//! A truncated collection with vector fields additionally emits one -//! `RecordType::Delete` per base row the truncate removes -//! ([`serialize_truncated_base_rows`]), the same per-row redo the autocommit -//! truncate mints from its write-set. Base rows superseded by an overlay -//! entry are covered by that entry instead. +//! A truncated collection additionally emits one `RecordType::Delete` per base +//! row the truncate removes ([`serialize_truncated_base_rows`]): the redo +//! record is all a replica installs, so every removed row travels in it. Base +//! rows superseded by an overlay entry are covered by that entry instead. //! //! ## Determinism //! @@ -143,25 +146,38 @@ pub(super) fn serialize_document_collection( payload, }); } - Staged::Tombstone => ops.push(delete_sub_record(collection, doc_id, surrogate)?), + Staged::Tombstone => ops.push(delete_sub_record( + collection, + doc_id, + surrogate, + overlay + .get_bitemporal(coll_key, surrogate) + .map(|stamp| stamp.sys_from_ms), + )?), } } Ok(()) } /// The `RecordType::Delete` redo sub-record for one row: -/// `(collection, document_id, Option, surrogate)`. +/// `(collection, document_id, Option, surrogate)`, with the +/// tombstone's system time appended when `sys_from_ms` is `Some` (a +/// `bitemporal=true` collection). fn delete_sub_record( collection: &str, doc_id: &RowIdentity, surrogate: u32, + sys_from_ms: Option, ) -> crate::Result { let prov: Option = None; - let payload = zerompk::to_msgpack_vec(&(collection, doc_id.as_str(), prov, surrogate)) - .map_err(|e| crate::Error::Serialization { - format: "msgpack".into(), - detail: format!("document resolve delete: {e}"), - })?; + let payload = match sys_from_ms { + Some(sys) => zerompk::to_msgpack_vec(&(collection, doc_id.as_str(), prov, surrogate, sys)), + None => zerompk::to_msgpack_vec(&(collection, doc_id.as_str(), prov, surrogate)), + } + .map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("document resolve delete: {e}"), + })?; Ok(RedoSubRecord { record_type: RecordType::Delete as u32, payload, @@ -177,13 +193,15 @@ pub(super) struct TruncatedBaseRows<'a> { pub strict_schema: Option<&'a StrictSchema>, /// The collection's declared `PRIMARY KEY` column, from the plan. pub declared_primary_key: Option<&'a str>, + /// The truncate's system time on a `bitemporal=true` collection. Every + /// removed row's tombstone carries it. + pub sys_from_ms: Option, } /// Append one `RecordType::Delete` redo sub-record per base row a staged -/// TRUNCATE removes, in deterministic doc-id order. Mirrors the per-row -/// `Delete` redo the autocommit truncate mints from `Response::write_set` on -/// a collection with vector fields, so a WAL-only restart does not replay -/// each row's original `Put` and resurrect its HNSW vector. +/// TRUNCATE removes, in deterministic doc-id order. Every replica removes the +/// same rows from the record, and a WAL-only restart does not replay a +/// removed row's original `Put` and resurrect it or its HNSW vector. pub(super) fn serialize_truncated_base_rows( params: TruncatedBaseRows<'_>, ops: &mut Vec, @@ -193,6 +211,7 @@ pub(super) fn serialize_truncated_base_rows( rows, strict_schema, declared_primary_key, + sys_from_ms, } = params; let mut entries: BTreeMap = BTreeMap::new(); for (key, body) in rows { @@ -200,7 +219,12 @@ pub(super) fn serialize_truncated_base_rows( entries.insert(identity, key.surrogate().as_u32()); } for (doc_id, surrogate) in &entries { - ops.push(delete_sub_record(collection, doc_id, *surrogate)?); + ops.push(delete_sub_record( + collection, + doc_id, + *surrogate, + sys_from_ms, + )?); } Ok(()) } @@ -326,6 +350,35 @@ mod tests { assert_eq!(surrogate, 11, "delete tuple must carry the surrogate"); } + #[test] + fn a_bitemporal_tombstone_carries_its_resolve_time_system_time() { + use crate::data::executor::handlers::transaction::overlay::BitemporalStamp; + + let mut overlay = TxnOverlay::new(); + overlay.insert_tombstone(coll_key("hist"), 11, &id("gone")); + overlay.set_bitemporal( + &coll_key("hist"), + 11, + BitemporalStamp { + sys_from_ms: 4_242, + valid_from_ms: i64::MIN, + valid_until_ms: i64::MAX, + }, + ); + + let mut ops = Vec::new(); + serialize_document_collection(&overlay, &coll_key("hist"), "hist", None, &mut ops) + .expect("serialize delete"); + let (_c, doc_id, _prov, surrogate, sys) = + zerompk::from_msgpack::<(String, String, Option, u32, i64)>( + &ops[0].payload, + ) + .expect("decode bitemporal delete tuple"); + assert_eq!(doc_id, "gone"); + assert_eq!(surrogate, 11); + assert_eq!(sys, 4_242, "the redo delete carries the resolve-time stamp"); + } + #[test] fn entries_emit_in_deterministic_doc_id_order() { let mut overlay = TxnOverlay::new(); diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index 56d9e420b..1b3c738be 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -2,30 +2,47 @@ //! `MetaOp::ResolveTxn`: turns a committing transaction's staged post-images //! into one replayable [`RedoRecord`], without mutating base. Overlay-driven -//! serializers (KV, Document, Graph) read post-images from the staging -//! overlay; plan-driven serializers (Vector, Array, Columnar, Timeseries, -//! Spatial) are unstaged and serialize from the plan node instead. +//! serializers (KV, Document, Graph, Columnar, vector-primary) read the +//! post-image the transaction was shown from the staging overlay, so replay +//! installs it verbatim. Plan-driven serializers (HNSW Vector, Array, +//! Timeseries, Spatial, CRDT, Text) carry absolute or append-only writes and +//! serialize from the plan node instead. use std::collections::{BTreeMap, BTreeSet}; -use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; use nodedb_types::RowIdentity; use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::handlers::transaction::overlay::{BitemporalStamp, Staged}; +use crate::data::executor::handlers::transaction::overlay::BitemporalStamp; use crate::data::executor::handlers::transaction::stage_write::GRAPH_LABEL_COLL_KEY; use crate::data::executor::task::ExecutionTask; use crate::types::{TenantId, TxnId}; use crate::wal::{RedoRecord, RedoSubRecord}; +use super::classify::{classify_document_op, classify_kv_op}; +use super::columnar_image::{ColumnarCollectionImages, ColumnarCollections}; use super::graph::EdgeIdentityKey; -use super::{array, columnar, document, graph, kv, spatial, vector}; +use super::vector_direct::DirectWrites; +use super::vector_primary::VectorPrimaryCollections; +use super::{array, columnar_image, crdt, document, graph, kv, spatial, text, vector}; + +/// Which writes of a transaction its statements staged into the overlay. +#[derive(Clone, Copy, PartialEq, Eq)] +pub(in crate::data::executor) enum StagedWrites { + /// A session transaction: every write is staged at its statement. + Session, + /// A Calvin transaction: document, KV, graph, timeseries and columnar + /// writes are staged; vector-primary direct writes are not, so they + /// resolve from their plan nodes. + Calvin, +} impl CoreLoop { - /// Resolve a committing transaction's staged writes into a [`RedoRecord`] - /// and return its encoded bytes in the response payload. Reads the overlay - /// by `&` and never mutates any base engine. + /// Resolve a committing session transaction's staged writes into a + /// [`RedoRecord`] and return its encoded bytes in the response payload. + /// Reads the overlay by `&` and never mutates any base engine. pub(in crate::data::executor) fn execute_resolve_txn( &mut self, task: &ExecutionTask, @@ -33,7 +50,19 @@ impl CoreLoop { txn_id: TxnId, plans: &[PhysicalPlan], ) -> Response { - let ops = match self.resolve_txn_ops(task, tid, txn_id, plans) { + self.execute_resolve_staged(task, tid, txn_id, plans, StagedWrites::Session) + } + + /// Resolve a transaction whose staging follows `staged`. + pub(in crate::data::executor) fn execute_resolve_staged( + &mut self, + task: &ExecutionTask, + tid: u64, + txn_id: TxnId, + plans: &[PhysicalPlan], + staged: StagedWrites, + ) -> Response { + let ops = match self.resolve_txn_ops(task, tid, txn_id, plans, staged) { Ok(ops) => ops, Err(e) => return self.response_error(task, e), }; @@ -57,11 +86,18 @@ impl CoreLoop { tid: u64, txn_id: TxnId, plans: &[PhysicalPlan], + staged: StagedWrites, ) -> crate::Result> { let mut kv_collections: BTreeSet = BTreeSet::new(); let mut doc_collections: BTreeSet = BTreeSet::new(); let mut graph_collections: BTreeSet = BTreeSet::new(); let mut edge_surrogates: BTreeMap = BTreeMap::new(); + let mut columnar_collections: ColumnarCollections = BTreeMap::new(); + let mut vector_primary_collections: VectorPrimaryCollections = BTreeMap::new(); + let mut direct_writes = match staged { + StagedWrites::Session => DirectWrites::Staged(&mut vector_primary_collections), + StagedWrites::Calvin => DirectWrites::Plan, + }; // Plan-driven serializers emit into `ops` during this walk; overlay-driven // serializers only collect collections here, serialized in phase two below. @@ -86,25 +122,34 @@ impl CoreLoop { graph::classify_graph_op(op, &mut graph_collections, &mut edge_surrogates)? } - // CRDT deltas ride their own `CrdtDelta` WAL record, never redo - // sub-records (see `replay_transaction_redo_wal`). - PhysicalPlan::Crdt(_) => {} + // Plan-driven: a CRDT write resolves to the intent record its + // autocommit form journals. + PhysicalPlan::Crdt(op) => crdt::serialize_crdt_op(op, &mut ops)?, - // FTS postings are re-derived from the owning document at - // install time, so a text op contributes no redo sub-record. - PhysicalPlan::Text(_) => {} + // Plan-driven: an FTS write resolves to the posting record its + // autocommit form journals. + PhysicalPlan::Text(op) => text::serialize_text_op(op, &mut ops)?, // Read-only families: scans, joins, aggregates, exchange, and // maintenance ops carry no persisted post-image. PhysicalPlan::Query(_) | PhysicalPlan::Meta(_) => {} // Plan-driven: each op serializes from the plan node, skips, or - // errors. A vector-primary write's overlay entry serves the - // transaction's own reads only, never the redo record. - PhysicalPlan::Vector(op) => vector::serialize_vector_op(op, &mut ops)?, + // errors. A vector-primary direct write only registers its + // collection; its staged row is serialized from the overlay. + PhysicalPlan::Vector(op) => { + vector::serialize_vector_op(op, &mut ops, &mut direct_writes)? + } PhysicalPlan::Array(op) => array::serialize_array_op(op, &mut ops)?, - PhysicalPlan::Columnar(op) => columnar::serialize_columnar_op(op, &mut ops)?, - PhysicalPlan::Timeseries(op) => columnar::serialize_timeseries_op(op, &mut ops)?, + PhysicalPlan::Timeseries(op) => { + self.serialize_timeseries_op(task, tid, txn_id, op, &mut ops)? + } + + // Columnar: every write is staged per surrogate, so the image + // the transaction was shown is serialized from the overlay. + PhysicalPlan::Columnar(op) => { + columnar_image::classify_columnar_op(op, &mut columnar_collections)? + } // Spatial `Insert`/`Delete` plan nodes carry the complete absolute // post-image, so they serialize from the plan node, not an overlay. @@ -120,9 +165,21 @@ impl CoreLoop { } } - // Pin the resolve-time bitemporal stamp once, in the overlay sidecar, so - // redo and install read the same version key. Deterministic (collection, - // doc-id) order keeps replicas resolving the same txn in sync. + if staged == StagedWrites::Session { + let reads_overlay = !kv_collections.is_empty() + || !doc_collections.is_empty() + || !graph_collections.is_empty() + || !columnar_collections.is_empty() + || !vector_primary_collections.is_empty() + || plans.iter().any(graph::is_label_write); + self.require_staging_overlay(txn_id, reads_overlay)?; + } + + // Pin the resolve-time bitemporal stamp once, in the overlay sidecar, for + // every staged put AND tombstone, so the redo carries it and every apply + // (install, replica, restart) writes the same version key. Deterministic + // (collection, doc-id) order keeps replicas resolving the same txn in + // sync. for collection in &doc_collections { if !self.is_bitemporal(task.request.database_id.as_u64(), tid, collection) { continue; @@ -132,20 +189,19 @@ impl CoreLoop { TenantId::new(tid), collection.clone(), ); - let mut puts: Vec<(&RowIdentity, u32)> = match self.txn_overlays.get(&txn_id) { + let mut writes: Vec<(&RowIdentity, u32)> = match self.txn_overlays.get(&txn_id) { Some(overlay) => overlay .iter_doc_entries_for_collection(&coll_key) - .filter_map(|(doc_id, staged)| match staged { - Staged::Put(_) => overlay + .filter_map(|(doc_id, _staged)| { + overlay .surrogate_for_doc_id(&coll_key, doc_id) - .map(|surrogate| (doc_id, surrogate)), - Staged::Tombstone => None, + .map(|surrogate| (doc_id, surrogate)) }) .collect(), None => Vec::new(), }; - puts.sort(); - let stamps: Vec<(u32, BitemporalStamp)> = puts + writes.sort(); + let stamps: Vec<(u32, BitemporalStamp)> = writes .into_iter() .map(|(_doc_id, surrogate)| { ( @@ -198,6 +254,9 @@ impl CoreLoop { declared_primary_key: truncate_primary_keys .get(collection.as_str()) .and_then(|pk| pk.as_deref()), + sys_from_ms: self + .is_bitemporal(task.request.database_id.as_u64(), tid, collection) + .then(|| self.bitemporal_now_ms()), }, &mut ops, )?; @@ -211,6 +270,14 @@ impl CoreLoop { )?; } } + self.serialize_staged_image_collections( + task, + tid, + txn_id, + &columnar_collections, + &vector_primary_collections, + &mut ops, + )?; if let Some(graph_overlay) = self.graph_txn_overlays.get(&txn_id) { // Freeze temporal identity independently from the overlay's lease // refresh stamp. Resolve retries reuse the exact same ordinal. @@ -242,12 +309,71 @@ impl CoreLoop { Ok(ops) } - /// The base rows a staged TRUNCATE of `collection` removes at COMMIT and - /// that need a per-row `Delete` redo: every base row with no overlay - /// entry, on a collection with vector fields. Empty for a collection - /// without vector fields, where the autocommit truncate mints no per-row - /// redo either (row durability is redb-synchronous). Read-only against - /// base. + /// Refuse a session resolve whose writes live in a staging overlay this + /// core does not hold. + /// + /// Every core that stages a write opens the transaction's overlay, so a + /// missing overlay means the writes were staged on another core or the + /// overlay was reaped. Resolving would then commit none of them. + fn require_staging_overlay(&self, txn_id: TxnId, reads_overlay: bool) -> crate::Result<()> { + if !reads_overlay + || self.txn_overlays.contains_key(&txn_id) + || self.graph_txn_overlays.contains_key(&txn_id) + { + return Ok(()); + } + Err(crate::Error::Internal { + detail: format!( + "{txn_id} has staged writes but core {} holds no staging overlay for it; \ + the commit is refused", + self.core_id + ), + }) + } + + /// Serialize the columnar and vector-primary collections the + /// transaction wrote from its overlay, in collection order. + fn serialize_staged_image_collections( + &self, + task: &ExecutionTask, + tid: u64, + txn_id: TxnId, + columnar: &ColumnarCollections, + vector_primary: &VectorPrimaryCollections, + ops: &mut Vec, + ) -> crate::Result<()> { + let Some(overlay) = self.txn_overlays.get(&txn_id) else { + return Ok(()); + }; + let coll_key = |collection: &str| { + ( + task.request.database_id, + TenantId::new(tid), + collection.to_string(), + ) + }; + for (collection, schema_bytes) in columnar { + let key = coll_key(collection); + columnar_image::serialize_columnar_collection( + ColumnarCollectionImages { + overlay, + coll_key: &key, + schema: self.columnar_engines.get(&key).map(|e| e.schema()), + schema_bytes, + }, + ops, + )?; + } + for (collection, writes) in vector_primary { + self.serialize_vector_primary_collection(overlay, &coll_key(collection), writes, ops)?; + } + Ok(()) + } + + /// The base rows a staged TRUNCATE of `collection` removes at COMMIT: + /// every base row with no overlay entry. The redo record is the only + /// thing a replica installs, so each removed row travels as its own + /// `Delete`. Read-only against base. fn truncated_base_rows( &self, task: &ExecutionTask, @@ -257,9 +383,6 @@ impl CoreLoop { coll_key: &(crate::types::DatabaseId, TenantId, String), ) -> crate::Result)>> { let database_id = task.request.database_id.as_u64(); - if !self.collection_has_vectors(database_id, tid, collection) { - return Ok(Vec::new()); - } let bitemporal = self.is_bitemporal(database_id, tid, collection); let mut rows = Vec::new(); for key in self.scan_matching_documents(database_id, tid, collection, &[])? { @@ -309,164 +432,6 @@ fn truncate_declared_primary_keys(plans: &[PhysicalPlan]) -> BTreeMap) -> crate::Result<()> { - match op { - // Row-level writes: the resolved post-image (value or tombstone) is in - // the overlay, keyed by collection. - KvOp::Put { collection, .. } - | KvOp::Insert { collection, .. } - | KvOp::InsertIfAbsent { collection, .. } - | KvOp::InsertOnConflictUpdate { collection, .. } - | KvOp::Delete { collection, .. } - | KvOp::BatchPut { collection, .. } - | KvOp::Incr { collection, .. } - | KvOp::IncrFloat { collection, .. } - | KvOp::Cas { collection, .. } - | KvOp::GetSet { collection, .. } - | KvOp::FieldSet { collection, .. } - | KvOp::Transfer { collection, .. } - // Predicate DML stages each matched row's post-image or tombstone - // at statement time, so the overlay carries it like a keyed write. - | KvOp::PredicateUpdate { collection, .. } - | KvOp::PredicateDelete { collection, .. } => { - collections.insert(collection.to_string()); - Ok(()) - } - // `TransferItem` moves a row across collections: the source holds a - // staged tombstone and the destination a staged value. - KvOp::TransferItem { - source_collection, - dest_collection, - .. - } => { - collections.insert(source_collection.to_string()); - collections.insert(dest_collection.to_string()); - Ok(()) - } - - // Read-only: nothing staged, nothing to persist. - KvOp::Get { .. } - | KvOp::BatchGet { .. } - | KvOp::Scan { .. } - | KvOp::FieldGet { .. } - | KvOp::GetTtl { .. } - | KvOp::MaterializeScan { .. } - | KvOp::SortedIndexRank { .. } - | KvOp::SortedIndexTopK { .. } - | KvOp::SortedIndexRange { .. } - | KvOp::SortedIndexCount { .. } - | KvOp::SortedIndexScore { .. } - // Read-only: reports what a governed write would apply, stages - // nothing. - | KvOp::ResolveWrite(_) => Ok(()), - - // Resolve-before-propose is an autocommit path: never staged into an - // overlay, so no row-level redo shape carries it. - KvOp::ResolvedWrite { .. } => Err(crate::Error::PlanError { - detail: "kv resolved write is not supported in transaction resolve".to_string(), - }), - - // A standalone TTL delta has no value post-image, and KV redo carries - // TTL only as part of a value put, so rejecting avoids a silent drop. - KvOp::Expire { .. } | KvOp::Persist { .. } => Err(crate::Error::PlanError { - detail: "kv EXPIRE/PERSIST is not supported in transaction resolve".to_string(), - }), - - // Truncate: staged as an overlay marker; the serializer emits the - // `kv_truncate` redo ahead of the collection's row entries. - KvOp::Truncate { collection, .. } => { - collections.insert(collection.to_string()); - Ok(()) - } - - // Index / DDL: never stageable into the overlay, so no row-level - // redo shape carries them. - KvOp::RegisterIndex { .. } - | KvOp::DropIndex { .. } - | KvOp::RegisterSortedIndex { .. } - | KvOp::DropSortedIndex { .. } => Err(crate::Error::PlanError { - detail: "kv index/DDL op is not supported in transaction resolve".to_string(), - }), - } -} - -/// Classify a Document op for transaction resolve: collect the collection of a -/// staged point/bulk write into `collections`, skip read-only ops, and reject -/// the writes that leave no overlay post-image. -fn classify_document_op(op: &DocumentOp, collections: &mut BTreeSet) -> crate::Result<()> { - match op { - // Staged writes: the resolved post-image is in the overlay, keyed by the - // user primary key. RETURNING doesn't affect staging, so these serialize - // from the overlay like any other point/bulk write. - DocumentOp::PointPut { collection, .. } - | DocumentOp::PointInsert { collection, .. } - | DocumentOp::Upsert { collection, .. } - | DocumentOp::PointDelete { collection, .. } - | DocumentOp::PointUpdate { collection, .. } - | DocumentOp::BulkUpdate { collection, .. } - | DocumentOp::BulkDelete { collection, .. } - // A balance write stages like any other point write: one target row, - // one absolute post-image, keyed by the row's own surrogate. - | DocumentOp::ApplyBalanceDelta { collection, .. } => { - collections.insert(collection.to_string()); - Ok(()) - } - // `INSERT ... SELECT` stages the copied rows into the target collection. - DocumentOp::InsertSelect { - target_collection, .. - } => { - collections.insert(target_collection.to_string()); - Ok(()) - } - - // Read-only families: scans, lookups, point-gets, and estimates carry - // no persisted post-image. - DocumentOp::ResolveWrite(_) - | DocumentOp::PointGet { .. } - | DocumentOp::Scan { .. } - | DocumentOp::RangeScan { .. } - | DocumentOp::IndexLookup { .. } - | DocumentOp::IndexedFetch { .. } - | DocumentOp::EstimateCount { .. } - | DocumentOp::MaterializeScan { .. } => Ok(()), - - // Resolve-before-propose is an autocommit path: never staged into an - // overlay, so no row-level redo shape carries it. - DocumentOp::ResolvedWrite { .. } => Err(crate::Error::PlanError { - detail: "document resolved write is not supported in transaction resolve".to_string(), - }), - - // Join/merge have no per-surrogate post-image; `BatchInsert` rides the - // buffered-plan path. None is staged, so rejecting avoids a lossy redo. - DocumentOp::UpdateFromJoin { .. } - | DocumentOp::Merge { .. } - | DocumentOp::BatchInsert { .. } => Err(crate::Error::PlanError { - detail: "document join/merge/batch DML has no staged post-image and is not \ - supported in transaction resolve" - .to_string(), - }), - - // Truncate: staged as an overlay marker; the serializer emits a - // `Delete` per removed base row on a vector collection ahead of the - // collection's overlay entries. - DocumentOp::Truncate { collection, .. } => { - collections.insert(collection.to_string()); - Ok(()) - } - - // Index / DDL: never stageable into the overlay, so no row-level - // redo shape carries them. - DocumentOp::Register { .. } - | DocumentOp::DropIndex { .. } - | DocumentOp::BackfillIndex { .. } => Err(crate::Error::PlanError { - detail: "document index/DDL op is not supported in transaction resolve".to_string(), - }), - } -} - #[cfg(test)] mod tests { use std::sync::Arc; @@ -1202,6 +1167,11 @@ mod tests { }), ]; + // A vector-primary direct write resolves from the row its statement + // staged. + let staged = core.execute_stage_write(&make_stage_task(txn), TID, &plans[0]); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); + for plan in plans { let resp = core.execute_resolve_txn(&task, TID, txn, std::slice::from_ref(&plan)); assert_eq!( @@ -1256,6 +1226,8 @@ mod tests { on_conflict_updates: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::decided_earlier_in_request(), }); + let staged = src.execute_stage_write(&make_stage_task(TxnId::new(51)), TID, &plan); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); let resp = src.execute_resolve_txn(&task, TID, TxnId::new(51), std::slice::from_ref(&plan)); let redo = decode_redo(&resp); let record = wrap_redo(&redo); @@ -1278,6 +1250,131 @@ mod tests { ); } + fn vp_fields(note: &str, v: i64) -> Vec { + let mut fields = std::collections::HashMap::new(); + fields.insert("note".to_string(), nodedb_types::Value::String(note.into())); + fields.insert("v".to_string(), nodedb_types::Value::Integer(v)); + zerompk::to_msgpack_vec(&fields).expect("encode payload fields") + } + + fn vp_upsert_plan( + surrogate: u32, + payload: Vec, + on_conflict_updates: Vec<(String, UpdateValue)>, + ) -> PhysicalPlan { + PhysicalPlan::Vector(VectorOp::DirectUpsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vp"), + field: "emb".to_string(), + surrogate: Surrogate::new(surrogate), + pk_bytes: Vec::new(), + vector: vec![1.0, 0.0, 0.0], + payload, + quantization: nodedb_types::VectorQuantization::None, + storage_dtype: nodedb_types::VectorStorageDtype::F32, + payload_indexes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + on_conflict_updates, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + }) + } + + /// Commit a vector-primary base row on `core` through the live handler. + fn vp_live_upsert(core: &mut CoreLoop, surrogate: u32, payload: &[u8]) { + let task = make_task(); + let resp = core.execute_vector_direct_upsert( + crate::data::executor::handlers::vector_upsert::VectorDirectUpsertParams { + task: &task, + tid: TID, + collection: "vp", + field: "emb", + surrogate: Surrogate::new(surrogate), + vector: &[1.0, 0.0, 0.0], + payload, + quantization: nodedb_types::VectorQuantization::None, + storage_dtype: nodedb_types::VectorStorageDtype::F32, + payload_indexes: &[], + intent: nodedb_physical::physical_plan::VectorDirectWriteIntent::Upsert, + on_conflict_updates: &[], + rls_write_check: &nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: &[], + }, + ); + assert_eq!(resp.status, Status::Ok, "seed vector-primary row: {resp:?}"); + } + + /// A vector-primary `ON CONFLICT DO UPDATE` stages the merged sidecar; the + /// redo carries exactly that sidecar, and a replica holding the same base + /// row installs it verbatim instead of re-running the merge. + #[test] + fn vector_primary_on_conflict_redo_installs_the_merged_sidecar_the_transaction_saw() { + let (mut src, _src_dir) = make_core(); + let (mut dst, _dst_dir) = make_core(); + vp_live_upsert(&mut src, 31, &vp_fields("orig", 1)); + vp_live_upsert(&mut dst, 31, &vp_fields("orig", 1)); + + let txn = TxnId::new(53); + let upsert = vp_upsert_plan( + 31, + vp_fields("new", 7), + vec![( + "v".to_string(), + UpdateValue::Literal( + nodedb_types::value_to_msgpack(&nodedb_types::Value::Integer(7)) + .expect("encode literal"), + ), + )], + ); + let staged = src.execute_stage_write(&make_stage_task(txn), TID, &upsert); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); + let staged_sidecar = match src + .txn_overlays + .get(&txn) + .and_then(|overlay| overlay.get(&coll_key("vp"), 31)) + { + Some(Staged::Put(body)) => { + crate::data::executor::handlers::transaction::overlay::StagedVectorRow::from_bytes( + body, + ) + .expect("decode staged row") + .sidecar + } + other => panic!("the upsert must stage a row, got {other:?}"), + }; + + let redo = decode_redo(&src.execute_resolve_txn(&make_task(), TID, txn, &[upsert])); + assert_eq!(redo.ops.len(), 1); + assert_eq!( + redo.ops[0].record_type, + RecordType::VectorResolvedDirectWrite as u32 + ); + + dst.replay_transaction_redo_wal( + std::slice::from_ref(&wrap_redo(&redo)), + 1, + &nodedb_wal::TombstoneSet::new(), + ) + .expect("redo replay must succeed"); + let installed = dst + .vector_sidecar_bytes(DatabaseId::DEFAULT.as_u64(), TID, "vp", Surrogate::new(31)) + .expect("read sidecar") + .expect("the row exists on the replica"); + assert_eq!( + installed, staged_sidecar, + "the replica holds the sidecar the transaction was shown" + ); + let fields = + crate::data::executor::handlers::vector_upsert::decode_payload_lowercased(&installed) + .expect("decode sidecar"); + assert_eq!( + fields.get("note"), + Some(&nodedb_types::Value::String("orig".into())), + "the column the SET list left alone keeps its stored value" + ); + assert_eq!(fields.get("v"), Some(&nodedb_types::Value::Integer(7))); + } + /// Resolve a `MultiVectorInsert` and replay it through the redo path. #[test] fn resolved_multi_vector_insert_replays_into_fresh_engine() { @@ -2581,70 +2678,248 @@ mod tests { ); } - /// A columnar `Insert` plan resolves to a `TimeseriesBatch` sub-record whose - /// payload is a map-shaped `ColumnarWalRecord` (`kind: "columnar"`) and - /// replays into the columnar engine's memtable. - #[test] - fn columnar_insert_resolves_to_columnar_batch_and_replays() { - let (mut src, _src_dir) = make_core(); - let task = make_task(); - let txn = TxnId::new(42); + fn columnar_schema() -> nodedb_types::columnar::ColumnarSchema { + nodedb_types::columnar::ColumnarSchema::new(vec![ + ColumnDef::required("id", ColumnType::String).with_primary_key(), + ColumnDef::nullable("v", ColumnType::Int64), + ColumnDef::nullable("note", ColumnType::String), + ]) + .expect("valid columnar schema") + } + fn columnar_row(id: &str, v: i64, note: &str) -> nodedb_types::Value { let mut row = std::collections::HashMap::new(); - row.insert("a".to_string(), nodedb_types::Value::Integer(1)); - row.insert("b".to_string(), nodedb_types::Value::Integer(2)); - let payload = nodedb_types::value_to_msgpack(&nodedb_types::Value::Array(vec![ - nodedb_types::Value::Object(row), - ])) - .expect("encode columnar payload"); + row.insert("id".to_string(), nodedb_types::Value::String(id.into())); + row.insert("v".to_string(), nodedb_types::Value::Integer(v)); + row.insert("note".to_string(), nodedb_types::Value::String(note.into())); + nodedb_types::Value::Object(row) + } - let plan = PhysicalPlan::Columnar(ColumnarOp::Insert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cevents"), - payload, + /// A columnar INSERT of one row at `surrogate`. `on_conflict` non-empty + /// makes it the `ON CONFLICT (pk) DO UPDATE` shape. + fn columnar_insert_plan( + collection: &str, + row: nodedb_types::Value, + surrogate: u32, + on_conflict: Vec<(String, UpdateValue)>, + ) -> PhysicalPlan { + let intent = if on_conflict.is_empty() { + ColumnarInsertIntent::Insert + } else { + ColumnarInsertIntent::Put + }; + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, collection), + payload: nodedb_types::value_to_msgpack(&nodedb_types::Value::Array(vec![row])) + .expect("encode columnar payload"), format: "msgpack".to_string(), - intent: ColumnarInsertIntent::Insert, - on_conflict_updates: Vec::new(), - surrogates: Vec::new(), - schema_bytes: Vec::new(), + intent, + on_conflict_updates: on_conflict, + surrogates: vec![Surrogate::new(surrogate)], + schema_bytes: zerompk::to_msgpack_vec(&columnar_schema()).expect("encode schema"), provenance: None, wal_lsn: None, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), - }); - let resp = src.execute_resolve_txn(&task, TID, txn, &[plan]); - let redo = decode_redo(&resp); - assert_eq!(redo.ops.len(), 1, "one columnar insert -> one sub-record"); - assert_eq!(redo.ops[0].record_type, RecordType::TimeseriesBatch as u32); + }) + } - // The payload decodes as a `ColumnarWalRecord` with kind "columnar". - let rec = zerompk::from_msgpack::( - &redo.ops[0].payload, - ) - .expect("decode columnar wal record"); - assert_eq!(rec.kind, "columnar"); + /// Commit `plan` on `core` the way an autocommit write does. + fn columnar_base_insert(core: &mut CoreLoop, plan: &PhysicalPlan) { + let (collection, payload, surrogates, schema_bytes) = match plan { + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection, + payload, + surrogates, + schema_bytes, + .. + }) => (collection, payload, surrogates, schema_bytes), + other => panic!("expected a columnar insert, got {other:?}"), + }; + let resp = core.execute_columnar_insert( + &make_task(), + crate::data::executor::handlers::columnar_write::ColumnarInsertParams { + collection: collection.as_str(), + payload, + format: "msgpack", + intent: ColumnarInsertIntent::Insert, + on_conflict_updates: &[], + surrogates, + schema_bytes, + provenance: None, + rls_write_check: &nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: &[], + spatial_undo: None, + }, + ); + assert_eq!(resp.status, Status::Ok, "seed base row: {resp:?}"); + } + + /// Every live `(surrogate, row)` of `collection` on `core`, by surrogate. + fn columnar_rows(core: &CoreLoop, collection: &str) -> Vec<(u32, Vec)> { + let key = ( + DatabaseId::DEFAULT, + TenantId::new(TID), + collection.to_string(), + ); + let mut rows: Vec<(u32, Vec)> = core + .columnar_engines + .get(&key) + .map(|engine| { + engine + .scan_memtable_rows_with_surrogates() + .filter_map(|(s, row)| s.map(|s| (s.as_u32(), row))) + .collect() + }) + .unwrap_or_default(); + rows.sort_by_key(|(s, _)| *s); + rows + } + + /// A staged columnar INSERT resolves to one `columnar_image` record whose + /// row replays into a fresh engine under its surrogate. + #[test] + fn columnar_insert_resolves_to_its_staged_image_and_replays() { + let (mut src, _src_dir) = make_core(); + let txn = TxnId::new(42); + let plan = columnar_insert_plan("cevents", columnar_row("a", 1, "x"), 7, Vec::new()); + let staged = src.execute_stage_write(&make_stage_task(txn), TID, &plan); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); + + let redo = decode_redo(&src.execute_resolve_txn(&make_task(), TID, txn, &[plan])); + assert_eq!(redo.ops.len(), 1, "one staged columnar row -> one record"); + assert_eq!(redo.ops[0].record_type, RecordType::TimeseriesBatch as u32); + let rec: nodedb_types::columnar::ColumnarImageWalRecord = + zerompk::from_msgpack(&redo.ops[0].payload).expect("decode image record"); + assert_eq!(rec.kind, nodedb_types::columnar::COLUMNAR_IMAGE_KIND); assert_eq!(rec.collection, "cevents"); - let record = wrap_redo(&redo); let (mut dst, _dst_dir) = make_core(); dst.replay_transaction_redo_wal( - std::slice::from_ref(&record), + std::slice::from_ref(&wrap_redo(&redo)), 1, &nodedb_wal::TombstoneSet::new(), ) .expect("redo replay must succeed"); + let rows = columnar_rows(&dst, "cevents"); + assert_eq!(rows.len(), 1, "the staged row replays: {rows:?}"); + assert_eq!(rows[0].0, 7, "under its own surrogate"); + } - let key = ( - DatabaseId::DEFAULT, - TenantId::new(TID), - "cevents".to_string(), + /// An `ON CONFLICT DO UPDATE` stages the merged row; the redo carries + /// that merged row, and a replica holding the same base row installs + /// exactly it. Replaying the submitted row instead would lose the + /// columns the SET list left untouched. + #[test] + fn columnar_on_conflict_redo_installs_the_merged_row_the_transaction_saw() { + let base = columnar_insert_plan("upserts", columnar_row("a", 1, "orig"), 9, Vec::new()); + let (mut src, _src_dir) = make_core(); + let (mut dst, _dst_dir) = make_core(); + columnar_base_insert(&mut src, &base); + columnar_base_insert(&mut dst, &base); + + let txn = TxnId::new(48); + let upsert = columnar_insert_plan( + "upserts", + columnar_row("a", 7, "new"), + 9, + vec![( + "v".to_string(), + UpdateValue::Literal( + nodedb_types::value_to_msgpack(&nodedb_types::Value::Integer(7)) + .expect("encode literal"), + ), + )], ); + let staged = src.execute_stage_write(&make_stage_task(txn), TID, &upsert); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); + let redo = decode_redo(&src.execute_resolve_txn(&make_task(), TID, txn, &[upsert])); + + dst.replay_transaction_redo_wal( + std::slice::from_ref(&wrap_redo(&redo)), + 1, + &nodedb_wal::TombstoneSet::new(), + ) + .expect("redo replay must succeed"); + let rows = columnar_rows(&dst, "upserts"); + assert_eq!(rows.len(), 1, "the upsert replaces the row: {rows:?}"); assert_eq!( - dst.columnar_engines - .get(&key) - .map(|e| e.memtable().row_count()), - Some(1), - "columnar row must replay into the memtable" + rows[0].1, + vec![ + nodedb_types::Value::String("a".into()), + nodedb_types::Value::Integer(7), + nodedb_types::Value::String("orig".into()), + ], + "the replica holds the merged row: v from the SET list, note untouched" + ); + } + + /// A key-changing UPDATE and a DELETE of base rows resolve to images + /// that name the base row they remove, so no row survives under the old + /// key on a replica. + #[test] + fn columnar_update_and_delete_redo_remove_the_base_rows_they_replace() { + let seed_a = columnar_insert_plan("moves", columnar_row("a", 1, "x"), 3, Vec::new()); + let seed_b = columnar_insert_plan("moves", columnar_row("b", 2, "y"), 4, Vec::new()); + let (mut src, _src_dir) = make_core(); + let (mut dst, _dst_dir) = make_core(); + for core in [&mut src, &mut dst] { + columnar_base_insert(core, &seed_a); + columnar_base_insert(core, &seed_b); + } + + let txn = TxnId::new(49); + let pk_filter = |id: &str| { + zerompk::to_msgpack_vec(&vec![nodedb_query::scan_filter::ScanFilter { + field: "id".to_string(), + op: nodedb_query::scan_filter::FilterOp::Eq, + value: nodedb_types::Value::String(id.into()), + clauses: Vec::new(), + expr: None, + }]) + .expect("encode filter") + }; + let update = PhysicalPlan::Columnar(ColumnarOp::Update { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "moves"), + filters: pk_filter("a"), + updates: vec![( + "id".to_string(), + nodedb_types::value_to_msgpack(&nodedb_types::Value::String("z".into())) + .expect("encode assignment"), + )], + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + }); + let delete = PhysicalPlan::Columnar(ColumnarOp::Delete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "moves"), + filters: pk_filter("b"), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + }); + for plan in [&update, &delete] { + let staged = src.execute_stage_write(&make_stage_task(txn), TID, plan); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); + } + let redo = decode_redo(&src.execute_resolve_txn(&make_task(), TID, txn, &[update, delete])); + + dst.replay_transaction_redo_wal( + std::slice::from_ref(&wrap_redo(&redo)), + 1, + &nodedb_wal::TombstoneSet::new(), + ) + .expect("redo replay must succeed"); + let rows = columnar_rows(&dst, "moves"); + assert_eq!( + rows, + vec![( + 3, + vec![ + nodedb_types::Value::String("z".into()), + nodedb_types::Value::Integer(1), + nodedb_types::Value::String("x".into()), + ] + )], + "the renamed row lives under its new key only, and the deleted row is gone" ); } @@ -2656,19 +2931,11 @@ mod tests { let task = make_task(); let txn = TxnId::new(43); - // A `TimeseriesWalBatch` is what `replay_timeseries_payload` decodes to - // ingest samples directly into the memtable. - let batch = nodedb_types::timeseries::TimeseriesWalBatch { - collection: "metrics".to_string(), - samples: vec![(11u64, 1_700_000_000_000i64, 42.0f64)], - provenance: None, - }; - let payload = zerompk::to_msgpack_vec(&batch).expect("encode ts batch"); - + // A line-protocol ingest: the format a timeseries INSERT stages. let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), - payload, - format: "samples".to_string(), + payload: b"metrics value=42 1700000000000000000".to_vec(), + format: "ilp".to_string(), wal_lsn: None, surrogates: Vec::new(), provenance: None, @@ -2683,14 +2950,13 @@ mod tests { // The payload is the format-preserving 5-element tuple tagged // "timeseries" (a msgpack array), distinct from the columnar map form. - let (kind, collection, _payload, _prov, format) = + let (kind, collection, _payload, _prov, _format) = zerompk::from_msgpack::<(String, String, Vec, Option, String)>( &redo.ops[0].payload, ) .expect("decode timeseries 5-tuple"); assert_eq!(kind, "timeseries"); assert_eq!(collection, "metrics"); - assert_eq!(format, "samples"); let record = wrap_redo(&redo); let (mut dst, _dst_dir) = make_core(); @@ -2778,49 +3044,6 @@ mod tests { assert_eq!(dictionary.1.get_id("west,1"), Some(0)); } - /// `Columnar::Update`/`Delete` are predicate DML: resolve emits the same - /// `columnar_dml` sub-record the autocommit path appends. - #[test] - fn columnar_update_and_delete_emit_columnar_dml_sub_record() { - use nodedb_types::columnar::ColumnarDmlWalRecord; - - let (mut core, _dir) = make_core(); - let task = make_task(); - - let update = PhysicalPlan::Columnar(ColumnarOp::Update { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cevents"), - filters: Vec::new(), - updates: vec![("a".to_string(), vec![1, 2, 3])], - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - }); - let resp = core.execute_resolve_txn(&task, TID, TxnId::new(44), &[update]); - assert_eq!(resp.status, Status::Ok); - let redo = decode_redo(&resp); - assert_eq!(redo.ops.len(), 1); - assert_eq!(redo.ops[0].record_type, RecordType::TimeseriesBatch as u32); - let rec: ColumnarDmlWalRecord = - zerompk::from_msgpack(&redo.ops[0].payload).expect("decode columnar_dml"); - assert_eq!(rec.kind, "columnar_dml"); - assert_eq!(rec.collection, "cevents"); - assert!(rec.is_update, "UPDATE must carry is_update = true"); - assert_eq!(rec.updates, vec![("a".to_string(), vec![1, 2, 3])]); - - let delete = PhysicalPlan::Columnar(ColumnarOp::Delete { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cevents"), - filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - }); - let resp = core.execute_resolve_txn(&task, TID, TxnId::new(45), &[delete]); - assert_eq!(resp.status, Status::Ok); - let redo = decode_redo(&resp); - assert_eq!(redo.ops.len(), 1); - let rec: ColumnarDmlWalRecord = - zerompk::from_msgpack(&redo.ops[0].payload).expect("decode columnar_dml"); - assert_eq!(rec.kind, "columnar_dml"); - assert!(!rec.is_update, "DELETE must carry is_update = false"); - assert!(rec.updates.is_empty(), "DELETE carries no assignments"); - } - /// Columnar-family truncates resolve to the same dedicated record the /// autocommit path appends, carrying the collection name only. #[test] @@ -2834,6 +3057,10 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cevents"), restart_identity: true, }); + // A columnar truncate resolves from the overlay marker its statement + // staged. + let staged = core.execute_stage_write(&make_stage_task(TxnId::new(46)), TID, &columnar); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); let resp = core.execute_resolve_txn(&task, TID, TxnId::new(46), &[columnar]); assert_eq!(resp.status, Status::Ok); let redo = decode_redo(&resp); @@ -2968,12 +3195,10 @@ mod tests { b"V".to_vec(), ); - let mut row = std::collections::HashMap::new(); - row.insert("a".to_string(), nodedb_types::Value::Integer(7)); - let col_payload = nodedb_types::value_to_msgpack(&nodedb_types::Value::Array(vec![ - nodedb_types::Value::Object(row), - ])) - .expect("encode columnar payload"); + // Stage the columnar row into the overlay (overlay-driven serializer). + let columnar = columnar_insert_plan("cevents", columnar_row("a", 7, "x"), 22, Vec::new()); + let staged = src.execute_stage_write(&make_stage_task(txn), TID, &columnar); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); let plans = [ kv_write_plan("kvc"), @@ -2986,20 +3211,7 @@ mod tests { pk_bytes: None, provenance: None, }), - PhysicalPlan::Columnar(ColumnarOp::Insert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cevents"), - payload: col_payload, - format: "msgpack".to_string(), - intent: ColumnarInsertIntent::Insert, - on_conflict_updates: Vec::new(), - surrogates: Vec::new(), - schema_bytes: Vec::new(), - provenance: None, - wal_lsn: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), + columnar, ]; let resp = src.execute_resolve_txn(&task, TID, txn, &plans); @@ -3395,4 +3607,169 @@ mod tests { ); assert!(resp.error_code.is_some()); } + + #[test] + fn a_session_resolve_of_staged_writes_on_a_core_without_their_overlay_is_refused() { + let (mut core, _dir) = make_core(); + let task = make_task(); + + let resp = core.execute_resolve_txn(&task, TID, TxnId::new(61), &[doc_put_plan("docs")]); + + assert_eq!( + resp.status, + Status::Error, + "a resolve must not commit nothing for writes staged elsewhere: {resp:?}" + ); + } + + #[test] + fn a_staged_write_that_stages_no_row_still_opens_the_overlay_its_resolve_reads() { + let (mut core, _dir) = make_core(); + let txn = TxnId::new(62); + let delete_absent = PhysicalPlan::Document(DocumentOp::PointDelete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), + document_id: "absent".to_string(), + surrogate: Surrogate::new(9_001), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + }); + + let staged = core.execute_stage_write(&make_stage_task(txn), TID, &delete_absent); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); + assert!(core.txn_overlays.contains_key(&txn)); + + let resp = core.execute_resolve_txn(&make_task(), TID, txn, &[delete_absent]); + assert_eq!(resp.status, Status::Ok, "resolve: {resp:?}"); + } + + #[test] + fn an_untimed_timeseries_row_resolves_with_the_instant_its_statement_read() { + let (mut core, _dir) = make_core(); + let txn = TxnId::new(63); + let mut row = std::collections::HashMap::new(); + row.insert("value".to_string(), nodedb_types::Value::Float(1.5)); + let payload = nodedb_types::value_to_msgpack(&nodedb_types::Value::Array(vec![ + nodedb_types::Value::Object(row), + ])) + .expect("encode rows"); + let ingest = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), + payload, + format: "msgpack".to_string(), + wal_lsn: None, + surrogates: vec![Surrogate::new(801)], + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }); + + core.epoch_system_ms = Some(1_700_000_000_000); + let staged = core.execute_stage_write(&make_stage_task(txn), TID, &ingest); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); + + // Resolve reads a later clock; the row keeps the statement's instant. + core.epoch_system_ms = Some(1_700_000_999_000); + let redo = decode_redo(&core.execute_resolve_txn(&make_task(), TID, txn, &[ingest])); + assert_eq!(redo.ops.len(), 1); + let (_kind, _collection, lines, _prov, format) = + zerompk::from_msgpack::<(String, String, Vec, Option, String)>( + &redo.ops[0].payload, + ) + .expect("decode timeseries redo"); + assert_eq!(format, "ilp-msgpack"); + let lines: Vec = zerompk::from_msgpack(&lines).expect("decode lines"); + assert_eq!(lines.len(), 1); + assert!( + lines[0].ends_with(" 1700000000000000000"), + "the row carries the statement's instant: {}", + lines[0] + ); + } + + fn seeded_columnar_core() -> (CoreLoop, tempfile::TempDir) { + let (mut core, dir) = make_core(); + let schema = nodedb_types::columnar::ColumnarSchema { + columns: vec![ + ColumnDef::required("id", ColumnType::Int64).with_primary_key(), + ColumnDef::required("v", ColumnType::Int64), + ], + version: 1, + }; + let mut engine = nodedb_columnar::MutationEngine::new("m".to_string(), schema); + engine + .insert_with_surrogate( + &[ + nodedb_types::Value::Integer(1), + nodedb_types::Value::Integer(10), + ], + Surrogate::new(5), + ) + .expect("seed base row"); + core.columnar_engines.insert(coll_key("m"), engine); + (core, dir) + } + + fn columnar_insert(intent: ColumnarInsertIntent) -> PhysicalPlan { + let mut row = std::collections::HashMap::new(); + row.insert("id".to_string(), nodedb_types::Value::Integer(1)); + row.insert("v".to_string(), nodedb_types::Value::Integer(20)); + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "m"), + payload: nodedb_types::value_to_msgpack(&nodedb_types::Value::Array(vec![ + nodedb_types::Value::Object(row), + ])) + .expect("encode row"), + format: "msgpack".to_string(), + intent, + on_conflict_updates: Vec::new(), + surrogates: vec![Surrogate::new(5)], + schema_bytes: Vec::new(), + provenance: None, + wal_lsn: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }) + } + + #[test] + fn a_staged_do_nothing_insert_of_an_existing_key_leaves_the_row_alone() { + let (mut core, _dir) = seeded_columnar_core(); + let txn = TxnId::new(64); + let insert = columnar_insert(ColumnarInsertIntent::InsertIfAbsent); + + let staged = core.execute_stage_write(&make_stage_task(txn), TID, &insert); + assert_eq!(staged.status, Status::Ok, "stage: {staged:?}"); + let overlay = core.txn_overlays.get(&txn).expect("overlay"); + assert!( + overlay.get(&coll_key("m"), 5).is_none(), + "DO NOTHING stages no row over an existing key" + ); + + let redo = decode_redo(&core.execute_resolve_txn(&make_task(), TID, txn, &[insert])); + assert!( + redo.ops.is_empty(), + "the redo writes nothing: {:?}", + redo.ops + ); + } + + #[test] + fn a_staged_unique_insert_of_an_existing_key_is_refused() { + let (mut core, _dir) = seeded_columnar_core(); + let txn = TxnId::new(65); + let insert = columnar_insert(ColumnarInsertIntent::InsertUnique); + + let staged = core.execute_stage_write(&make_stage_task(txn), TID, &insert); + + assert_eq!(staged.status, Status::Error); + assert!(matches!( + staged.error_code.as_deref(), + Some(crate::bridge::envelope::ErrorCode::RejectedConstraint { .. }) + )); + } } diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/graph.rs b/nodedb/src/data/executor/handlers/transaction/resolve/graph.rs index 7f690f27b..e2a0288ef 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/graph.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/graph.rs @@ -9,7 +9,7 @@ use std::collections::{BTreeMap, BTreeSet}; -use nodedb_physical::physical_plan::GraphOp; +use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; use nodedb_wal::record::RecordType; use crate::control::server::wal_dispatch::encode_graph_node_label_payload; @@ -234,6 +234,15 @@ pub(super) fn classify_graph_op( } } +/// Whether `plan` writes node labels, which stage into the graph overlay +/// under its fixed label key. +pub(super) fn is_label_write(plan: &PhysicalPlan) -> bool { + matches!( + plan, + PhysicalPlan::Graph(GraphOp::SetNodeLabels { .. } | GraphOp::RemoveNodeLabels { .. }) + ) +} + #[cfg(test)] mod tests { use super::*; diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/mod.rs b/nodedb/src/data/executor/handlers/transaction/resolve/mod.rs index cc1b0ead1..2544dcb85 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/mod.rs @@ -1,10 +1,18 @@ // SPDX-License-Identifier: BUSL-1.1 mod array; -mod columnar; +mod classify; +mod columnar_image; +mod crdt; mod document; mod entry; mod graph; mod kv; mod spatial; +mod text; +mod timeseries; mod vector; +mod vector_direct; +mod vector_primary; + +pub(in crate::data::executor) use entry::StagedWrites; diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/text.rs b/nodedb/src/data/executor/handlers/transaction/resolve/text.rs new file mode 100644 index 000000000..719e63376 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/resolve/text.rs @@ -0,0 +1,48 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Full-text serializer for transaction resolve. +//! +//! **Plan-driven.** An `FtsIndexDoc` / `FtsDeleteDoc` buffered inside a +//! transaction carries the complete posting input (collection, surrogate, +//! text), so it resolves to the exact `FtsIndex` / `FtsDelete` record its +//! autocommit form journals (`wal_dispatch::encode_text_op_record`). A row the +//! same transaction also writes as a document re-derives the same postings at +//! install; an index upsert for one document is idempotent, so the two agree. +//! Searches and analyzer configuration emit nothing. + +use nodedb_physical::physical_plan::TextOp; + +use crate::control::server::wal_dispatch::encode_text_op_record; +use crate::wal::RedoSubRecord; + +/// Append the redo sub-record for a single text plan op to `ops`. +pub(super) fn serialize_text_op(op: &TextOp, ops: &mut Vec) -> crate::Result<()> { + if let Some((record_type, payload)) = encode_text_op_record(op)? { + ops.push(RedoSubRecord { + record_type: record_type as u32, + payload, + }); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate}; + use nodedb_wal::record::RecordType; + + #[test] + fn fts_index_resolves_to_an_fts_index_sub_record() { + let op = TextOp::FtsIndexDoc { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), + surrogate: Surrogate::new(7), + text: "hello world".to_string(), + provenance: None, + }; + let mut ops = Vec::new(); + serialize_text_op(&op, &mut ops).expect("resolve fts index"); + assert_eq!(ops.len(), 1); + assert_eq!(ops[0].record_type, RecordType::FtsIndex as u32); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/timeseries.rs b/nodedb/src/data/executor/handlers/transaction/resolve/timeseries.rs new file mode 100644 index 000000000..d095a1acd --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/resolve/timeseries.rs @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Timeseries serializer for transaction resolve. Plan-driven: a timeseries +//! ingest is append-only, so the redo carries the ingested rows through the +//! autocommit path's `RecordType::TimeseriesBatch` encoder +//! (`control::server::wal_dispatch`) and replay appends the same samples. +//! Emission is in plan order, already deterministic. Columnar writes resolve +//! from the overlay instead (`columnar_image`). +//! +//! The rows are resolved to canonical line protocol, and every row with no +//! timestamp is stamped here, once, with the instant its statement read. The +//! sub-record therefore stores the same rows on every replica and on every +//! restart. + +use nodedb_physical::physical_plan::TimeseriesOp; +use nodedb_wal::record::RecordType; + +use crate::control::server::wal_dispatch::{ + encode_columnar_truncate_payload, encode_timeseries_batch_payload_with_format, +}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::timeseries::StampedIngest; +use crate::data::executor::task::ExecutionTask; +use crate::types::{TenantId, TxnId}; +use crate::wal::RedoSubRecord; + +/// The ingest format of a resolved timeseries sub-record: canonical line +/// protocol, one string per row. +const RESOLVED_INGEST_FORMAT: &str = "ilp-msgpack"; + +impl CoreLoop { + /// Append the redo sub-record for a single timeseries plan op to `ops`. + /// `Ingest` tags `"timeseries"`; the scan op emits nothing. + pub(super) fn serialize_timeseries_op( + &self, + task: &ExecutionTask, + tid: u64, + txn_id: TxnId, + op: &TimeseriesOp, + ops: &mut Vec, + ) -> crate::Result<()> { + match op { + TimeseriesOp::Ingest { + collection, + payload, + format, + wal_lsn: _, + surrogates, + provenance, + rls_write_check: _, + // Redo carries the ingested rows, not one caller's projected + // response shape — replay reconstructs state, nothing else. + returning: _, + rls_filters: _, + } => { + let tenant = TenantId::new(tid); + let coll_key = ( + task.request.database_id, + tenant, + collection.as_str().to_string(), + ); + // The instant the statement read. An ingest staged with no + // surrogate recorded none, and resolve reads the clock now. + let now_ms = surrogates + .first() + .and_then(|first| { + self.txn_overlays + .get(&txn_id)? + .ingest_now(&coll_key, first.as_u32()) + }) + .unwrap_or_else(|| self.ingest_now_ms()); + let lines = self + .stamped_ingest_lines(StampedIngest { + database_id: task.request.database_id, + tid: tenant, + collection: collection.as_str(), + payload, + format, + now_ms, + }) + .map_err(crate::Error::DataPlane)?; + let resolved = + zerompk::to_msgpack_vec(&lines).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("resolved timeseries lines: {e}"), + })?; + let sub_payload = encode_timeseries_batch_payload_with_format( + collection.as_str(), + &resolved, + provenance.as_ref(), + RESOLVED_INGEST_FORMAT, + )?; + ops.push(RedoSubRecord { + record_type: RecordType::TimeseriesBatch as u32, + payload: sub_payload, + }); + Ok(()) + } + + // Same record the autocommit path appends + // (`RecordType::TimeseriesTruncate`), replayed via + // `replay_timeseries_truncate`. + TimeseriesOp::Truncate { + collection, + restart_identity: _, + } => { + let sub_payload = encode_columnar_truncate_payload(collection.as_str())?; + ops.push(RedoSubRecord { + record_type: RecordType::TimeseriesTruncate as u32, + payload: sub_payload, + }); + Ok(()) + } + + // Read family: no persisted post-image. The resolve pass is read-only + // too — the ingest it reports is proposed as its own plan. + TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_) => Ok(()), + } + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/vector.rs b/nodedb/src/data/executor/handlers/transaction/resolve/vector.rs index db1735ec2..1618ae7ca 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/vector.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/vector.rs @@ -2,18 +2,14 @@ //! Vector serializer for transaction resolve. //! -//! Unlike the KV / document / graph serializers, the vector serializer is -//! **plan-driven**, not overlay-driven. A vector-primary direct write does -//! stage a `StagedVectorRow` (`stage_write/stage_vector.rs`), but only so the -//! transaction's own reads see it; the redo record still comes from the plan. -//! A vector post-image is inexpressible — the HNSW graph mutation has no -//! compact absolute form — so the redo record logs the INSERT itself and -//! replay rebuilds the index -//! (`replay_vector_wal`, dispatched from the redo reconstitute path). This -//! module therefore reads the [`VectorOp`] plan node directly and emits the -//! SAME engine-native WAL sub-record shape the autocommit vector path produces, -//! reusing its encoders (`control::server::wal_dispatch::vector`) so producer -//! and replay never drift: +//! The HNSW-indexed vector ops are **plan-driven**. A vector post-image has +//! no compact absolute form, so the redo record logs the INSERT itself and +//! replay rebuilds the index (`replay_vector_wal`, dispatched from the redo +//! reconstitute path). This module reads the [`VectorOp`] plan node directly +//! and emits the SAME engine-native WAL sub-record shape the autocommit +//! vector path produces, reusing its encoders +//! (`control::server::wal_dispatch::vector`) so producer and replay never +//! drift: //! //! * `Insert` → `RecordType::VectorPut`, the 7-element //! `(collection, vector, dim, field_name, doc_id_compat, surrogate, provenance)` @@ -23,15 +19,12 @@ //! * `Delete` → `RecordType::VectorDelete`, `(collection, vector_id, None)`. //! * `DeleteBySurrogate` → `RecordType::VectorDelete`, //! `(collection, surrogate, field_name, provenance)`. -//! * `DirectInsert` / `DirectInsertIfAbsent` / `DirectUpsert` → -//! `RecordType::VectorDirectUpsert`, the 11-element vector-primary -//! post-image with its intent (`replay_direct_upsert`). -//! * `DirectDelete` → `RecordType::VectorDirectDelete`, -//! `(collection, field, targets)` (`replay_direct_delete`). -//! * `DirectTruncate` → `RecordType::VectorDirectTruncate`, -//! `(collection, field)` (`replay_direct_truncate`). -//! * `DirectUpdate` → `RecordType::VectorDirectUpdate`, the 8-element -//! vector-primary patch (`replay_direct_update`). +//! * `DirectInsert` / `DirectInsertIfAbsent` / `DirectUpsert` / +//! `DirectUpdate` / `DirectDelete` / `DirectTruncate` in a session +//! transaction register their collection only: a vector-primary row is +//! staged whole, so those resolve from the overlay (`vector_primary`). In +//! a Calvin transaction, which stages none of them, each serializes to its +//! autocommit record shape (`vector_direct`). //! * `MultiVectorInsert` → `RecordType::MultiVectorPut`, the 6-element //! flattened multi-vector shape (`replay_multi_vector_put`). //! * `MultiVectorDelete` → `RecordType::MultiVectorDelete`, @@ -41,7 +34,7 @@ //! * `SparseDelete` → `RecordType::SparseVectorDelete`, //! `(collection, field_name, doc_id)` (`replay_sparse_delete`). //! -//! These five share the autocommit WAL shapes emitted by `wal_append_vector_op` +//! The multi-vector and sparse ops share the autocommit WAL shapes emitted by `wal_append_vector_op` //! and decoded by `replay_vector_extended_wal`, which the redo replay path //! invokes after `replay_vector_wal`, so producer and replay never drift. //! @@ -65,32 +58,31 @@ //! puts on replay, but `SetParams` is rejected here, so ordering reduces to the //! given plan order. -use nodedb_physical::physical_plan::VectorOp; +use nodedb_physical::physical_plan::{VectorDirectWriteIntent, VectorOp}; use nodedb_wal::record::RecordType; +use super::vector_direct::{DirectInsert, DirectUpdate, DirectWrites}; +use super::vector_primary::VectorPrimarySpec; use crate::control::server::wal_dispatch::{ - VectorDirectUpdatePayload, VectorDirectUpsertPayload, VectorResolvedDirectWritePayload, - encode_multi_vector_delete_payload, encode_multi_vector_put_payload, - encode_sparse_vector_delete_payload, encode_sparse_vector_put_payload, - encode_vector_batch_put_payload, encode_vector_delete_by_surrogate_payload, - encode_vector_delete_payload, encode_vector_direct_delete_payload, - encode_vector_direct_truncate_payload, encode_vector_direct_update_payload, - encode_vector_direct_upsert_payload, encode_vector_put_payload, - encode_vector_resolved_direct_write_payload, + VectorResolvedDirectWritePayload, encode_multi_vector_delete_payload, + encode_multi_vector_put_payload, encode_sparse_vector_delete_payload, + encode_sparse_vector_put_payload, encode_vector_batch_put_payload, + encode_vector_delete_by_surrogate_payload, encode_vector_delete_payload, + encode_vector_put_payload, encode_vector_resolved_direct_write_payload, }; use crate::wal::RedoSubRecord; -use nodedb_physical::physical_plan::VectorDirectWriteIntent; /// Append the redo sub-record(s) for a single vector plan op to `ops`. /// /// Writes serialize to their engine-native record shape (`VectorPut` / -/// `VectorDelete` / `VectorDirectUpsert` / `MultiVectorPut` / -/// `MultiVectorDelete` / `SparseVectorPut` / `SparseVectorDelete`); read and -/// index-maintenance ops emit nothing; vector-index DDL (`SetParams`) raises a -/// typed error (see module docs). +/// `VectorDelete` / `MultiVectorPut` / `MultiVectorDelete` / `SparseVectorPut` +/// / `SparseVectorDelete`); a vector-primary direct write goes through +/// `direct`; read and index-maintenance ops emit nothing; vector-index DDL +/// (`SetParams`) raises a typed error (see module docs). pub(super) fn serialize_vector_op( op: &VectorOp, ops: &mut Vec, + direct: &mut DirectWrites<'_>, ) -> crate::Result<()> { match op { VectorOp::Insert { @@ -186,11 +178,9 @@ pub(super) fn serialize_vector_op( .to_string(), }), - // Vector-primary direct writes: full post-image plus the row's - // existence intent, replayed via `replay_direct_upsert`. A redo record - // replays a write, and a replayed write answers nobody — no client - // session is behind it to receive rows, so the projection, its read - // gate, and the already-decided write check are not carried. + // Vector-primary direct writes: a session transaction resolves the + // staged rows from the overlay, a Calvin transaction serializes each + // op from its plan node (`vector_direct`). VectorOp::DirectUpsert { collection, field, @@ -205,26 +195,20 @@ pub(super) fn serialize_vector_op( rls_filters: _, on_conflict_updates, rls_write_check: _, - } => { - let payload = encode_vector_direct_upsert_payload(VectorDirectUpsertPayload { + } => direct.insert( + DirectInsert { collection: collection.as_str(), field, surrogate: *surrogate, pk_bytes, vector, payload, - quantization: *quantization, - storage_dtype: *storage_dtype, - payload_indexes, + spec: spec(*quantization, *storage_dtype, payload_indexes), intent: VectorDirectWriteIntent::Upsert, on_conflict_updates, - })?; - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectUpsert as u32, - payload, - }); - Ok(()) - } + }, + ops, + ), VectorOp::DirectInsert { collection, field, @@ -237,27 +221,8 @@ pub(super) fn serialize_vector_op( payload_indexes, returning: _, rls_filters: _, - } => { - let payload = encode_vector_direct_upsert_payload(VectorDirectUpsertPayload { - collection: collection.as_str(), - field, - surrogate: *surrogate, - pk_bytes, - vector, - payload, - quantization: *quantization, - storage_dtype: *storage_dtype, - payload_indexes, - intent: VectorDirectWriteIntent::Insert, - on_conflict_updates: &[], - })?; - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectUpsert as u32, - payload, - }); - Ok(()) } - VectorOp::DirectInsertIfAbsent { + | VectorOp::DirectInsertIfAbsent { collection, field, surrogate, @@ -269,58 +234,24 @@ pub(super) fn serialize_vector_op( payload_indexes, returning: _, rls_filters: _, - } => { - let payload = encode_vector_direct_upsert_payload(VectorDirectUpsertPayload { + } => direct.insert( + DirectInsert { collection: collection.as_str(), field, surrogate: *surrogate, pk_bytes, vector, payload, - quantization: *quantization, - storage_dtype: *storage_dtype, - payload_indexes, - intent: VectorDirectWriteIntent::InsertIfAbsent, + spec: spec(*quantization, *storage_dtype, payload_indexes), + intent: if matches!(op, VectorOp::DirectInsertIfAbsent { .. }) { + VectorDirectWriteIntent::InsertIfAbsent + } else { + VectorDirectWriteIntent::Insert + }, on_conflict_updates: &[], - })?; - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectUpsert as u32, - payload, - }); - Ok(()) - } - // Vector-primary delete, replayed via `replay_direct_delete`. - VectorOp::DirectDelete { - collection, - field, - targets, - returning: _, - rls_filters: _, - rls_write_check: _, - } => { - let payload = encode_vector_direct_delete_payload(collection.as_str(), field, targets)?; - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectDelete as u32, - payload, - }); - Ok(()) - } - // Vector-primary truncate, replayed via `replay_direct_truncate`. - // `restart_identity` is a Control-Plane sequence concern and never - // enters the redo record. - VectorOp::DirectTruncate { - collection, - field, - restart_identity: _, - } => { - let payload = encode_vector_direct_truncate_payload(collection.as_str(), field)?; - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectTruncate as u32, - payload, - }); - Ok(()) - } - // Vector-primary update, replayed via `replay_direct_update`. + }, + ops, + ), VectorOp::DirectUpdate { collection, field, @@ -333,23 +264,30 @@ pub(super) fn serialize_vector_op( returning: _, rls_filters: _, rls_write_check: _, - } => { - let payload = encode_vector_direct_update_payload(VectorDirectUpdatePayload { + } => direct.update( + DirectUpdate { collection: collection.as_str(), field, targets, new_vector: new_vector.as_deref(), payload_patch, - quantization: *quantization, - storage_dtype: *storage_dtype, - payload_indexes, - })?; - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectUpdate as u32, - payload, - }); - Ok(()) - } + spec: spec(*quantization, *storage_dtype, payload_indexes), + }, + ops, + ), + VectorOp::DirectDelete { + collection, + field, + targets, + returning: _, + rls_filters: _, + rls_write_check: _, + } => direct.delete(collection.as_str(), field, targets, ops), + VectorOp::DirectTruncate { + collection, + field, + restart_identity: _, + } => direct.truncate(collection.as_str(), field, ops), // Resolved vector-primary write, replayed via // `replay_vector_resolved_direct_write`: the same record the // autocommit path appends, carrying every row's stored image. @@ -453,3 +391,16 @@ pub(super) fn serialize_vector_op( } } } + +/// The index settings a direct insert-family or update plan carries. +fn spec( + quantization: nodedb_types::VectorQuantization, + storage_dtype: nodedb_types::VectorStorageDtype, + payload_indexes: &[(String, nodedb_types::PayloadIndexKind)], +) -> VectorPrimarySpec { + VectorPrimarySpec { + quantization, + storage_dtype, + payload_indexes: payload_indexes.to_vec(), + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/vector_direct.rs b/nodedb/src/data/executor/handlers/transaction/resolve/vector_direct.rs new file mode 100644 index 000000000..bd07dfd92 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/resolve/vector_direct.rs @@ -0,0 +1,174 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Vector-primary direct writes in transaction resolve, by staging source. +//! +//! A session transaction stages every direct write into its overlay, so its +//! redo carries the staged rows (`vector_primary`): each op here only +//! registers its collection. A Calvin transaction stages no vector-primary +//! write, so its redo carries each op as the autocommit record shape and +//! replay re-runs it through the live handler in the epoch's deterministic +//! order. + +use nodedb_physical::physical_plan::{UpdateValue, VectorDirectWriteIntent, VectorWriteTargets}; +use nodedb_types::Surrogate; +use nodedb_wal::record::RecordType; + +use super::vector_primary::{VectorPrimaryCollections, VectorPrimarySpec, note_direct_write}; +use crate::control::server::wal_dispatch::{ + VectorDirectUpdatePayload, VectorDirectUpsertPayload, encode_vector_direct_delete_payload, + encode_vector_direct_truncate_payload, encode_vector_direct_update_payload, + encode_vector_direct_upsert_payload, +}; +use crate::wal::RedoSubRecord; + +/// Where a transaction's vector-primary direct writes resolve from. +pub(super) enum DirectWrites<'a> { + /// The overlay holds the staged rows; ops register their collection. + Staged(&'a mut VectorPrimaryCollections), + /// Nothing is staged; each op serializes from its plan node. + Plan, +} + +/// One insert-family direct write (`DirectInsert` / `DirectInsertIfAbsent` / +/// `DirectUpsert`). +pub(super) struct DirectInsert<'a> { + pub collection: &'a str, + pub field: &'a str, + pub surrogate: Surrogate, + pub pk_bytes: &'a [u8], + pub vector: &'a [f32], + pub payload: &'a [u8], + pub spec: VectorPrimarySpec, + pub intent: VectorDirectWriteIntent, + pub on_conflict_updates: &'a [(String, UpdateValue)], +} + +/// One `DirectUpdate`. +pub(super) struct DirectUpdate<'a> { + pub collection: &'a str, + pub field: &'a str, + pub targets: &'a VectorWriteTargets, + pub new_vector: Option<&'a [f32]>, + pub payload_patch: &'a [(String, UpdateValue)], + pub spec: VectorPrimarySpec, +} + +impl DirectWrites<'_> { + pub(super) fn insert( + &mut self, + write: DirectInsert<'_>, + ops: &mut Vec, + ) -> crate::Result<()> { + match self { + Self::Staged(collections) => { + note_direct_write( + collections, + write.collection, + write.field, + Some(write.spec), + Some((write.surrogate, write.pk_bytes)), + ); + Ok(()) + } + Self::Plan => { + let payload = encode_vector_direct_upsert_payload(VectorDirectUpsertPayload { + collection: write.collection, + field: write.field, + surrogate: write.surrogate, + pk_bytes: write.pk_bytes, + vector: write.vector, + payload: write.payload, + quantization: write.spec.quantization, + storage_dtype: write.spec.storage_dtype, + payload_indexes: &write.spec.payload_indexes, + intent: write.intent, + on_conflict_updates: write.on_conflict_updates, + })?; + ops.push(RedoSubRecord { + record_type: RecordType::VectorDirectUpsert as u32, + payload, + }); + Ok(()) + } + } + } + + pub(super) fn update( + &mut self, + write: DirectUpdate<'_>, + ops: &mut Vec, + ) -> crate::Result<()> { + match self { + Self::Staged(collections) => { + note_direct_write( + collections, + write.collection, + write.field, + Some(write.spec), + None, + ); + Ok(()) + } + Self::Plan => { + let payload = encode_vector_direct_update_payload(VectorDirectUpdatePayload { + collection: write.collection, + field: write.field, + targets: write.targets, + new_vector: write.new_vector, + payload_patch: write.payload_patch, + quantization: write.spec.quantization, + storage_dtype: write.spec.storage_dtype, + payload_indexes: &write.spec.payload_indexes, + })?; + ops.push(RedoSubRecord { + record_type: RecordType::VectorDirectUpdate as u32, + payload, + }); + Ok(()) + } + } + } + + pub(super) fn delete( + &mut self, + collection: &str, + field: &str, + targets: &VectorWriteTargets, + ops: &mut Vec, + ) -> crate::Result<()> { + match self { + Self::Staged(collections) => { + note_direct_write(collections, collection, field, None, None); + Ok(()) + } + Self::Plan => { + ops.push(RedoSubRecord { + record_type: RecordType::VectorDirectDelete as u32, + payload: encode_vector_direct_delete_payload(collection, field, targets)?, + }); + Ok(()) + } + } + } + + pub(super) fn truncate( + &mut self, + collection: &str, + field: &str, + ops: &mut Vec, + ) -> crate::Result<()> { + match self { + Self::Staged(collections) => { + note_direct_write(collections, collection, field, None, None); + Ok(()) + } + Self::Plan => { + ops.push(RedoSubRecord { + record_type: RecordType::VectorDirectTruncate as u32, + payload: encode_vector_direct_truncate_payload(collection, field)?, + }); + Ok(()) + } + } + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/vector_primary.rs b/nodedb/src/data/executor/handlers/transaction/resolve/vector_primary.rs new file mode 100644 index 000000000..396840e3c --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/resolve/vector_primary.rs @@ -0,0 +1,194 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Vector-primary serializer for transaction resolve. Overlay-driven. +//! +//! A vector-primary direct write (`DirectInsert`, `DirectInsertIfAbsent`, +//! `DirectUpsert`, `DirectUpdate`, `DirectDelete`, `DirectTruncate`) stages +//! the row it leaves behind per surrogate: a [`StagedVectorRow`] holding the +//! vector and the exact sidecar bytes, or a tombstone. An `ON CONFLICT DO +//! UPDATE` stages the merged sidecar and an UPDATE stages the patched one. +//! That staged row is what the transaction's own reads showed, so the redo +//! carries it verbatim as one `VectorResolvedDirectWrite` record per +//! collection, the record the governed autocommit path already ships. Replay +//! installs each row through the same apply and never re-runs a merge or a +//! patch against the replaying node's state. +//! +//! * A staged row → `VectorResolvedMutation::Upsert` with the staged vector +//! and sidecar. `old_payload` is the base sidecar the row replaces, `None` +//! when no base row is bound. +//! * A staged tombstone of a bound base row → `VectorResolvedMutation::Delete`. +//! A tombstone of a row the transaction itself inserted emits nothing. +//! * A staged TRUNCATE → the `VectorDirectTruncate` record ahead of the rows. +//! +//! Rows are emitted in surrogate order, so two resolves of one transaction +//! produce byte-identical records. +//! +//! This covers session transactions. A Calvin transaction stages no +//! vector-primary write and resolves from its plans instead (`vector_direct`). + +use std::collections::BTreeMap; + +use nodedb_physical::physical_plan::VectorResolvedMutation; +use nodedb_types::{PayloadIndexKind, Surrogate, VectorQuantization, VectorStorageDtype}; +use nodedb_wal::record::RecordType; + +use crate::control::server::wal_dispatch::{ + VectorResolvedDirectWritePayload, encode_vector_direct_truncate_payload, + encode_vector_resolved_direct_write_payload, +}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::transaction::overlay::{Staged, StagedVectorRow, TxnOverlay}; +use crate::types::{DatabaseId, TenantId}; +use crate::wal::RedoSubRecord; + +/// The index settings a direct insert-family or update plan carries. +#[derive(Clone)] +pub(super) struct VectorPrimarySpec { + pub quantization: VectorQuantization, + pub storage_dtype: VectorStorageDtype, + pub payload_indexes: Vec<(String, PayloadIndexKind)>, +} + +/// What the plans say about one vector-primary collection the transaction +/// wrote. +#[derive(Default)] +pub(super) struct VectorPrimaryWrites { + pub field: String, + /// `None` when only DELETE / TRUNCATE plans named the collection. + pub spec: Option, + /// The declared primary-key bytes each inserted surrogate carried. + pub pk_by_surrogate: BTreeMap>, +} + +/// Vector-primary collections a transaction wrote, keyed by collection. +pub(super) type VectorPrimaryCollections = BTreeMap; + +/// Record one vector-primary direct write against its collection. `spec` is +/// the index settings an insert-family or update plan carries. `pk` is the +/// surrogate and declared primary-key bytes an insert-family plan carries. +pub(super) fn note_direct_write( + collections: &mut VectorPrimaryCollections, + collection: &str, + field: &str, + spec: Option, + pk: Option<(Surrogate, &[u8])>, +) { + let writes = collections.entry(collection.to_string()).or_default(); + if writes.field.is_empty() { + writes.field = field.to_string(); + } + if writes.spec.is_none() { + writes.spec = spec; + } + if let Some((surrogate, pk_bytes)) = pk { + writes + .pk_by_surrogate + .entry(surrogate.as_u32()) + .or_insert_with(|| pk_bytes.to_vec()); + } +} + +impl CoreLoop { + /// Append the truncate record (when the transaction truncated the + /// collection) and the collection's `VectorResolvedDirectWrite` record + /// to `ops`. Reads base by `&` only. + pub(super) fn serialize_vector_primary_collection( + &self, + overlay: &TxnOverlay, + coll_key: &(DatabaseId, TenantId, String), + writes: &VectorPrimaryWrites, + ops: &mut Vec, + ) -> crate::Result<()> { + let (database_id, tid, collection) = + (coll_key.0.as_u64(), coll_key.1.as_u64(), &coll_key.2); + let truncated = overlay.is_truncated(coll_key); + if truncated { + ops.push(RedoSubRecord { + record_type: RecordType::VectorDirectTruncate as u32, + payload: encode_vector_direct_truncate_payload(collection, &writes.field)?, + }); + } + + let index_key = CoreLoop::vector_index_key(database_id, tid, collection, &writes.field); + let entries: BTreeMap = overlay.iter_for_collection(coll_key).collect(); + let mut mutations = Vec::with_capacity(entries.len()); + for (surrogate_u32, staged) in entries { + let surrogate = Surrogate::new(surrogate_u32); + let base = if !truncated && self.vector_direct_node(&index_key, surrogate).is_some() { + Some( + self.vector_sidecar_bytes(database_id, tid, collection, surrogate) + .map_err(|e| crate::Error::Internal { + detail: format!( + "vector-primary resolve of '{collection}': base sidecar of \ + surrogate {surrogate_u32}: {e:?}" + ), + })? + .unwrap_or_default(), + ) + } else { + None + }; + match staged { + Staged::Put(body) => { + let row = StagedVectorRow::from_bytes(body)?; + mutations.push(VectorResolvedMutation::Upsert { + surrogate, + pk_bytes: writes + .pk_by_surrogate + .get(&surrogate_u32) + .cloned() + .unwrap_or_default(), + vector: row.vector, + payload: row.sidecar, + old_payload: base, + }); + } + Staged::Tombstone => { + if let Some(old_payload) = base { + mutations.push(VectorResolvedMutation::Delete { + surrogate, + old_payload, + }); + } + } + } + } + if mutations.is_empty() { + return Ok(()); + } + + let stores_rows = mutations + .iter() + .any(|m| matches!(m, VectorResolvedMutation::Upsert { .. })); + let spec = match (&writes.spec, stores_rows) { + (Some(spec), _) => spec.clone(), + (None, false) => VectorPrimarySpec { + quantization: VectorQuantization::default(), + storage_dtype: VectorStorageDtype::default(), + payload_indexes: Vec::new(), + }, + (None, true) => { + return Err(crate::Error::Internal { + detail: format!( + "vector-primary resolve of '{collection}': a staged row has no \ + insert or update plan carrying its index settings" + ), + }); + } + }; + let payload = + encode_vector_resolved_direct_write_payload(VectorResolvedDirectWritePayload { + collection, + field: &writes.field, + quantization: spec.quantization, + storage_dtype: spec.storage_dtype, + payload_indexes: &spec.payload_indexes, + mutations: &mutations, + })?; + ops.push(RedoSubRecord { + record_type: RecordType::VectorResolvedDirectWrite as u32, + payload, + }); + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs index d2310a3df..8a47026f7 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/dispatch.rs @@ -32,6 +32,10 @@ impl CoreLoop { }, ); }; + // Every core that stages a write for this transaction holds its + // overlay, even when the write stages no row. COMMIT's resolve refuses + // a transaction with staged writes whose overlay is missing. + self.txn_overlay_mut(txn_id); let doc_op = match plan { PhysicalPlan::Document(op) => op, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs index b27f05028..590c7c139 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs @@ -20,6 +20,7 @@ mod stage_array; mod stage_bulk_delete; mod stage_bulk_update; mod stage_columnar; +mod stage_columnar_base_key; mod stage_columnar_dml; mod stage_columnar_family; mod stage_columnar_resolved_dml; @@ -37,6 +38,7 @@ mod stage_point_document; mod stage_rls; mod stage_spatial; mod stage_timeseries; +mod stage_timeseries_now; mod stage_truncate; mod stage_upsert; mod stage_vector; diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar.rs index 4193003a7..c91c25b45 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar.rs @@ -5,9 +5,8 @@ //! A columnar batch INSERT issued inside a `BEGIN..COMMIT` block is staged //! here, one overlay `Put` per row, so a later same-transaction columnar //! SELECT observes the newly inserted rows (read-your-own-writes) before -//! COMMIT. COMMIT durable replay is unchanged: the buffered `ColumnarOp::Insert` -//! plan is still replayed through `execute_columnar_insert` inside the -//! COMMIT `TransactionBatch`, which remains the sole durable apply. +//! COMMIT. COMMIT resolves the staged rows into the transaction's redo record +//! (`resolve::columnar_image`), and every replica installs exactly those rows. //! //! Row identity: the overlay's identity side-map holds the row's primary-key //! value when the schema declares one, else the decimal surrogate @@ -16,19 +15,19 @@ //! //! Row body encoding: each row's schema-ordered `Vec` is wrapped as a //! `Value::Array` and encoded via `nodedb_types::value_to_msgpack` — decoded -//! the same way by `merge_overlay_into_columnar_scan`. This is a -//! staging-only representation; it plays no part in the durable segment -//! format written at COMMIT by `execute_columnar_insert`. +//! the same way by `merge_overlay_into_columnar_scan` and by COMMIT resolve. +//! +//! Conflict intent: a row whose primary key this transaction already sees is +//! skipped under `ON CONFLICT DO NOTHING` and refuses the statement under a +//! declared natural primary key, the decisions the autocommit insert makes. //! //! ON CONFLICT DO UPDATE: the staged body is the MERGED row, not the submitted //! one. The overlay exists to show what this transaction has written, and after -//! a conflict merge that is the stored row with the assignments applied — -//! exactly what `execute_columnar_insert` persists at COMMIT. Staging the -//! submitted body instead made the overlay and the eventual durable state -//! disagree, so a same-transaction `SELECT` showed a row the COMMIT would never -//! produce. The merge is resolved against this transaction's own overlay first -//! and the engine second, so an earlier statement's staged row is the one it -//! merges against. +//! a conflict merge that is the stored row with the assignments applied. The +//! redo carries that merged row, so COMMIT persists exactly what a +//! same-transaction `SELECT` showed. The merge is resolved against this +//! transaction's own overlay first and the engine second, so an earlier +//! statement's staged row is the one it merges against. //! //! Row-level security: the write policy decides the batch here, at the //! statement, not only at COMMIT — otherwise a refused row would be reported as @@ -39,10 +38,12 @@ //! //! Field coercion mirrors `execute_columnar_insert` exactly (same //! `ndb_field_to_value` / bitemporal column population) via the shared -//! `columnar_write::schema` helpers, so a staged row's values match what the -//! durable COMMIT replay will eventually store. +//! `columnar_write::schema` helpers, so a staged row holds the values an +//! autocommit insert of the same row stores. + +use std::collections::HashSet; -use nodedb_physical::physical_plan::UpdateValue; +use nodedb_physical::physical_plan::{ColumnarInsertIntent, UpdateValue}; use nodedb_types::Surrogate; use nodedb_types::columnar::schema::{TS_SYSTEM, TS_VALID_FROM, TS_VALID_UNTIL}; use nodedb_types::value::Value; @@ -52,6 +53,7 @@ use super::stage_columnar_dml::columnar_row_identity; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::columnar_write::ndb_field_to_value; +use crate::data::executor::handlers::transaction::overlay::Staged; use crate::data::executor::task::ExecutionTask; use crate::types::{TenantId, TxnId}; @@ -65,6 +67,9 @@ pub(in crate::data::executor) struct StageColumnarInsertParams<'a> { pub payload: &'a [u8], pub surrogates: &'a [Surrogate], pub schema_bytes: &'a [u8], + /// What a row whose primary key already exists does: replace it, merge + /// into it, stay out (`ON CONFLICT DO NOTHING`), or refuse the statement. + pub intent: ColumnarInsertIntent, /// `ON CONFLICT (pk) DO UPDATE SET` assignments carried by the plan. /// Needed here only to resolve the row image the write policy decides — /// the merged row, not the submitted one. The staged body itself is @@ -95,6 +100,7 @@ impl CoreLoop { payload, surrogates, schema_bytes, + intent, on_conflict_updates, rls_write_check, } = params; @@ -179,6 +185,9 @@ impl CoreLoop { // overlay. Splitting it the other way would make a refusal partially // durable in the overlay, which is the worse of the two. let mut resolved: Vec<(Surrogate, Vec)> = Vec::with_capacity(ndb_rows.len()); + // Rows earlier in this statement, by surrogate. A primary key maps to + // one surrogate, so a repeated key within the statement is caught here. + let mut written: HashSet = HashSet::with_capacity(ndb_rows.len()); for (row_idx, row) in ndb_rows.iter().enumerate() { let obj = match row { @@ -226,6 +235,42 @@ impl CoreLoop { } }; + // A key that already exists keeps its row under `DO NOTHING` and + // refuses the statement under a declared natural primary key, + // exactly as the autocommit insert decides. + if matches!( + intent, + ColumnarInsertIntent::InsertIfAbsent | ColumnarInsertIntent::InsertUnique + ) { + let exists = written.contains(&surrogate) + || match self.staged_columnar_row_exists( + txn_id, + &engine_key, + surrogate, + &values, + ) { + Ok(exists) => exists, + Err(error) => return self.response_error(task, error), + }; + if exists { + if intent == ColumnarInsertIntent::InsertIfAbsent { + continue; + } + return self.response_error( + task, + crate::Error::RejectedConstraint { + collection: collection.to_string(), + constraint: "unique".to_string(), + detail: format!( + "duplicate primary key violates primary-key uniqueness on \ + '{collection}'" + ), + }, + ); + } + } + written.insert(surrogate); + // The row that will exist afterwards: the incoming row for a plain // insert, the merged row for the ON CONFLICT branch. This is the // body staged as well as the image decided — the overlay is @@ -281,4 +326,33 @@ impl CoreLoop { self.stage_count_response(task, staged) } + + /// Whether the row keyed like `values` exists as this transaction sees + /// it: its staged put or tombstone first, then the engine, unless the + /// transaction truncated the collection. + fn staged_columnar_row_exists( + &self, + txn_id: TxnId, + engine_key: &(nodedb_types::DatabaseId, TenantId, String), + surrogate: Surrogate, + values: &[Value], + ) -> Result { + if let Some(overlay) = self.txn_overlays.get(&txn_id) { + match overlay.get(engine_key, surrogate.as_u32()) { + Some(Staged::Put(_)) => return Ok(true), + Some(Staged::Tombstone) => return Ok(false), + None if overlay.is_truncated(engine_key) => return Ok(false), + None => {} + } + } + let Some(engine) = self.columnar_engines.get(engine_key) else { + return Ok(false); + }; + let pk_bytes = engine + .encode_pk_from_row(values) + .map_err(|e| ErrorCode::Internal { + detail: format!("columnar insert: pk encode failed: {e}"), + })?; + Ok(engine.pk_index().contains(&pk_bytes)) + } } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_base_key.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_base_key.rs new file mode 100644 index 000000000..1c8cf3272 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_base_key.rs @@ -0,0 +1,70 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Records which base row a staged columnar UPDATE or DELETE displaces. +//! +//! The overlay is keyed by surrogate and holds only the row's final image or a +//! tombstone. COMMIT resolve turns that into a row-image redo record, and the +//! record must name the base row the transaction replaced by its primary key. +//! A key-changing UPDATE and a DELETE remove a row the final image does not +//! name. The key is recorded here, at the statement, from the pre-image the +//! statement matched. It is recorded only when that pre-image is a base row: +//! the transaction has staged nothing for the surrogate yet. + +use nodedb_types::columnar::ColumnarSchema; +use nodedb_types::value::Value; + +use crate::data::executor::core_loop::CoreLoop; +use crate::types::{DatabaseId, TenantId, TxnId}; + +impl CoreLoop { + /// Record the primary key of `row` as the base key of `surrogate` when + /// the transaction has staged nothing for `surrogate` yet. Call it before + /// staging the statement's put or tombstone for that surrogate. + pub(super) fn stage_note_columnar_base_row( + &mut self, + txn_id: TxnId, + coll_key: &(DatabaseId, TenantId, String), + schema: &ColumnarSchema, + surrogate: u32, + row: &[Value], + ) -> crate::Result<()> { + let pk = schema + .columns + .iter() + .position(|c| c.primary_key) + .and_then(|idx| row.get(idx)) + .ok_or_else(|| crate::Error::Internal { + detail: format!( + "columnar staging on '{}': matched row carries no primary-key value", + coll_key.2 + ), + })?; + self.stage_note_columnar_base_pk(txn_id, coll_key, surrogate, pk) + } + + /// Record `pk` as the base key of `surrogate` when the transaction has + /// staged nothing for `surrogate` yet. + pub(super) fn stage_note_columnar_base_pk( + &mut self, + txn_id: TxnId, + coll_key: &(DatabaseId, TenantId, String), + surrogate: u32, + pk: &Value, + ) -> crate::Result<()> { + let staged_already = self + .txn_overlays + .get(&txn_id) + .is_some_and(|overlay| overlay.get(coll_key, surrogate).is_some()); + if staged_already { + return Ok(()); + } + let pk_msgpack = + nodedb_types::value_to_msgpack(pk).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("columnar base key of '{}': {e}", coll_key.2), + })?; + self.txn_overlay_mut(txn_id) + .note_base_pk(coll_key, surrogate, pk_msgpack); + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs index f55d34856..53aea9f10 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs @@ -35,13 +35,13 @@ //! COMMIT would report `{"affected": N}` for a statement the transaction can //! never keep, and expose the refused image to its own reads meanwhile. //! -//! COMMIT durable replay is unchanged: the buffered `ColumnarOp::Delete` / -//! `ColumnarOp::Update` plan is still replayed through -//! `execute_columnar_delete` / `execute_columnar_update` inside the COMMIT -//! `TransactionBatch`, which remains the sole durable apply. The staged set is -//! resolved from the live memtable (plus overlay) exactly as those durable -//! handlers resolve their matching set, so the in-transaction view matches the -//! post-commit view. +//! COMMIT resolves the staged post-images and tombstones into the +//! transaction's redo record (`resolve::columnar_image`). The first statement +//! that stages a base row records that row's primary key +//! (`stage_columnar_base_key`), so the redo names the base row a key-changing +//! UPDATE or a DELETE removes. The staged set is resolved from the live +//! memtable (plus overlay), the scope the autocommit handlers +//! (`execute_columnar_delete` / `execute_columnar_update`) match against. use nodedb_types::columnar::ColumnarSchema; use nodedb_types::value::Value; @@ -157,6 +157,11 @@ impl CoreLoop { let affected = affected_rows.len(); for (surrogate, row) in affected_rows { + if let Err(e) = + self.stage_note_columnar_base_row(txn_id, &coll_key, &schema, surrogate, &row) + { + return self.response_error(task, e); + } let identity = columnar_row_identity(&schema, &row, surrogate); self.txn_overlay_mut(txn_id) .insert_tombstone(coll_key.clone(), surrogate, &identity); @@ -211,13 +216,15 @@ impl CoreLoop { // transaction's own reads. let affected = affected_rows.len(); let mut new_rows: Vec<(u32, Vec)> = Vec::with_capacity(affected); + let mut base_rows: Vec<(u32, Vec)> = Vec::with_capacity(affected); for (surrogate, row) in affected_rows { - match apply_columnar_updates(&schema, row, updates) { + match apply_columnar_updates(&schema, row.clone(), updates) { Ok(r) => new_rows.push((surrogate, r)), Err(detail) => { return self.response_error(task, ErrorCode::Internal { detail }); } } + base_rows.push((surrogate, row)); } if let Err(response) = self.stage_admit_columnar_rows( task, @@ -230,6 +237,15 @@ impl CoreLoop { return response; } + // The key the pre-image carried names the base row a key-changing + // update removes; recorded before the put below stages the surrogate. + for (surrogate, row) in &base_rows { + if let Err(e) = + self.stage_note_columnar_base_row(txn_id, &coll_key, &schema, *surrogate, row) + { + return self.response_error(task, e); + } + } for (surrogate, new_row) in new_rows { let identity = columnar_row_identity(&schema, &new_row, surrogate); let body = match nodedb_types::value_to_msgpack(&Value::Array(new_row)) { diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_family.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_family.rs index bd0e0d97b..317f04abd 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_family.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_family.rs @@ -17,7 +17,7 @@ use crate::types::TxnId; impl CoreLoop { /// Stage a `ColumnarOp` write into `txn_id`'s overlay. - pub(super) fn execute_stage_columnar( + pub(in crate::data::executor) fn execute_stage_columnar( &mut self, task: &ExecutionTask, tid: u64, @@ -30,6 +30,7 @@ impl CoreLoop { payload, surrogates, schema_bytes, + intent, on_conflict_updates, rls_write_check, .. @@ -41,6 +42,7 @@ impl CoreLoop { payload, surrogates, schema_bytes, + intent: *intent, on_conflict_updates, rls_write_check, }), diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_resolved_dml.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_resolved_dml.rs index 13cdc7e82..8988839b9 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_resolved_dml.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_resolved_dml.rs @@ -29,10 +29,10 @@ //! runs at COMMIT replay — both refuse a shipped-but-vanished row rather than //! silently dropping it from the affected count. //! -//! COMMIT durable replay is unchanged: the buffered `ColumnarOp::ResolvedUpdate` -//! / `ColumnarOp::ResolvedDelete` plan is still replayed through -//! `execute_columnar_resolved_update` / `execute_columnar_resolved_delete` -//! inside the COMMIT `TransactionBatch`, which remains the sole durable apply. +//! COMMIT resolves the staged post-images and tombstones into the +//! transaction's redo record (`resolve::columnar_image`), with each shipped +//! primary key recorded as the base row it replaces +//! (`stage_columnar_base_key`). use std::collections::HashMap; @@ -170,6 +170,11 @@ impl CoreLoop { } let affected = resolved.len(); + for ((surrogate, _, _), (pk, _)) in resolved.iter().zip(rows) { + if let Err(e) = self.stage_note_columnar_base_pk(txn_id, &coll_key, *surrogate, pk) { + return self.response_error(task, e); + } + } for (surrogate, identity, new_row) in resolved { let body = match nodedb_types::value_to_msgpack(&Value::Array(new_row.clone())) { Ok(b) => b, @@ -263,6 +268,11 @@ impl CoreLoop { } let affected = surrogates.len(); + for ((surrogate, _), pk) in surrogates.iter().zip(pks) { + if let Err(e) = self.stage_note_columnar_base_pk(txn_id, &coll_key, *surrogate, pk) { + return self.response_error(task, e); + } + } for (surrogate, identity) in surrogates { self.txn_overlay_mut(txn_id) .insert_tombstone(coll_key.clone(), surrogate, &identity); diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs index db5445acd..38cfe7368 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs @@ -41,12 +41,12 @@ use nodedb_types::Surrogate; use super::context::StageCtx; use super::stage_kv::kv_row_identity; -use crate::bridge::envelope::{ErrorCode, Response}; +use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::StagedTtl; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::{AtomicError, atomic_compute, current_ms}; +use crate::engine::kv::{atomic_compute, current_ms}; use crate::types::TxnId; /// FNV-1a 32-bit hash, used only to derive a stable, collection-local overlay @@ -201,7 +201,7 @@ impl CoreLoop { } self.kv_atomic_json_response(ctx.task, &serde_json::json!({ "value": new_i64 })) } - Err(e) => self.kv_atomic_error(ctx.task, ctx.collection, e), + Err(e) => self.response_atomic_error(ctx.task, ctx.collection, e), } } @@ -223,7 +223,7 @@ impl CoreLoop { } self.kv_atomic_json_response(ctx.task, &serde_json::json!({ "value": new_f64 })) } - Err(e) => self.kv_atomic_error(ctx.task, ctx.collection, e), + Err(e) => self.response_atomic_error(ctx.task, ctx.collection, e), } } @@ -260,7 +260,11 @@ impl CoreLoop { rls_write_check: &nodedb_types::RlsWriteCheck, ) -> Response { let current = self.resolve_kv_current(ctx, key); - let (matches, write_bytes) = atomic_compute::cas(current.as_deref(), expected, new_value); + let (matches, write_bytes) = + match atomic_compute::cas(current.as_deref(), expected, new_value) { + Ok(outcome) => outcome, + Err(e) => return self.response_atomic_error(ctx.task, ctx.collection, e), + }; if matches { if let Err(e) = self.stage_admit_kv_image(ctx, &write_bytes, rls_write_check) { @@ -294,7 +298,10 @@ impl CoreLoop { rls_write_check: &nodedb_types::RlsWriteCheck, ) -> Response { let current = self.resolve_kv_current(ctx, key); - let write_bytes = atomic_compute::getset(current.as_deref(), new_value); + let write_bytes = match atomic_compute::getset(current.as_deref(), new_value) { + Ok(bytes) => bytes, + Err(e) => return self.response_atomic_error(ctx.task, ctx.collection, e), + }; if let Err(e) = self.stage_admit_kv_image(ctx, &write_bytes, rls_write_check) { return self.response_error(ctx.task, e); } @@ -370,30 +377,4 @@ impl CoreLoop { Err(e) => self.response_error(task, e), } } - - fn kv_atomic_error(&self, task: &ExecutionTask, collection: &str, e: AtomicError) -> Response { - match e { - AtomicError::TypeMismatch { detail } => self.response_error( - task, - ErrorCode::TypeMismatch { - collection: collection.to_string(), - detail, - }, - ), - AtomicError::Overflow => self.response_error( - task, - ErrorCode::OverflowError { - collection: collection.to_string(), - }, - ), - AtomicError::Encode { detail } => { - self.response_error(task, ErrorCode::Internal { detail }) - } - // Staging computes its image through `atomic_compute` and decides - // the policy itself, so the engine's own admission gate never - // reaches this path — the arm exists so a new engine-side refusal - // cannot be silently dropped here. - AtomicError::Rejected(error) => self.response_error(task, *error), - } - } } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs index 95731299a..719f379ad 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs @@ -5,10 +5,11 @@ //! A timeseries INSERT issued inside a `BEGIN..COMMIT` block is staged here, //! one overlay `Put` per row, so a later same-transaction RAW timeseries //! SELECT observes the newly inserted rows (read-your-own-writes) before -//! COMMIT. COMMIT durable replay is unchanged: the buffered -//! `TimeseriesOp::Ingest` plan is still replayed through -//! `execute_timeseries_ingest` inside the COMMIT `TransactionBatch`, which -//! remains the sole durable apply. +//! COMMIT. COMMIT resolves the buffered `TimeseriesOp::Ingest` plan into a +//! redo sub-record, which is the sole durable apply. The staged batch records +//! the instant it read as its default row timestamp, and resolve stamps the +//! batch's untimed rows with that instant, so every replica stores the rows +//! the statement decided. //! //! No memtable mutation at statement time: staging writes ONLY into the //! per-transaction overlay (`txn_overlays`), never into `columnar_memtables`. @@ -144,6 +145,11 @@ impl CoreLoop { ); } + // The default timestamp of every untimed row in this batch. The write + // policy decides against it here, and COMMIT resolve stamps the rows + // with it, so the stored image is the one decided. + let now_ms = self.ingest_now_ms(); + // Decide the whole batch before the first staged put, so a refusal // leaves the overlay untouched and reports no affected count. // @@ -166,7 +172,7 @@ impl CoreLoop { crate::types::TenantId::new(tid), collection, ), - self.ingest_now_ms(), + now_ms, tid, collection, ) { @@ -216,6 +222,7 @@ impl CoreLoop { } staged += 1; } + self.note_staged_ingest_now(task, tid, txn_id, collection, surrogates, now_ms); self.stage_count_response(task, staged) } @@ -307,6 +314,7 @@ impl CoreLoop { }, ); } + let now_ms = self.ingest_now_ms(); // The write policy decides the parsed lines, through the very same // helper the Data-Plane ingest gate uses, so the statement-time // decision and the COMMIT-time one are made on a byte-identical image. @@ -320,7 +328,7 @@ impl CoreLoop { crate::types::TenantId::new(tid), collection, ), - self.ingest_now_ms(), + now_ms, tid, collection, ) { @@ -436,6 +444,7 @@ impl CoreLoop { return self.response_error(task, error); } } + self.note_staged_ingest_now(task, tid, txn_id, collection, surrogates, now_ms); self.stage_count_response(task, lines.len()) } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries_now.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries_now.rs new file mode 100644 index 000000000..ecc47a907 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries_now.rs @@ -0,0 +1,34 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The default timestamp a staged timeseries batch records. + +use nodedb_types::Surrogate; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; +use crate::types::TxnId; + +impl CoreLoop { + /// Record `now_ms` as the default timestamp of the staged batch whose + /// first row is `surrogates[0]`. + pub(super) fn note_staged_ingest_now( + &mut self, + task: &ExecutionTask, + tid: u64, + txn_id: TxnId, + collection: &str, + surrogates: &[Surrogate], + now_ms: i64, + ) { + let Some(first) = surrogates.first() else { + return; + }; + let coll_key = ( + task.request.database_id, + crate::types::TenantId::new(tid), + collection.to_string(), + ); + self.txn_overlay_mut(txn_id) + .note_ingest_now(&coll_key, first.as_u32(), now_ms); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_vector.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_vector.rs index 382bb4409..f9c53bcdf 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_vector.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_vector.rs @@ -12,8 +12,9 @@ //! same-transaction point read or scan renders the sidecar. Each op decides //! its outcome against BASE ∪ OVERLAY with the live handler's own rules: //! a duplicate key refuses, `ON CONFLICT DO NOTHING` skips, a conflict -//! patch merges through the live handler's merge. COMMIT replays the -//! buffered plan through the live handler, which stays the durable apply. +//! patch merges through the live handler's merge. COMMIT resolves the staged +//! row into the transaction's redo record (`resolve::vector_primary`), and +//! every replica installs exactly that row. use std::collections::HashMap; diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs index d4987ef01..4e4717170 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs @@ -8,6 +8,9 @@ use crate::data::executor::enforcement::funnel::WriteEnforcementOutcome; use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, WriteImages}; use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::handlers::transaction::undo::document_outcome::{ + DocumentRow, push_delete_undo, push_target_undo, +}; use crate::data::executor::task::ExecutionTask; /// Parameters for [`CoreLoop::tx_point_delete`]. @@ -123,112 +126,23 @@ impl CoreLoop { })?; self.checkpoint_coordinator.mark_dirty("sparse", 1); - // Reverse every derived materialized-sum target write with the SAME set - // of undo entries a source row uses: the target write is a full document - // write, so it has index, vector, spatial and stats side-effects of its - // own to reverse. - for target in target_writes { - undo_log.push(UndoEntry::PutDocument { - collection: target.collection, - document_id: nodedb_types::StorageKey::for_surrogate(target.surrogate), - identity: target.identity, - old_value: target.outcome.prior_value, - bitemporal_sys_from_ms: target.outcome.bitemporal_sys_from_ms, - bitemporal_index_tuples: target.outcome.bitemporal_index_tuples, - secondary_index_added: target.outcome.secondary_index_added, - secondary_index_removed: target.outcome.secondary_index_removed, - chain_hash_prior: None, - }); - for delta in target.outcome.vector_inserts { - undo_log.push(UndoEntry::InsertVector { - index_key: delta.index_key, - vector_id: delta.vector_id, - collection: delta.collection, - field: delta.field, - doc_id: Some(delta.doc_id), - }); - } - for (key, entry_id) in target.outcome.spatial_inserts { - undo_log.push(UndoEntry::SpatialInsert { key, entry_id }); - } - for (key, prior) in target.outcome.stats_prior { - undo_log.push(UndoEntry::StatsRestore { key, prior }); - } - } - - // Only push an undo entry when a row was actually removed — a delete - // against a non-existent key has nothing to reverse. - if let Some(old) = outcome.prior_value { - undo_log.push(UndoEntry::DeleteDocument { - collection: collection.to_string(), - document_id: nodedb_types::StorageKey::for_surrogate(surrogate), - // The plan's `document_id` is the row's client identity. - identity: nodedb_types::RowIdentity::from_user_key(document_id), - old_value: old, - bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, - bitemporal_index_tuples: outcome.bitemporal_index_tuples, - // NON-empty on non-bitemporal deletes: the cascade removed these - // plain secondary-index entries, so a rolled-back DELETE restores - // them (closes the pre-existing tx-DELETE rollback hole). - secondary_index_tuples: outcome.secondary_index_tuples, - chain_hash_prior: None, - }); - } - - // The delete-cleanup soft-deleted this document's vectors unconditionally - // (fixing the orphan leak even in autocommit). In the transactional path - // a rollback must restore them, so push one `DeleteVector` undo per - // soft-deleted vector — `apply_undo_vector` `undelete`s each on rollback. - for delta in outcome.vector_deletes { - undo_log.push(UndoEntry::DeleteVector { - index_key: delta.index_key, - vector_id: delta.vector_id, - collection: delta.collection, - field: delta.field, - doc_id: Some(delta.doc_id), - }); - } - - // Reverse any spatial R-tree removals on rollback (one `SpatialDelete` - // undo per per-field R-tree entry the delete removed, re-inserting it - // with its captured bbox). - for (key, entry_id, bbox, document_id) in outcome.spatial_deletes { - undo_log.push(UndoEntry::SpatialDelete { - key, - entry_id, - bbox, - document_id, - }); - } - - // Reverse the `mark_node_deleted` bookkeeping on rollback: un-mark the - // node in the in-memory `deleted_nodes` tracker. `Some` only when this - // delete NEWLY marked the node (a pre-existing tombstone from a prior - // committed op is never resurrected — see `apply_point_delete`). - if let Some(node_id) = outcome.mark_node_deleted { - undo_log.push(UndoEntry::MarkNodeDeleted { + // A target write is a full document write with side effects of its + // own. The delete's own entries reverse the row, its index entries, + // vectors, R-tree entries, the node tombstone, and every edge the + // cascade removed. + push_target_undo(undo_log, &target_writes); + push_delete_undo( + undo_log, + DocumentRow { database_id, tid, - node_id, - }); - } - - // The graph-edge cascade unconditionally removed every edge incident on - // this document from BOTH the CSR partition and the persistent edge - // store. In the transactional path a rollback must restore them, so push - // one `DeleteEdge` undo per cascaded edge — `apply_undo_edge` re-inserts - // each into both stores with its captured old properties. NON-empty - // whenever the deleted document had edges: this closes the pre-existing - // hole where a rolled-back tx DELETE permanently lost cascaded edges. - for (collection, src_id, label, dst_id, old_properties) in outcome.edge_deletes { - undo_log.push(UndoEntry::DeleteEdge { collection, - src_id, - label, - dst_id, - old_properties, - }); - } + storage_key: nodedb_types::StorageKey::for_surrogate(surrogate), + // The plan's `document_id` is the row's client identity. + identity: nodedb_types::RowIdentity::from_user_key(document_id), + }, + outcome, + ); // `PointDelete` renders a `DELETE ` command tag, so its response // carries the count — 0 when the key was absent — exactly as the diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs index bb2ddc9e6..f1a98d86a 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs @@ -9,6 +9,9 @@ use crate::data::executor::enforcement::funnel::WriteEnforcementOutcome; use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, WriteImages}; use crate::data::executor::handlers::point::apply_put::PointPutParams; use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::handlers::transaction::undo::document_outcome::{ + DocumentRow, push_put_undo, push_target_undo, +}; use crate::data::executor::task::ExecutionTask; /// Parameters for [`CoreLoop::tx_point_put`]. @@ -285,77 +288,22 @@ impl CoreLoop { })?; self.checkpoint_coordinator.mark_dirty("sparse", 1); - // Reverse every derived materialized-sum target write with the SAME set - // of undo entries the source row uses: the target write is a full - // document write, so it has index, vector, spatial and stats - // side-effects of its own to reverse. - for target in target_writes { - undo_log.push(UndoEntry::PutDocument { - collection: target.collection, - document_id: nodedb_types::StorageKey::for_surrogate(target.surrogate), - identity: target.identity, - old_value: target.outcome.prior_value, - bitemporal_sys_from_ms: target.outcome.bitemporal_sys_from_ms, - bitemporal_index_tuples: target.outcome.bitemporal_index_tuples, - secondary_index_added: target.outcome.secondary_index_added, - secondary_index_removed: target.outcome.secondary_index_removed, - chain_hash_prior: None, - }); - for delta in target.outcome.vector_inserts { - undo_log.push(UndoEntry::InsertVector { - index_key: delta.index_key, - vector_id: delta.vector_id, - collection: delta.collection, - field: delta.field, - doc_id: Some(delta.doc_id), - }); - } - for (key, entry_id) in target.outcome.spatial_inserts { - undo_log.push(UndoEntry::SpatialInsert { key, entry_id }); - } - for (key, prior) in target.outcome.stats_prior { - undo_log.push(UndoEntry::StatsRestore { key, prior }); - } - } - - undo_log.push(UndoEntry::PutDocument { - collection: collection.to_string(), - document_id: storage_key, - // The plan's `document_id` is the row's client identity. - identity: nodedb_types::RowIdentity::from_user_key(document_id), - old_value: outcome.prior_value, - bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, - bitemporal_index_tuples: outcome.bitemporal_index_tuples, - // Plain secondary-index entries this put added/removed; reversed on - // rollback so the index returns to its pre-tx state. - secondary_index_added: outcome.secondary_index_added, - secondary_index_removed: outcome.secondary_index_removed, - chain_hash_prior: chain.prior(), - }); - - // Reverse any HNSW vector inserts on rollback (one `InsertVector` undo - // per vector this put added to a per-field index). - for delta in outcome.vector_inserts { - undo_log.push(UndoEntry::InsertVector { - index_key: delta.index_key, - vector_id: delta.vector_id, - collection: delta.collection, - field: delta.field, - doc_id: Some(delta.doc_id), - }); - } - - // Reverse any spatial R-tree inserts on rollback (one `SpatialInsert` - // undo per per-field R-tree entry this put added). - for (key, entry_id) in outcome.spatial_inserts { - undo_log.push(UndoEntry::SpatialInsert { key, entry_id }); - } - - // Reverse the column-stats read-modify-write on rollback by restoring - // each captured pre-image. - for (key, prior) in outcome.stats_prior { - undo_log.push(UndoEntry::StatsRestore { key, prior }); - } + // A target write is a full document write with side effects of its + // own, reversed with the same entries the source row uses. + push_target_undo(undo_log, &target_writes); + push_put_undo( + undo_log, + DocumentRow { + database_id, + tid, + collection, + storage_key, + // The plan's `document_id` is the row's client identity. + identity: nodedb_types::RowIdentity::from_user_key(document_id), + }, + outcome, + chain.prior(), + ); // One row was written, and the count is REPORTED — `PointPut` and // `PointInsert` both render an `INSERT ` command tag, so their diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs index 145968df7..3c6e8785a 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs @@ -17,11 +17,11 @@ use crate::types::TenantId; use nodedb_physical::physical_plan::ColumnarInsertIntent; use nodedb_physical::physical_plan::document::UpdateValue; -use super::undo::{TimeseriesIngestUndo, UndoEntry}; +use super::undo::UndoEntry; /// Captured undo state for a pending columnar insert: the list of new PK bytes /// to insert, paired with the prior `RowLocation` of any displaced memtable rows. -type ColumnarUndoState = (Vec>, Vec<(Vec, RowLocation)>); +pub(in crate::data::executor) type ColumnarUndoState = (Vec>, Vec<(Vec, RowLocation)>); /// Parameters for [`CoreLoop::execute_tx_columnar_insert`]. pub(super) struct TxColumnarInsertParams<'a> { @@ -133,39 +133,53 @@ impl CoreLoop { payload: &[u8], intent: ColumnarInsertIntent, ) -> ColumnarUndoState { - let mut inserted_pks: Vec> = Vec::new(); - let mut displaced: Vec<(Vec, RowLocation)> = Vec::new(); - let Some(engine) = self.columnar_engines.get(collection_key) else { // Engine doesn't exist yet; execute_columnar_insert will create it. // row_count_before will be 0, so truncate_to(0) handles rollback. - return (inserted_pks, displaced); + return (Vec::new(), Vec::new()); }; - let ndb_rows: Vec = match nodedb_types::value_from_msgpack(payload) { Ok(nodedb_types::Value::Array(arr)) => arr, Ok(v @ nodedb_types::Value::Object(_)) => vec![v], - _ => return (inserted_pks, displaced), + _ => return (Vec::new(), Vec::new()), }; + let schema = engine.schema(); + let rows: Vec> = ndb_rows + .iter() + .filter_map(|row| match row { + nodedb_types::Value::Object(obj) => Some( + schema + .columns + .iter() + .map(|col| { + obj.get(&col.name) + .cloned() + .unwrap_or(nodedb_types::Value::Null) + }) + .collect(), + ), + _ => None, + }) + .collect(); + self.columnar_insert_undo_state(collection_key, &rows, intent) + } - let schema = engine.schema().clone(); - for row in &ndb_rows { - let obj = match row { - nodedb_types::Value::Object(m) => m, - _ => continue, - }; - - let values: Vec = schema - .columns - .iter() - .map(|col| { - obj.get(&col.name) - .cloned() - .unwrap_or(nodedb_types::Value::Null) - }) - .collect(); - - let Ok(pk_bytes) = engine.encode_pk_from_row(&values) else { + /// The PK bytes a columnar insert of `rows` (schema-ordered values) will + /// bind, and the memtable rows it will displace, captured before the + /// insert runs. + pub(in crate::data::executor) fn columnar_insert_undo_state( + &self, + collection_key: &(nodedb_types::DatabaseId, TenantId, String), + rows: &[Vec], + intent: ColumnarInsertIntent, + ) -> ColumnarUndoState { + let mut inserted_pks: Vec> = Vec::new(); + let mut displaced: Vec<(Vec, RowLocation)> = Vec::new(); + let Some(engine) = self.columnar_engines.get(collection_key) else { + return (inserted_pks, displaced); + }; + for values in rows { + let Ok(pk_bytes) = engine.encode_pk_from_row(values) else { continue; }; @@ -219,28 +233,7 @@ impl CoreLoop { } = params; let collection_key = (task.request.database_id, tid, collection.to_string()); - let undo = TimeseriesIngestUndo { - collection_key: collection_key.clone(), - memtable_before: self - .columnar_memtables - .get(&collection_key) - .map(|memtable| memtable.export_snapshot()), - memtable_config_before: self - .columnar_memtables - .get(&collection_key) - .map(|memtable| memtable.config()), - memtable_memory_bytes_before: self - .columnar_memtables - .get(&collection_key) - .map(|memtable| memtable.memory_bytes()), - last_value_cache_before: self.ts_last_value_caches.get(&collection_key).cloned(), - max_ingested_lsn_before: self.ts_max_ingested_lsn.get(&collection_key).copied(), - last_ts_ingest_before: self.last_ts_ingest, - reservation_bytes_before: self - .columnar_memtable_mem - .get(&collection_key) - .map(nodedb_mem::ReservationToken::size), - }; + let undo = self.capture_timeseries_ingest_undo(&collection_key); // Push before mutation. A panic in ingest is caught by the batch // driver, which can then restore this exact pre-image. diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs index c298aa477..24f45fe33 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs @@ -43,7 +43,7 @@ impl CoreLoop { let now_ms = current_ms(); let prior = self .kv_engine - .get(did, tid, collection.as_str(), key, now_ms); + .entry_image(did, tid, collection.as_str(), key, now_ms); let resp = self.execute_kv_incr( crate::data::executor::handlers::kv::atomic::KvAtomicCtx { task, @@ -65,7 +65,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: prior, + prior, }); Ok(resp) } @@ -80,7 +80,7 @@ impl CoreLoop { let now_ms = current_ms(); let prior = self .kv_engine - .get(did, tid, collection.as_str(), key, now_ms); + .entry_image(did, tid, collection.as_str(), key, now_ms); let resp = self.execute_kv_incr_float( crate::data::executor::handlers::kv::atomic::KvAtomicCtx { task, @@ -101,7 +101,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: prior, + prior, }); Ok(resp) } @@ -117,7 +117,7 @@ impl CoreLoop { let now_ms = current_ms(); let prior = self .kv_engine - .get(did, tid, collection.as_str(), key, now_ms); + .entry_image(did, tid, collection.as_str(), key, now_ms); let resp = self.execute_kv_cas( crate::data::executor::handlers::kv::atomic::KvAtomicCtx { task, @@ -140,7 +140,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: prior, + prior, }); Ok(resp) } @@ -156,7 +156,7 @@ impl CoreLoop { let now_ms = current_ms(); let prior = self .kv_engine - .get(did, tid, collection.as_str(), key, now_ms); + .entry_image(did, tid, collection.as_str(), key, now_ms); let resp = self.execute_kv_getset( crate::data::executor::handlers::kv::atomic::KvAtomicCtx { task, @@ -178,7 +178,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: prior, + prior, }); Ok(resp) } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs index f82662eb0..fb6688420 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs @@ -36,7 +36,7 @@ impl CoreLoop { let now_ms = current_ms(); let prior = self .kv_engine - .get(did, tid, collection.as_str(), key, now_ms); + .entry_image(did, tid, collection.as_str(), key, now_ms); let resp = self.execute_kv_put( task, crate::data::executor::handlers::kv::crud::KvWriteParams { @@ -59,7 +59,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: prior, + prior, }); Ok(resp) } @@ -95,7 +95,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: None, + prior: None, }); Ok(resp) } @@ -137,7 +137,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: None, + prior: None, }); } Ok(resp) @@ -149,7 +149,7 @@ impl CoreLoop { let now_ms = current_ms(); let prior = self .kv_engine - .get(did, tid, collection.as_str(), key, now_ms); + .entry_image(did, tid, collection.as_str(), key, now_ms); let resp = self.execute_kv(task, did, tid, op); if resp.status == Status::Error { return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { @@ -159,7 +159,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: prior, + prior, }); Ok(resp) } @@ -172,13 +172,13 @@ impl CoreLoop { } => { let now_ms = current_ms(); // Capture prior values for all keys that exist before deleting. - let priors: Vec<(Vec, Vec)> = keys + let priors: Vec<(Vec, crate::engine::kv::KvEntryImage)> = keys .iter() .filter_map(|k| { - let v = self - .kv_engine - .get(did, tid, collection.as_str(), k, now_ms)?; - Some((k.clone(), v)) + let image = + self.kv_engine + .entry_image(did, tid, collection.as_str(), k, now_ms)?; + Some((k.clone(), image)) }) .collect(); // In-transaction writes never carry `RETURNING`: the Control @@ -200,11 +200,11 @@ impl CoreLoop { detail: "kv delete failed".into(), })); } - for (key, prior_value) in priors { + for (key, prior) in priors { undo_log.push(UndoEntry::KvDelete { collection: collection.to_string(), key, - prior_value, + prior, }); } Ok(resp) @@ -218,13 +218,20 @@ impl CoreLoop { .. } => { let now_ms = current_ms(); - let prior_entries: Vec<(Vec, Option>)> = entries - .iter() - .map(|(k, _v)| { - let prior = self.kv_engine.get(did, tid, collection.as_str(), k, now_ms); - (k.clone(), prior) - }) - .collect(); + let prior_entries: Vec<(Vec, Option)> = + entries + .iter() + .map(|(k, _v)| { + let prior = self.kv_engine.entry_image( + did, + tid, + collection.as_str(), + k, + now_ms, + ); + (k.clone(), prior) + }) + .collect(); let resp = self.execute_kv_batch_put( task, crate::data::executor::handlers::kv::batch::KvBatchPutArgs { @@ -262,7 +269,7 @@ impl CoreLoop { let now_ms = current_ms(); let prior = self .kv_engine - .get(did, tid, collection.as_str(), key, now_ms); + .entry_image(did, tid, collection.as_str(), key, now_ms); let resp = self.execute_kv_field_set( crate::data::executor::handlers::kv::atomic::KvAtomicCtx { task, @@ -288,7 +295,7 @@ impl CoreLoop { undo_log.push(UndoEntry::KvPut { collection: collection.to_string(), key: key.clone(), - prior_value: prior, + prior, }); Ok(resp) } @@ -307,10 +314,10 @@ impl CoreLoop { let now_ms = current_ms(); let source_prior = self.kv_engine - .get(did, tid, collection.as_str(), source_key, now_ms); + .entry_image(did, tid, collection.as_str(), source_key, now_ms); let dest_prior = self.kv_engine - .get(did, tid, collection.as_str(), dest_key, now_ms); + .entry_image(did, tid, collection.as_str(), dest_key, now_ms); let resp = self.execute_kv(task, did, tid, op); if resp.status == Status::Error { return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { @@ -343,12 +350,20 @@ impl CoreLoop { dest_rls_write_check, } => { let now_ms = current_ms(); - let source_prior = - self.kv_engine - .get(did, tid, source_collection.as_str(), item_key, now_ms); - let dest_prior = - self.kv_engine - .get(did, tid, dest_collection.as_str(), dest_key, now_ms); + let source_prior = self.kv_engine.entry_image( + did, + tid, + source_collection.as_str(), + item_key, + now_ms, + ); + let dest_prior = self.kv_engine.entry_image( + did, + tid, + dest_collection.as_str(), + dest_key, + now_ms, + ); let resp = self.execute_kv_transfer_item( task, crate::data::executor::handlers::kv::transfer::TransferItemParams { diff --git a/nodedb/src/data/executor/handlers/transaction/undo/crdt_collection.rs b/nodedb/src/data/executor/handlers/transaction/undo/crdt_collection.rs new file mode 100644 index 000000000..b27d0cb9d --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/crdt_collection.rs @@ -0,0 +1,78 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Undo of one CRDT write a committed redo record installs. +//! +//! A Loro import merges, so a write cannot be withdrawn by importing +//! anything. The pre-image is the collection's Loro snapshot before the +//! write, and the undo replaces the collection's document with it. A tenant +//! engine the write created is removed again. The sparse projection of a +//! document row records its own undo where it is written. + +use crate::data::executor::core_loop::CoreLoop; +use crate::types::{DatabaseId, TenantId}; + +use super::UndoEntry; + +/// The Loro state of one CRDT collection before a write. +pub(in crate::data::executor) struct CrdtCollectionUndo { + pub database_id: DatabaseId, + pub tenant_id: TenantId, + pub collection: String, + /// Whether the tenant's CRDT engine existed. + pub engine_existed: bool, + /// The collection's snapshot, `None` when it held no document. + pub snapshot: Option>, +} + +impl CoreLoop { + /// Capture the Loro state of `collection` before a write. + pub(in crate::data::executor) fn capture_crdt_collection_undo( + &self, + database_id: DatabaseId, + tenant_id: TenantId, + collection: &str, + ) -> crate::Result { + let engine = self.crdt_engines.get(&(database_id, tenant_id)); + let snapshot = match engine { + Some(engine) => engine.export_snapshot_bytes(collection)?, + None => None, + }; + Ok(UndoEntry::CrdtCollection(Box::new(CrdtCollectionUndo { + database_id, + tenant_id, + collection: collection.to_string(), + engine_existed: engine.is_some(), + snapshot, + }))) + } + + /// Put the collection's Loro document back. + pub(super) fn apply_undo_crdt_collection( + &mut self, + entry_index: usize, + undo: CrdtCollectionUndo, + ) -> Result<(), (usize, String)> { + let key = (undo.database_id, undo.tenant_id); + if !undo.engine_existed { + self.crdt_engines.remove(&key); + return Ok(()); + } + let Some(engine) = self.crdt_engines.get_mut(&key) else { + return Err(( + entry_index, + format!( + "the CRDT engine of '{}' vanished before its write was rolled back", + undo.collection + ), + )); + }; + engine + .restore_collection_snapshot(&undo.collection, undo.snapshot.as_deref()) + .map_err(|e| { + ( + entry_index, + format!("restoring the CRDT collection '{}': {e}", undo.collection), + ) + }) + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs b/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs new file mode 100644 index 000000000..71949b057 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs @@ -0,0 +1,168 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Undo entries that reverse one document write, built from the outcome the +//! shared write path (`apply_point_put` / `apply_point_delete`) returned. +//! +//! The transaction batch and the committed-redo apply build them here, so a +//! rolled-back write reverses the same side effects on both paths: the row, +//! its secondary and versioned index entries, its HNSW vectors, its R-tree +//! entries, its column stats, its hash-chain head, the edges a delete +//! cascaded, and every materialized-sum target row the write folded into. +//! +//! Entries are pushed in the order the writes happened. Rollback runs the log +//! in reverse. + +use nodedb_types::{RowIdentity, StorageKey}; + +use crate::data::executor::enforcement::materialized_sum::apply::TargetWrite; +use crate::data::executor::handlers::point::apply_delete::PointDeleteOutcome; +use crate::data::executor::handlers::point::apply_put::PointPutOutcome; + +use super::UndoEntry; + +/// The document row one write touched. +pub(in crate::data::executor::handlers) struct DocumentRow<'a> { + pub database_id: u64, + pub tid: u64, + pub collection: &'a str, + pub storage_key: StorageKey, + /// The row's client identity, as its event names it. + pub identity: RowIdentity, +} + +/// Push the undo entries that reverse every materialized-sum target row the +/// write folded into. A target write is a full document write, so it has +/// index, vector, spatial and stats side effects of its own. The targets stay +/// with the caller, which also reports them in its response. +pub(in crate::data::executor::handlers) fn push_target_undo( + undo_log: &mut Vec, + targets: &[TargetWrite], +) { + for target in targets { + let outcome = &target.outcome; + undo_log.push(UndoEntry::PutDocument { + collection: target.collection.clone(), + document_id: StorageKey::for_surrogate(target.surrogate), + identity: target.identity.clone(), + old_value: outcome.prior_value.clone(), + bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, + bitemporal_index_tuples: outcome.bitemporal_index_tuples.clone(), + secondary_index_added: outcome.secondary_index_added.clone(), + secondary_index_removed: outcome.secondary_index_removed.clone(), + chain_hash_prior: None, + }); + push_put_side_effects( + undo_log, + outcome.vector_inserts.clone(), + outcome.spatial_inserts.clone(), + outcome.stats_prior.clone(), + ); + } +} + +/// Push the undo entries that reverse one document put. `chain_hash_prior` +/// is the hash-chain head before the put, `None` when the put did not touch +/// the chain. +pub(in crate::data::executor::handlers) fn push_put_undo( + undo_log: &mut Vec, + row: DocumentRow<'_>, + outcome: PointPutOutcome, + chain_hash_prior: Option>, +) { + undo_log.push(UndoEntry::PutDocument { + collection: row.collection.to_string(), + document_id: row.storage_key, + identity: row.identity, + old_value: outcome.prior_value, + bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, + bitemporal_index_tuples: outcome.bitemporal_index_tuples, + secondary_index_added: outcome.secondary_index_added, + secondary_index_removed: outcome.secondary_index_removed, + chain_hash_prior, + }); + push_put_side_effects( + undo_log, + outcome.vector_inserts, + outcome.spatial_inserts, + outcome.stats_prior, + ); +} + +/// Push the undo entries that reverse one document delete. A delete that +/// removed no row reverses only the side effects it still had. +pub(in crate::data::executor::handlers) fn push_delete_undo( + undo_log: &mut Vec, + row: DocumentRow<'_>, + outcome: PointDeleteOutcome, +) { + if let Some(old_value) = outcome.prior_value { + undo_log.push(UndoEntry::DeleteDocument { + collection: row.collection.to_string(), + document_id: row.storage_key, + identity: row.identity, + old_value, + bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, + bitemporal_index_tuples: outcome.bitemporal_index_tuples, + secondary_index_tuples: outcome.secondary_index_tuples, + chain_hash_prior: None, + }); + } + for delta in outcome.vector_deletes { + undo_log.push(UndoEntry::DeleteVector { + index_key: delta.index_key, + vector_id: delta.vector_id, + collection: delta.collection, + field: delta.field, + doc_id: Some(delta.doc_id), + }); + } + for (key, entry_id, bbox, document_id) in outcome.spatial_deletes { + undo_log.push(UndoEntry::SpatialDelete { + key, + entry_id, + bbox, + document_id, + }); + } + // `Some` only when this delete newly marked the node: a tombstone a prior + // committed write left is never un-marked. + if let Some(node_id) = outcome.mark_node_deleted { + undo_log.push(UndoEntry::MarkNodeDeleted { + database_id: row.database_id, + tid: row.tid, + node_id, + }); + } + for (collection, src_id, label, dst_id, old_properties) in outcome.edge_deletes { + undo_log.push(UndoEntry::DeleteEdge { + collection, + src_id, + label, + dst_id, + old_properties, + }); + } +} + +fn push_put_side_effects( + undo_log: &mut Vec, + vector_inserts: Vec, + spatial_inserts: Vec<(crate::data::executor::spatial_key::SpatialIndexKey, u64)>, + stats_prior: Vec, +) { + for delta in vector_inserts { + undo_log.push(UndoEntry::InsertVector { + index_key: delta.index_key, + vector_id: delta.vector_id, + collection: delta.collection, + field: delta.field, + doc_id: Some(delta.doc_id), + }); + } + for (key, entry_id) in spatial_inserts { + undo_log.push(UndoEntry::SpatialInsert { key, entry_id }); + } + for (key, prior) in stats_prior { + undo_log.push(UndoEntry::StatsRestore { key, prior }); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs index 4635daf54..6a263e1eb 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs @@ -204,52 +204,57 @@ pub(in crate::data::executor) enum UndoEntry { old_properties: Vec, }, /// Undo a KV write (Put / Insert / InsertIfAbsent / InsertOnConflictUpdate / - /// FieldSet / Incr / IncrFloat / Cas / GetSet) by restoring the prior value. + /// FieldSet / Incr / IncrFloat / Cas / GetSet) by reinstating the key's + /// prior state. /// - /// `prior_value == None` means the key did not exist before — undo deletes it. - /// `prior_value == Some(bytes)` means the key was overwritten — undo restores it. - /// - /// The KV hash table preserves existing non-ZERO surrogate bindings on `put`, - /// so passing `Surrogate::ZERO` during undo is safe: the original surrogate - /// remains bound in the entry. + /// `prior == None` means the key did not exist before: undo deletes it. + /// `prior == Some(image)` reinstalls the value, the absolute expiry + /// instant and the surrogate the key held. KvPut { collection: String, key: Vec, - prior_value: Option>, + prior: Option, }, - /// Undo a KV Delete by restoring one key's prior value. + /// Undo a KV Delete by reinstalling one key's prior state. /// /// One entry per key that was actually deleted. If a batch delete removed /// N keys, N `KvDelete` entries are pushed. KvDelete { collection: String, key: Vec, - prior_value: Vec, + prior: crate::engine::kv::KvEntryImage, }, - /// Undo a KV BatchPut by restoring prior values for all affected keys. + /// Undo a KV BatchPut by reinstating the prior state of every key. /// - /// Each element is `(key, prior_value)` where `prior_value == None` - /// means the key was newly inserted. + /// Each element is `(key, prior)` where `prior == None` means the key was + /// newly inserted. KvBatchPut { collection: String, - entries: Vec<(Vec, Option>)>, + entries: Vec<(Vec, Option)>, }, - /// Undo a KV Transfer (fungible) by restoring source and destination prior values. + /// Undo a KV Transfer (fungible) by reinstating the source and destination + /// prior state. KvTransfer { collection: String, source_key: Vec, - source_prior: Vec, + source_prior: crate::engine::kv::KvEntryImage, dest_key: Vec, - dest_prior: Option>, + dest_prior: Option, }, - /// Undo a KV TransferItem by restoring source and destination prior values. + /// Undo a KV TransferItem by reinstating the source and destination prior + /// state. KvTransferItem { source_collection: String, dest_collection: String, item_key: Vec, dest_key: Vec, - source_prior: Vec, - dest_prior: Option>, + source_prior: crate::engine::kv::KvEntryImage, + dest_prior: Option, + }, + /// Undo a KV `TRUNCATE` by reinstalling every row the collection held. + KvTruncate { + collection: String, + rows: Vec, }, /// Undo a KV `Expire`/`Persist` by restoring the key's prior TTL state. /// @@ -300,6 +305,51 @@ pub(in crate::data::executor) enum UndoEntry { tid: u64, node_id: String, }, + /// Undo a CRDT write by putting the collection's Loro document back. + CrdtCollection(Box), + /// Undo an array cell write by putting back the memtable tiles it + /// touched. + ArrayTiles { + array_id: nodedb_array::types::ArrayId, + snapshot: crate::engine::array::ArrayTileSnapshot, + }, + /// Undo a vector write a committed redo record installed: withdraw the + /// nodes it inserted and put every binding, tombstone, sidecar and + /// bitmap entry back. + VectorWrite(Box), + /// Undo a sparse-vector write: put the document back to its prior entries + /// and the index's id counter back. `next_id == None` means the index did + /// not exist before the write. + SparseDoc { + key: (nodedb_types::DatabaseId, TenantId, String, String), + doc_id: String, + prior: Option, + next_id: Option, + }, + /// Undo a vector-primary truncate: put the detached collection and every + /// sidecar row back. + VectorTruncate(Box), + /// Undo a sync-ingested spatial write by reinstalling the row, the R-tree + /// entry and the reverse-map record it replaced. + SpatialRow(Box), + /// Undo a sync-ingested full-text write by putting the document's index + /// footprint back. + FtsDocument(Box), + /// Undo the sync high-water-mark advance of a sync-ingested write. `prior + /// == None` means the stream had no mark before. + SyncHwm { + producer_id: u64, + stream_id: u64, + prior: Option, + }, + /// Undo a node-label set or removal: each label the op touched goes back + /// to whether the node carried it before (`true` = it did). + NodeLabels { + database_id: u64, + tid: u64, + node_id: String, + prior: Vec<(String, bool)>, + }, /// Undo a columnar insert by rolling back in-memory state. /// /// `row_count_before` is the memtable row count snapshot taken before the @@ -346,6 +396,11 @@ pub(in crate::data::executor) enum UndoEntry { collection_key: (nodedb_types::DatabaseId, TenantId, String), restored: Vec<(Vec, nodedb_columnar::pk_index::RowLocation)>, }, + /// Undo the creation of a columnar engine by the write: the collection + /// held no engine before it. + ColumnarEngineCreated { + collection_key: (nodedb_types::DatabaseId, TenantId, String), + }, /// Undo a transaction-deferred timeseries ingest from its complete /// pre-image. Row-count truncation is insufficient: ingest can evolve /// schema/dictionaries and update the last-value cache before a later diff --git a/nodedb/src/data/executor/handlers/transaction/undo/fts_doc.rs b/nodedb/src/data/executor/handlers/transaction/undo/fts_doc.rs new file mode 100644 index 000000000..4b8a1b5e9 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/fts_doc.rs @@ -0,0 +1,82 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Undo of one sync-ingested full-text write (`TextOp::FtsIndexDoc` / +//! `FtsDeleteDoc`). +//! +//! The write replaces the document's whole index footprint. Its pre-image is +//! the footprint before the write (`InvertedIndex::document_image`), and the +//! undo re-indexes it, or removes the document when it was not indexed. + +use nodedb_types::Surrogate; +use nodedb_types::sync::wire::SyncProvenance; + +use crate::data::executor::core_loop::CoreLoop; +use crate::engine::sparse::inverted::FtsDocImage; +use crate::types::{DatabaseId, TenantId}; + +use super::UndoEntry; + +/// The pre-image of one document's index footprint. +pub(in crate::data::executor) struct FtsDocUndo { + pub database_id: u64, + pub tid: u64, + pub collection: String, + pub surrogate: Surrogate, + /// `None` when the document was not indexed before the write. + pub prior: Option, +} + +impl CoreLoop { + /// The undo entries of one full-text write: the document's footprint and + /// the sync high-water mark its provenance advances. + pub(in crate::data::executor) fn capture_fts_doc_undo( + &self, + database_id: DatabaseId, + tid: u64, + collection: &str, + surrogate: Surrogate, + provenance: Option<&SyncProvenance>, + ) -> crate::Result> { + let prior = self.inverted.document_image( + database_id.as_u64(), + TenantId::new(tid), + collection, + surrogate, + )?; + let mut undo = vec![UndoEntry::FtsDocument(Box::new(FtsDocUndo { + database_id: database_id.as_u64(), + tid, + collection: collection.to_string(), + surrogate, + prior, + }))]; + undo.extend(self.capture_sync_hwm_undo(provenance)); + Ok(undo) + } + + /// Put the document's index footprint back. + pub(super) fn apply_undo_fts_doc( + &mut self, + entry_index: usize, + undo: FtsDocUndo, + ) -> Result<(), (usize, String)> { + self.inverted + .restore_document_image( + undo.database_id, + TenantId::new(undo.tid), + &undo.collection, + undo.surrogate, + undo.prior.as_ref(), + ) + .map_err(|e| { + ( + entry_index, + format!( + "restoring the full-text footprint of {} in '{}': {e}", + undo.surrogate.as_u32(), + undo.collection + ), + ) + }) + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs index f1abba4f0..5e0bf276b 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs @@ -38,6 +38,31 @@ impl CoreLoop { _ => unreachable!("apply_undo_mark_node called with non-mark-node entry"), } } + + /// Put every label a node-label op touched back to its prior state. + pub(super) fn apply_undo_node_labels( + &mut self, + entry_index: usize, + database_id: u64, + tid: u64, + node_id: &str, + prior: Vec<(String, bool)>, + ) -> Result<(), (usize, String)> { + let partition = self.csr_partition_mut(database_id, tid); + for (label, carried) in prior { + if carried { + partition.add_node_label(node_id, &label).map_err(|e| { + ( + entry_index, + format!("restoring label '{label}' on node '{node_id}': {e}"), + ) + })?; + } else { + partition.remove_node_label(node_id, &label); + } + } + Ok(()) + } } #[cfg(test)] diff --git a/nodedb/src/data/executor/handlers/transaction/undo/kv.rs b/nodedb/src/data/executor/handlers/transaction/undo/kv.rs index 297c7a4be..e2031622f 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/kv.rs @@ -13,6 +13,20 @@ use crate::engine::kv::current_ms; use super::UndoEntry; +fn kv_key<'a>( + did: u64, + tid: u64, + collection: &'a str, + key: &'a [u8], +) -> crate::engine::kv::KvKeyRef<'a> { + crate::engine::kv::KvKeyRef { + database_id: did, + tenant_id: tid, + collection, + key, + } +} + impl CoreLoop { pub(super) fn apply_undo_kv( &mut self, @@ -25,47 +39,25 @@ impl CoreLoop { UndoEntry::KvPut { collection, key, - prior_value, + prior, } => { - let now_ms = current_ms(); - if let Some(old) = prior_value { - self.kv_engine.put(crate::engine::kv::KvPutParams { - database_id: did, - tenant_id: tid, - collection: &collection, - key: &key, - value: &old, - ttl_ms: 0, - now_ms, - surrogate: nodedb_types::Surrogate::ZERO, - }); - } else { - self.kv_engine.delete( - did, - tid, - &collection, - std::slice::from_ref(&key), - now_ms, - ); - } + self.kv_engine.reinstate_entry( + kv_key(did, tid, &collection, &key), + prior.as_ref(), + current_ms(), + ); Ok(()) } UndoEntry::KvDelete { collection, key, - prior_value, + prior, } => { - let now_ms = current_ms(); - self.kv_engine.put(crate::engine::kv::KvPutParams { - database_id: did, - tenant_id: tid, - collection: &collection, - key: &key, - value: &prior_value, - ttl_ms: 0, - now_ms, - surrogate: nodedb_types::Surrogate::ZERO, - }); + self.kv_engine.restore_entry_image( + kv_key(did, tid, &collection, &key), + &prior, + current_ms(), + ); Ok(()) } UndoEntry::KvBatchPut { @@ -73,21 +65,12 @@ impl CoreLoop { entries, } => { let now_ms = current_ms(); - for (key, prior_value) in entries { - if let Some(old) = prior_value { - self.kv_engine.put(crate::engine::kv::KvPutParams { - database_id: did, - tenant_id: tid, - collection: &collection, - key: &key, - value: &old, - ttl_ms: 0, - now_ms, - surrogate: nodedb_types::Surrogate::ZERO, - }); - } else { - self.kv_engine.delete(did, tid, &collection, &[key], now_ms); - } + for (key, prior) in entries { + self.kv_engine.reinstate_entry( + kv_key(did, tid, &collection, &key), + prior.as_ref(), + now_ms, + ); } Ok(()) } @@ -99,31 +82,16 @@ impl CoreLoop { dest_prior, } => { let now_ms = current_ms(); - self.kv_engine.put(crate::engine::kv::KvPutParams { - database_id: did, - tenant_id: tid, - collection: &collection, - key: &source_key, - value: &source_prior, - ttl_ms: 0, + self.kv_engine.restore_entry_image( + kv_key(did, tid, &collection, &source_key), + &source_prior, now_ms, - surrogate: nodedb_types::Surrogate::ZERO, - }); - if let Some(old) = dest_prior { - self.kv_engine.put(crate::engine::kv::KvPutParams { - database_id: did, - tenant_id: tid, - collection: &collection, - key: &dest_key, - value: &old, - ttl_ms: 0, - now_ms, - surrogate: nodedb_types::Surrogate::ZERO, - }); - } else { - self.kv_engine - .delete(did, tid, &collection, &[dest_key], now_ms); - } + ); + self.kv_engine.reinstate_entry( + kv_key(did, tid, &collection, &dest_key), + dest_prior.as_ref(), + now_ms, + ); Ok(()) } UndoEntry::KvTransferItem { @@ -134,39 +102,39 @@ impl CoreLoop { source_prior, dest_prior, } => { - let now_ms = current_ms(); // Cross-collection move: the forward op deleted `item_key` from - // `source_collection` and wrote to `dest_key` in `dest_collection` - // (e.g. inventory → archive). Reverse both halves: re-insert the - // source row, then undo the destination write below. `source_prior` - // is always Some because the forward op required the source to - // exist; `dest_prior` is None when the dest key was a new insert - // and Some(old) when it overwrote an existing row. - self.kv_engine.put(crate::engine::kv::KvPutParams { - database_id: did, - tenant_id: tid, - collection: &source_collection, - key: &item_key, - value: &source_prior, - ttl_ms: 0, + // `source_collection` and wrote `dest_key` in `dest_collection`. + // Both halves are reinstated: the source row always existed, + // the destination key may have been absent. + let now_ms = current_ms(); + self.kv_engine.restore_entry_image( + kv_key(did, tid, &source_collection, &item_key), + &source_prior, now_ms, - surrogate: nodedb_types::Surrogate::ZERO, - }); - // Undo the dest write. - if let Some(old) = dest_prior { - self.kv_engine.put(crate::engine::kv::KvPutParams { - database_id: did, - tenant_id: tid, - collection: &dest_collection, - key: &dest_key, - value: &old, - ttl_ms: 0, + ); + self.kv_engine.reinstate_entry( + kv_key(did, tid, &dest_collection, &dest_key), + dest_prior.as_ref(), + now_ms, + ); + Ok(()) + } + UndoEntry::KvTruncate { collection, rows } => { + // Every write after the truncate was reversed first, so the + // collection holds what the truncate left. Empty it and + // reinstall every row it held. + let now_ms = current_ms(); + self.kv_engine.truncate(did, tid, &collection); + for row in rows { + self.kv_engine.restore_entry_image( + kv_key(did, tid, &collection, &row.key), + &crate::engine::kv::KvEntryImage { + value: row.value, + expire_at_ms: row.expire_at_ms, + surrogate: row.surrogate, + }, now_ms, - surrogate: nodedb_types::Surrogate::ZERO, - }); - } else { - self.kv_engine - .delete(did, tid, &dest_collection, &[dest_key], now_ms); + ); } Ok(()) } @@ -646,4 +614,100 @@ mod tests { "restored index must rank identically to the original" ); } + + fn seed_with_expiry_and_surrogate(core: &mut CoreLoop) -> crate::engine::kv::KvEntryImage { + let now_ms = current_ms(); + core.kv_engine.put_with_absolute_expiry( + crate::engine::kv::KvPutParams { + database_id: DB, + tenant_id: TID, + collection: "cache", + key: b"k", + value: b"old", + ttl_ms: 0, + now_ms, + surrogate: nodedb_types::Surrogate::new(9), + }, + now_ms + 3_600_000, + ); + core.kv_engine + .entry_image(DB, TID, "cache", b"k", now_ms) + .expect("seeded key") + } + + #[test] + fn a_rolled_back_overwrite_restores_the_value_expiry_and_surrogate() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let before = seed_with_expiry_and_surrogate(&mut core); + put_kv(&mut core, "cache", b"k", b"new", 0); + + core.rollback_undo_log( + DB, + TID, + vec![UndoEntry::KvPut { + collection: "cache".into(), + key: b"k".to_vec(), + prior: Some(before.clone()), + }], + ) + .expect("rollback"); + + assert_eq!( + core.kv_engine + .entry_image(DB, TID, "cache", b"k", current_ms()), + Some(before) + ); + } + + #[test] + fn a_rolled_back_delete_restores_the_expiry_and_surrogate() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let before = seed_with_expiry_and_surrogate(&mut core); + core.kv_engine + .delete(DB, TID, "cache", &[b"k".to_vec()], current_ms()); + + core.rollback_undo_log( + DB, + TID, + vec![UndoEntry::KvDelete { + collection: "cache".into(), + key: b"k".to_vec(), + prior: before.clone(), + }], + ) + .expect("rollback"); + + assert_eq!( + core.kv_engine + .entry_image(DB, TID, "cache", b"k", current_ms()), + Some(before) + ); + } + + #[test] + fn a_rolled_back_truncate_reinstalls_every_row() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let before = seed_with_expiry_and_surrogate(&mut core); + let rows = core.kv_engine.export_collection(DB, TID, "cache"); + core.kv_engine.truncate(DB, TID, "cache"); + + core.rollback_undo_log( + DB, + TID, + vec![UndoEntry::KvTruncate { + collection: "cache".into(), + rows, + }], + ) + .expect("rollback"); + + assert_eq!( + core.kv_engine + .entry_image(DB, TID, "cache", b"k", current_ms()), + Some(before) + ); + } } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/mod.rs b/nodedb/src/data/executor/handlers/transaction/undo/mod.rs index f517a297f..591dea3b7 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/mod.rs @@ -4,16 +4,23 @@ pub(super) mod apply; pub(super) mod balanced; +pub(in crate::data::executor) mod crdt_collection; pub(super) mod document; pub(super) mod document_fts; +pub(in crate::data::executor::handlers) mod document_outcome; pub(super) mod entry; +pub(in crate::data::executor) mod fts_doc; pub(super) mod graph_node; pub(super) mod kv; pub(super) mod rollback; pub(super) mod spatial; +pub(in crate::data::executor) mod spatial_row; pub(super) mod stats; +pub(super) mod sync_hwm; pub(super) mod timeseries; pub(super) mod truncate_columnar; +pub(in crate::data::executor) mod vector_truncate; +pub(in crate::data::executor) mod vector_write; pub(in crate::data::executor) use entry::{ ColumnarTruncateUndo, TimeseriesIngestUndo, TimeseriesTruncateUndo, UndoEntry, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index d68103185..5fe856d60 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -84,11 +84,18 @@ impl CoreLoop { | UndoEntry::KvBatchPut { .. } | UndoEntry::KvTransfer { .. } | UndoEntry::KvTransferItem { .. } + | UndoEntry::KvTruncate { .. } | UndoEntry::KvTtl { .. } | UndoEntry::SortedIndexDdl { .. } => self.apply_undo_kv(did, tid, entry_index, entry), UndoEntry::ColumnarInsert { .. } | UndoEntry::ColumnarUpdate { .. } | UndoEntry::ColumnarDelete { .. } => self.apply_undo_columnar(entry_index, entry), + UndoEntry::ColumnarEngineCreated { collection_key } => { + self.columnar_engines.remove(&collection_key); + self.columnar_flushed_segments.remove(&collection_key); + self.columnar_flushed_surrogates.remove(&collection_key); + Ok(()) + } UndoEntry::TimeseriesIngest(_) => self.apply_undo_timeseries(entry_index, entry), UndoEntry::ColumnarTruncate(undo) => { self.apply_undo_columnar_truncate(entry_index, undo) @@ -98,6 +105,55 @@ impl CoreLoop { } UndoEntry::StatsRestore { .. } => self.apply_undo_stats(entry_index, entry), UndoEntry::MarkNodeDeleted { .. } => self.apply_undo_mark_node(entry_index, entry), + UndoEntry::SpatialRow(undo) => self.apply_undo_spatial_row(entry_index, *undo), + UndoEntry::VectorWrite(undo) => self.apply_undo_vector_write(entry_index, *undo), + UndoEntry::CrdtCollection(undo) => self.apply_undo_crdt_collection(entry_index, *undo), + UndoEntry::ArrayTiles { array_id, snapshot } => self + .array_engine + .restore_tiles(&array_id, snapshot) + .map_err(|e| { + ( + entry_index, + format!( + "restoring the memtable tiles of array '{}': {e}", + array_id.name + ), + ) + }), + UndoEntry::SparseDoc { + key, + doc_id, + prior, + next_id, + } => { + match next_id { + Some(next_id) => { + if let Some(index) = self.sparse_vector_indexes.get_mut(&key) { + index.roll_back_doc(&doc_id, prior.as_ref(), next_id); + } + } + None => { + self.sparse_vector_indexes.remove(&key); + } + } + Ok(()) + } + UndoEntry::VectorTruncate(undo) => self.apply_undo_vector_truncate(entry_index, *undo), + UndoEntry::FtsDocument(undo) => self.apply_undo_fts_doc(entry_index, *undo), + UndoEntry::SyncHwm { + producer_id, + stream_id, + prior, + } => { + self.apply_undo_sync_hwm(producer_id, stream_id, prior); + Ok(()) + } + UndoEntry::NodeLabels { + database_id, + tid: label_tid, + node_id, + prior, + } => self.apply_undo_node_labels(entry_index, database_id, label_tid, &node_id, prior), } } } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/spatial_row.rs b/nodedb/src/data/executor/handlers/transaction/undo/spatial_row.rs new file mode 100644 index 000000000..a67b5bd34 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/spatial_row.rs @@ -0,0 +1,133 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Undo of one sync-ingested spatial write (`SpatialOp::Insert` / `Delete`). +//! +//! Such a write replaces three things at once: the geometry row in the +//! sparse store, the entry in the field's R-tree, and its reverse +//! `spatial_doc_map` record. The pre-image of all three is captured before +//! the write, and the undo puts every one back. + +use nodedb_types::{BoundingBox, StorageKey, Surrogate}; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::spatial_key::SpatialIndexKey; +use crate::types::{DatabaseId, TenantId}; +use crate::util::fnv1a_hash; + +use super::UndoEntry; + +/// The pre-image of one spatial row. +pub(in crate::data::executor) struct SpatialRowUndo { + pub key: SpatialIndexKey, + pub storage_key: StorageKey, + /// The sparse-store row before the write, `None` when absent. + pub prior_row: Option>, + /// The R-tree entry id the write keys on. + pub entry_id: u64, + /// The R-tree entry and its reverse-map value before the write, `None` + /// when the field held no entry for the row. + pub prior_entry: Option<(BoundingBox, String)>, +} + +impl CoreLoop { + /// Capture the pre-image of the spatial row `surrogate` in `field` of + /// `collection`. + pub(in crate::data::executor) fn capture_spatial_row_undo( + &self, + database_id: DatabaseId, + tid: u64, + collection: &str, + field: &str, + surrogate: Surrogate, + ) -> crate::Result { + let storage_key = StorageKey::for_surrogate(surrogate); + let prior_row = self + .sparse + .get(database_id.as_u64(), tid, collection, &storage_key)?; + let key: SpatialIndexKey = ( + database_id, + TenantId::new(tid), + collection.to_string(), + field.to_string(), + ); + let entry_id = fnv1a_hash(storage_key.to_string().as_bytes()); + let doc_map_key = (key.0, key.1, key.2.clone(), key.3.clone(), entry_id); + let prior_entry = self + .spatial_doc_map + .get(&doc_map_key) + .and_then(|document_id| { + let bbox = self + .spatial_indexes + .get(&key)? + .entries() + .into_iter() + .find(|entry| entry.id == entry_id) + .map(|entry| entry.bbox)?; + Some((bbox, document_id.clone())) + }); + Ok(UndoEntry::SpatialRow(Box::new(SpatialRowUndo { + key, + storage_key, + prior_row, + entry_id, + prior_entry, + }))) + } + + /// Put the row, the R-tree entry and the reverse-map record back. + pub(super) fn apply_undo_spatial_row( + &mut self, + entry_index: usize, + undo: SpatialRowUndo, + ) -> Result<(), (usize, String)> { + let SpatialRowUndo { + key, + storage_key, + prior_row, + entry_id, + prior_entry, + } = undo; + let database_id = key.0.as_u64(); + let tid = key.1.as_u64(); + let restored = match &prior_row { + Some(row) => self + .sparse + .put(database_id, tid, &key.2, &storage_key, row) + .map(drop), + None => self + .sparse + .delete(database_id, tid, &key.2, &storage_key) + .map(drop), + }; + restored.map_err(|e| { + ( + entry_index, + format!("restoring spatial row in '{}': {e}", key.2), + ) + })?; + + if let Some(rtree) = self.spatial_indexes.get_mut(&key) { + rtree.delete(entry_id); + } + let doc_map_key = (key.0, key.1, key.2.clone(), key.3.clone(), entry_id); + match prior_entry { + Some((bbox, document_id)) => { + let memory = nodedb_mem::ScopedMemory::new( + self.governor.clone(), + key.0, + key.1, + nodedb_mem::EngineId::Spatial, + ); + self.spatial_indexes + .entry(key) + .or_insert_with(|| crate::engine::spatial::RTree::new(memory)) + .insert(crate::engine::spatial::RTreeEntry { id: entry_id, bbox }); + self.spatial_doc_map.insert(doc_map_key, document_id); + } + None => { + self.spatial_doc_map.remove(&doc_map_key); + } + } + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/sync_hwm.rs b/nodedb/src/data/executor/handlers/transaction/undo/sync_hwm.rs new file mode 100644 index 000000000..560d65ce2 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/sync_hwm.rs @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Undo of a sync high-water-mark advance. +//! +//! A sync-ingested write carries its producer's sequence number, and applying +//! it advances the in-memory high-water mark of that `(producer, stream)`. +//! Rolling the write back puts the mark back, so the producer's frame is +//! admitted again instead of being taken for a duplicate. + +use nodedb_types::sync::wire::SyncProvenance; + +use crate::data::executor::core_loop::CoreLoop; + +use super::UndoEntry; + +impl CoreLoop { + /// The undo entry for the high-water-mark advance a write with + /// `provenance` makes. `None` when the write carries no identified + /// producer: nothing advances. + pub(in crate::data::executor) fn capture_sync_hwm_undo( + &self, + provenance: Option<&SyncProvenance>, + ) -> Option { + let prov = provenance.filter(|prov| prov.producer_id != 0)?; + Some(UndoEntry::SyncHwm { + producer_id: prov.producer_id, + stream_id: prov.stream_id, + prior: self + .sync_hwm + .get(&(prov.producer_id, prov.stream_id)) + .copied(), + }) + } + + /// Put the high-water mark of `(producer_id, stream_id)` back to `prior`. + pub(super) fn apply_undo_sync_hwm( + &mut self, + producer_id: u64, + stream_id: u64, + prior: Option, + ) { + match prior { + Some(seq) => { + self.sync_hwm.insert((producer_id, stream_id), seq); + } + None => { + self.sync_hwm.remove(&(producer_id, stream_id)); + } + } + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs b/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs index 8e12fd99b..06c0aa907 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs @@ -1,8 +1,9 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Undo of a timeseries `TRUNCATE` inside a transaction batch: reinstall the -//! in-memory state `execute_timeseries_truncate` moved out and rename the -//! partition directory back to its live name. +//! Timeseries undo: the pre-image capture of an ingest, and the undo of a +//! `TRUNCATE` inside a transaction batch, which reinstalls the in-memory +//! state `execute_timeseries_truncate` moved out and renames the partition +//! directory back to its live name. //! //! The rename is the only durable step. A rename that fails is fatal to the //! rollback (`Err` → `RollbackFailed`): the memory state would say the rows @@ -10,10 +11,35 @@ //! would answer from half a collection. use crate::data::executor::core_loop::CoreLoop; +use crate::types::{DatabaseId, TenantId}; -use super::TimeseriesTruncateUndo; +use super::{TimeseriesIngestUndo, TimeseriesTruncateUndo}; impl CoreLoop { + /// The complete in-memory pre-image of a timeseries collection before an + /// ingest mutates it: the memtable, its config and resident footprint, + /// the last-value cache, the ingest watermark, the ingest timer, and the + /// memtable's reservation. + pub(in crate::data::executor) fn capture_timeseries_ingest_undo( + &self, + collection_key: &(DatabaseId, TenantId, String), + ) -> TimeseriesIngestUndo { + let memtable = self.columnar_memtables.get(collection_key); + TimeseriesIngestUndo { + collection_key: collection_key.clone(), + memtable_before: memtable.map(|memtable| memtable.export_snapshot()), + memtable_config_before: memtable.map(|memtable| memtable.config()), + memtable_memory_bytes_before: memtable.map(|memtable| memtable.memory_bytes()), + last_value_cache_before: self.ts_last_value_caches.get(collection_key).cloned(), + max_ingested_lsn_before: self.ts_max_ingested_lsn.get(collection_key).copied(), + last_ts_ingest_before: self.last_ts_ingest, + reservation_bytes_before: self + .columnar_memtable_mem + .get(collection_key) + .map(nodedb_mem::ReservationToken::size), + } + } + pub(super) fn apply_undo_timeseries_truncate( &mut self, entry_index: usize, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/vector_truncate.rs b/nodedb/src/data/executor/handlers/transaction/undo/vector_truncate.rs new file mode 100644 index 000000000..2ef9cd200 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/vector_truncate.rs @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Undo of a vector-primary `TRUNCATE` a committed redo record installs. +//! +//! A truncate removes the mmap files of the collection's sealed segments, so +//! it cannot be reversed after the fact. Before the truncate runs, the +//! install detaches the collection whole and puts an empty collection with +//! the same configuration in its place; the truncate then empties that one. +//! The undo puts the detached collection and every sidecar row back. Once +//! the record installed, the detached collection is truncated for good, +//! which removes its segment files. + +use nodedb_types::{StorageKey, Surrogate}; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::vector_direct_row::VectorIndexKey; +use crate::engine::vector::collection::VectorCollection; + +use super::UndoEntry; + +/// The collection and sidecar rows a truncate would remove. +pub(in crate::data::executor) struct VectorTruncateUndo { + pub index_key: VectorIndexKey, + pub tid: u64, + pub collection: String, + /// `None` when the index did not exist. + pub prior: Option, + pub sidecars: Vec<(Surrogate, Vec)>, +} + +impl CoreLoop { + /// Detach the collection at `index_key` and capture every sidecar row of + /// `collection`, leaving an empty collection with the same configuration + /// for the truncate to empty. + pub(in crate::data::executor) fn detach_vector_collection_for_truncate( + &mut self, + index_key: &VectorIndexKey, + tid: u64, + collection: &str, + ) -> crate::Result { + let database_id = index_key.0.as_u64(); + let surrogates = self + .scan_vector_sidecar_matches(database_id, tid, collection, &[]) + .map_err(crate::Error::DataPlane)?; + let mut sidecars = Vec::with_capacity(surrogates.len()); + for surrogate in surrogates { + let key = StorageKey::for_surrogate(surrogate); + if let Some(bytes) = self.sparse.get(database_id, tid, collection, &key)? { + sidecars.push((surrogate, bytes)); + } + } + let prior = match self.vector_collections.get(index_key) { + Some(coll) => { + let empty = coll.detached_empty(); + self.vector_collections.insert(index_key.clone(), empty) + } + None => None, + }; + Ok(UndoEntry::VectorTruncate(Box::new(VectorTruncateUndo { + index_key: index_key.clone(), + tid, + collection: collection.to_string(), + prior, + sidecars, + }))) + } + + /// Put the detached collection and every sidecar row back. + pub(super) fn apply_undo_vector_truncate( + &mut self, + entry_index: usize, + undo: VectorTruncateUndo, + ) -> Result<(), (usize, String)> { + let VectorTruncateUndo { + index_key, + tid, + collection, + prior, + sidecars, + } = undo; + let database_id = index_key.0.as_u64(); + match prior { + Some(coll) => { + self.vector_collections.insert(index_key, coll); + } + None => { + self.vector_collections.remove(&index_key); + } + } + for (surrogate, bytes) in sidecars { + let key = StorageKey::for_surrogate(surrogate); + self.sparse + .put(database_id, tid, &collection, &key, &bytes) + .map_err(|e| { + ( + entry_index, + format!("restoring the sidecar of {key} in '{collection}': {e}"), + ) + })?; + self.doc_cache + .invalidate(database_id, tid, &collection, &key); + } + Ok(()) + } + + /// Truncate for good every collection a committed install detached, + /// removing the mmap files of its sealed segments. + pub(in crate::data::executor) fn finalize_vector_truncates( + &mut self, + undo_log: &mut [UndoEntry], + ) { + for entry in undo_log { + if let UndoEntry::VectorTruncate(undo) = entry + && let Some(prior) = undo.prior.as_mut() + { + prior.truncate(); + } + } + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs b/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs new file mode 100644 index 000000000..17df75ad9 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Undo of one vector write a committed redo record installs: an HNSW or +//! IVF-PQ insert, a delete by node or by surrogate, a multi-vector write, or a +//! vector-primary row write. +//! +//! The pre-image is taken before the write: the collection's write mark +//! (`VectorCollection::write_mark`), the IVF-PQ add counter, and for a +//! vector-primary collection the sidecar row and payload bitmap entries of +//! every named row. The undo withdraws every node the write inserted, puts +//! every binding and tombstone back, and restores the sidecars and bitmap +//! entries. A collection the write created is removed again. + +use std::collections::HashMap; + +use nodedb_types::{StorageKey, Surrogate, Value}; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::vector_direct_row::VectorIndexKey; +use crate::engine::vector::collection::VectorWriteMark; + +use super::UndoEntry; + +/// The IVF-PQ state of a collection before a write. +pub(in crate::data::executor) struct IvfMark { + /// Vectors the index held. + pub count: u32, + pub trained: bool, +} + +/// The pre-image of one vector write. +pub(in crate::data::executor) struct VectorWriteUndo { + pub index_key: VectorIndexKey, + pub tid: u64, + /// The collection name the sidecars are stored under. + pub collection: String, + /// `None` when the collection did not exist before the write. + pub mark: Option, + /// Whether the write found no `vector_params` entry for the key. + pub params_absent: bool, + /// `None` when the collection had no IVF-PQ index. + pub ivf: Option, + /// Sidecar bytes of every named surrogate, `None` when absent. Empty for + /// a write that stores no sidecar. + pub sidecars: Vec<(Surrogate, Option>)>, + /// Payload bitmap rows of the named nodes that were live. + pub payload_rows: Vec<(u32, HashMap)>, +} + +/// What one vector write names. +pub(in crate::data::executor) struct VectorWriteTarget<'a> { + pub index_key: &'a VectorIndexKey, + pub tid: u64, + pub collection: &'a str, + pub surrogates: &'a [Surrogate], + pub ids: &'a [u32], + /// Whether the write stores sidecar rows (a vector-primary write). + pub sidecars: bool, +} + +impl CoreLoop { + /// Capture the pre-image of the vector write `target` names. + pub(in crate::data::executor) fn capture_vector_write_undo( + &self, + target: VectorWriteTarget<'_>, + ) -> crate::Result { + let VectorWriteTarget { + index_key, + tid, + collection, + surrogates, + ids, + sidecars, + } = target; + let database_id = index_key.0.as_u64(); + let coll = self.vector_collections.get(index_key); + let mut sidecar_rows = Vec::new(); + if sidecars { + for &surrogate in surrogates { + let key = StorageKey::for_surrogate(surrogate); + let bytes = self.sparse.get(database_id, tid, collection, &key)?; + sidecar_rows.push((surrogate, bytes)); + } + } + let mut payload_rows = Vec::new(); + if let Some(coll) = coll.filter(|coll| !coll.payload.is_empty()) { + let named = surrogates + .iter() + .filter_map(|s| coll.local_for_surrogate(*s).map(|id| (id, *s))) + .chain( + ids.iter() + .filter_map(|id| coll.get_surrogate(*id).map(|s| (*id, s))), + ); + for (id, surrogate) in named { + if !coll.is_live(id) { + continue; + } + let key = StorageKey::for_surrogate(surrogate); + if let Some(bytes) = self.sparse.get(database_id, tid, collection, &key)? { + let fields = + crate::data::executor::handlers::vector_upsert::decode_payload_lowercased( + &bytes, + ) + .map_err(|e| crate::Error::Internal { + detail: format!("vector sidecar of {key} does not decode: {e}"), + })?; + payload_rows.push((id, fields)); + } + } + } + Ok(UndoEntry::VectorWrite(Box::new(VectorWriteUndo { + index_key: index_key.clone(), + tid, + collection: collection.to_string(), + mark: coll.map(|coll| coll.write_mark(surrogates, ids)), + params_absent: !self.vector_params.contains_key(index_key), + ivf: self.ivf_indexes.get(index_key).map(|ivf| IvfMark { + count: ivf.len() as u32, + trained: ivf.is_trained(), + }), + sidecars: sidecar_rows, + payload_rows, + }))) + } + + /// Reverse one vector write. + pub(super) fn apply_undo_vector_write( + &mut self, + entry_index: usize, + undo: VectorWriteUndo, + ) -> Result<(), (usize, String)> { + let VectorWriteUndo { + index_key, + tid, + collection, + mark, + params_absent, + ivf, + sidecars, + payload_rows, + } = undo; + let database_id = index_key.0.as_u64(); + let fail = |detail: String| (entry_index, detail); + + // The bitmap entries of the rows the write left behind go first: their + // fields sit in the sidecars the write stored. + for (surrogate, _) in &sidecars { + let Some(coll) = self.vector_collections.get(&index_key) else { + break; + }; + let Some(id) = coll.local_for_surrogate(*surrogate) else { + continue; + }; + let key = StorageKey::for_surrogate(*surrogate); + let current = self + .sparse + .get(database_id, tid, &collection, &key) + .map_err(|e| fail(format!("reading the sidecar of {key}: {e}")))?; + if let Some(bytes) = current + && let Ok(fields) = + crate::data::executor::handlers::vector_upsert::decode_payload_lowercased( + &bytes, + ) + && let Some(coll) = self.vector_collections.get_mut(&index_key) + { + coll.payload.delete_row(id, &fields); + } + } + + match mark { + Some(mark) => { + let Some(coll) = self.vector_collections.get_mut(&index_key) else { + return Err(fail(format!( + "vector index {:?} vanished before its write was rolled back", + index_key + ))); + }; + if !coll.roll_back_to(mark) { + return Err(fail(format!( + "vector index {:?} sealed the nodes a rolled-back write inserted", + index_key + ))); + } + for (id, fields) in &payload_rows { + coll.payload.insert_row(*id, fields); + } + } + None => { + self.vector_collections.remove(&index_key); + } + } + if params_absent { + self.vector_params.remove(&index_key); + } + match ivf { + Some(IvfMark { count, trained }) => { + if let Some(index) = self.ivf_indexes.get_mut(&index_key) { + index.roll_back_to(count, trained); + } + } + None => { + self.ivf_indexes.remove(&index_key); + } + } + + for (surrogate, prior) in sidecars { + let key = StorageKey::for_surrogate(surrogate); + let restored = match &prior { + Some(bytes) => self + .sparse + .put(database_id, tid, &collection, &key, bytes) + .map(drop), + None => self + .sparse + .delete(database_id, tid, &collection, &key) + .map(drop), + }; + restored.map_err(|e| fail(format!("restoring the sidecar of {key}: {e}")))?; + self.doc_cache + .invalidate(database_id, tid, &collection, &key); + } + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/vector.rs b/nodedb/src/data/executor/handlers/vector.rs index a99c747b7..3428939df 100644 --- a/nodedb/src/data/executor/handlers/vector.rs +++ b/nodedb/src/data/executor/handlers/vector.rs @@ -218,7 +218,10 @@ impl CoreLoop { return self.ivf_insert(task, tid, &key, vector, dim, surrogate); } - // Default: HNSW (with or without PQ). + // Default: HNSW (with or without PQ). A committed-redo install seals + // once the whole record landed, so a rollback finds its inserts in + // the growing segment. + let defer_seal = self.recording_redo_undo(); match self.get_or_create_vector_index(database_id, tid, collection, dim, field_name) { Ok(collection_ref) => { collection_ref.insert_with_surrogate(vector.to_vec(), surrogate); @@ -231,7 +234,8 @@ impl CoreLoop { collection_ref.note_checkpoint_lsn(lsn.as_u64()); } let seal_key = CoreLoop::vector_build_key(&index_key); - if collection_ref.needs_seal() + if !defer_seal + && collection_ref.needs_seal() && let Some(req) = collection_ref.seal(&seal_key) && let Some(tx) = &self.build_tx && let Err(e) = tx.send(req) diff --git a/nodedb/src/data/executor/handlers/vector_direct_row.rs b/nodedb/src/data/executor/handlers/vector_direct_row.rs index f894a35c0..19578db48 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_row.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_row.rs @@ -290,7 +290,10 @@ impl CoreLoop { surrogates: &[Surrogate], ) { let seal_key = CoreLoop::vector_build_key(index_key); - if let Some(coll) = self.vector_collections.get_mut(index_key) + // A committed-redo install seals once the whole record landed, so a + // rollback finds its inserts in the growing segment. + if !self.recording_redo_undo() + && let Some(coll) = self.vector_collections.get_mut(index_key) && coll.needs_seal() && let Some(req) = coll.seal(&seal_key) && let Some(tx) = &self.build_tx diff --git a/nodedb/src/data/executor/handlers/vector_multi.rs b/nodedb/src/data/executor/handlers/vector_multi.rs index d9b7f62e5..9125c2887 100644 --- a/nodedb/src/data/executor/handlers/vector_multi.rs +++ b/nodedb/src/data/executor/handlers/vector_multi.rs @@ -112,6 +112,8 @@ impl CoreLoop { .get(&index_key) .cloned() .unwrap_or_default(); + // A committed-redo install seals once the whole record landed. + let defer_seal = self.recording_redo_undo(); let coll = self .vector_collections .entry(index_key.clone()) @@ -143,7 +145,8 @@ impl CoreLoop { // Auto-seal if needed. let seal_key = CoreLoop::vector_build_key(&index_key); - if coll.needs_seal() + if !defer_seal + && coll.needs_seal() && let Some(req) = coll.seal(&seal_key) && let Some(tx) = &self.build_tx && let Err(e) = tx.send(req) diff --git a/nodedb/src/data/executor/handlers/vector_write.rs b/nodedb/src/data/executor/handlers/vector_write.rs index 315fe1125..4758ee5d7 100644 --- a/nodedb/src/data/executor/handlers/vector_write.rs +++ b/nodedb/src/data/executor/handlers/vector_write.rs @@ -28,6 +28,8 @@ impl CoreLoop { debug!(core = self.core_id, %collection, dim, count = vectors.len(), "vector batch insert"); let database_id = task.request.database_id.as_u64(); let index_key = CoreLoop::vector_index_key(database_id, tid, collection, ""); + // A committed-redo install seals once the whole record landed. + let defer_seal = self.recording_redo_undo(); match self.get_or_create_vector_index(database_id, tid, collection, dim, "") { Ok(collection_ref) => { for (i, vector) in vectors.iter().enumerate() { @@ -53,7 +55,8 @@ impl CoreLoop { collection_ref.note_checkpoint_lsn(lsn.as_u64()); } let seal_key = CoreLoop::vector_build_key(&index_key); - if collection_ref.needs_seal() + if !defer_seal + && collection_ref.needs_seal() && let Some(req) = collection_ref.seal(&seal_key) && let Some(tx) = &self.build_tx && let Err(e) = tx.send(req) diff --git a/nodedb/src/data/executor/kv_checkpoint/load.rs b/nodedb/src/data/executor/kv_checkpoint/load.rs index 91cd407fa..b7b4196f6 100644 --- a/nodedb/src/data/executor/kv_checkpoint/load.rs +++ b/nodedb/src/data/executor/kv_checkpoint/load.rs @@ -73,6 +73,7 @@ impl CoreLoop { .replay_floors .kv .set(Lsn::new(manifest.durable_through_lsn)); + self.floors.kv_published_lsn = Lsn::new(manifest.durable_through_lsn); info!( core = self.core_id, diff --git a/nodedb/src/data/executor/kv_checkpoint/write.rs b/nodedb/src/data/executor/kv_checkpoint/write.rs index b526feedc..3684809e2 100644 --- a/nodedb/src/data/executor/kv_checkpoint/write.rs +++ b/nodedb/src/data/executor/kv_checkpoint/write.rs @@ -39,7 +39,10 @@ impl CoreLoop { /// no-op (`add_index` reports the field as already indexed and skips the /// backfill), and replaying a drop whose registration the export therefore /// never saw is a no-op too. - pub(in crate::data::executor) fn checkpoint_kv_engines(&self) -> crate::Result { + /// + /// Every published generation raises `kv_published_lsn`, the LSN restart + /// restores KV from (see `redo_apply::cover`). + pub(in crate::data::executor) fn checkpoint_kv_engines(&mut self) -> crate::Result { let durable_through = self.watermark; let ckpt_dir = kv_ckpt_dir(&self.data_dir, self.core_id); @@ -63,6 +66,7 @@ impl CoreLoop { let written = self.write_kv_generation(&gen_dir)?; self.publish_kv_generation(&ckpt_dir, generation, durable_through)?; + self.floors.kv_published_lsn = self.floors.kv_published_lsn.max(durable_through); // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what diff --git a/nodedb/src/data/executor/mod.rs b/nodedb/src/data/executor/mod.rs index cd6010a5a..75a3ed728 100644 --- a/nodedb/src/data/executor/mod.rs +++ b/nodedb/src/data/executor/mod.rs @@ -18,6 +18,8 @@ pub(crate) mod kv_checkpoint; pub(super) mod msgpack_utils; pub(crate) mod replay_abort; pub(crate) mod replay_floors; +mod replay_policy; +mod replay_task; pub mod response_codec; mod row_shape; mod scan_normalize; @@ -37,6 +39,7 @@ pub(crate) mod vector_string; mod wal_replay; mod wal_replay_all; mod wal_replay_columnar_dml; +mod wal_replay_columnar_image; mod wal_replay_columnar_truncate; mod wal_replay_document_vector; mod wal_replay_fts; @@ -55,8 +58,12 @@ mod wal_replay_redo_document; mod wal_replay_redo_graph; mod wal_replay_spatial; mod wal_replay_vector; +mod wal_replay_vector_delete; mod wal_replay_vector_direct; mod wal_replay_vector_extended; mod wal_replay_vector_index_drop; mod wal_replay_vector_params; +mod wal_replay_vector_redo; mod wal_replay_vector_resolved; +mod wal_replay_vector_sparse; +mod wal_replay_vector_task; diff --git a/nodedb/src/data/executor/replay_policy.rs b/nodedb/src/data/executor/replay_policy.rs new file mode 100644 index 000000000..5522cd0d0 --- /dev/null +++ b/nodedb/src/data/executor/replay_policy.rs @@ -0,0 +1,181 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! How a replay arm settles a per-record decision in each of its two passes. +//! +//! The per-engine replay arms run in two passes: +//! +//! * **Restart replay** rebuilds engine state from the WAL at boot. +//! * **The committed-redo apply** (`handlers::transaction::redo_apply`) +//! installs one committed transaction's redo record on a live core. It +//! opens a redo-apply scope for the duration of the record. +//! +//! The two passes differ in four decisions. Every arm routes them through +//! this file so the difference lives in one place. +//! +//! ## Validate, then install +//! +//! A committed-redo apply runs the arms twice over the record. In the +//! validate pass an arm decodes, routes and checks each sub-record it owns, +//! then claims it and skips the write (`claim_for_validation`). In the +//! install pass an arm writes, and records the undo entries that reverse the +//! write (`record_redo_undo`). A failure in the install pass rolls the +//! record back from them, so every replica holds all of the record or none +//! of it. Restart replay runs neither. +//! +//! ## Watermark skips +//! +//! A restart skips a record at or below an engine watermark: a restored +//! checkpoint's LSN, a flushed timeseries partition's LSN, an array +//! manifest's durable LSN. Each watermark says what engine state already +//! holds before replay starts. An online apply installs a record exactly once, +//! and the apply loop's proposal ledger guarantees that. A live flush or +//! checkpoint can mint a watermark above the LSN of a record that has not +//! applied yet. Such a watermark says nothing about that record, so an online +//! apply never consults one. +//! +//! ## Records that cannot be applied +//! +//! Restart policy: a committed record an arm cannot decode, cannot route, or +//! whose handler rejects stops recovery (`replay_abort::abort_replay`). +//! Starting with a hole in the replayed suffix is worse than not starting. +//! Online policy: the error fails the apply response. The funnel then refuses +//! or aborts the entry, and the process keeps serving. +//! +//! ## Records restart skips with a warning +//! +//! Restart policy: some arms skip a record whose live re-execution fails, log +//! it with its LSN, and continue booting. Each site states why a skip is the +//! chosen outcome there. Examples are a record inapplicable to an index +//! rebuilt at a new dimension, and a resolved-row drift that replay has no +//! caller to retry against. Online policy: the same failure is an error of +//! this apply and fails its response. + +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::transaction::redo_apply::RedoApplyPass; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::replay_abort::abort_replay; + +impl CoreLoop { + /// Whether a committed-redo apply is driving the replay arms. + pub(in crate::data::executor) fn applying_committed_redo(&self) -> bool { + self.redo_apply.scope.is_some() + } + + /// Whether the arm must skip the write it reached because the apply is + /// validating. Claims the sub-record when it is. Call once per + /// sub-record, after its decode, routing and checks. + pub(in crate::data::executor) fn claim_for_validation(&mut self) -> bool { + match self.redo_apply.scope.as_mut() { + Some(scope) if scope.pass == RedoApplyPass::Validate => { + scope.claimed += 1; + true + } + _ => false, + } + } + + /// Whether the arm must record undo entries for the writes it makes: + /// `true` only in the install pass of a committed-redo apply. + pub(in crate::data::executor) fn recording_redo_undo(&self) -> bool { + self.redo_apply + .scope + .as_ref() + .is_some_and(|scope| scope.pass == RedoApplyPass::Install) + } + + /// Record the entries that reverse one write of the install pass. Restart + /// replay records nothing. + pub(in crate::data::executor) fn record_redo_undo( + &mut self, + entries: impl IntoIterator, + ) { + if let Some(scope) = self.redo_apply.scope.as_mut() + && scope.pass == RedoApplyPass::Install + { + scope.undo.extend(entries); + } + } + + /// Record the undo entries a pre-image capture produced, or keep its + /// error on the apply. Returns whether the write may proceed: `false` + /// when the pre-image could not be read, so the write is skipped and the + /// install pass rolls back. + pub(in crate::data::executor) fn record_redo_capture( + &mut self, + captured: crate::Result, + ) -> bool + where + I: IntoIterator, + { + match captured { + Ok(entries) => { + self.record_redo_undo(entries); + true + } + Err(error) => { + if let Some(scope) = self.redo_apply.scope.as_mut() { + scope.record_error(error); + } + false + } + } + } + + /// Whether an engine watermark skips a record. `covered` is the arm's + /// own watermark test. Always `false` during a committed-redo apply. + pub(in crate::data::executor) fn replay_watermark_skips(&self, covered: bool) -> bool { + covered && !self.applying_committed_redo() + } + + /// A committed record cannot be applied. Restart replay stops recovery + /// and never returns. A committed-redo apply records the error for its + /// response and returns, and the caller skips the record. + pub(in crate::data::executor) fn replay_record_unapplied( + &mut self, + engine: &str, + stage: &str, + record_lsn: u64, + detail: &str, + ) { + match self.redo_apply.scope.as_mut() { + Some(scope) => scope.record_error(ErrorCode::Internal { + detail: format!( + "committed redo record at lsn {record_lsn} cannot be applied ({engine} \ + {stage}): {detail}" + ), + }), + None => abort_replay(engine, stage, self.core_id, record_lsn, detail), + } + } + + /// A handler rejected a record. Restart replay logs it with its LSN and + /// skips it. A committed-redo apply records `error` for its response. + /// `error` is the handler's own error code when it reported one. + pub(in crate::data::executor) fn replay_record_rejected( + &mut self, + engine: &str, + record_lsn: u64, + error: Option>, + detail: &str, + ) { + match self.redo_apply.scope.as_mut() { + Some(scope) => scope.record_error(error.map_or_else( + || ErrorCode::Internal { + detail: format!( + "committed redo record at lsn {record_lsn} was rejected ({engine}): \ + {detail}" + ), + }, + |error| *error, + )), + None => tracing::warn!( + core = self.core_id, + engine, + lsn = record_lsn, + error = ?error, + "{detail}; skipping WAL record" + ), + } + } +} diff --git a/nodedb/src/data/executor/replay_task.rs b/nodedb/src/data/executor/replay_task.rs new file mode 100644 index 000000000..d223b5604 --- /dev/null +++ b/nodedb/src/data/executor/replay_task.rs @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The synthetic `ExecutionTask` a replay arm hands a live handler. + +use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::{ExecutionTask, TaskState}; +use crate::types::{DatabaseId, ReadConsistency}; + +impl CoreLoop { + /// Build a synthetic replay `ExecutionTask` embedding `plan`. + /// + /// Shared with `wal_replay_columnar_dml` — every replay handler that + /// re-invokes a live execute_* method needs the same minimal task shape. + pub(in crate::data::executor) fn replay_task( + tenant_id: crate::types::TenantId, + database_id: DatabaseId, + vshard_id: crate::types::VShardId, + plan: PhysicalPlan, + wal_lsn: Option, + ) -> ExecutionTask { + ExecutionTask { + request: Request { + request_id: crate::types::RequestId::new(0), + tenant_id, + database_id, + vshard_id, + plan, + deadline: std::time::Instant::now() + + crate::data::executor::deadline::REPLAY_DEADLINE, + priority: Priority::Normal, + trace_id: crate::types::TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn, + resolved_now_ms: None, + admission: crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::AlreadyOrdered, + ), + }, + state: TaskState::Running, + wal_lsn, + resolved_now_ms: None, + } + } +} diff --git a/nodedb/src/data/executor/task.rs b/nodedb/src/data/executor/task.rs index 59ed29e4e..141b29986 100644 --- a/nodedb/src/data/executor/task.rs +++ b/nodedb/src/data/executor/task.rs @@ -26,15 +26,13 @@ pub struct ExecutionTask { pub wal_lsn: Option, /// Wall-clock instant (ms since epoch) the Control Plane resolved at - /// WAL-append time for a TTL-bearing KV write, carried alongside the - /// request so live apply installs the SAME instant the durable WAL record - /// carries rather than re-reading the clock. Re-reading it at apply time - /// would let the live value disagree with the durable one by the dispatch - /// latency — harmless day to day, but a crash between the two would have - /// replay recompute `now_ms` at restart time instead of installing the - /// original instant, pushing the TTL's expiry forward by the - /// crash-to-restart delay. `None` for non-TTL writes, reads, and writes - /// whose resolved instant is not (yet) threaded. + /// WAL-append time, carried alongside the request so live apply installs + /// the SAME instant the durable WAL record carries rather than re-reading + /// the clock. A TTL-bearing KV write resolves its expiry against it, and a + /// timeseries ingest stamps its untimed rows with it. Re-reading the clock + /// at apply time would let the live value disagree with the durable one, + /// and replay would recompute it at restart time. `None` for other + /// writes and reads. /// Copied from [`Request::resolved_now_ms`](crate::bridge::envelope::Request) /// in [`ExecutionTask::new`], same as `wal_lsn`. pub resolved_now_ms: Option, @@ -89,8 +87,8 @@ impl ExecutionTask { self.wal_lsn } - /// Wall-clock instant the Control Plane resolved for a TTL-bearing KV - /// write, if any. See the field doc on [`ExecutionTask::resolved_now_ms`]. + /// Wall-clock instant the Control Plane resolved for this write, if any. + /// See the field doc on [`ExecutionTask::resolved_now_ms`]. pub fn resolved_now_ms(&self) -> Option { self.resolved_now_ms } diff --git a/nodedb/src/data/executor/vector_checkpoint/load.rs b/nodedb/src/data/executor/vector_checkpoint/load.rs index b71135814..d32cfdf9a 100644 --- a/nodedb/src/data/executor/vector_checkpoint/load.rs +++ b/nodedb/src/data/executor/vector_checkpoint/load.rs @@ -68,6 +68,7 @@ impl CoreLoop { // clamps truncation to, so claiming it over a half-restored generation // would authorise deleting the records that would have completed it. self.floors.vector_durable_lsn = Lsn::new(manifest.durable_through_lsn); + self.floors.vector_published_lsn = Lsn::new(manifest.durable_through_lsn); info!( core = self.core_id, diff --git a/nodedb/src/data/executor/vector_checkpoint/write.rs b/nodedb/src/data/executor/vector_checkpoint/write.rs index 22276ef30..b804bdef3 100644 --- a/nodedb/src/data/executor/vector_checkpoint/write.rs +++ b/nodedb/src/data/executor/vector_checkpoint/write.rs @@ -40,7 +40,13 @@ impl CoreLoop { /// runs on the core's own thread between tasks, and a vector write raises /// the watermark only after the collection has already been mutated, so /// every write with `lsn <= watermark` is in the bytes written below. - pub(crate) fn checkpoint_vector_indexes(&self) -> crate::Result { + /// + /// Every published generation raises `vector_published_lsn`, whichever + /// caller published it: restart restores from the newest generation, so + /// a committed record applied at or below it must reach a newer one + /// (`redo_apply::cover`). `vector_durable_lsn` stays the caller's to + /// raise. + pub(crate) fn checkpoint_vector_indexes(&mut self) -> crate::Result { let durable_lsn = self.watermark; let ckpt_dir = vector_ckpt_dir(&self.data_dir, self.core_id); @@ -64,6 +70,7 @@ impl CoreLoop { let files_written = self.write_vector_generation(&gen_dir)?; publish_vector_generation(&ckpt_dir, generation, durable_lsn)?; + self.floors.vector_published_lsn = self.floors.vector_published_lsn.max(durable_lsn); // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what diff --git a/nodedb/src/data/executor/wal_replay/array.rs b/nodedb/src/data/executor/wal_replay/array.rs index 83896070a..9e6c22472 100644 --- a/nodedb/src/data/executor/wal_replay/array.rs +++ b/nodedb/src/data/executor/wal_replay/array.rs @@ -16,7 +16,7 @@ //! must be re-applied into the memtable. use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::replay_abort::abort_replay; +use crate::data::executor::handlers::transaction::undo::UndoEntry; use std::sync::Arc; /// Outcome of preparing an array's store for replay. @@ -63,11 +63,43 @@ impl CoreLoop { Ok(ArrayOpen::Ready) } + /// In the install pass of a committed-redo apply, record the memtable + /// tiles a cell write touches and the sync high-water mark it advances. + /// Returns `false` when the tiles cannot be read; the error is kept on + /// the apply. + fn record_array_tiles_undo<'a>( + &mut self, + array_id: &nodedb_array::types::ArrayId, + cells: impl IntoIterator, + provenance: Option<&nodedb_types::sync::wire::SyncProvenance>, + ) -> bool { + let captured = self + .array_engine + .snapshot_tiles(array_id, cells) + .map_err(|e| crate::Error::Internal { + detail: format!( + "reading the memtable tiles of array '{}': {e}", + array_id.name + ), + }) + .map(|snapshot| { + std::iter::once(UndoEntry::ArrayTiles { + array_id: array_id.clone(), + snapshot, + }) + .chain(self.capture_sync_hwm_undo(provenance)) + }); + self.record_redo_capture(captured) + } + /// The LSN this array's flushed segments are already durable through. /// /// `0` when the store is not open, which gates nothing — the safe /// direction, matching every other engine's unset replay floor. - fn array_durable_lsn(&self, array_id: &nodedb_array::types::ArrayId) -> u64 { + pub(in crate::data::executor) fn array_durable_lsn( + &self, + array_id: &nodedb_array::types::ArrayId, + ) -> u64 { self.array_engine .store(array_id) .map_or(0, |store| store.manifest().durable_lsn) @@ -111,13 +143,16 @@ impl CoreLoop { if is_put { let payload = match decode_put_with_version(&record.payload) { Ok(p) => p, - Err(e) => abort_replay( - "array", - "decode_put", - self.core_id, - record_lsn, - &format!("ArrayPut payload could not be decoded: {e}"), - ), + Err(e) => { + self.replay_record_unapplied( + "array", + "decode_put", + record_lsn, + &format!("ArrayPut payload could not be decoded: {e}"), + ); + skipped += 1; + continue; + } }; if tombstones.is_tombstoned( record.header.database_id, @@ -131,43 +166,78 @@ impl CoreLoop { match self.ensure_array_open_for_replay(&payload.array_id) { Ok(ArrayOpen::Ready) => {} Ok(ArrayOpen::NoCatalogEntry) => { - tracing::warn!( - core = self.core_id, - array = %payload.array_id.name, - lsn = record_lsn, - "WAL array replay: no catalog entry for this array; \ - skipping its retained cells" + self.replay_record_rejected( + "array", + record_lsn, + None, + &format!( + "array '{}' has no catalog entry; its retained cells are skipped", + payload.array_id.name + ), + ); + skipped += 1; + continue; + } + Err(e) => { + self.replay_record_unapplied( + "array", + "open", + record_lsn, + &format!("array '{}' could not be opened: {e}", payload.array_id.name), ); skipped += 1; continue; } - Err(e) => abort_replay( - "array", - "open", - self.core_id, - record_lsn, - &format!("array '{}' could not be opened: {e}", payload.array_id.name), - ), } - if record_lsn <= self.array_durable_lsn(&payload.array_id) { + if self + .replay_watermark_skips(record_lsn <= self.array_durable_lsn(&payload.array_id)) + { + skipped += 1; + continue; + } + if self.claim_for_validation() { + continue; + } + let installing = self.recording_redo_undo(); + if installing + && !self.record_array_tiles_undo( + &payload.array_id, + payload + .cells + .iter() + .map(|c| (c.coord.as_slice(), c.system_from_ms)), + payload.provenance.as_ref(), + ) + { skipped += 1; continue; } let cell_count = payload.cells.len(); let prov = payload.provenance.clone(); - if let Err(e) = + // An install flushes once the whole record landed, so a + // rollback finds its cells in the memtable. + let stamped = if installing { + self.array_engine.put_cells_unflushed( + &payload.array_id, + payload.cells, + record_lsn, + ) + } else { self.array_engine .put_cells(&payload.array_id, payload.cells, record_lsn) - { - abort_replay( + }; + if let Err(e) = stamped { + self.replay_record_unapplied( "array", "put_cells", - self.core_id, record_lsn, &format!("committed cells could not be re-applied: {e}"), ); + skipped += 1; + continue; } puts += cell_count; + self.note_redo_array_written(&payload.array_id); // Rebuild the per-core HWM frontier from the WAL record's // provenance. No fence check here — replay records are already // durable and ordered; just advance the frontier. @@ -179,13 +249,16 @@ impl CoreLoop { let payload = match decode_delete_with_version(&record.payload) { Ok(p) => p, - Err(e) => abort_replay( - "array", - "decode_delete", - self.core_id, - record_lsn, - &format!("ArrayDelete payload could not be decoded: {e}"), - ), + Err(e) => { + self.replay_record_unapplied( + "array", + "decode_delete", + record_lsn, + &format!("ArrayDelete payload could not be decoded: {e}"), + ); + skipped += 1; + continue; + } }; if tombstones.is_tombstoned( record.header.database_id, @@ -199,43 +272,75 @@ impl CoreLoop { match self.ensure_array_open_for_replay(&payload.array_id) { Ok(ArrayOpen::Ready) => {} Ok(ArrayOpen::NoCatalogEntry) => { - tracing::warn!( - core = self.core_id, - array = %payload.array_id.name, - lsn = record_lsn, - "WAL array replay: no catalog entry for this array; \ - skipping its retained tombstones" + self.replay_record_rejected( + "array", + record_lsn, + None, + &format!( + "array '{}' has no catalog entry; its retained tombstones are skipped", + payload.array_id.name + ), + ); + skipped += 1; + continue; + } + Err(e) => { + self.replay_record_unapplied( + "array", + "open", + record_lsn, + &format!("array '{}' could not be opened: {e}", payload.array_id.name), ); skipped += 1; continue; } - Err(e) => abort_replay( - "array", - "open", - self.core_id, - record_lsn, - &format!("array '{}' could not be opened: {e}", payload.array_id.name), - ), } - if record_lsn <= self.array_durable_lsn(&payload.array_id) { + if self.replay_watermark_skips(record_lsn <= self.array_durable_lsn(&payload.array_id)) + { + skipped += 1; + continue; + } + if self.claim_for_validation() { + continue; + } + let installing = self.recording_redo_undo(); + if installing + && !self.record_array_tiles_undo( + &payload.array_id, + payload + .cells + .iter() + .map(|c| (c.coord.as_slice(), c.system_from_ms)), + payload.provenance.as_ref(), + ) + { skipped += 1; continue; } let cell_count = payload.cells.len(); let prov = payload.provenance.clone(); - if let Err(e) = + let stamped = if installing { + self.array_engine.delete_cells_unflushed( + &payload.array_id, + payload.cells, + record_lsn, + ) + } else { self.array_engine .delete_cells(&payload.array_id, payload.cells, record_lsn) - { - abort_replay( + }; + if let Err(e) = stamped { + self.replay_record_unapplied( "array", "delete_cells", - self.core_id, record_lsn, &format!("committed tombstones could not be re-applied: {e}"), ); + skipped += 1; + continue; } deletes += cell_count; + self.note_redo_array_written(&payload.array_id); if let Some(p) = &prov { self.sync_commit(p); } diff --git a/nodedb/src/data/executor/wal_replay/crdt_doc.rs b/nodedb/src/data/executor/wal_replay/crdt_doc.rs index 8252517ec..f72c5d63a 100644 --- a/nodedb/src/data/executor/wal_replay/crdt_doc.rs +++ b/nodedb/src/data/executor/wal_replay/crdt_doc.rs @@ -12,8 +12,6 @@ //! `wal::CrdtDocOpWalRecord`'s doc comment for the full rationale and why this //! is deliberately NOT `RecordType::CrdtDelta`. -use tracing::warn; - use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; @@ -54,10 +52,11 @@ impl CoreLoop { } let Ok(payload) = zerompk::from_msgpack::(&record.payload) else { - warn!( - core = self.core_id, - lsn = record.header.lsn, - "malformed CrdtDocOp WAL record; skipping" + self.replay_record_rejected( + "crdt", + record.header.lsn, + None, + "malformed CrdtDocOp WAL record", ); return Some(0); }; @@ -157,13 +156,11 @@ impl CoreLoop { }; if response.status != Status::Ok { - warn!( - core = self.core_id, - collection = %collection, - document_id = %document_id, - lsn = record_lsn, - error = ?response.error_code, - "CRDT doc-op WAL replay failed; skipping record" + self.replay_record_rejected( + "crdt", + record_lsn, + response.error_code, + &format!("CRDT doc op on '{collection}' / '{document_id}' failed"), ); return Some(0); } diff --git a/nodedb/src/data/executor/wal_replay/crdt_list.rs b/nodedb/src/data/executor/wal_replay/crdt_list.rs index ced7b1297..9ab9f902a 100644 --- a/nodedb/src/data/executor/wal_replay/crdt_list.rs +++ b/nodedb/src/data/executor/wal_replay/crdt_list.rs @@ -12,8 +12,6 @@ //! diverge. See `wal::CrdtListOpWalRecord`'s doc comment for the full //! rationale and why this is deliberately NOT `RecordType::CrdtDelta`. -use tracing::warn; - use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; @@ -22,24 +20,14 @@ use nodedb_physical::physical_plan::CrdtOp; use nodedb_types::{RowIdentity, Surrogate}; /// Narrow a WAL-logged `u64` list position to the `usize` the live -/// `execute_crdt_list_*` handlers take. Returns `None` (with the record -/// skipped by the caller) on a platform where `usize` is narrower than -/// `u64` and the logged position doesn't fit — never truncates via `as`, -/// which would silently replay at the wrong position. -fn wal_list_index(core_id: usize, lsn: u64, field: &str, value: u64) -> Option { - match usize::try_from(value) { - Ok(v) => Some(v), - Err(_) => { - warn!( - core = core_id, - lsn, - field, - value, - "CrdtListOp WAL record position does not fit usize; skipping record" - ); - None - } - } +/// `execute_crdt_list_*` handlers take. The error names the field on a +/// platform where `usize` is narrower than `u64` and the logged position +/// does not fit. It never truncates via `as`, which would silently replay at +/// the wrong position. +fn wal_list_index(field: &str, value: u64) -> crate::Result { + usize::try_from(value).map_err(|_| crate::Error::Internal { + detail: format!("CrdtListOp position {field} = {value} does not fit usize"), + }) } impl CoreLoop { @@ -76,10 +64,11 @@ impl CoreLoop { } let Ok(payload) = zerompk::from_msgpack::(&record.payload) else { - warn!( - core = self.core_id, - lsn = record.header.lsn, - "malformed CrdtListOp WAL record; skipping" + self.replay_record_rejected( + "crdt", + record.header.lsn, + None, + "malformed CrdtListOp WAL record", ); return Some(0); }; @@ -98,7 +87,6 @@ impl CoreLoop { let tid = TenantId::new(tenant_id); let database_id = DatabaseId::new(record.header.database_id); let vshard = VShardId::new(record.header.vshard_id); - let core_id = self.core_id; // The task carries the real intent even though today's handlers read // only the explicit args passed alongside it, not the plan itself — @@ -127,8 +115,17 @@ impl CoreLoop { index, fields_json, } => { - let Some(index) = wal_list_index(core_id, record_lsn, "index", index) else { - return Some(0); + let index = match wal_list_index("index", index) { + Ok(index) => index, + Err(e) => { + self.replay_record_rejected( + "crdt", + record_lsn, + Some(Box::new(crate::bridge::envelope::ErrorCode::from(e))), + "CrdtListOp position out of range", + ); + return Some(0); + } }; let document_id = RowIdentity::from_user_key(document_id); let plan = PhysicalPlan::Crdt(CrdtOp::ListInsert { @@ -157,8 +154,17 @@ impl CoreLoop { list_path, index, } => { - let Some(index) = wal_list_index(core_id, record_lsn, "index", index) else { - return Some(0); + let index = match wal_list_index("index", index) { + Ok(index) => index, + Err(e) => { + self.replay_record_rejected( + "crdt", + record_lsn, + Some(Box::new(crate::bridge::envelope::ErrorCode::from(e))), + "CrdtListOp position out of range", + ); + return Some(0); + } }; let document_id = RowIdentity::from_user_key(document_id); let plan = PhysicalPlan::Crdt(CrdtOp::ListDelete { @@ -186,14 +192,19 @@ impl CoreLoop { from_index, to_index, } => { - let Some(from_index) = - wal_list_index(core_id, record_lsn, "from_index", from_index) - else { - return Some(0); - }; - let Some(to_index) = wal_list_index(core_id, record_lsn, "to_index", to_index) - else { - return Some(0); + let (from_index, to_index) = match wal_list_index("from_index", from_index) + .and_then(|from| wal_list_index("to_index", to_index).map(|to| (from, to))) + { + Ok(indexes) => indexes, + Err(e) => { + self.replay_record_rejected( + "crdt", + record_lsn, + Some(Box::new(crate::bridge::envelope::ErrorCode::from(e))), + "CrdtListOp position out of range", + ); + return Some(0); + } }; let document_id = RowIdentity::from_user_key(document_id); let plan = PhysicalPlan::Crdt(CrdtOp::ListMove { @@ -219,14 +230,13 @@ impl CoreLoop { }; if response.status != Status::Ok { - warn!( - core = self.core_id, - collection = %collection, - document_id = %document_id, - list_path = %list_path, - lsn = record_lsn, - error = ?response.error_code, - "CRDT list-op WAL replay failed; skipping record" + self.replay_record_rejected( + "crdt", + record_lsn, + response.error_code, + &format!( + "CRDT list op on '{collection}' / '{document_id}' at '{list_path}' failed" + ), ); return Some(0); } diff --git a/nodedb/src/data/executor/wal_replay/crdt_ordered.rs b/nodedb/src/data/executor/wal_replay/crdt_ordered.rs index 94d2b11ab..e4c93d317 100644 --- a/nodedb/src/data/executor/wal_replay/crdt_ordered.rs +++ b/nodedb/src/data/executor/wal_replay/crdt_ordered.rs @@ -9,15 +9,63 @@ use nodedb_wal::record::RecordType; use crate::data::executor::core_loop::CoreLoop; +use crate::types::{DatabaseId, TenantId}; +use crate::wal::{CrdtDeltaWalPayload, CrdtDocOpWalRecord, CrdtListOpWalRecord}; + +/// The collection a CRDT record writes, `None` when its payload does not +/// decode or names none. +fn crdt_record_collection(record: &nodedb_wal::WalRecord) -> Option { + match RecordType::from_raw(record.logical_record_type())? { + RecordType::CrdtDelta => { + CrdtDeltaWalPayload::decode(&record.payload) + .ok()? + .collection + } + RecordType::CrdtListOp => { + match zerompk::from_msgpack::(&record.payload).ok()? { + CrdtListOpWalRecord::Insert { collection, .. } + | CrdtListOpWalRecord::Delete { collection, .. } + | CrdtListOpWalRecord::Move { collection, .. } => Some(collection), + } + } + RecordType::CrdtDocOp => { + match zerompk::from_msgpack::(&record.payload).ok()? { + CrdtDocOpWalRecord::Upsert { collection, .. } + | CrdtDocOpWalRecord::Delete { collection, .. } => Some(collection), + } + } + _ => None, + } +} impl CoreLoop { - /// Replay standalone CRDT WAL records in stable global LSN order. + /// The committed-redo step before a CRDT record's write. In the validate + /// pass it claims a record that names its collection and returns `false`. + /// In the install pass it records the collection's Loro pre-image and + /// returns whether the write proceeds. A record naming no collection is + /// left unclaimed, so the validate pass refuses the record. + fn redo_crdt_prelude(&mut self, record: &nodedb_wal::WalRecord) -> bool { + let Some(collection) = crdt_record_collection(record) else { + return false; + }; + if self.claim_for_validation() { + return false; + } + let captured = self.capture_crdt_collection_undo( + DatabaseId::new(record.header.database_id), + TenantId::new(record.header.tenant_id), + &collection, + ); + self.record_redo_capture(captured.map(std::iter::once)) + } + + /// Replay CRDT WAL records in stable global LSN order. /// - /// `TransactionRedo` has no CRDT subrecords: CRDT writes are deliberately - /// excluded from transaction redo encoding because raw applies require the - /// serialized admission boundary and CRDT intent is independently durable. - /// Consequently this method handles only the three standalone CRDT record - /// classes and must run exactly once during startup. + /// `records` holds the standalone CRDT records and the CRDT intents a + /// committed transaction journalled as `TransactionRedo` sub-records. A + /// transaction never journals a raw delta apply: that is refused inside a + /// transaction, because a raw apply requires the serialized admission + /// boundary. pub fn replay_crdt_wal_ordered( &mut self, records: &[nodedb_wal::WalRecord], @@ -37,6 +85,9 @@ impl CoreLoop { crdt_records.sort_by_key(|(original_index, record)| (record.header.lsn, *original_index)); for (_, record) in crdt_records { + if self.applying_committed_redo() && !self.redo_crdt_prelude(record) { + continue; + } match RecordType::from_raw(record.logical_record_type()) { Some(RecordType::CrdtDelta) => { let _ = self.try_replay_crdt_delta(record, num_cores, tombstones); diff --git a/nodedb/src/data/executor/wal_replay/kv.rs b/nodedb/src/data/executor/wal_replay/kv.rs index 782e83edf..c9f91ec0d 100644 --- a/nodedb/src/data/executor/wal_replay/kv.rs +++ b/nodedb/src/data/executor/wal_replay/kv.rs @@ -4,6 +4,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::write_index::KeyRepr; +use crate::data::executor::handlers::transaction::undo::UndoEntry; use crate::data::executor::wal_replay::kv_put::KvReplayRecord; impl CoreLoop { @@ -19,7 +20,7 @@ impl CoreLoop { record_lsn: u64, ) -> bool { tombstones.is_tombstoned(tenant_id, collection, record_lsn) - || self.floors.replay_floors.kv.covers(record_lsn) + || self.replay_watermark_skips(self.floors.replay_floors.kv.covers(record_lsn)) } /// Replay WAL KV records to rebuild in-memory hash tables after crash. @@ -81,6 +82,12 @@ impl CoreLoop { puts += applied; continue; } + // A committed redo record carries only absolute `kv_put` + // post-images. Every other shape is left unclaimed, so the + // validate pass refuses the record before any arm writes. + if self.applying_committed_redo() { + continue; + } if let Some(applied) = self.try_replay_kv_batch_put(&kv_record, tombstones) { puts += applied; @@ -245,6 +252,29 @@ impl CoreLoop { if self.skip_kv_replay_record(tombstones, tenant_id, &collection, record_lsn) { continue; } + if self.claim_for_validation() { + continue; + } + if self.recording_redo_undo() { + let undo: Vec = keys + .iter() + .filter_map(|key| { + let prior = self.kv_engine.entry_image( + database_id, + tenant_id, + &collection, + key, + now_ms, + )?; + Some(UndoEntry::KvDelete { + collection: collection.clone(), + key: key.clone(), + prior, + }) + }) + .collect(); + self.record_redo_undo(undo); + } self.kv_engine .delete(database_id, tenant_id, &collection, &keys, now_ms); for deleted_key in &keys { @@ -268,6 +298,18 @@ impl CoreLoop { if self.skip_kv_replay_record(tombstones, tenant_id, &collection, record_lsn) { continue; } + if self.claim_for_validation() { + continue; + } + if self.recording_redo_undo() { + let rows = + self.kv_engine + .export_collection(database_id, tenant_id, &collection); + self.record_redo_undo([UndoEntry::KvTruncate { + collection: collection.clone(), + rows, + }]); + } self.kv_engine.truncate(database_id, tenant_id, &collection); self.note_replay_write_lsn( database_id, @@ -279,6 +321,9 @@ impl CoreLoop { deletes += 1; continue; } + if self.applying_committed_redo() { + continue; + } // kv_predicate_delete (delta): re-resolves the row set // against current state. diff --git a/nodedb/src/data/executor/wal_replay/kv_put.rs b/nodedb/src/data/executor/wal_replay/kv_put.rs index 75a27784a..37cac8eac 100644 --- a/nodedb/src/data/executor/wal_replay/kv_put.rs +++ b/nodedb/src/data/executor/wal_replay/kv_put.rs @@ -31,7 +31,7 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::write_index::KeyRepr; -use crate::data::executor::replay_abort::abort_replay; +use crate::data::executor::handlers::transaction::undo::UndoEntry; use nodedb_types::Surrogate; /// Inputs shared by both arms, bundled so each stays under the @@ -68,6 +68,19 @@ impl CoreLoop { if self.skip_kv_replay_record(tombstones, tenant_id, &collection, record_lsn) { return Some(0); } + if self.claim_for_validation() { + return Some(0); + } + if self.recording_redo_undo() { + let prior = + self.kv_engine + .entry_image(database_id, tenant_id, &collection, &key, now_ms); + self.record_redo_undo([UndoEntry::KvPut { + collection: collection.clone(), + key: key.clone(), + prior, + }]); + } let params = crate::engine::kv::KvPutParams { database_id, @@ -118,10 +131,9 @@ impl CoreLoop { // lengths disagree cannot be applied without guessing which row owns // which identity, so it aborts rather than binding the wrong one. if surrogates.len() != entries.len() { - abort_replay( + self.replay_record_unapplied( "kv", "batch_put_surrogates", - self.core_id, record_lsn, &format!( "kv_batch_put into '{collection}' carries {} entries but {} surrogates", @@ -129,6 +141,7 @@ impl CoreLoop { surrogates.len() ), ); + return Some(0); } let params = crate::engine::kv::KvBatchPutParams { diff --git a/nodedb/src/data/executor/wal_replay_columnar_dml.rs b/nodedb/src/data/executor/wal_replay_columnar_dml.rs index 437cdbcde..05319b563 100644 --- a/nodedb/src/data/executor/wal_replay_columnar_dml.rs +++ b/nodedb/src/data/executor/wal_replay_columnar_dml.rs @@ -43,8 +43,6 @@ //! used by the separate `ts_registries` / bucketed-partition machinery, which //! this op pair never targets. -use tracing::warn; - use super::core_loop::CoreLoop; use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; @@ -99,7 +97,7 @@ impl CoreLoop { // trying to detect the duplicate afterwards. Returns `Some(0)` and not // `None`: the record decoded as this shape, so the caller must not fall // through to its row-payload decoders and mis-classify it. - if self.floors.replay_floors.columnar.covers(record_lsn) { + if self.replay_watermark_skips(self.floors.replay_floors.columnar.covers(record_lsn)) { return Some(0); } @@ -172,13 +170,14 @@ impl CoreLoop { }; if response.status != Status::Ok { - warn!( - core = self.core_id, - collection = %record.collection, - lsn = record_lsn, - is_update = record.is_update, - error = ?response.error_code, - "columnar predicate DML WAL replay failed; skipping record" + self.replay_record_rejected( + "columnar", + record_lsn, + response.error_code, + &format!( + "columnar predicate DML replay on '{}' failed (update: {})", + record.collection, record.is_update + ), ); return Some(0); } @@ -252,7 +251,7 @@ impl CoreLoop { return Some(0); } - if self.floors.replay_floors.columnar.covers(record_lsn) { + if self.replay_watermark_skips(self.floors.replay_floors.columnar.covers(record_lsn)) { return Some(0); } @@ -266,12 +265,14 @@ impl CoreLoop { let pk = match nodedb_types::value_from_msgpack(&wal_row.pk_msgpack) { Ok(v) => v, Err(e) => { - warn!( - core = self.core_id, - collection = %record.collection, - lsn = record_lsn, - error = %e, - "columnar resolved-row-set DML WAL replay: malformed PK; skipping record" + self.replay_record_rejected( + "columnar", + record_lsn, + None, + &format!( + "columnar resolved-row-set DML on '{}': malformed PK: {e}", + record.collection + ), ); return Some(0); } @@ -279,12 +280,14 @@ impl CoreLoop { let new_row = match nodedb_types::value_from_msgpack(&wal_row.new_row_msgpack) { Ok(Value::Array(arr)) => arr, Ok(_) | Err(_) => { - warn!( - core = self.core_id, - collection = %record.collection, - lsn = record_lsn, - "columnar resolved-row-set DML WAL replay: malformed post-image row; \ - skipping record" + self.replay_record_rejected( + "columnar", + record_lsn, + None, + &format!( + "columnar resolved-row-set DML on '{}': malformed post-image row", + record.collection + ), ); return Some(0); } @@ -318,12 +321,14 @@ impl CoreLoop { let pk = match nodedb_types::value_from_msgpack(&wal_row.pk_msgpack) { Ok(v) => v, Err(e) => { - warn!( - core = self.core_id, - collection = %record.collection, - lsn = record_lsn, - error = %e, - "columnar resolved-row-set DML WAL replay: malformed PK; skipping record" + self.replay_record_rejected( + "columnar", + record_lsn, + None, + &format!( + "columnar resolved-row-set DML on '{}': malformed PK: {e}", + record.collection + ), ); return Some(0); } @@ -354,13 +359,14 @@ impl CoreLoop { }; if response.status != Status::Ok { - warn!( - core = self.core_id, - collection = %record.collection, - lsn = record_lsn, - is_update = record.is_update, - error = ?response.error_code, - "columnar resolved-row-set DML WAL replay failed; skipping record" + self.replay_record_rejected( + "columnar", + record_lsn, + response.error_code, + &format!( + "columnar resolved-row-set DML replay on '{}' failed (update: {})", + record.collection, record.is_update + ), ); return Some(0); } diff --git a/nodedb/src/data/executor/wal_replay_columnar_image.rs b/nodedb/src/data/executor/wal_replay_columnar_image.rs new file mode 100644 index 000000000..54a059441 --- /dev/null +++ b/nodedb/src/data/executor/wal_replay_columnar_image.rs @@ -0,0 +1,330 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Install of a transaction's columnar row images (`columnar_image` +//! records), for restart replay and the committed-redo apply alike. +//! +//! Each row names a surrogate, the primary key of the base row the +//! transaction displaced (when it displaced one), and the row's post-image. +//! The install removes every displaced base row by key, then writes every +//! image verbatim under its surrogate, then runs the side effects a live +//! columnar insert runs (memtable flush, geometry index, aggregate cache, +//! checkpoint dirtiness, collection write floor). +//! +//! The install is all-or-nothing up to engine errors: every row decodes, +//! every image coerces to the schema, and every displaced key is bound before +//! the first mutation. + +use nodedb_columnar::pk_index::encode_pk; +use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp}; +use nodedb_types::columnar::{COLUMNAR_IMAGE_KIND, ColumnarImageWalRecord, ColumnarSchema}; +use nodedb_types::{Surrogate, Value}; + +use super::core_loop::CoreLoop; +use super::wal_replay_columnar_truncate::TruncateFloors; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan}; +use crate::data::executor::handlers::columnar_write::ndb_field_to_value; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::task::ExecutionTask; +use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + +/// One decoded row of a `columnar_image` record. +struct ImageRow { + surrogate: Surrogate, + prior_pk: Option, + image: Option, +} + +/// Decode every row, or name the first one that does not decode. +fn decode_rows(record: &ColumnarImageWalRecord) -> crate::Result> { + let decode = |bytes: &[u8], what: &str, surrogate: u32| -> crate::Result> { + if bytes.is_empty() { + return Ok(None); + } + nodedb_types::value_from_msgpack(bytes) + .map(Some) + .map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("{what} of surrogate {surrogate} does not decode: {e}"), + }) + }; + record + .rows + .iter() + .map(|row| { + let image = decode(&row.image_msgpack, "image", row.surrogate)?; + if let Some(image) = &image + && !matches!(image, Value::Object(_)) + { + return Err(crate::Error::Internal { + detail: format!("image of surrogate {} is not an object", row.surrogate), + }); + } + Ok(ImageRow { + surrogate: Surrogate::new(row.surrogate), + prior_pk: decode(&row.prior_pk_msgpack, "base key", row.surrogate)?, + image, + }) + }) + .collect() +} + +/// The schema-ordered values of `image`, with bitemporal columns taken from +/// the image as written. +fn image_values(schema: &ColumnarSchema, image: &Value) -> Result, ErrorCode> { + let Value::Object(fields) = image else { + return Err(ErrorCode::Internal { + detail: "columnar image is not an object".into(), + }); + }; + schema + .columns + .iter() + .map(|col| ndb_field_to_value(fields.get(&col.name), &col.column_type)) + .collect::>>() + .map_err(ErrorCode::from) +} + +impl CoreLoop { + /// Try to decode `payload` as a `columnar_image` record and, if it is + /// one, install it. `None` when the payload is another shape. `Some(n)` + /// with the number of rows written otherwise, `Some(0)` when a gate + /// skipped the record or the install failed (reported through + /// `replay_policy`). + pub(in crate::data::executor) fn try_replay_columnar_image( + &mut self, + payload: &[u8], + tenant_id: u64, + database_id: DatabaseId, + record_lsn: u64, + tombstones: &nodedb_wal::TombstoneSet, + truncate_floors: &TruncateFloors, + ) -> Option { + let record: ColumnarImageWalRecord = zerompk::from_msgpack(payload).ok()?; + if record.kind != COLUMNAR_IMAGE_KIND { + return None; + } + let collection = record.collection.as_str(); + if tombstones.is_tombstoned(database_id.as_u64(), tenant_id, collection, record_lsn) { + return Some(0); + } + let tid = TenantId::new(tenant_id); + if truncate_floors.covers_collection(database_id, tid, collection, record_lsn) { + return Some(0); + } + // Already folded into the restored checkpoint. Re-installing would + // append a second version on a `bitemporal=true` collection. + if self.replay_watermark_skips(self.floors.replay_floors.columnar.covers(record_lsn)) { + return Some(0); + } + let rows = match decode_rows(&record) { + Ok(rows) => rows, + Err(error) => { + self.replay_record_unapplied( + "columnar", + "image_decode", + record_lsn, + &format!("image record of '{collection}': {error}"), + ); + return Some(0); + } + }; + + if self.claim_for_validation() { + return Some(0); + } + + let images: Vec = rows.iter().filter_map(|r| r.image.clone()).collect(); + let plan_payload = match nodedb_types::value_to_msgpack(&Value::Array(images)) { + Ok(bytes) => bytes, + Err(e) => { + self.replay_record_unapplied( + "columnar", + "image_encode", + record_lsn, + &format!("images of '{collection}' do not re-encode: {e}"), + ); + return Some(0); + } + }; + let task = Self::replay_task( + tid, + database_id, + VShardId::from_collection_in_database(database_id, collection), + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), + payload: plan_payload, + format: "msgpack".into(), + intent: ColumnarInsertIntent::Put, + on_conflict_updates: Vec::new(), + surrogates: rows + .iter() + .filter(|r| r.image.is_some()) + .map(|r| r.surrogate) + .collect(), + schema_bytes: record.schema_bytes.clone(), + provenance: None, + wal_lsn: Some(record_lsn), + rls_write_check: nodedb_types::RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: Vec::new(), + }), + Some(Lsn::new(record_lsn)), + ); + match self.install_columnar_images(&task, collection, &record.schema_bytes, &rows) { + Ok(written) => Some(written), + Err(error) => { + self.replay_record_rejected( + "columnar", + record_lsn, + Some(Box::new(error)), + &format!("columnar image install into '{collection}' failed"), + ); + Some(0) + } + } + } + + /// Remove every displaced base row, then write every image under its + /// surrogate. Returns the number of rows removed and written. + fn install_columnar_images( + &mut self, + task: &ExecutionTask, + collection: &str, + schema_bytes: &[u8], + rows: &[ImageRow], + ) -> Result { + let key = ( + task.request.database_id, + task.request.tenant_id, + collection.to_string(), + ); + let recording = self.recording_redo_undo(); + let schema = match self.columnar_engines.get(&key) { + Some(engine) => engine.schema().clone(), + None => { + if recording { + self.record_redo_undo([UndoEntry::ColumnarEngineCreated { + collection_key: key.clone(), + }]); + } + if rows.iter().any(|r| r.prior_pk.is_some()) { + return Err(ErrorCode::Internal { + detail: format!( + "columnar image record removes base rows of '{collection}', which \ + has no engine on this core" + ), + }); + } + let Some(first) = rows.iter().find_map(|r| r.image.as_ref()) else { + return Ok(0); + }; + let bitemporal = + self.is_bitemporal(key.0.as_u64(), task.request.tenant_id.as_u64(), collection); + self.ensure_columnar_engine_schema( + &key, + collection, + bitemporal, + first, + schema_bytes, + ) + } + }; + + let mut images: Vec<(Surrogate, Vec, Value)> = Vec::with_capacity(rows.len()); + for row in rows { + if let Some(image) = &row.image { + images.push((row.surrogate, image_values(&schema, image)?, image.clone())); + } + } + let priors: Vec = rows.iter().filter_map(|r| r.prior_pk.clone()).collect(); + if let Some(engine) = self.columnar_engines.get(&key) + && let Some(missing) = priors + .iter() + .find(|pk| !engine.pk_index().contains(&encode_pk(pk))) + { + return Err(ErrorCode::Internal { + detail: format!( + "columnar image record removes base row {missing:?} of '{collection}', \ + which this core does not hold" + ), + }); + } + + let mut undo = Vec::new(); + let removed = + self.apply_columnar_delete_pks(&key, &schema, &priors, recording.then_some(&mut undo)); + self.record_redo_undo(undo); + if removed.affected != priors.len() as u64 { + return Err(ErrorCode::Internal { + detail: format!( + "columnar image record removed {} of {} base rows of '{collection}'", + removed.affected, + priors.len() + ), + }); + } + if recording { + let row_count_before = self + .columnar_engines + .get(&key) + .map_or(0, |engine| engine.memtable().row_count()); + let rows: Vec> = + images.iter().map(|(_, values, _)| values.clone()).collect(); + let (inserted_pks, displaced) = + self.columnar_insert_undo_state(&key, &rows, ColumnarInsertIntent::Put); + self.record_redo_undo([UndoEntry::ColumnarInsert { + collection_key: key.clone(), + row_count_before, + inserted_pks, + displaced, + }]); + } + let Some(engine) = self.columnar_engines.get_mut(&key) else { + return Err(ErrorCode::Internal { + detail: format!("columnar engine of '{collection}' vanished during install"), + }); + }; + for (surrogate, values, _) in &images { + engine + .insert_with_surrogate(values, *surrogate) + .map_err(|e| ErrorCode::Internal { + detail: format!("columnar image install into '{collection}': {e}"), + })?; + } + + // A flush drains the memtable rows the undo would truncate, so the + // install flushes once the whole record landed. + if recording { + self.note_redo_columnar_written(key.clone()); + } else { + self.flush_columnar_memtable_if_needed(task, &key, collection) + .map_err(|response| { + response.error_code.map_or_else( + || ErrorCode::Internal { + detail: format!("columnar flush of '{collection}' failed"), + }, + |error| *error, + ) + })?; + } + let objects: Vec = images.into_iter().map(|(_, _, image)| image).collect(); + let delta = self.index_columnar_geometry_columns(task, &schema, collection, &objects); + if recording { + let mut undo = Vec::new(); + Self::push_geometry_index_undo(&mut undo, delta); + self.record_redo_undo(undo); + } + + let written = priors.len() + objects.len(); + if written > 0 { + self.invalidate_aggregate_cache_for_collection( + key.0.as_u64(), + task.request.tenant_id.as_u64(), + collection, + ); + self.checkpoint_coordinator.mark_dirty("columnar", written); + self.note_collection_write_lsn(task, collection); + } + Ok(written) + } +} diff --git a/nodedb/src/data/executor/wal_replay_columnar_truncate.rs b/nodedb/src/data/executor/wal_replay_columnar_truncate.rs index bbe7350ff..edaab36bc 100644 --- a/nodedb/src/data/executor/wal_replay_columnar_truncate.rs +++ b/nodedb/src/data/executor/wal_replay_columnar_truncate.rs @@ -70,12 +70,18 @@ impl TruncateFloors { /// Whether a record at `lsn` for `key` precedes a truncate of that /// collection and is therefore already removed. + /// + /// Strictly below: a record at the truncate's own LSN is a sibling + /// sub-record of the same transaction redo group. Replay applies a group's + /// sub-records in the order the transaction wrote them, so a row staged + /// before the truncate is removed by the truncate's own replay, and a row + /// staged after it must apply. pub(in crate::data::executor) fn covers( &self, key: &(DatabaseId, TenantId, String), lsn: u64, ) -> bool { - self.floors.get(key).is_some_and(|floor| lsn <= *floor) + self.floors.get(key).is_some_and(|floor| lsn < *floor) } /// Same as [`Self::covers`], keyed by the collection's parts. The key @@ -138,8 +144,17 @@ impl CoreLoop { record_lsn: u64, tombstones: &nodedb_wal::TombstoneSet, ) -> bool { - let Ok(record) = zerompk::from_msgpack::(payload) else { - return false; + let record = match zerompk::from_msgpack::(payload) { + Ok(record) => record, + Err(e) => { + self.replay_record_unapplied( + "columnar", + "truncate_decode", + record_lsn, + &format!("truncate record does not decode: {e}"), + ); + return false; + } }; if tombstones.is_tombstoned( database_id.as_u64(), @@ -149,7 +164,10 @@ impl CoreLoop { ) { return false; } - if self.floors.replay_floors.columnar.covers(record_lsn) { + if self.replay_watermark_skips(self.floors.replay_floors.columnar.covers(record_lsn)) { + return false; + } + if self.claim_for_validation() { return false; } let task = Self::replay_task( @@ -164,14 +182,23 @@ impl CoreLoop { }), Some(Lsn::new(record_lsn)), ); - let response = self.execute_columnar_truncate(&task, &record.collection, None); + let mut undo = Vec::new(); + let recording = self.recording_redo_undo(); + let response = self.execute_columnar_truncate( + &task, + &record.collection, + recording.then_some(&mut undo), + ); + self.record_redo_undo(undo); if response.status != Status::Ok { - tracing::warn!( - core = self.core_id, - collection = %record.collection, - lsn = record_lsn, - error = ?response.error_code, - "WAL replay: columnar truncate handler returned error; skipping" + self.replay_record_rejected( + "columnar", + record_lsn, + response.error_code, + &format!( + "columnar truncate handler rejected truncate of '{}'", + record.collection + ), ); return false; } @@ -190,8 +217,17 @@ impl CoreLoop { record_lsn: u64, tombstones: &nodedb_wal::TombstoneSet, ) -> bool { - let Ok(record) = zerompk::from_msgpack::(payload) else { - return false; + let record = match zerompk::from_msgpack::(payload) { + Ok(record) => record, + Err(e) => { + self.replay_record_unapplied( + "timeseries", + "truncate_decode", + record_lsn, + &format!("truncate record does not decode: {e}"), + ); + return false; + } }; if tombstones.is_tombstoned( database_id.as_u64(), @@ -208,7 +244,10 @@ impl CoreLoop { .iter() .any(|(_, e)| e.meta.last_flushed_wal_lsn >= record_lsn) }); - if flushed_after_truncate { + if self.replay_watermark_skips(flushed_after_truncate) { + return false; + } + if self.claim_for_validation() { return false; } let task = Self::replay_task( @@ -223,14 +262,25 @@ impl CoreLoop { }), Some(Lsn::new(record_lsn)), ); - let response = self.execute_timeseries_truncate(&task, &record.collection, None); + // The install keeps the partition directory aside until the whole + // record landed, so a rollback can rename it back. + let mut undo = Vec::new(); + let recording = self.recording_redo_undo(); + let response = self.execute_timeseries_truncate( + &task, + &record.collection, + recording.then_some(&mut undo), + ); + self.record_redo_undo(undo); if response.status != Status::Ok { - tracing::warn!( - core = self.core_id, - collection = %record.collection, - lsn = record_lsn, - error = ?response.error_code, - "WAL replay: timeseries truncate handler returned error; skipping" + self.replay_record_rejected( + "timeseries", + record_lsn, + response.error_code, + &format!( + "timeseries truncate handler rejected truncate of '{}'", + record.collection + ), ); return false; } @@ -261,6 +311,18 @@ mod tests { .expect("wal record") } + #[test] + fn a_row_at_the_truncates_own_lsn_is_not_covered() { + let records = vec![truncate_record(RecordType::ColumnarTruncate, 9, 0, "c")]; + let floors = TruncateFloors::collect(&records, 1, 0); + let c = (DatabaseId::DEFAULT, TenantId::new(1), "c".to_string()); + assert!(floors.covers(&c, 8)); + assert!( + !floors.covers(&c, 9), + "a sibling sub-record of the truncate's redo group applies in group order" + ); + } + #[test] fn floors_keep_the_highest_truncate_lsn_per_collection_on_this_core() { let records = vec![ @@ -273,9 +335,9 @@ mod tests { let floors = TruncateFloors::collect(&records, 2, 0); let c = (DatabaseId::DEFAULT, TenantId::new(1), "c".to_string()); let ts = (DatabaseId::DEFAULT, TenantId::new(1), "ts".to_string()); - assert!(floors.covers(&c, 9)); + assert!(floors.covers(&c, 8)); assert!(!floors.covers(&c, 10)); - assert!(floors.covers(&ts, 7)); + assert!(floors.covers(&ts, 6)); assert!( !floors.covers(&ts, 8), "the other core's truncate is not ours" diff --git a/nodedb/src/data/executor/wal_replay_fts.rs b/nodedb/src/data/executor/wal_replay_fts.rs index 12c8a3185..9f22e556a 100644 --- a/nodedb/src/data/executor/wal_replay_fts.rs +++ b/nodedb/src/data/executor/wal_replay_fts.rs @@ -42,7 +42,6 @@ use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::replay_abort::abort_replay; use crate::data::executor::task::{ExecutionTask, TaskState}; use crate::types::{DatabaseId, ReadConsistency}; use nodedb_physical::physical_plan::TextOp; @@ -136,13 +135,16 @@ impl CoreLoop { if is_fts_index { let payload = match FtsIndexPayload::from_bytes(&record.payload) { Ok(p) => p, - Err(e) => abort_replay( - "fts", - "decode_index", - self.core_id, - record_lsn, - &format!("FtsIndexPayload could not be decoded: {e}"), - ), + Err(e) => { + self.replay_record_unapplied( + "fts", + "decode_index", + record_lsn, + &format!("FtsIndexPayload could not be decoded: {e}"), + ); + skipped += 1; + continue; + } }; if tombstones.is_tombstoned( @@ -158,19 +160,38 @@ impl CoreLoop { // Re-derive surrogate from the hex doc_id stored in the WAL. let surrogate = match u32::from_str_radix(&payload.doc_id, 16) { Ok(raw) => Surrogate::new(raw), - Err(e) => abort_replay( - "fts", - "doc_id", - self.core_id, - record_lsn, - &format!( - "doc_id '{}' is not the hex surrogate the index path writes: {e}", - payload.doc_id - ), - ), + Err(e) => { + self.replay_record_unapplied( + "fts", + "doc_id", + record_lsn, + &format!( + "doc_id '{}' is not the hex surrogate the index path writes: {e}", + payload.doc_id + ), + ); + skipped += 1; + continue; + } }; let prov = payload.provenance.clone(); + if self.claim_for_validation() { + continue; + } + if self.recording_redo_undo() { + let captured = self.capture_fts_doc_undo( + database_id, + tenant_id, + &payload.collection, + surrogate, + Some(&prov), + ); + if !self.record_redo_capture(captured) { + skipped += 1; + continue; + } + } let vshard = crate::types::VShardId::from_collection_in_database( database_id, @@ -200,29 +221,33 @@ impl CoreLoop { ); if response.status != crate::bridge::envelope::Status::Ok { - abort_replay( + self.replay_record_unapplied( "fts", "index_handler", - self.core_id, record_lsn, &format!( - "the FtsIndexDoc handler rejected a committed write into '{}'", - payload.collection + "the FtsIndexDoc handler rejected a committed write into '{}': {:?}", + payload.collection, response.error_code ), ); + skipped += 1; + continue; } indexed += 1; } else { // FtsDelete let payload = match FtsDeletePayload::from_bytes(&record.payload) { Ok(p) => p, - Err(e) => abort_replay( - "fts", - "decode_delete", - self.core_id, - record_lsn, - &format!("FtsDeletePayload could not be decoded: {e}"), - ), + Err(e) => { + self.replay_record_unapplied( + "fts", + "decode_delete", + record_lsn, + &format!("FtsDeletePayload could not be decoded: {e}"), + ); + skipped += 1; + continue; + } }; if tombstones.is_tombstoned( @@ -237,19 +262,38 @@ impl CoreLoop { let surrogate = match u32::from_str_radix(&payload.doc_id, 16) { Ok(raw) => Surrogate::new(raw), - Err(e) => abort_replay( - "fts", - "doc_id", - self.core_id, - record_lsn, - &format!( - "doc_id '{}' is not the hex surrogate the index path writes: {e}", - payload.doc_id - ), - ), + Err(e) => { + self.replay_record_unapplied( + "fts", + "doc_id", + record_lsn, + &format!( + "doc_id '{}' is not the hex surrogate the index path writes: {e}", + payload.doc_id + ), + ); + skipped += 1; + continue; + } }; let prov = payload.provenance.clone(); + if self.claim_for_validation() { + continue; + } + if self.recording_redo_undo() { + let captured = self.capture_fts_doc_undo( + database_id, + tenant_id, + &payload.collection, + surrogate, + Some(&prov), + ); + if !self.record_redo_capture(captured) { + skipped += 1; + continue; + } + } let vshard = crate::types::VShardId::from_collection_in_database( database_id, @@ -277,16 +321,17 @@ impl CoreLoop { ); if response.status != crate::bridge::envelope::Status::Ok { - abort_replay( + self.replay_record_unapplied( "fts", "delete_handler", - self.core_id, record_lsn, &format!( - "the FtsDeleteDoc handler rejected a committed delete in '{}'", - payload.collection + "the FtsDeleteDoc handler rejected a committed delete in '{}': {:?}", + payload.collection, response.error_code ), ); + skipped += 1; + continue; } deleted += 1; } diff --git a/nodedb/src/data/executor/wal_replay_graph_labels.rs b/nodedb/src/data/executor/wal_replay_graph_labels.rs index 9619cbf95..086bb3b7b 100644 --- a/nodedb/src/data/executor/wal_replay_graph_labels.rs +++ b/nodedb/src/data/executor/wal_replay_graph_labels.rs @@ -36,12 +36,11 @@ //! `Err`. `remove_node_label` never vivifies (it no-ops on an unknown node or //! unknown label), so it was already identical to live and needs no change. -use tracing::warn; - use nodedb_wal::WalRecord; use nodedb_wal::record::RecordType; use super::core_loop::CoreLoop; +use super::handlers::transaction::undo::UndoEntry; use crate::types::DatabaseId; impl CoreLoop { @@ -69,15 +68,27 @@ impl CoreLoop { let Ok((node_id, labels)) = zerompk::from_msgpack::<(String, Vec)>(&record.payload) else { - warn!( - core = self.core_id, - lsn = record_lsn, - "WAL graph node-label replay: malformed payload; skipping" + self.replay_record_rejected( + "graph", + record_lsn, + None, + "malformed graph node-label WAL record", ); return Some(0); }; + if self.claim_for_validation() { + return Some(0); + } + if self.recording_redo_undo() { + let prior = self.node_label_prior(database_id.as_u64(), tenant_id, &node_id, &labels); + self.record_redo_undo([UndoEntry::NodeLabels { + database_id: database_id.as_u64(), + tid: tenant_id, + node_id: node_id.clone(), + prior, + }]); + } - let partition = self.csr_partition_mut(database_id.as_u64(), tenant_id); if is_set { for label in &labels { // `add_node_label` vivifies `node_id` via `ensure_node` exactly @@ -85,18 +96,21 @@ impl CoreLoop { // 64-distinct-label bitset limit) is discarded here, mirroring // the live handler, which also never inspects the returned // bool on `Ok`. - if let Err(e) = partition.add_node_label(&node_id, label) { - warn!( - core = self.core_id, - %node_id, - lsn = record_lsn, - error = %e, - "WAL graph node-label replay: set label failed; skipping" + let added = self + .csr_partition_mut(database_id.as_u64(), tenant_id) + .add_node_label(&node_id, label); + if let Err(e) = added { + self.replay_record_rejected( + "graph", + record_lsn, + None, + &format!("setting label '{label}' on node '{node_id}' failed: {e}"), ); return Some(0); } } } else { + let partition = self.csr_partition_mut(database_id.as_u64(), tenant_id); // `remove_node_label` no-ops on an unknown node or unknown label — // it never vivifies, so calling it unconditionally is already // identical to the live `RemoveNodeLabels` handler. @@ -107,6 +121,28 @@ impl CoreLoop { Some(1) } + /// Whether `node_id` carries each of `labels` now. + fn node_label_prior( + &self, + database_id: u64, + tenant_id: u64, + node_id: &str, + labels: &[String], + ) -> Vec<(String, bool)> { + let partition = self.csr_partition(database_id, tenant_id); + let local = partition.and_then(|p| p.node_id_raw(node_id)); + labels + .iter() + .map(|label| { + let carried = match (partition, local) { + (Some(p), Some(id)) => p.node_has_label(id, label), + _ => false, + }; + (label.clone(), carried) + }) + .collect() + } + /// Replay every `GraphNodeLabelSet` / `GraphNodeLabelRemove` record in /// `records`, routing each through [`CoreLoop::try_replay_graph_node_label`]. /// diff --git a/nodedb/src/data/executor/wal_replay_kv_atomic.rs b/nodedb/src/data/executor/wal_replay_kv_atomic.rs index 29e561e8e..149669608 100644 --- a/nodedb/src/data/executor/wal_replay_kv_atomic.rs +++ b/nodedb/src/data/executor/wal_replay_kv_atomic.rs @@ -105,8 +105,18 @@ impl CoreLoop { }, &expected, &new_value, + // Replay re-applies a write the policy already admitted when it + // was first accepted. + &crate::engine::kv::admit_any, ); - if result.success { + let swapped = match result { + Ok(result) => result.success(), + Err(error) => { + self.replay_swap_error("cas", &collection, &key, record_lsn, error); + false + } + }; + if swapped { self.note_replay_write_lsn( database_id, tenant_id, @@ -115,7 +125,7 @@ impl CoreLoop { record_lsn, ); } - Some(usize::from(result.success)) + Some(usize::from(swapped)) } /// Decode + tombstone-gate + replay one `kv_incr_float` WAL record. @@ -242,7 +252,7 @@ impl CoreLoop { if self.skip_kv_replay_record(tombstones, tenant_id, &collection, record_lsn) { return Some(0); } - self.kv_engine.getset( + let result = self.kv_engine.getset( AtomicKeyCtx { database_id, tenant_id, @@ -252,7 +262,14 @@ impl CoreLoop { surrogate: nodedb_types::Surrogate::new(surrogate), }, &new_value, + // Replay re-applies a write the policy already admitted when it + // was first accepted. + &crate::engine::kv::admit_any, ); + if let Err(error) = result { + self.replay_swap_error("getset", &collection, &key, record_lsn, error); + return Some(0); + } self.note_replay_write_lsn( database_id, tenant_id, @@ -262,6 +279,42 @@ impl CoreLoop { ); Some(1) } + + /// Handle a `cas` or `getset` record whose replay computed no value. + /// + /// The live apply of the same record against the same pre-state failed + /// the same way and wrote nothing, so replay skips it. A refusal by the + /// write gate cannot happen here: replay hands the engine `admit_any`. + /// Reaching it means a redo path re-decides writes that were already + /// admitted, so replay stops and files a forensic report. + fn replay_swap_error( + &self, + op: &'static str, + collection: &str, + key: &[u8], + record_lsn: u64, + error: AtomicError, + ) { + let detail = match error { + AtomicError::TypeMismatch { detail } | AtomicError::Encode { detail } => detail, + AtomicError::Overflow => "overflow".to_string(), + AtomicError::Rejected(error) => abort_replay( + "kv", + "swap_admission", + self.core_id, + record_lsn, + &format!("the RLS write gate refused a committed {op} on '{collection}': {error}"), + ), + }; + warn!( + core = self.core_id, + collection = %collection, + key = %String::from_utf8_lossy(key), + op, + %detail, + "WAL kv swap replay: no value computed, skipping record" + ); + } } #[cfg(test)] diff --git a/nodedb/src/data/executor/wal_replay_redo_document.rs b/nodedb/src/data/executor/wal_replay_redo_document.rs index 64caa0de1..c2afb5323 100644 --- a/nodedb/src/data/executor/wal_replay_redo_document.rs +++ b/nodedb/src/data/executor/wal_replay_redo_document.rs @@ -25,7 +25,10 @@ //! The autocommit delete shape `(collection, document_id, prov)` omits the //! surrogate; replay needs it (the redb storage key is //! `StorageKey::for_surrogate(surrogate)`, and the delete cascade keys on -//! it), so the redo shape appends it as a fourth element. +//! it), so the redo shape appends it as a fourth element. A +//! `bitemporal=true` collection's delete appends its resolve-time +//! `sys_from_ms` as a fifth element, and the decoded stamp forces the +//! versioned tombstone at that exact version key. //! //! ## Idempotency //! @@ -65,6 +68,14 @@ //! //! Both paths run exactly once over their own node's state; they differ only in //! whether that state already contains the effect. +//! +//! ### Committed-redo apply +//! +//! A replica applying a committed `TransactionRedo` Raft entry drives this arm +//! with a redo-apply scope open. The source rows it writes then fold into +//! their targets and link their hash chain, exactly as replication does (see +//! `handlers::transaction::redo_apply`). With no scope open this arm is plain +//! restart replay. use nodedb_types::Surrogate; use nodedb_types::sync::wire::SyncProvenance; @@ -75,6 +86,7 @@ use super::core_loop::CoreLoop; use super::handlers::point::apply_delete::PointDeleteParams; use super::handlers::point::apply_put::PointPutParams; use super::handlers::transaction::overlay::BitemporalStamp; +use super::handlers::transaction::redo_apply::CommittedDocWrite; use crate::data::executor::core_loop::write_index::KeyRepr; use crate::engine::document::store::StorageKey; @@ -134,12 +146,14 @@ impl CoreLoop { ); type PlainPut = (String, String, Vec, Option, u32); // Replay keys the row by its surrogate; the record's text - // `document_id` is the client key and stays unread. + // `document_id` is the client key, read only by a committed + // redo apply for the row's event identity. let decoded = zerompk::from_msgpack::(&record.payload) .map( - |(collection, _document_id, value, _prov, surrogate, sys, vf, vu)| { + |(collection, document_id, value, _prov, surrogate, sys, vf, vu)| { ( collection, + document_id, value, surrogate, Some(BitemporalStamp { @@ -152,17 +166,20 @@ impl CoreLoop { ) .or_else(|_| { zerompk::from_msgpack::(&record.payload).map( - |(collection, _document_id, value, _prov, surrogate)| { - (collection, value, surrogate, None) + |(collection, document_id, value, _prov, surrogate)| { + (collection, document_id, value, surrogate, None) }, ) }); - let Ok((collection, value, surrogate_u32, stamp)) = decoded else { + let Ok((collection, document_id, value, surrogate_u32, stamp)) = decoded else { continue; }; if tombstones.is_tombstoned(database_id, tenant_id, &collection, record_lsn) { continue; } + if self.claim_for_validation() { + continue; + } // Carry the stamp into apply scratch (forcing the versioned // branch at the exact stamp the commit-time install used) and // advance the per-core HLC so post-restart writes stay monotonic. @@ -170,14 +187,28 @@ impl CoreLoop { self.observe_bitemporal_stamp(s.sys_from_ms); self.active_bitemporal_stamps.insert(surrogate_u32, s); } - let applied = self.apply_document_put( - database_id, - tenant_id, - &collection, - surrogate_u32, - &value, - record_lsn, - ); + let applied = if self.redo_apply.scope.is_some() { + self.apply_committed_document_put( + CommittedDocWrite { + database_id, + tenant_id, + collection: &collection, + document_id: &document_id, + surrogate: surrogate_u32, + record_lsn, + }, + &value, + ) + } else { + self.apply_document_put( + database_id, + tenant_id, + &collection, + surrogate_u32, + &value, + record_lsn, + ) + }; if stamp.is_some() { self.active_bitemporal_stamps.remove(&surrogate_u32); } @@ -193,18 +224,61 @@ impl CoreLoop { } } else { // Replay keys the row by its surrogate; the record's text - // `document_id` is the client key and stays unread. - let Ok((collection, _document_id, _prov, surrogate_u32)) = - zerompk::from_msgpack::<(String, String, Option, u32)>( - &record.payload, - ) - else { + // `document_id` is the client key, read only by a committed + // redo apply for the row's event identity. + // A `bitemporal=true` collection's delete carries its + // resolve-time system time as a fifth element; the plain form + // has four. The stamp forces the versioned tombstone at that + // exact version key. + type BitemporalDelete = (String, String, Option, u32, i64); + type PlainDelete = (String, String, Option, u32); + let decoded = zerompk::from_msgpack::(&record.payload) + .map(|(collection, document_id, _prov, surrogate, sys)| { + (collection, document_id, surrogate, Some(sys)) + }) + .or_else(|_| { + zerompk::from_msgpack::(&record.payload).map( + |(collection, document_id, _prov, surrogate)| { + (collection, document_id, surrogate, None) + }, + ) + }); + let Ok((collection, document_id, surrogate_u32, sys_from_ms)) = decoded else { continue; }; if tombstones.is_tombstoned(database_id, tenant_id, &collection, record_lsn) { continue; } - if self.apply_document_delete(database_id, tenant_id, &collection, surrogate_u32) { + if self.claim_for_validation() { + continue; + } + if let Some(sys) = sys_from_ms { + self.observe_bitemporal_stamp(sys); + self.active_bitemporal_stamps.insert( + surrogate_u32, + BitemporalStamp { + sys_from_ms: sys, + valid_from_ms: i64::MIN, + valid_until_ms: i64::MAX, + }, + ); + } + let removed = if self.redo_apply.scope.is_some() { + self.apply_committed_document_delete(CommittedDocWrite { + database_id, + tenant_id, + collection: &collection, + document_id: &document_id, + surrogate: surrogate_u32, + record_lsn, + }) + } else { + self.apply_document_delete(database_id, tenant_id, &collection, surrogate_u32) + }; + if sys_from_ms.is_some() { + self.active_bitemporal_stamps.remove(&surrogate_u32); + } + if removed { deletes += 1; self.note_replay_write_lsn( database_id, @@ -849,4 +923,86 @@ mod tests { "kv sub-record must be replayed" ); } + + fn bitemporal_put_sub(collection: &str, surrogate: u32, sys_from_ms: i64) -> RedoSubRecord { + let prov: Option = None; + let payload = zerompk::to_msgpack_vec(&( + collection, + "userpk", + doc_value("alice"), + prov, + surrogate, + sys_from_ms, + i64::MIN, + i64::MAX, + )) + .expect("encode bitemporal put sub-record"); + RedoSubRecord { + record_type: RecordType::Put as u32, + payload, + } + } + + fn bitemporal_delete_sub(collection: &str, surrogate: u32, sys_from_ms: i64) -> RedoSubRecord { + let prov: Option = None; + let payload = + zerompk::to_msgpack_vec(&(collection, "userpk", prov, surrogate, sys_from_ms)) + .expect("encode bitemporal delete sub-record"); + RedoSubRecord { + record_type: RecordType::Delete as u32, + payload, + } + } + + /// A bitemporal redo delete carries its resolve-time system time, so two + /// independent applies of the same record (two replicas, or a replica and + /// a restart) write the tombstone at the same version key. + #[test] + fn a_bitemporal_redo_delete_tombstones_at_its_carried_system_time_on_every_apply() { + const PUT_MS: i64 = 1_000; + const DELETE_MS: i64 = 2_000; + let surrogate = 42u32; + let record = redo_record( + 7, + 0, + vec![ + bitemporal_put_sub("hist", surrogate, PUT_MS), + bitemporal_delete_sub("hist", surrogate, DELETE_MS), + ], + ); + let row_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); + + for _replica in 0..2 { + let mut h = make_core(); + h.core + .replay_transaction_redo_wal( + std::slice::from_ref(&record), + 1, + &nodedb_wal::TombstoneSet::new(), + ) + .expect("redo replay must succeed"); + let before = h + .core + .sparse + .versioned_get_as_of(0, 7, "hist", &row_key, Some(DELETE_MS - 1), None) + .expect("read before the tombstone"); + assert!( + before.is_some(), + "the row is live before the carried tombstone" + ); + let at = h + .core + .sparse + .versioned_get_as_of(0, 7, "hist", &row_key, Some(DELETE_MS), None) + .expect("read at the tombstone"); + assert!( + at.is_none(), + "the tombstone sits at the carried system time, not a locally minted one" + ); + assert!( + h.core.active_bitemporal_stamps.is_empty(), + "the carried stamp is scoped to its own apply" + ); + } + } } diff --git a/nodedb/src/data/executor/wal_replay_redo_graph.rs b/nodedb/src/data/executor/wal_replay_redo_graph.rs index a9315c3f5..631582448 100644 --- a/nodedb/src/data/executor/wal_replay_redo_graph.rs +++ b/nodedb/src/data/executor/wal_replay_redo_graph.rs @@ -100,6 +100,9 @@ impl CoreLoop { ) { continue; } + if self.claim_for_validation() { + continue; + } let task = Self::replay_graph_task( tenant_id, database_id, @@ -118,7 +121,9 @@ impl CoreLoop { }), ); self.active_graph_system_from = system_from; - let response = self.execute_edge_put( + let mut undo = Vec::new(); + let recording = self.recording_redo_undo(); + let response = self.execute_edge_put_with_undo( &task, EdgePutParams { tid: tenant_id, @@ -130,16 +135,18 @@ impl CoreLoop { src_surrogate: nodedb_types::Surrogate::new(src_sur), dst_surrogate: nodedb_types::Surrogate::new(dst_sur), }, + recording.then_some(&mut undo), ); self.active_graph_system_from = None; + self.record_redo_undo(undo); if response.status == crate::bridge::envelope::Status::Ok { puts += 1; } else { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - "WAL graph redo: edge put handler returned error; skipping" + self.replay_record_rejected( + "graph", + record_lsn, + response.error_code, + &format!("graph edge put into '{collection}' failed"), ); } } else { @@ -164,6 +171,9 @@ impl CoreLoop { ) { continue; } + if self.claim_for_validation() { + continue; + } let task = Self::replay_graph_task( tenant_id, database_id, @@ -185,7 +195,9 @@ impl CoreLoop { }), ); self.active_graph_system_from = system_from; - let response = self.execute_edge_delete( + let mut undo = Vec::new(); + let recording = self.recording_redo_undo(); + let response = self.execute_edge_delete_with_undo( &task, crate::data::executor::handlers::graph::EdgeDeleteParams { tid: tenant_id, @@ -198,16 +210,18 @@ impl CoreLoop { // identity is not present at boot. rls_write_check: &nodedb_types::RlsWriteCheck::already_decided_elsewhere(), }, + recording.then_some(&mut undo), ); self.active_graph_system_from = None; + self.record_redo_undo(undo); if response.status == crate::bridge::envelope::Status::Ok { deletes += 1; } else { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - "WAL graph redo: edge delete handler returned error; skipping" + self.replay_record_rejected( + "graph", + record_lsn, + response.error_code, + &format!("graph edge delete in '{collection}' failed"), ); } } diff --git a/nodedb/src/data/executor/wal_replay_spatial.rs b/nodedb/src/data/executor/wal_replay_spatial.rs index 85467a67f..6681c5c8e 100644 --- a/nodedb/src/data/executor/wal_replay_spatial.rs +++ b/nodedb/src/data/executor/wal_replay_spatial.rs @@ -43,7 +43,6 @@ use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::spatial_sync::SpatialInsertExec; -use crate::data::executor::replay_abort::abort_replay; use crate::data::executor::task::{ExecutionTask, TaskState}; use crate::types::{DatabaseId, ReadConsistency}; use nodedb_physical::physical_plan::SpatialOp; @@ -51,6 +50,27 @@ use nodedb_types::Surrogate; use nodedb_wal::record::RecordType; impl CoreLoop { + /// In the install pass of a committed-redo apply, record the undo of one + /// spatial write before it runs: the row's pre-image and the sync + /// high-water mark. Returns `false` when the pre-image cannot be read; + /// the error is kept on the apply and the write is skipped. + fn record_spatial_row_undo( + &mut self, + database_id: DatabaseId, + tenant_id: u64, + (collection, field): (&str, &str), + surrogate: Surrogate, + provenance: &nodedb_types::sync::wire::SyncProvenance, + ) -> bool { + if !self.recording_redo_undo() { + return true; + } + let captured = self + .capture_spatial_row_undo(database_id, tenant_id, collection, field, surrogate) + .map(|row| std::iter::once(row).chain(self.capture_sync_hwm_undo(Some(provenance)))); + self.record_redo_capture(captured) + } + /// Build a synthetic `ExecutionTask` for Spatial WAL replay. fn replay_spatial_task( tenant_id: crate::types::TenantId, @@ -133,13 +153,16 @@ impl CoreLoop { if is_spatial_put { let payload = match SpatialPutPayload::from_bytes(&record.payload) { Ok(p) => p, - Err(e) => abort_replay( - "spatial", - "decode_put", - self.core_id, - record_lsn, - &format!("SpatialPutPayload could not be decoded: {e}"), - ), + Err(e) => { + self.replay_record_unapplied( + "spatial", + "decode_put", + record_lsn, + &format!("SpatialPutPayload could not be decoded: {e}"), + ); + skipped += 1; + continue; + } }; if tombstones.is_tombstoned( @@ -154,35 +177,54 @@ impl CoreLoop { let surrogate = match u32::from_str_radix(&payload.doc_id, 16) { Ok(raw) => Surrogate::new(raw), - Err(e) => abort_replay( - "spatial", - "doc_id", - self.core_id, - record_lsn, - &format!( - "doc_id '{}' is not the hex surrogate the insert path writes: {e}", - payload.doc_id - ), - ), + Err(e) => { + self.replay_record_unapplied( + "spatial", + "doc_id", + record_lsn, + &format!( + "doc_id '{}' is not the hex surrogate the insert path writes: {e}", + payload.doc_id + ), + ); + skipped += 1; + continue; + } }; // Decode geometry from msgpack bytes stored in the WAL payload. let geometry: nodedb_types::geometry::Geometry = match zerompk::from_msgpack(&payload.geometry_bytes) { Ok(g) => g, - Err(e) => abort_replay( - "spatial", - "geometry", - self.core_id, - record_lsn, - &format!( - "the geometry committed into '{}' could not be decoded: {e}", - payload.collection - ), - ), + Err(e) => { + self.replay_record_unapplied( + "spatial", + "geometry", + record_lsn, + &format!( + "the geometry committed into '{}' could not be decoded: {e}", + payload.collection + ), + ); + skipped += 1; + continue; + } }; let prov = payload.provenance.clone(); + if self.claim_for_validation() { + continue; + } + if !self.record_spatial_row_undo( + database_id, + tenant_id, + (&payload.collection, &payload.field), + surrogate, + &prov, + ) { + skipped += 1; + continue; + } let vshard = crate::types::VShardId::from_collection_in_database( database_id, @@ -214,29 +256,33 @@ impl CoreLoop { }); if response.status != crate::bridge::envelope::Status::Ok { - abort_replay( + self.replay_record_unapplied( "spatial", "insert_handler", - self.core_id, record_lsn, &format!( - "the SpatialInsert handler rejected a committed write into '{}'", - payload.collection + "the SpatialInsert handler rejected a committed write into '{}': {:?}", + payload.collection, response.error_code ), ); + skipped += 1; + continue; } inserted += 1; } else { // SpatialDelete let payload = match SpatialDeletePayload::from_bytes(&record.payload) { Ok(p) => p, - Err(e) => abort_replay( - "spatial", - "decode_delete", - self.core_id, - record_lsn, - &format!("SpatialDeletePayload could not be decoded: {e}"), - ), + Err(e) => { + self.replay_record_unapplied( + "spatial", + "decode_delete", + record_lsn, + &format!("SpatialDeletePayload could not be decoded: {e}"), + ); + skipped += 1; + continue; + } }; if tombstones.is_tombstoned( @@ -251,19 +297,35 @@ impl CoreLoop { let surrogate = match u32::from_str_radix(&payload.doc_id, 16) { Ok(raw) => Surrogate::new(raw), - Err(e) => abort_replay( - "spatial", - "doc_id", - self.core_id, - record_lsn, - &format!( - "doc_id '{}' is not the hex surrogate the insert path writes: {e}", - payload.doc_id - ), - ), + Err(e) => { + self.replay_record_unapplied( + "spatial", + "doc_id", + record_lsn, + &format!( + "doc_id '{}' is not the hex surrogate the insert path writes: {e}", + payload.doc_id + ), + ); + skipped += 1; + continue; + } }; let prov = payload.provenance.clone(); + if self.claim_for_validation() { + continue; + } + if !self.record_spatial_row_undo( + database_id, + tenant_id, + (&payload.collection, &payload.field), + surrogate, + &prov, + ) { + skipped += 1; + continue; + } let vshard = crate::types::VShardId::from_collection_in_database( database_id, @@ -293,16 +355,17 @@ impl CoreLoop { ); if response.status != crate::bridge::envelope::Status::Ok { - abort_replay( + self.replay_record_unapplied( "spatial", "delete_handler", - self.core_id, record_lsn, &format!( - "the SpatialDelete handler rejected a committed delete in '{}'", - payload.collection + "the SpatialDelete handler rejected a committed delete in '{}': {:?}", + payload.collection, response.error_code ), ); + skipped += 1; + continue; } deleted += 1; } diff --git a/nodedb/src/data/executor/wal_replay_vector.rs b/nodedb/src/data/executor/wal_replay_vector.rs index 4d1a34565..e02eba487 100644 --- a/nodedb/src/data/executor/wal_replay_vector.rs +++ b/nodedb/src/data/executor/wal_replay_vector.rs @@ -2,55 +2,13 @@ //! WAL replay for vector engine startup recovery. -use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; -use crate::data::executor::replay_abort::abort_replay; -use crate::data::executor::task::{ExecutionTask, TaskState}; -use crate::types::{DatabaseId, ReadConsistency}; +use crate::bridge::envelope::PhysicalPlan; +use crate::types::DatabaseId; use super::core_loop::CoreLoop; +use super::wal_replay_vector_redo::RedoVectorWrite; impl CoreLoop { - /// Build a synthetic `ExecutionTask` for WAL replay. - /// - /// Mirrors the equivalent helper in `timeseries_wal.rs`. The task carries - /// no meaningful request semantics — it is only needed so that the handler - /// methods can return a typed `Response`. - pub(in crate::data::executor) fn replay_vector_task( - tenant_id: crate::types::TenantId, - database_id: DatabaseId, - vshard_id: crate::types::VShardId, - plan: PhysicalPlan, - ) -> ExecutionTask { - ExecutionTask { - request: Request { - request_id: crate::types::RequestId::new(0), - tenant_id, - database_id, - vshard_id, - plan, - deadline: std::time::Instant::now() - + crate::data::executor::deadline::REPLAY_DEADLINE, - priority: Priority::Normal, - trace_id: crate::types::TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::AlreadyOrdered, - ), - }, - state: TaskState::Running, - wal_lsn: None, - resolved_now_ms: None, - } - } - /// Replay WAL vector records to rebuild in-memory HNSW indexes after crash. /// /// Called once during startup, after `open()` but before the event loop. @@ -101,6 +59,12 @@ impl CoreLoop { let record_lsn = record.header.lsn; let tombstones = tombstones.for_database(database_id); + // A committed redo record carries no index DDL; left unclaimed, + // the validate pass refuses a record that does. + if (is_index_drop || is_vector_params) && self.applying_committed_redo() { + continue; + } + if is_index_drop { // Applied in LSN order, so it wipes the params / puts that // preceded it and leaves a later re-CREATE to rebuild. @@ -158,10 +122,9 @@ impl CoreLoop { // payload, so a disagreement is not a schema change the // record predates — it is a record whose two halves // cannot both be what the writer wrote. - abort_replay( + self.replay_record_unapplied( "vector", "dim", - self.core_id, record_lsn, &format!( "record for '{collection}' declares dim {dim} but carries {} \ @@ -169,6 +132,8 @@ impl CoreLoop { vector.len() ), ); + skipped += 1; + continue; } // Checkpoint watermark gate: a restored checkpoint already // contains every write at or below its `checkpoint_wal_lsn`. @@ -182,13 +147,31 @@ impl CoreLoop { &collection, &field_name, ); - if let Some(existing) = self.vector_collections.get(&insert_index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&insert_index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { skipped += 1; continue; } let surrogate = nodedb_types::Surrogate::new(surrogate_u32); + if !self.redo_vector_prelude( + RedoVectorWrite { + index_key: &insert_index_key, + tid: tenant_id, + collection: &collection, + dim, + surrogates: &[surrogate], + ids: &[], + sidecars: false, + }, + provenance.as_ref(), + record_lsn, + ) { + skipped += 1; + continue; + } // Local replay rebinds by the carried surrogate; the // compat doc-id slot (always `None` on this write path) // maps straight through to `pk_bytes` for fidelity. @@ -226,16 +209,18 @@ impl CoreLoop { }, ); if response.status != crate::bridge::envelope::Status::Ok { - abort_replay( + self.replay_record_unapplied( "vector", "insert_handler", - self.core_id, record_lsn, &format!( "the vector insert handler rejected a committed write into \ - '{collection}'" + '{collection}': {:?}", + response.error_code ), ); + skipped += 1; + continue; } // Advance the (possibly freshly created) collection's // watermark so the next checkpoint records this replayed @@ -249,6 +234,11 @@ impl CoreLoop { &record.payload, ) { + // A committed redo record never carries this shape; left + // unclaimed, the validate pass refuses the record. + if self.applying_committed_redo() { + continue; + } if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { skipped += 1; continue; @@ -258,10 +248,9 @@ impl CoreLoop { // payload, so a disagreement is not a schema change the // record predates — it is a record whose two halves // cannot both be what the writer wrote. - abort_replay( + self.replay_record_unapplied( "vector", "dim", - self.core_id, record_lsn, &format!( "record for '{collection}' declares dim {dim} but carries {} \ @@ -269,6 +258,8 @@ impl CoreLoop { vector.len() ), ); + skipped += 1; + continue; } let index_key = CoreLoop::vector_index_key( database_id, @@ -277,9 +268,11 @@ impl CoreLoop { &field_name, ); // Checkpoint watermark gate (see the surrogate arm above). - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { skipped += 1; continue; } @@ -307,12 +300,15 @@ impl CoreLoop { // than malformed, so this stays a skip: aborting here would // wedge the boot on a retained pre-rebuild tail. if index.dim() != dim { - tracing::warn!( - core = self.core_id, - %collection, - index_dim = index.dim(), - record_dim = dim, - "skipping WAL vector record: index dimension mismatch" + let index_dim = index.dim(); + self.replay_record_rejected( + "vector", + record_lsn, + None, + &format!( + "vector record for '{collection}' has dim {dim}, the index has \ + dim {index_dim}" + ), ); continue; } @@ -327,6 +323,11 @@ impl CoreLoop { } else if let Ok((collection, vector, dim)) = zerompk::from_msgpack::<(String, Vec, usize)>(&record.payload) { + // A committed redo record never carries this shape; left + // unclaimed, the validate pass refuses the record. + if self.applying_committed_redo() { + continue; + } if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { skipped += 1; continue; @@ -336,10 +337,9 @@ impl CoreLoop { // payload, so a disagreement is not a schema change the // record predates — it is a record whose two halves // cannot both be what the writer wrote. - abort_replay( + self.replay_record_unapplied( "vector", "dim", - self.core_id, record_lsn, &format!( "record for '{collection}' declares dim {dim} but carries {} \ @@ -347,13 +347,17 @@ impl CoreLoop { vector.len() ), ); + skipped += 1; + continue; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, ""); // Checkpoint watermark gate (see the surrogate arm above). - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { skipped += 1; continue; } @@ -381,12 +385,15 @@ impl CoreLoop { // than malformed, so this stays a skip: aborting here would // wedge the boot on a retained pre-rebuild tail. if index.dim() != dim { - tracing::warn!( - core = self.core_id, - %collection, - index_dim = index.dim(), - record_dim = dim, - "skipping WAL vector record: index dimension mismatch" + let index_dim = index.dim(); + self.replay_record_rejected( + "vector", + record_lsn, + None, + &format!( + "vector record for '{collection}' has dim {dim}, the index has \ + dim {index_dim}" + ), ); continue; } @@ -403,9 +410,27 @@ impl CoreLoop { let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, ""); // Checkpoint watermark gate (see the surrogate arm above). - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { + skipped += 1; + continue; + } + if !self.redo_vector_prelude( + RedoVectorWrite { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim, + surrogates: &[], + ids: &[], + sidecars: false, + }, + None, + record_lsn, + ) { skipped += 1; continue; } @@ -432,89 +457,16 @@ impl CoreLoop { inserted += 1; } } else if is_vector_delete { - // Decode order (longest shape first for backward compatibility): - // - // 4-element: (collection, surrogate_u32, field_name, Option) - // → sync-path delete-by-surrogate; routes through the handler so the - // idempotency gate fires on replay. - // - // 3-element: (collection, vector_id, Option) - // → local delete-by-node-id with provenance (discarded here). - // - // 2-element: (collection, vector_id) - // → legacy shape; direct node-id deletion. - if let Ok((collection, surrogate_u32, field_name, provenance)) = - zerompk::from_msgpack::<( - String, - u32, - String, - Option, - )>(&record.payload) - { - if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { - skipped += 1; - continue; - } - let surrogate = nodedb_types::Surrogate::new(surrogate_u32); - let vshard = crate::types::VShardId::from_collection_in_database( - DatabaseId::new(database_id), - &collection, - ); - let task = Self::replay_vector_task( - nodedb_types::TenantId::new(tenant_id), - DatabaseId::new(database_id), - vshard, - PhysicalPlan::Vector( - nodedb_physical::physical_plan::VectorOp::DeleteBySurrogate { - collection: nodedb_types::QualifiedCollection::from_stored( - collection.clone(), - ), - surrogate, - field_name: field_name.clone(), - provenance: provenance.clone(), - }, - ), - ); - let response = self.execute_vector_delete_by_surrogate( - &task, - tenant_id, - &collection, - surrogate, - &field_name, - provenance.as_ref(), - ); - if response.status != crate::bridge::envelope::Status::Ok { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - "WAL vector replay: delete-by-surrogate handler returned error; skipping" - ); - skipped += 1; - continue; - } + if self.replay_vector_delete_record( + &record.payload, + tenant_id, + database_id, + record_lsn, + &tombstones, + ) { deleted += 1; } else { - // Legacy: 3-element (with discarded provenance) or 2-element. - let delete_decoded = zerompk::from_msgpack::<( - String, - u32, - Option, - )>(&record.payload) - .map(|(c, id, _prov)| (c, id)) - .or_else(|_| zerompk::from_msgpack::<(String, u32)>(&record.payload)); - if let Ok((collection, vector_id)) = delete_decoded { - if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { - skipped += 1; - continue; - } - let index_key = - CoreLoop::vector_index_key(database_id, tenant_id, &collection, ""); - if let Some(index) = self.vector_collections.get_mut(&index_key) { - index.delete(vector_id); - deleted += 1; - } - } + skipped += 1; } } } diff --git a/nodedb/src/data/executor/wal_replay_vector_delete.rs b/nodedb/src/data/executor/wal_replay_vector_delete.rs new file mode 100644 index 000000000..87af61862 --- /dev/null +++ b/nodedb/src/data/executor/wal_replay_vector_delete.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! WAL replay of one `VectorDelete` record. +//! +//! Decode order, longest shape first: +//! +//! * `(collection, surrogate, field_name, Option)`: the +//! delete-by-surrogate shape. It routes through the live handler so the +//! sync idempotency gate runs on replay exactly as on the live path. +//! * `(collection, vector_id, Option)`: a delete by local node +//! id. The provenance is not used. +//! * `(collection, vector_id)`: the same delete without provenance. + +use crate::bridge::envelope::PhysicalPlan; +use crate::types::DatabaseId; + +use super::core_loop::CoreLoop; +use super::wal_replay_vector_redo::RedoVectorWrite; + +impl CoreLoop { + /// Replay one `VectorDelete` record. Returns whether a vector was + /// deleted. A record no shape decodes is reported through + /// `replay_policy` as unapplied. + pub(in crate::data::executor) fn replay_vector_delete_record( + &mut self, + payload: &[u8], + tenant_id: u64, + database_id: u64, + record_lsn: u64, + tombstones: &nodedb_wal::DatabaseTombstones<'_>, + ) -> bool { + if let Ok((collection, surrogate_u32, field_name, provenance)) = zerompk::from_msgpack::<( + String, + u32, + String, + Option, + )>(payload) + { + if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { + return false; + } + let surrogate = nodedb_types::Surrogate::new(surrogate_u32); + // The handler resolves the field-suffixed index first, then the + // plain one; the undo covers the index it will write. + let field_key = + CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field_name); + let index_key = if self.vector_collections.contains_key(&field_key) { + field_key + } else { + CoreLoop::vector_index_key(database_id, tenant_id, &collection, "") + }; + if !self.redo_vector_prelude( + RedoVectorWrite { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim: 0, + surrogates: &[surrogate], + ids: &[], + sidecars: false, + }, + provenance.as_ref(), + record_lsn, + ) { + return false; + } + let vshard = crate::types::VShardId::from_collection_in_database( + DatabaseId::new(database_id), + &collection, + ); + let task = Self::replay_vector_task( + nodedb_types::TenantId::new(tenant_id), + DatabaseId::new(database_id), + vshard, + PhysicalPlan::Vector( + nodedb_physical::physical_plan::VectorOp::DeleteBySurrogate { + collection: nodedb_types::QualifiedCollection::from_stored( + collection.clone(), + ), + surrogate, + field_name: field_name.clone(), + provenance: provenance.clone(), + }, + ), + ); + let response = self.execute_vector_delete_by_surrogate( + &task, + tenant_id, + &collection, + surrogate, + &field_name, + provenance.as_ref(), + ); + if response.status != crate::bridge::envelope::Status::Ok { + self.replay_record_rejected( + "vector", + record_lsn, + response.error_code, + &format!("vector delete-by-surrogate on '{collection}' failed"), + ); + return false; + } + return true; + } + + let decoded = zerompk::from_msgpack::<( + String, + u32, + Option, + )>(payload) + .map(|(collection, vector_id, _prov)| (collection, vector_id)) + .or_else(|_| zerompk::from_msgpack::<(String, u32)>(payload)); + let Ok((collection, vector_id)) = decoded else { + self.replay_record_unapplied( + "vector", + "delete_decode", + record_lsn, + "VectorDelete payload matched none of its record shapes", + ); + return false; + }; + if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { + return false; + } + let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, ""); + if !self.redo_vector_prelude( + RedoVectorWrite { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim: 0, + surrogates: &[], + ids: &[vector_id], + sidecars: false, + }, + None, + record_lsn, + ) { + return false; + } + match self.vector_collections.get_mut(&index_key) { + Some(index) => { + index.delete(vector_id); + true + } + None => false, + } + } +} diff --git a/nodedb/src/data/executor/wal_replay_vector_direct.rs b/nodedb/src/data/executor/wal_replay_vector_direct.rs index a1f07167f..48499d36a 100644 --- a/nodedb/src/data/executor/wal_replay_vector_direct.rs +++ b/nodedb/src/data/executor/wal_replay_vector_direct.rs @@ -15,6 +15,7 @@ use crate::control::server::wal_dispatch::{ VectorDirectDeleteRecord, VectorDirectTruncateRecord, VectorDirectUpdateRecord, }; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::wal_replay_vector_redo::RedoVectorTargets; use crate::types::DatabaseId; impl CoreLoop { @@ -33,15 +34,35 @@ impl CoreLoop { let Ok((collection, field, targets)) = zerompk::from_msgpack::(payload) else { + self.replay_record_unapplied( + "vector", + "direct_delete_decode", + record_lsn, + "VectorDirectDelete payload does not decode", + ); return false; }; if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { + return false; + } + if !self.redo_vector_targets_prelude( + RedoVectorTargets { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim: 0, + targets: &targets, + }, + record_lsn, + ) { return false; } let vshard = crate::types::VShardId::from_collection_in_database( @@ -74,11 +95,11 @@ impl CoreLoop { }, ); if response.status != Status::Ok { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - "WAL replay: direct-delete handler returned error; skipping" + self.replay_record_rejected( + "vector", + record_lsn, + response.error_code, + &format!("vector direct delete on '{collection}' failed"), ); return false; } @@ -102,17 +123,35 @@ impl CoreLoop { let tombstones = tombstones.for_database(database_id); let Ok((collection, field)) = zerompk::from_msgpack::(payload) else { + self.replay_record_unapplied( + "vector", + "direct_truncate_decode", + record_lsn, + "VectorDirectTruncate payload does not decode", + ); return false; }; if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { + return false; + } + if self.claim_for_validation() { return false; } + if self.recording_redo_undo() { + let captured = + self.detach_vector_collection_for_truncate(&index_key, tenant_id, &collection); + if !self.record_redo_capture(captured.map(std::iter::once)) { + return false; + } + } let vshard = crate::types::VShardId::from_collection_in_database( DatabaseId::new(database_id), &collection, @@ -129,11 +168,11 @@ impl CoreLoop { ); let response = self.execute_vector_direct_truncate(&task, tenant_id, &collection, &field); if response.status != Status::Ok { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - "WAL replay: direct-truncate handler returned error; skipping" + self.replay_record_rejected( + "vector", + record_lsn, + response.error_code, + &format!("vector direct truncate on '{collection}' failed"), ); return false; } @@ -165,15 +204,35 @@ impl CoreLoop { payload_indexes, )) = zerompk::from_msgpack::(payload) else { + self.replay_record_unapplied( + "vector", + "direct_update_decode", + record_lsn, + "VectorDirectUpdate payload does not decode", + ); return false; }; if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { + return false; + } + if !self.redo_vector_targets_prelude( + RedoVectorTargets { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim: new_vector.as_ref().map_or(0, Vec::len), + targets: &targets, + }, + record_lsn, + ) { return false; } let vshard = crate::types::VShardId::from_collection_in_database( @@ -216,11 +275,11 @@ impl CoreLoop { }, ); if response.status != Status::Ok { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - "WAL replay: direct-update handler returned error; skipping" + self.replay_record_rejected( + "vector", + record_lsn, + response.error_code, + &format!("vector direct update on '{collection}' failed"), ); return false; } diff --git a/nodedb/src/data/executor/wal_replay_vector_extended.rs b/nodedb/src/data/executor/wal_replay_vector_extended.rs index 1964a316d..78dccdd7c 100644 --- a/nodedb/src/data/executor/wal_replay_vector_extended.rs +++ b/nodedb/src/data/executor/wal_replay_vector_extended.rs @@ -29,6 +29,7 @@ use nodedb_wal::record::RecordType; use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::server::wal_dispatch::VectorDirectUpsertRecord; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::wal_replay_vector_redo::RedoVectorWrite; use crate::types::DatabaseId; impl CoreLoop { @@ -189,6 +190,12 @@ impl CoreLoop { on_conflict_updates, )) = zerompk::from_msgpack::(payload) else { + self.replay_record_unapplied( + "vector", + "direct_upsert_decode", + record_lsn, + "VectorDirectUpsert payload does not decode", + ); return false; }; if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { @@ -197,12 +204,29 @@ impl CoreLoop { let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); // Watermark gate: a restored checkpoint already holds every write at or // below its watermark; re-applying would append a duplicate HNSW node. - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { return false; } let surrogate = nodedb_types::Surrogate::new(surrogate_u32); + if !self.redo_vector_prelude( + RedoVectorWrite { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim: vector.len(), + surrogates: &[surrogate], + ids: &[], + sidecars: true, + }, + None, + record_lsn, + ) { + return false; + } let vshard = crate::types::VShardId::from_collection_in_database( DatabaseId::new(database_id), &collection, @@ -250,11 +274,11 @@ impl CoreLoop { }, ); if response.status != Status::Ok { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - "WAL replay: direct-upsert handler returned error; skipping" + self.replay_record_rejected( + "vector", + record_lsn, + response.error_code, + &format!("vector direct upsert into '{collection}' failed"), ); return false; } @@ -280,6 +304,12 @@ impl CoreLoop { let Ok((collection, field_name, doc_surrogate_u32, vectors_flat, count, dim)) = zerompk::from_msgpack::<(String, String, u32, Vec, usize, usize)>(payload) else { + self.replay_record_unapplied( + "vector", + "multi_vector_put_decode", + record_lsn, + "MultiVectorPut payload does not decode", + ); return false; }; if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { @@ -287,12 +317,29 @@ impl CoreLoop { } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field_name); - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { return false; } let document_surrogate = nodedb_types::Surrogate::new(doc_surrogate_u32); + if !self.redo_vector_prelude( + RedoVectorWrite { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim, + surrogates: &[document_surrogate], + ids: &[], + sidecars: false, + }, + None, + record_lsn, + ) { + return false; + } let vshard = crate::types::VShardId::from_collection_in_database( DatabaseId::new(database_id), &collection, @@ -323,11 +370,11 @@ impl CoreLoop { }, ); if response.status != Status::Ok { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - "WAL replay: multi-vector insert handler returned error; skipping" + self.replay_record_rejected( + "vector", + record_lsn, + response.error_code, + &format!("multi-vector insert into '{collection}' failed"), ); return false; } @@ -352,6 +399,12 @@ impl CoreLoop { let Ok((collection, field_name, doc_surrogate_u32)) = zerompk::from_msgpack::<(String, String, u32)>(payload) else { + self.replay_record_unapplied( + "vector", + "multi_vector_delete_decode", + record_lsn, + "MultiVectorDelete payload does not decode", + ); return false; }; if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { @@ -359,12 +412,29 @@ impl CoreLoop { } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field_name); - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { return false; } let document_surrogate = nodedb_types::Surrogate::new(doc_surrogate_u32); + if !self.redo_vector_prelude( + RedoVectorWrite { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim: 0, + surrogates: &[document_surrogate], + ids: &[], + sidecars: false, + }, + None, + record_lsn, + ) { + return false; + } let vshard = crate::types::VShardId::from_collection_in_database( DatabaseId::new(database_id), &collection, @@ -394,90 +464,6 @@ impl CoreLoop { } true } - - /// Replay one `SparseVectorPut` record. Idempotent upsert-by-`doc_id`, so - /// no watermark gate is required. - fn replay_sparse_put( - &mut self, - payload: &[u8], - tenant_id: u64, - database_id: u64, - record_lsn: u64, - tombstones: &nodedb_wal::TombstoneSet, - ) -> bool { - let tombstones = tombstones.for_database(database_id); - let Ok((collection, field_name, doc_id, entries)) = - zerompk::from_msgpack::<(String, String, String, Vec<(u32, f32)>)>(payload) - else { - return false; - }; - if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { - return false; - } - let vshard = crate::types::VShardId::from_collection_in_database( - DatabaseId::new(database_id), - &collection, - ); - let task = Self::replay_vector_task( - nodedb_types::TenantId::new(tenant_id), - DatabaseId::new(database_id), - vshard, - PhysicalPlan::Vector(VectorOp::SparseInsert { - collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), - field_name: field_name.clone(), - doc_id: doc_id.clone(), - entries: entries.clone(), - }), - ); - let response = self.execute_sparse_insert( - &task, - tenant_id, - &collection, - &field_name, - &doc_id, - &entries, - ); - response.status == Status::Ok - } - - /// Replay one `SparseVectorDelete` record. Idempotent (an absent document - /// is a no-op), so no watermark gate is required. - fn replay_sparse_delete( - &mut self, - payload: &[u8], - tenant_id: u64, - database_id: u64, - record_lsn: u64, - tombstones: &nodedb_wal::TombstoneSet, - ) -> bool { - let tombstones = tombstones.for_database(database_id); - let Ok((collection, field_name, doc_id)) = - zerompk::from_msgpack::<(String, String, String)>(payload) - else { - return false; - }; - if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { - return false; - } - let vshard = crate::types::VShardId::from_collection_in_database( - DatabaseId::new(database_id), - &collection, - ); - let task = Self::replay_vector_task( - nodedb_types::TenantId::new(tenant_id), - DatabaseId::new(database_id), - vshard, - PhysicalPlan::Vector(VectorOp::SparseDelete { - collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), - field_name: field_name.clone(), - doc_id: doc_id.clone(), - }), - ); - // An absent document yields NotFound; that is an expected idempotent - // no-op on replay, not a failure. - let _ = self.execute_sparse_delete(&task, tenant_id, &collection, &field_name, &doc_id); - true - } } #[cfg(test)] diff --git a/nodedb/src/data/executor/wal_replay_vector_redo.rs b/nodedb/src/data/executor/wal_replay_vector_redo.rs new file mode 100644 index 000000000..69b966cb4 --- /dev/null +++ b/nodedb/src/data/executor/wal_replay_vector_redo.rs @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The committed-redo step every vector replay arm runs before its write. +//! +//! In the validate pass the arm refuses a write whose vectors do not fit the +//! existing index, claims the sub-record, and writes nothing. In the install +//! pass it records the write's undo (`capture_vector_write_undo`) and the +//! sync high-water mark its provenance advances, then writes. Restart replay +//! writes with neither. + +use nodedb_physical::physical_plan::VectorWriteTargets; +use nodedb_types::Surrogate; +use nodedb_types::sync::wire::SyncProvenance; + +use super::core_loop::CoreLoop; +use super::handlers::transaction::undo::vector_write::VectorWriteTarget; +use super::handlers::vector_direct_row::VectorIndexKey; + +/// One vector write a replayed record carries. +pub(in crate::data::executor) struct RedoVectorWrite<'a> { + pub index_key: &'a VectorIndexKey, + pub tid: u64, + pub collection: &'a str, + /// The dimension of the write's vectors, `0` when it carries none. + pub dim: usize, + pub surrogates: &'a [Surrogate], + pub ids: &'a [u32], + /// Whether the write stores sidecar rows (a vector-primary write). + pub sidecars: bool, +} + +impl CoreLoop { + /// Run the committed-redo step before a vector write. Returns whether the + /// arm writes: `false` in the validate pass, and when the install could + /// not read the write's pre-image (the error is kept on the apply). + pub(in crate::data::executor) fn redo_vector_prelude( + &mut self, + write: RedoVectorWrite<'_>, + provenance: Option<&SyncProvenance>, + record_lsn: u64, + ) -> bool { + if !self.applying_committed_redo() { + return true; + } + if write.dim != 0 + && let Some(existing) = self.vector_collections.get(write.index_key) + && existing.dim() != write.dim + { + let index_dim = existing.dim(); + self.replay_record_unapplied( + "vector", + "dim", + record_lsn, + &format!( + "record for '{}' has dim {}, the index has dim {index_dim}", + write.collection, write.dim + ), + ); + return false; + } + if self.claim_for_validation() { + return false; + } + let captured = self + .capture_vector_write_undo(VectorWriteTarget { + index_key: write.index_key, + tid: write.tid, + collection: write.collection, + surrogates: write.surrogates, + ids: write.ids, + sidecars: write.sidecars, + }) + .map(|undo| std::iter::once(undo).chain(self.capture_sync_hwm_undo(provenance))); + self.record_redo_capture(captured) + } + + /// [`Self::redo_vector_prelude`] for a vector-primary write that names + /// its rows by `targets`. The rows are resolved the way the handler + /// resolves them. A resolution error is kept on the apply. + pub(in crate::data::executor) fn redo_vector_targets_prelude( + &mut self, + write: RedoVectorTargets<'_>, + record_lsn: u64, + ) -> bool { + if !self.applying_committed_redo() { + return true; + } + let surrogates = match self.resolve_vector_direct_targets( + write.index_key.0.as_u64(), + write.tid, + write.collection, + write.targets, + ) { + Ok(surrogates) => surrogates, + Err(error) => { + self.replay_record_unapplied( + "vector", + "resolve_targets", + record_lsn, + &format!("rows of '{}' do not resolve: {error:?}", write.collection), + ); + return false; + } + }; + self.redo_vector_prelude( + RedoVectorWrite { + index_key: write.index_key, + tid: write.tid, + collection: write.collection, + dim: write.dim, + surrogates: &surrogates, + ids: &[], + sidecars: true, + }, + None, + record_lsn, + ) + } +} + +/// A vector-primary write that names its rows by [`VectorWriteTargets`]. +pub(in crate::data::executor) struct RedoVectorTargets<'a> { + pub index_key: &'a VectorIndexKey, + pub tid: u64, + pub collection: &'a str, + /// The dimension of the write's vector, `0` when it carries none. + pub dim: usize, + pub targets: &'a VectorWriteTargets, +} diff --git a/nodedb/src/data/executor/wal_replay_vector_resolved.rs b/nodedb/src/data/executor/wal_replay_vector_resolved.rs index 5117214d2..ad124d0cc 100644 --- a/nodedb/src/data/executor/wal_replay_vector_resolved.rs +++ b/nodedb/src/data/executor/wal_replay_vector_resolved.rs @@ -16,6 +16,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::server::wal_dispatch::VectorResolvedDirectWriteRecord; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::vector_direct_resolve::VectorResolvedIndexSpec; +use crate::data::executor::wal_replay_vector_redo::RedoVectorWrite; use crate::types::DatabaseId; impl CoreLoop { @@ -32,15 +33,46 @@ impl CoreLoop { let Ok((collection, field, quantization, storage_dtype, payload_indexes, mutations)) = zerompk::from_msgpack::(payload) else { + self.replay_record_unapplied( + "vector", + "resolved_direct_write_decode", + record_lsn, + "VectorResolvedDirectWrite payload does not decode", + ); return false; }; if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - if let Some(existing) = self.vector_collections.get(&index_key) - && record_lsn <= existing.checkpoint_wal_lsn() - { + if self.replay_watermark_skips( + self.vector_collections + .get(&index_key) + .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), + ) { + return false; + } + let surrogates: Vec = mutations + .iter() + .map(|mutation| mutation.surrogate()) + .collect(); + let dim = mutations + .iter() + .find_map(|mutation| mutation.stored_vector()) + .map_or(0, <[f32]>::len); + if !self.redo_vector_prelude( + RedoVectorWrite { + index_key: &index_key, + tid: tenant_id, + collection: &collection, + dim, + surrogates: &surrogates, + ids: &[], + sidecars: true, + }, + None, + record_lsn, + ) { return false; } let vshard = crate::types::VShardId::from_collection_in_database( @@ -79,12 +111,11 @@ impl CoreLoop { ) { Ok(touched) => touched, Err(e) => { - tracing::warn!( - core = self.core_id, - %collection, - lsn = record_lsn, - error = ?e, - "WAL replay: resolved direct write apply returned error; skipping" + self.replay_record_rejected( + "vector", + record_lsn, + Some(Box::new(e)), + &format!("vector resolved direct write on '{collection}' failed"), ); return false; } diff --git a/nodedb/src/data/executor/wal_replay_vector_sparse.rs b/nodedb/src/data/executor/wal_replay_vector_sparse.rs new file mode 100644 index 000000000..435f73a56 --- /dev/null +++ b/nodedb/src/data/executor/wal_replay_vector_sparse.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! WAL replay for sparse-vector inserts and deletes. Both upsert or remove by +//! `doc_id`, so full re-application over a restored checkpoint reproduces the +//! same state and no watermark gates them. + +use nodedb_physical::physical_plan::VectorOp; + +use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::types::DatabaseId; + +impl CoreLoop { + /// In the install pass of a committed-redo apply, record the prior state + /// of `doc_id` in the sparse index of `field` before a write replaces it. + fn record_sparse_doc_undo( + &mut self, + database_id: u64, + tenant_id: u64, + (collection, field): (&str, &str), + doc_id: &str, + ) { + if !self.recording_redo_undo() { + return; + } + let key = Self::sparse_index_key(database_id, tenant_id, collection, field); + let (prior, next_id) = match self.sparse_vector_indexes.get(&key) { + Some(index) => (index.doc_image(doc_id), Some(index.next_internal_id())), + None => (None, None), + }; + self.record_redo_undo([UndoEntry::SparseDoc { + key, + doc_id: doc_id.to_string(), + prior, + next_id, + }]); + } + + /// Replay one `SparseVectorPut` record. Idempotent upsert-by-`doc_id`, so + /// no watermark gate is required. + pub(in crate::data::executor) fn replay_sparse_put( + &mut self, + payload: &[u8], + tenant_id: u64, + database_id: u64, + record_lsn: u64, + tombstones: &nodedb_wal::TombstoneSet, + ) -> bool { + let tombstones = tombstones.for_database(database_id); + let Ok((collection, field_name, doc_id, entries)) = + zerompk::from_msgpack::<(String, String, String, Vec<(u32, f32)>)>(payload) + else { + self.replay_record_unapplied( + "vector", + "sparse_put_decode", + record_lsn, + "SparseVectorPut payload does not decode", + ); + return false; + }; + if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { + return false; + } + if self.applying_committed_redo() + && let Err(e) = nodedb_types::SparseVector::from_entries(entries.clone()) + { + self.replay_record_unapplied( + "vector", + "sparse_entries", + record_lsn, + &format!("sparse vector for '{collection}' is invalid: {e}"), + ); + return false; + } + if self.claim_for_validation() { + return false; + } + self.record_sparse_doc_undo(database_id, tenant_id, (&collection, &field_name), &doc_id); + let vshard = crate::types::VShardId::from_collection_in_database( + DatabaseId::new(database_id), + &collection, + ); + let task = Self::replay_vector_task( + nodedb_types::TenantId::new(tenant_id), + DatabaseId::new(database_id), + vshard, + PhysicalPlan::Vector(VectorOp::SparseInsert { + collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), + field_name: field_name.clone(), + doc_id: doc_id.clone(), + entries: entries.clone(), + }), + ); + let response = self.execute_sparse_insert( + &task, + tenant_id, + &collection, + &field_name, + &doc_id, + &entries, + ); + if response.status != Status::Ok { + self.replay_record_rejected( + "vector", + record_lsn, + response.error_code, + &format!("sparse vector insert into '{collection}' failed"), + ); + return false; + } + true + } + + /// Replay one `SparseVectorDelete` record. Idempotent (an absent document + /// is a no-op), so no watermark gate is required. + pub(in crate::data::executor) fn replay_sparse_delete( + &mut self, + payload: &[u8], + tenant_id: u64, + database_id: u64, + record_lsn: u64, + tombstones: &nodedb_wal::TombstoneSet, + ) -> bool { + let tombstones = tombstones.for_database(database_id); + let Ok((collection, field_name, doc_id)) = + zerompk::from_msgpack::<(String, String, String)>(payload) + else { + self.replay_record_unapplied( + "vector", + "sparse_delete_decode", + record_lsn, + "SparseVectorDelete payload does not decode", + ); + return false; + }; + if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { + return false; + } + if self.claim_for_validation() { + return false; + } + self.record_sparse_doc_undo(database_id, tenant_id, (&collection, &field_name), &doc_id); + let vshard = crate::types::VShardId::from_collection_in_database( + DatabaseId::new(database_id), + &collection, + ); + let task = Self::replay_vector_task( + nodedb_types::TenantId::new(tenant_id), + DatabaseId::new(database_id), + vshard, + PhysicalPlan::Vector(VectorOp::SparseDelete { + collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), + field_name: field_name.clone(), + doc_id: doc_id.clone(), + }), + ); + // An absent document yields NotFound; that is an expected idempotent + // no-op on replay, not a failure. + let _ = self.execute_sparse_delete(&task, tenant_id, &collection, &field_name, &doc_id); + true + } +} diff --git a/nodedb/src/data/executor/wal_replay_vector_task.rs b/nodedb/src/data/executor/wal_replay_vector_task.rs new file mode 100644 index 000000000..af9ecbeb4 --- /dev/null +++ b/nodedb/src/data/executor/wal_replay_vector_task.rs @@ -0,0 +1,52 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The synthetic task the vector replay arms hand the live handlers. + +use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; +use crate::data::executor::task::{ExecutionTask, TaskState}; +use crate::types::{DatabaseId, ReadConsistency}; + +use super::core_loop::CoreLoop; + +impl CoreLoop { + /// Build a synthetic `ExecutionTask` for WAL replay. + /// + /// Mirrors `CoreLoop::replay_task` (`replay_task.rs`). The task carries + /// no meaningful request semantics — it is only needed so that the handler + /// methods can return a typed `Response`. + pub(in crate::data::executor) fn replay_vector_task( + tenant_id: crate::types::TenantId, + database_id: DatabaseId, + vshard_id: crate::types::VShardId, + plan: PhysicalPlan, + ) -> ExecutionTask { + ExecutionTask { + request: Request { + request_id: crate::types::RequestId::new(0), + tenant_id, + database_id, + vshard_id, + plan, + deadline: std::time::Instant::now() + + crate::data::executor::deadline::REPLAY_DEADLINE, + priority: Priority::Normal, + trace_id: crate::types::TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::AlreadyOrdered, + ), + }, + state: TaskState::Running, + wal_lsn: None, + resolved_now_ms: None, + } + } +} diff --git a/nodedb/src/data/runtime/spawn.rs b/nodedb/src/data/runtime/spawn.rs index 33024afb8..ca27e89b7 100644 --- a/nodedb/src/data/runtime/spawn.rs +++ b/nodedb/src/data/runtime/spawn.rs @@ -92,6 +92,9 @@ pub fn spawn_core( // (Duration is Copy). let checkpoint_interval = compaction_config.checkpoint_interval; + // 2b. The committed-redo apply routes records by `vshard % num_cores`. + core.set_num_cores(num_cores); + // 2c. Apply compaction config. core.set_compaction_config( compaction_config.interval, diff --git a/nodedb/src/engine/array/memtable/tile_buffer.rs b/nodedb/src/engine/array/memtable/tile_buffer.rs index 03395488e..3647ce793 100644 --- a/nodedb/src/engine/array/memtable/tile_buffer.rs +++ b/nodedb/src/engine/array/memtable/tile_buffer.rs @@ -263,6 +263,23 @@ impl Memtable { Ok(tile) } + /// The buffer of `tile`, if the memtable holds one. + pub fn tile(&self, tile: &TileId) -> Option<&TileBuffer> { + self.tiles.get(tile) + } + + /// Put `tile` back to `buffer`, or drop it when `buffer` is `None`. + pub fn restore_tile(&mut self, tile: TileId, buffer: Option) { + match buffer { + Some(buffer) => { + self.tiles.insert(tile, buffer); + } + None => { + self.tiles.remove(&tile); + } + } + } + pub fn stats(&self) -> MemtableStats { let mut s = MemtableStats::default(); for b in self.tiles.values() { diff --git a/nodedb/src/engine/array/mod.rs b/nodedb/src/engine/array/mod.rs index 441f7a754..64ab7157d 100644 --- a/nodedb/src/engine/array/mod.rs +++ b/nodedb/src/engine/array/mod.rs @@ -16,6 +16,7 @@ pub mod memtable; pub mod purge; pub mod read; pub mod recovery; +pub mod rollback; pub mod store; #[cfg(test)] mod test_support; @@ -23,4 +24,5 @@ pub mod wal; pub mod write; pub use engine::{ArrayEngine, ArrayEngineConfig}; +pub use rollback::ArrayTileSnapshot; pub use wal::{ArrayDeletePayload, ArrayFlushPayload, ArrayPutPayload}; diff --git a/nodedb/src/engine/array/rollback.rs b/nodedb/src/engine/array/rollback.rs new file mode 100644 index 000000000..0995ca173 --- /dev/null +++ b/nodedb/src/engine/array/rollback.rs @@ -0,0 +1,92 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Withdraw a cell write from an array's memtable. +//! +//! Every cell write lands in the memtable tile its coordinate and system +//! time map to. [`ArrayEngine::snapshot_tiles`] copies those tiles before the +//! write, and [`ArrayEngine::restore_tiles`] puts them back. The write must +//! not flush in between: [`ArrayEngine::put_cells_unflushed`] and +//! [`ArrayEngine::delete_cells_unflushed`] stamp the memtable without the +//! threshold flush, and [`ArrayEngine::flush_if_full`] runs it afterwards. + +use nodedb_array::tile::tile_id_for_cell; +use nodedb_array::types::coord::value::CoordValue; +use nodedb_array::types::{ArrayId, TileId}; + +use super::engine::{ArrayEngine, ArrayEngineResult}; +use super::memtable::TileBuffer; +use super::wal::{ArrayDeleteCell, ArrayPutCell}; +use super::write::{stamp_delete_cells, stamp_put_cells}; + +/// The memtable tiles a write touches, as they were before it. +#[derive(Debug)] +pub struct ArrayTileSnapshot { + tiles: Vec<(TileId, Option)>, +} + +impl ArrayEngine { + /// Copy the memtable tiles the cells at `cells` (coordinate and system + /// time) map to. + pub fn snapshot_tiles<'a>( + &self, + id: &ArrayId, + cells: impl IntoIterator, + ) -> ArrayEngineResult { + let store = self.store(id)?; + let schema = store.schema().clone(); + let mut tiles: Vec<(TileId, Option)> = Vec::new(); + for (coord, system_from_ms) in cells { + let tile = tile_id_for_cell(&schema, coord, system_from_ms)?; + if tiles.iter().any(|(seen, _)| *seen == tile) { + continue; + } + tiles.push((tile, store.memtable.tile(&tile).cloned())); + } + Ok(ArrayTileSnapshot { tiles }) + } + + /// Put every tile of `snapshot` back. + pub fn restore_tiles( + &mut self, + id: &ArrayId, + snapshot: ArrayTileSnapshot, + ) -> ArrayEngineResult<()> { + let store = self.store_mut(id)?; + for (tile, buffer) in snapshot.tiles { + store.memtable.restore_tile(tile, buffer); + } + Ok(()) + } + + /// Stamp `cells` into the memtable without the threshold flush. + pub fn put_cells_unflushed( + &mut self, + id: &ArrayId, + cells: Vec, + wal_lsn: u64, + ) -> ArrayEngineResult<()> { + if cells.is_empty() { + return Ok(()); + } + stamp_put_cells(self.store_mut(id)?, cells, wal_lsn) + } + + /// Stamp tombstones for `cells` into the memtable without the threshold + /// flush. + pub fn delete_cells_unflushed( + &mut self, + id: &ArrayId, + cells: Vec, + wal_lsn: u64, + ) -> ArrayEngineResult<()> { + if cells.is_empty() { + return Ok(()); + } + stamp_delete_cells(self.store_mut(id)?, cells, wal_lsn) + } + + /// Run the threshold flush the unflushed writes skipped. + pub fn flush_if_full(&mut self, id: &ArrayId) -> ArrayEngineResult<()> { + self.maybe_flush(id) + } +} diff --git a/nodedb/src/engine/kv/engine/mod.rs b/nodedb/src/engine/kv/engine/mod.rs index 75894c79d..323a6eb94 100644 --- a/nodedb/src/engine/kv/engine/mod.rs +++ b/nodedb/src/engine/kv/engine/mod.rs @@ -17,4 +17,5 @@ mod write_epoch; pub use checkpoint_export::KvCollectionRef; pub use checkpoint_restore::{RestoreCompositeIndexParams, RestoreFieldIndexParams}; +pub use reads::{KvEntryImage, KvKeyRef}; pub use state::{KvEngine, ScanResult}; diff --git a/nodedb/src/engine/kv/engine/reads.rs b/nodedb/src/engine/kv/engine/reads.rs index 75df25015..97e131f7b 100644 --- a/nodedb/src/engine/kv/engine/reads.rs +++ b/nodedb/src/engine/kv/engine/reads.rs @@ -3,8 +3,30 @@ use nodedb_types::Surrogate; use super::KvEngine; +use crate::engine::kv::KvPutParams; use crate::engine::kv::engine_helpers::table_key; -use crate::engine::kv::hash_table::EntryMeta; +use crate::engine::kv::entry::NO_EXPIRY; +use crate::engine::kv::hash_table::{EntryMeta, KvExportEntry}; + +/// One key of one KV collection. +#[derive(Debug, Clone, Copy)] +pub struct KvKeyRef<'a> { + pub database_id: u64, + pub tenant_id: u64, + pub collection: &'a str, + pub key: &'a [u8], +} + +/// A key's complete stored state: its value, its absolute expiry, and the +/// surrogate bound to it. A rollback reinstalls it exactly, so the key keeps +/// its TTL instant and its identity. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct KvEntryImage { + pub value: Vec, + /// Absolute expiry instant in ms since epoch, [`NO_EXPIRY`] for none. + pub expire_at_ms: u64, + pub surrogate: Surrogate, +} impl KvEngine { /// Look up the user primary key bytes for a given surrogate within @@ -105,6 +127,84 @@ impl KvEngine { self.tables.get(&tkey)?.get_entry_meta(key) } + /// The complete stored state of `key`, or `None` when it is absent or + /// expired. + pub fn entry_image( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + key: &[u8], + now_ms: u64, + ) -> Option { + let tkey = table_key(database_id, tenant_id, collection); + let table = self.tables.get(&tkey)?; + let (value, surrogate) = table.get_with_surrogate(key, now_ms)?; + let expire_at_ms = table + .get_entry_meta(key) + .map_or(NO_EXPIRY, |meta| meta.expire_at_ms); + Some(KvEntryImage { + value: value.to_vec(), + expire_at_ms, + surrogate, + }) + } + + /// Reinstall `image` under `key`: the value, the absolute expiry instant + /// and the surrogate it held, with every index maintained. + pub fn restore_entry_image(&mut self, target: KvKeyRef<'_>, image: &KvEntryImage, now_ms: u64) { + self.put_with_absolute_expiry( + KvPutParams { + database_id: target.database_id, + tenant_id: target.tenant_id, + collection: target.collection, + key: target.key, + value: &image.value, + ttl_ms: 0, + now_ms, + surrogate: image.surrogate, + }, + image.expire_at_ms, + ); + } + + /// Put `key` back to `image`, or remove it when `image` is `None`: the + /// key was absent before. + pub fn reinstate_entry( + &mut self, + target: KvKeyRef<'_>, + image: Option<&KvEntryImage>, + now_ms: u64, + ) { + match image { + Some(image) => self.restore_entry_image(target, image, now_ms), + None => { + let key = [target.key.to_vec()]; + self.delete( + target.database_id, + target.tenant_id, + target.collection, + &key, + now_ms, + ); + } + } + } + + /// Every live row of a collection with its complete stored state. + pub fn export_collection( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + ) -> Vec { + let tkey = table_key(database_id, tenant_id, collection); + self.tables + .get(&tkey) + .map(|table| table.export_entries_with_surrogates()) + .unwrap_or_default() + } + /// BATCH GET: fetch multiple keys. Returns values in order (None for missing). pub fn batch_get( &self, diff --git a/nodedb/src/engine/kv/engine_atomic.rs b/nodedb/src/engine/kv/engine_atomic.rs index a16c3be0f..becbb6164 100644 --- a/nodedb/src/engine/kv/engine_atomic.rs +++ b/nodedb/src/engine/kv/engine_atomic.rs @@ -14,13 +14,30 @@ use super::hash_table::KvHashTable; /// Result of a compare-and-swap operation. pub struct CasResult { - /// Whether the swap succeeded (current == expected). - pub success: bool, + /// The bytes the swap stored. `None` when the compare failed and nothing + /// was written. + pub written: Option>, /// The value that was present at the time of the CAS. /// `None` if the key did not exist. pub current_value: Option>, } +impl CasResult { + /// Whether the swap succeeded (current == expected). + pub fn success(&self) -> bool { + self.written.is_some() + } +} + +/// Result of a get-and-set operation. +pub struct GetSetResult { + /// The value that was present before the write. `None` if the key did + /// not exist. + pub old: Option>, + /// The bytes the write stored. + pub written: Vec, +} + /// Errors specific to atomic KV operations. #[derive(Debug)] pub enum AtomicError { @@ -30,20 +47,20 @@ pub enum AtomicError { Overflow, /// The computed new value failed to re-encode as MessagePack. Encode { detail: String }, - /// The [`IncrAdmission`] gate refused the computed post-image, so nothing + /// The [`AtomicAdmission`] gate refused the computed post-image, so nothing /// was written. Boxed to keep the error small on the success path. Rejected(Box), } -/// A gate consulted with the computed post-image before an increment commits. +/// A gate consulted with the computed post-image before an atomic commits. /// -/// INCR computes the value it stores from the stored one, so the row a -/// row-level-security write policy has to decide does not exist until the -/// arithmetic has run — and the arithmetic runs here, inside the engine, in the -/// same pass that persists the result. Passing the decision in is what keeps -/// that arithmetic in one place: pre-computing the increment at the call site -/// just to check it would leave two copies of it to drift apart. -pub type IncrAdmission<'a> = &'a dyn Fn(&[u8]) -> crate::Result<()>; +/// Every atomic computes the value it stores from the stored one: INCR runs +/// arithmetic, and CAS and GETSET swap one column of a typed row. The row a +/// row-level-security write policy has to decide does not exist until that +/// computation has run, and it runs here, inside the engine, in the same pass +/// that persists the result. Passing the decision in keeps the computation in +/// one place. +pub type AtomicAdmission<'a> = &'a dyn Fn(&[u8]) -> crate::Result<()>; /// An admission that accepts every image. /// @@ -88,7 +105,7 @@ impl KvEngine { ctx: AtomicKeyCtx<'_>, delta: i64, ttl_ms: u64, - admit: IncrAdmission<'_>, + admit: AtomicAdmission<'_>, ) -> Result { self.incr_resolved(ctx, delta, ttl_ms, None, admit) } @@ -111,7 +128,7 @@ impl KvEngine { delta: i64, ttl_ms: u64, expire_at_ms: u64, - admit: IncrAdmission<'_>, + admit: AtomicAdmission<'_>, ) -> Result { self.incr_resolved(ctx, delta, ttl_ms, Some(expire_at_ms), admit) } @@ -125,7 +142,7 @@ impl KvEngine { delta: i64, ttl_ms: u64, expire_override: Option, - admit: IncrAdmission<'_>, + admit: AtomicAdmission<'_>, ) -> Result { let tkey = table_key(ctx.database_id, ctx.tenant_id, ctx.collection); let table = self.ensure_table(tkey, ctx.tenant_id, ctx.collection); @@ -159,7 +176,7 @@ impl KvEngine { &mut self, ctx: AtomicKeyCtx<'_>, delta: f64, - admit: IncrAdmission<'_>, + admit: AtomicAdmission<'_>, ) -> Result { let tkey = table_key(ctx.database_id, ctx.tenant_id, ctx.collection); let table = self.ensure_table(tkey, ctx.tenant_id, ctx.collection); @@ -179,41 +196,61 @@ impl KvEngine { /// If current value equals `expected`, sets to `new_value` and returns success. /// If current value differs, returns the actual current value. /// If key doesn't exist and `expected` is empty, creates the key (create-if-not-exists). - pub fn cas(&mut self, ctx: AtomicKeyCtx<'_>, expected: &[u8], new_value: &[u8]) -> CasResult { + /// If `admit` refuses the bytes the swap would store: returns `Rejected` + /// and writes nothing. + pub fn cas( + &mut self, + ctx: AtomicKeyCtx<'_>, + expected: &[u8], + new_value: &[u8], + admit: AtomicAdmission<'_>, + ) -> Result { let tkey = table_key(ctx.database_id, ctx.tenant_id, ctx.collection); let table = self.ensure_table(tkey, ctx.tenant_id, ctx.collection); let current = table.get(ctx.key, ctx.now_ms).map(|v| v.to_vec()); - let (matches, write_bytes) = compute::cas(current.as_deref(), expected, new_value); - - if matches { - self.atomic_put(ctx, tkey, &write_bytes, 0, current.is_none(), None); - CasResult { - success: true, + let (matches, write_bytes) = compute::cas(current.as_deref(), expected, new_value)?; + if !matches { + return Ok(CasResult { + written: None, current_value: current, - } - } else { - CasResult { - success: false, - current_value: current, - } + }); } + // Decided before the value is installed — see `incr_resolved`. + admit(&write_bytes).map_err(|error| AtomicError::Rejected(Box::new(error)))?; + self.atomic_put(ctx, tkey, &write_bytes, 0, current.is_none(), None); + Ok(CasResult { + written: Some(write_bytes), + current_value: current, + }) } /// Atomic get-and-set: sets new value, returns old value. /// - /// If key didn't exist, returns `None`. + /// If key didn't exist, `old` is `None`. /// Preserves existing TTL. - pub fn getset(&mut self, ctx: AtomicKeyCtx<'_>, new_value: &[u8]) -> Option> { + /// If `admit` refuses the bytes the write would store: returns `Rejected` + /// and writes nothing. + pub fn getset( + &mut self, + ctx: AtomicKeyCtx<'_>, + new_value: &[u8], + admit: AtomicAdmission<'_>, + ) -> Result { let tkey = table_key(ctx.database_id, ctx.tenant_id, ctx.collection); let table = self.ensure_table(tkey, ctx.tenant_id, ctx.collection); let old = table.get(ctx.key, ctx.now_ms).map(|v| v.to_vec()); - let write_bytes = compute::getset(old.as_deref(), new_value); + let write_bytes = compute::getset(old.as_deref(), new_value)?; + // Decided before the value is installed — see `incr_resolved`. + admit(&write_bytes).map_err(|error| AtomicError::Rejected(Box::new(error)))?; // GetSet preserves existing TTL (ttl_ms = 0). self.atomic_put(ctx, tkey, &write_bytes, 0, old.is_none(), None); - old + Ok(GetSetResult { + old, + written: write_bytes, + }) } /// Ensure a hash table exists for (tenant, collection), creating if needed. @@ -585,8 +622,10 @@ mod tests { #[test] fn cas_create_if_not_exists() { let mut engine = make_engine(); - let result = engine.cas(ctx("state", b"player1"), b"", b"idle"); - assert!(result.success); + let result = engine + .cas(ctx("state", b"player1"), b"", b"idle", &admit_any) + .expect("cas"); + assert!(result.success()); assert!(result.current_value.is_none()); // Verify key was created. let val = engine.get(0, 1, "state", b"player1", 1000); @@ -606,8 +645,10 @@ mod tests { now_ms: 1000, surrogate: Surrogate::ZERO, }); - let result = engine.cas(ctx("state", b"p1"), b"idle", b"in_match"); - assert!(result.success); + let result = engine + .cas(ctx("state", b"p1"), b"idle", b"in_match", &admit_any) + .expect("cas"); + assert!(result.success()); assert_eq!(result.current_value.as_deref(), Some(b"idle".as_slice())); let val = engine.get(0, 1, "state", b"p1", 1000); assert_eq!(val.as_deref(), Some(b"in_match".as_slice())); @@ -626,8 +667,10 @@ mod tests { now_ms: 1000, surrogate: Surrogate::ZERO, }); - let result = engine.cas(ctx("state", b"p1"), b"idle", b"in_match"); - assert!(!result.success); + let result = engine + .cas(ctx("state", b"p1"), b"idle", b"in_match", &admit_any) + .expect("cas"); + assert!(!result.success()); assert_eq!( result.current_value.as_deref(), Some(b"fighting".as_slice()) @@ -640,7 +683,10 @@ mod tests { #[test] fn getset_new_key() { let mut engine = make_engine(); - let old = engine.getset(ctx("session", b"tok"), b"new-token"); + let old = engine + .getset(ctx("session", b"tok"), b"new-token", &admit_any) + .expect("getset") + .old; assert!(old.is_none()); let val = engine.get(0, 1, "session", b"tok", 1000); assert_eq!(val.as_deref(), Some(b"new-token".as_slice())); @@ -659,7 +705,10 @@ mod tests { now_ms: 1000, surrogate: Surrogate::ZERO, }); - let old = engine.getset(ctx("session", b"tok"), b"new-token"); + let old = engine + .getset(ctx("session", b"tok"), b"new-token", &admit_any) + .expect("getset") + .old; assert_eq!(old.as_deref(), Some(b"old-token".as_slice())); let val = engine.get(0, 1, "session", b"tok", 1000); assert_eq!(val.as_deref(), Some(b"new-token".as_slice())); diff --git a/nodedb/src/engine/kv/engine_atomic_compute.rs b/nodedb/src/engine/kv/engine_atomic_compute.rs index 16907010c..846d8b243 100644 --- a/nodedb/src/engine/kv/engine_atomic_compute.rs +++ b/nodedb/src/engine/kv/engine_atomic_compute.rs @@ -7,12 +7,104 @@ //! code. Split out of `engine_atomic.rs` to keep that file under the //! file-size limit. +use std::collections::HashMap; + +use nodedb_query::msgpack_scan::{KvBodyShape, row_to_kv_body}; +use nodedb_types::Value; + use super::engine_atomic::AtomicError; -/// Decode a MessagePack-encoded value as i64. +/// The field of a typed row an atomic never targets. +const KEY_FIELD: &str = "key"; + +/// Decode a map-shaped body into its typed columns. Returns `None` for a +/// body of any other shape. +fn decode_map(bytes: &[u8]) -> Option> { + match nodedb_types::value_from_msgpack(bytes) { + Ok(Value::Object(map)) => Some(map), + _ => None, + } +} + +/// Encode typed columns back into a map-shaped body. The fields are written +/// in key order, so every replica and every WAL replay stores the same +/// bytes. +fn encode_map(map: HashMap) -> Result, AtomicError> { + row_to_kv_body(&Value::Object(map), KvBodyShape::Map).map_err(|e| AtomicError::Encode { + detail: format!("typed row re-encode: {e}"), + }) +} + +/// The column an atomic reads and writes in a typed row: the first column +/// in key order that `pick` accepts, never the `key` column. /// -/// If the value is a map (typed KV entry), extracts the first numeric field. -fn decode_msgpack_i64(bytes: &[u8]) -> Result { +/// Key order is the order the row is stored in. A `HashMap` iterates in a +/// per-process random order, so choosing by iteration order lets two +/// replicas move two different columns. +fn target_field( + map: &HashMap, + pick: impl Fn(&Value) -> Option, +) -> Option<(String, T)> { + let mut chosen: Option<(&String, T)> = None; + for (name, value) in map { + if name == KEY_FIELD { + continue; + } + if chosen.as_ref().is_some_and(|(best, _)| *best <= name) { + continue; + } + if let Some(picked) = pick(value) { + chosen = Some((name, picked)); + } + } + chosen.map(|(name, picked)| (name.clone(), picked)) +} + +/// The i64 an `INCR` reads from a typed column. +fn column_i64(value: &Value) -> Option { + match value { + Value::Integer(i) => Some(*i), + Value::Float(f) => integral_f64_to_i64(*f), + _ => None, + } +} + +/// The f64 an `INCR_FLOAT` reads from a typed column. +fn column_f64(value: &Value) -> Option { + match value { + Value::Float(f) => Some(*f), + Value::Integer(i) => Some(*i as f64), + _ => None, + } +} + +/// The string a `CAS` or `GETSET` addresses in a typed column. +fn column_string(value: &Value) -> Option { + match value { + Value::String(s) => Some(s.clone()), + _ => None, + } +} + +/// A whole `f64` inside the i64 range, as an i64. +fn integral_f64_to_i64(v: f64) -> Option { + (v.fract() == 0.0 && v >= i64::MIN as f64 && v <= i64::MAX as f64).then_some(v as i64) +} + +fn not_an_integer() -> AtomicError { + AtomicError::TypeMismatch { + detail: "value is not an integer".into(), + } +} + +fn not_numeric() -> AtomicError { + AtomicError::TypeMismatch { + detail: "value is not numeric".into(), + } +} + +/// Decode a bare MessagePack scalar as i64. +fn decode_scalar_i64(bytes: &[u8]) -> Result { // Try i64 first, then u64 (MessagePack encodes small positive as u64). if let Ok(v) = zerompk::from_msgpack::(bytes) { return Ok(v); @@ -20,36 +112,15 @@ fn decode_msgpack_i64(bytes: &[u8]) -> Result { if let Ok(v) = zerompk::from_msgpack::(bytes) { return i64::try_from(v).map_err(|_| AtomicError::Overflow); } - // Try f64 → i64 truncation for values stored as float. - if let Ok(v) = zerompk::from_msgpack::(bytes) - && v.fract() == 0.0 - && v >= i64::MIN as f64 - && v <= i64::MAX as f64 - { - return Ok(v as i64); - } - // If value is a map (typed KV entry), find the first numeric field. - if let Ok(nodedb_types::Value::Object(map)) = nodedb_types::value_from_msgpack(bytes) { - for (k, v) in &map { - if k == "key" { - continue; - } - match v { - nodedb_types::Value::Integer(i) => return Ok(*i), - nodedb_types::Value::Float(f) if f.fract() == 0.0 => return Ok(*f as i64), - _ => {} - } - } - } - Err(AtomicError::TypeMismatch { - detail: "value is not an integer".into(), - }) + // A float with no fractional part truncates to i64. + zerompk::from_msgpack::(bytes) + .ok() + .and_then(integral_f64_to_i64) + .ok_or(not_an_integer()) } -/// Decode a MessagePack-encoded value as f64. -/// -/// If the value is a map (typed KV entry), extracts the first numeric field. -fn decode_msgpack_f64(bytes: &[u8]) -> Result { +/// Decode a bare MessagePack scalar as f64. +fn decode_scalar_f64(bytes: &[u8]) -> Result { if let Ok(v) = zerompk::from_msgpack::(bytes) { return Ok(v); } @@ -60,22 +131,7 @@ fn decode_msgpack_f64(bytes: &[u8]) -> Result { if let Ok(v) = zerompk::from_msgpack::(bytes) { return Ok(v as f64); } - // If value is a map, find the first numeric field. - if let Ok(nodedb_types::Value::Object(map)) = nodedb_types::value_from_msgpack(bytes) { - for (k, v) in &map { - if k == "key" { - continue; - } - match v { - nodedb_types::Value::Float(f) => return Ok(*f), - nodedb_types::Value::Integer(i) => return Ok(*i as f64), - _ => {} - } - } - } - Err(AtomicError::TypeMismatch { - detail: "value is not numeric".into(), - }) + Err(not_numeric()) } /// Encode an `i64` as MessagePack, wrapping the (practically unreachable, but @@ -95,146 +151,217 @@ fn encode_f64(v: f64) -> Result, AtomicError> { } /// Compute the new value for `INCR`, given the current raw bytes (if -/// any). Returns `(new_i64, new_bytes)` -- mirrors the typed-map / -/// plain-i64 branches of [`super::engine_atomic::KvEngine::incr`] exactly. +/// any). Returns `(new_i64, new_bytes)`. +/// +/// A typed row keeps its shape: the numeric column [`target_field`] picks +/// moves, and every other column stays. A bare scalar stays a bare scalar. pub fn incr(current: Option<&[u8]>, delta: i64) -> Result<(i64, Vec), AtomicError> { + if let Some(mut map) = current.and_then(decode_map) { + let (field, old_i64) = target_field(&map, column_i64).ok_or(not_an_integer())?; + let new_i64 = old_i64.checked_add(delta).ok_or(AtomicError::Overflow)?; + map.insert(field, Value::Integer(new_i64)); + return Ok((new_i64, encode_map(map)?)); + } let old_i64 = match current { None => 0i64, - Some(bytes) => decode_msgpack_i64(bytes)?, + Some(bytes) => decode_scalar_i64(bytes)?, }; let new_i64 = old_i64.checked_add(delta).ok_or(AtomicError::Overflow)?; - - // If value is a map (typed KV entry), update the numeric field in-place. - let new_bytes = if let Some(cur) = current - && let Ok(nodedb_types::Value::Object(mut map)) = nodedb_types::value_from_msgpack(cur) - && map.len() > 1 - { - let mut updated = false; - for (k, v) in map.iter_mut() { - if k == "key" { - continue; - } - if matches!( - v, - nodedb_types::Value::Integer(_) | nodedb_types::Value::Float(_) - ) { - *v = nodedb_types::Value::Integer(new_i64); - updated = true; - break; - } - } - if updated { - match nodedb_types::value_to_msgpack(&nodedb_types::Value::Object(map)) { - Ok(bytes) => bytes, - Err(_) => encode_i64(new_i64)?, - } - } else { - encode_i64(new_i64)? - } - } else { - encode_i64(new_i64)? - }; - Ok((new_i64, new_bytes)) + Ok((new_i64, encode_i64(new_i64)?)) } -/// Compute the new value for `INCR_FLOAT`. Returns `(new_f64, new_bytes)` -/// -- mirrors [`super::engine_atomic::KvEngine::incr_float`] exactly (no -/// typed-map branch: float counters are always stored as a bare f64). +/// Compute the new value for `INCR_FLOAT`. Returns `(new_f64, new_bytes)`. +/// +/// A typed row keeps its shape, as in [`incr`]. A bare scalar is stored as +/// a bare f64. pub fn incr_float(current: Option<&[u8]>, delta: f64) -> Result<(f64, Vec), AtomicError> { - let old_f64 = match current { - None => 0.0f64, - Some(bytes) => decode_msgpack_f64(bytes)?, + let mut row = current.and_then(decode_map); + let (field, old_f64) = match (&row, current) { + (Some(map), _) => { + let (field, old) = target_field(map, column_f64).ok_or(not_numeric())?; + (Some(field), old) + } + (None, None) => (None, 0.0f64), + (None, Some(bytes)) => (None, decode_scalar_f64(bytes)?), }; let new_f64 = old_f64 + delta; if new_f64.is_nan() || new_f64.is_infinite() { return Err(AtomicError::Overflow); } - let new_bytes = encode_f64(new_f64)?; + let new_bytes = match (row.take(), field) { + (Some(mut map), Some(field)) => { + map.insert(field, Value::Float(new_f64)); + encode_map(map)? + } + _ => encode_f64(new_f64)?, + }; Ok((new_f64, new_bytes)) } -/// Compute the CAS outcome: whether `expected` matches the current value -/// (with the typed-map first-string-field fallback), and the bytes to -/// write when it does. Mirrors [`super::engine_atomic::KvEngine::cas`] -/// exactly. -pub fn cas(current: Option<&[u8]>, expected: &[u8], new_value: &[u8]) -> (bool, Vec) { - let matches = match current { - None => expected.is_empty(), - Some(v) => { - if v == expected { - true - } else if let Ok(nodedb_types::Value::Object(map)) = nodedb_types::value_from_msgpack(v) - { - let expected_str = String::from_utf8_lossy(expected); - map.iter().any(|(k, val)| { - k != "key" - && matches!(val, nodedb_types::Value::String(s) if s == expected_str.as_ref()) - }) - } else { - false - } - } - }; +/// Write `new_value` into the string column of the typed row `row` and +/// encode it. `column` is the column [`target_field`] picked. +fn swap_string_column( + mut row: HashMap, + column: String, + new_value: &[u8], +) -> Result, AtomicError> { + row.insert( + column, + Value::String(String::from_utf8_lossy(new_value).into_owned()), + ); + encode_map(row) +} - if !matches { - return (false, Vec::new()); - } - - let write_bytes = if let Some(cur) = current - && let Ok(nodedb_types::Value::Object(mut map)) = nodedb_types::value_from_msgpack(cur) - && map.len() > 1 - { - let new_str = String::from_utf8_lossy(new_value).to_string(); - let mut updated = false; - for (k, v) in map.iter_mut() { - if k == "key" { - continue; - } - if matches!(v, nodedb_types::Value::String(_)) { - *v = nodedb_types::Value::String(new_str.clone()); - updated = true; - break; - } - } - if updated { - nodedb_types::value_to_msgpack(&nodedb_types::Value::Object(map)) - .unwrap_or_else(|_| new_value.to_vec()) +/// A typed row and its string column, when `current` is a typed row with +/// one. [`cas`] and [`getset`] address the same column. +fn string_column(current: Option<&[u8]>) -> Option<(HashMap, String, String)> { + let row = current.and_then(decode_map)?; + let (column, text) = target_field(&row, column_string)?; + Some((row, column, text)) +} + +/// Compute the CAS outcome: whether `expected` matches the current value, +/// and the bytes to write when it does. +/// +/// The current value matches when its bytes equal `expected`, or when it is +/// a typed row whose string column holds `expected`. A typed row with a +/// string column keeps its shape: only that column is swapped. +pub fn cas( + current: Option<&[u8]>, + expected: &[u8], + new_value: &[u8], +) -> Result<(bool, Vec), AtomicError> { + let Some(cur) = current else { + return Ok(if expected.is_empty() { + (true, new_value.to_vec()) } else { - new_value.to_vec() - } - } else { - new_value.to_vec() + (false, Vec::new()) + }); }; - (true, write_bytes) -} - -/// Compute the bytes to write for `GETSET`. Mirrors -/// [`super::engine_atomic::KvEngine::getset`] exactly (typed-map -/// first-string-field update, or a plain overwrite). -pub fn getset(current: Option<&[u8]>, new_value: &[u8]) -> Vec { - if let Some(cur) = current - && let Ok(nodedb_types::Value::Object(mut map)) = nodedb_types::value_from_msgpack(cur) - && map.len() > 1 - { - let new_str = String::from_utf8_lossy(new_value).to_string(); - let mut updated = false; - for (k, v) in map.iter_mut() { - if k == "key" { - continue; - } - if matches!(v, nodedb_types::Value::String(_)) { - *v = nodedb_types::Value::String(new_str.clone()); - updated = true; - break; - } - } - if updated { - nodedb_types::value_to_msgpack(&nodedb_types::Value::Object(map)) - .unwrap_or_else(|_| new_value.to_vec()) - } else { - new_value.to_vec() + let typed = string_column(current); + let column_matches = typed + .as_ref() + .is_some_and(|(_, _, text)| *text == String::from_utf8_lossy(expected)); + if cur != expected && !column_matches { + return Ok((false, Vec::new())); + } + let write_bytes = match typed { + Some((row, column, _)) => swap_string_column(row, column, new_value)?, + None => new_value.to_vec(), + }; + Ok((true, write_bytes)) +} + +/// Compute the bytes to write for `GETSET`: the string column of a typed +/// row swapped in place, or a plain overwrite. +pub fn getset(current: Option<&[u8]>, new_value: &[u8]) -> Result, AtomicError> { + match string_column(current) { + Some((row, column, _)) => swap_string_column(row, column, new_value), + None => Ok(new_value.to_vec()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn row(fields: &[(&str, Value)]) -> Vec { + let map: HashMap = fields + .iter() + .map(|(k, v)| ((*k).to_string(), v.clone())) + .collect(); + nodedb_types::value_to_msgpack(&Value::Object(map)).expect("encode row") + } + + fn columns(bytes: &[u8]) -> HashMap { + decode_map(bytes).expect("a typed row stays a typed row") + } + + #[test] + fn incr_on_a_one_column_typed_row_keeps_the_row() { + let current = row(&[("n", Value::Integer(5))]); + let (new_i64, bytes) = incr(Some(¤t), 3).expect("incr"); + assert_eq!(new_i64, 8); + assert_eq!(columns(&bytes).get("n"), Some(&Value::Integer(8))); + } + + #[test] + fn incr_moves_the_first_numeric_column_in_key_order() { + let current = row(&[ + ("b", Value::Integer(100)), + ("a", Value::Integer(1)), + ("label", Value::String("x".into())), + ]); + let (new_i64, bytes) = incr(Some(¤t), 1).expect("incr"); + assert_eq!(new_i64, 2); + let cols = columns(&bytes); + assert_eq!(cols.get("a"), Some(&Value::Integer(2))); + assert_eq!(cols.get("b"), Some(&Value::Integer(100))); + assert_eq!(cols.get("label"), Some(&Value::String("x".into()))); + } + + #[test] + fn incr_on_a_typed_row_encodes_the_same_bytes_every_time() { + let current = row(&[ + ("a", Value::Integer(1)), + ("b", Value::Integer(2)), + ("c", Value::Integer(3)), + ]); + let (_, first) = incr(Some(¤t), 1).expect("incr"); + for _ in 0..16 { + let (_, again) = incr(Some(¤t), 1).expect("incr"); + assert_eq!(again, first); } - } else { - new_value.to_vec() + } + + #[test] + fn incr_on_a_typed_row_without_a_numeric_column_is_a_type_mismatch() { + let current = row(&[("label", Value::String("x".into()))]); + assert!(matches!( + incr(Some(¤t), 1), + Err(AtomicError::TypeMismatch { .. }) + )); + } + + #[test] + fn incr_on_a_bare_scalar_stays_a_bare_scalar() { + let current = zerompk::to_msgpack_vec(&5i64).expect("encode"); + let (new_i64, bytes) = incr(Some(¤t), 3).expect("incr"); + assert_eq!(new_i64, 8); + assert_eq!(zerompk::from_msgpack::(&bytes).expect("decode"), 8); + let (fresh, _) = incr(None, 4).expect("incr"); + assert_eq!(fresh, 4); + } + + #[test] + fn incr_float_on_a_one_column_typed_row_keeps_the_row() { + let current = row(&[("score", Value::Float(1.5))]); + let (new_f64, bytes) = incr_float(Some(¤t), 1.0).expect("incr_float"); + assert_eq!(new_f64, 2.5); + assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); + } + + #[test] + fn cas_on_a_one_column_typed_row_swaps_the_column() { + let current = row(&[("state", Value::String("idle".into()))]); + let (matched, bytes) = cas(Some(¤t), b"idle", b"busy").expect("cas"); + assert!(matched); + assert_eq!( + columns(&bytes).get("state"), + Some(&Value::String("busy".into())) + ); + let (matched, _) = cas(Some(¤t), b"busy", b"idle").expect("cas"); + assert!(!matched); + } + + #[test] + fn getset_on_a_one_column_typed_row_swaps_the_column() { + let current = row(&[("token", Value::String("old".into()))]); + let bytes = getset(Some(¤t), b"new").expect("getset"); + assert_eq!( + columns(&bytes).get("token"), + Some(&Value::String("new".into())) + ); + assert_eq!(getset(None, b"raw").expect("getset"), b"raw".to_vec()); } } diff --git a/nodedb/src/engine/kv/mod.rs b/nodedb/src/engine/kv/mod.rs index 65624d4fd..2b182f567 100644 --- a/nodedb/src/engine/kv/mod.rs +++ b/nodedb/src/engine/kv/mod.rs @@ -22,8 +22,12 @@ pub mod sorted_index; pub use batch_put::KvBatchPutParams; pub use clock::current_ms; -pub use engine::{KvEngine, RestoreCompositeIndexParams, RestoreFieldIndexParams}; -pub use engine_atomic::{AtomicError, AtomicKeyCtx, CasResult, IncrAdmission, admit_any}; +pub use engine::{ + KvEngine, KvEntryImage, KvKeyRef, RestoreCompositeIndexParams, RestoreFieldIndexParams, +}; +pub use engine_atomic::{ + AtomicAdmission, AtomicError, AtomicKeyCtx, CasResult, GetSetResult, admit_any, +}; pub use engine_atomic_compute as atomic_compute; pub use engine_index::RegisterIndexParams; pub use engine_rename::RenameCollectionParams; diff --git a/nodedb/src/engine/sparse/inverted/doc_image.rs b/nodedb/src/engine/sparse/inverted/doc_image.rs new file mode 100644 index 000000000..245cff996 --- /dev/null +++ b/nodedb/src/engine/sparse/inverted/doc_image.rs @@ -0,0 +1,130 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A document's complete index footprint, captured so a rollback can put it +//! back exactly. +//! +//! The index does not keep a document's text. It keeps, per term, the +//! positions the term occupies in the document's analyzed token stream, and +//! the stream's length. Those positions are dense (`0..len`), so the token +//! stream rebuilds from them exactly. Re-indexing that stream reproduces the +//! document's postings, its length row, its term set, and the collection's +//! corpus counters. + +use nodedb_fts::posting::Posting; +use nodedb_types::{Surrogate, TenantId}; +use redb::ReadableTable as _; + +use super::core::InvertedIndex; +use super::doc_terms; +use super::errors::inverted_err; +use super::indexing::{IndexDocScope, prior_doc_length}; +use crate::engine::sparse::fts_redb::tables::POSTINGS; + +/// The analyzed token stream one document is indexed with. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct FtsDocImage { + tokens: Vec, +} + +impl InvertedIndex { + /// The index footprint of `surrogate` in `collection`, or `None` when the + /// document is not indexed. Reads only: the write transaction it opens to + /// reach the term-set fallback is aborted. + pub fn document_image( + &self, + database_id: u64, + tid: TenantId, + collection: &str, + surrogate: Surrogate, + ) -> crate::Result> { + let scope = IndexDocScope { + database_id, + tid, + collection, + surrogate, + }; + let db = self.inner.backend().db(); + let txn = db.begin_write().map_err(|e| inverted_err("write txn", e))?; + let image = Self::read_document_image(&txn, scope)?; + txn.abort() + .map_err(|e| inverted_err("abort image read", e))?; + Ok(image) + } + + /// Put the index footprint of `surrogate` back to `image`, or remove the + /// document when `image` is `None`: it was not indexed before. + pub fn restore_document_image( + &self, + database_id: u64, + tid: TenantId, + collection: &str, + surrogate: Surrogate, + image: Option<&FtsDocImage>, + ) -> crate::Result<()> { + let Some(image) = image.filter(|image| !image.tokens.is_empty()) else { + return self.remove_document(database_id, tid, collection, surrogate); + }; + let scope = IndexDocScope { + database_id, + tid, + collection, + surrogate, + }; + let db = self.inner.backend().db(); + let txn = db.begin_write().map_err(|e| inverted_err("write txn", e))?; + self.write_index_data(&txn, scope, &image.tokens)?; + txn.commit() + .map_err(|e| inverted_err("commit image restore", e))?; + Ok(()) + } + + fn read_document_image( + txn: &redb::WriteTransaction, + scope: IndexDocScope<'_>, + ) -> crate::Result> { + let Some(doc_len) = prior_doc_length(txn, scope)? else { + return Ok(None); + }; + let terms = doc_terms::occupied_terms(txn, scope, true)?; + let postings = txn + .open_table(POSTINGS) + .map_err(|e| inverted_err("open postings", e))?; + let mut tokens = vec![String::new(); doc_len as usize]; + for term in terms { + let list: Vec = postings + .get(( + scope.database_id, + scope.tid.as_u64(), + scope.collection, + term.as_str(), + )) + .map_err(|e| inverted_err("read postings", e))? + .map(|bytes| zerompk::from_msgpack(bytes.value())) + .transpose() + .map_err(|e| inverted_err("decode postings", e))? + .unwrap_or_default(); + let Some(posting) = list.iter().find(|p| p.doc_id == scope.surrogate) else { + continue; + }; + for &position in &posting.positions { + let slot = tokens.get_mut(position as usize).ok_or_else(|| { + inverted_err( + "document image", + format!( + "term '{term}' sits at position {position} past the document \ + length {doc_len}" + ), + ) + })?; + slot.clone_from(&term); + } + } + if let Some(position) = tokens.iter().position(String::is_empty) { + return Err(inverted_err( + "document image", + format!("no posting covers position {position} of {doc_len}"), + )); + } + Ok(Some(FtsDocImage { tokens })) + } +} diff --git a/nodedb/src/engine/sparse/inverted/indexing.rs b/nodedb/src/engine/sparse/inverted/indexing.rs index 14b24742e..ebf1a5701 100644 --- a/nodedb/src/engine/sparse/inverted/indexing.rs +++ b/nodedb/src/engine/sparse/inverted/indexing.rs @@ -140,7 +140,7 @@ impl InvertedIndex { /// Core indexing logic: writes postings, doc length, and stats within /// a transaction. Bypasses the LSM memtable so Origin transactions can /// stay atomic with the document write. - fn write_index_data( + pub(super) fn write_index_data( &self, txn: &WriteTransaction, scope: IndexDocScope<'_>, diff --git a/nodedb/src/engine/sparse/inverted/mod.rs b/nodedb/src/engine/sparse/inverted/mod.rs index 9689c26c0..aa47293cb 100644 --- a/nodedb/src/engine/sparse/inverted/mod.rs +++ b/nodedb/src/engine/sparse/inverted/mod.rs @@ -16,6 +16,7 @@ mod compaction; mod core; mod corpus_stats; +mod doc_image; mod doc_terms; mod errors; mod indexing; @@ -24,6 +25,7 @@ mod search; mod synonyms; pub use core::InvertedIndex; +pub use doc_image::FtsDocImage; pub use indexing::IndexDocScope; pub use nodedb_fts::FtsSearchParams; pub use nodedb_fts::posting::{MatchOffset, Posting, QueryMode, TextSearchResult}; diff --git a/nodedb/src/engine/vector/sparse/index.rs b/nodedb/src/engine/vector/sparse/index.rs index 5664e73f7..3c75c84a0 100644 --- a/nodedb/src/engine/vector/sparse/index.rs +++ b/nodedb/src/engine/vector/sparse/index.rs @@ -11,6 +11,14 @@ use std::collections::HashMap; use nodedb_types::SparseVector; use serde::{Deserialize, Serialize}; +/// One document of a [`SparseInvertedIndex`]: its internal id and its +/// `(dimension, weight)` entries. +#[derive(Debug, Clone, PartialEq)] +pub struct SparseDocImage { + internal_id: u32, + entries: Vec<(u32, f32)>, +} + /// Inverted index for sparse vectors. /// /// For each dimension that appears in any document's sparse vector, stores @@ -110,6 +118,59 @@ impl SparseInvertedIndex { } } + /// The internal id and entries of `doc_id`, or `None` when absent. + pub fn doc_image(&self, doc_id: &str) -> Option { + let internal_id = *self.doc_id_forward.get(doc_id)?; + let entries = self + .doc_dims + .get(&internal_id) + .map(|dims| { + dims.iter() + .filter_map(|dim| { + let list = self.postings.get(dim)?; + let weight = list.iter().find(|(id, _)| *id == internal_id)?.1; + Some((*dim, weight)) + }) + .collect() + }) + .unwrap_or_default(); + Some(SparseDocImage { + internal_id, + entries, + }) + } + + /// The internal id the next insert takes. + pub fn next_internal_id(&self) -> u32 { + self.next_id + } + + /// Put `doc_id` back to `prior` (absent when `None`) and the internal id + /// counter back to `next_id`. A rollback uses it to withdraw a write, so + /// the next insert takes the id it would have taken without it. + pub fn roll_back_doc(&mut self, doc_id: &str, prior: Option<&SparseDocImage>, next_id: u32) { + if let Some(current) = self.doc_id_forward.remove(doc_id) { + self.remove_internal(current); + self.doc_id_reverse.remove(¤t); + self.doc_count -= 1; + } + if let Some(prior) = prior { + let id = prior.internal_id; + let mut dims = Vec::with_capacity(prior.entries.len()); + for &(dim, weight) in &prior.entries { + let list = self.postings.entry(dim).or_default(); + let at = list.partition_point(|(existing, _)| *existing < id); + list.insert(at, (id, weight)); + dims.push(dim); + } + self.doc_dims.insert(id, dims); + self.doc_id_forward.insert(doc_id.to_string(), id); + self.doc_id_reverse.insert(id, doc_id.to_string()); + self.doc_count += 1; + } + self.next_id = self.next_id.min(next_id); + } + /// Get the posting list for a dimension. pub fn get_postings(&self, dim: u32) -> Option<&[(u32, f32)]> { self.postings.get(&dim).map(|v| v.as_slice()) @@ -294,4 +355,36 @@ mod tests { let postings = idx.get_postings(5).unwrap(); assert_eq!(postings.len(), 3); } + + #[test] + fn a_rolled_back_upsert_restores_the_prior_entries_and_id_counter() { + let mut idx = SparseInvertedIndex::new(); + idx.insert("doc1", &make_sv(&[(10, 0.5), (20, 0.8)])); + idx.insert("doc2", &make_sv(&[(10, 0.1)])); + let prior = idx.doc_image("doc1"); + let next_id = idx.next_internal_id(); + + idx.insert("doc1", &make_sv(&[(30, 0.9)])); + idx.roll_back_doc("doc1", prior.as_ref(), next_id); + + assert_eq!(idx.doc_image("doc1"), prior); + assert_eq!(idx.next_internal_id(), next_id); + assert!(idx.get_postings(30).is_none()); + assert_eq!( + idx.get_postings(10).map(<[(u32, f32)]>::to_vec), + Some(vec![(0, 0.5), (1, 0.1)]), + "the restored posting keeps its id order" + ); + } + + #[test] + fn a_rolled_back_insert_of_a_new_document_removes_it() { + let mut idx = SparseInvertedIndex::new(); + let next_id = idx.next_internal_id(); + idx.insert("doc1", &make_sv(&[(10, 0.5)])); + idx.roll_back_doc("doc1", None, next_id); + assert!(idx.doc_image("doc1").is_none()); + assert_eq!(idx.doc_count(), 0); + assert!(idx.get_postings(10).is_none()); + } } diff --git a/nodedb/src/engine/vector/sparse/mod.rs b/nodedb/src/engine/vector/sparse/mod.rs index 513442172..a75cfd2bc 100644 --- a/nodedb/src/engine/vector/sparse/mod.rs +++ b/nodedb/src/engine/vector/sparse/mod.rs @@ -3,5 +3,5 @@ pub mod index; pub mod search; -pub use index::SparseInvertedIndex; +pub use index::{SparseDocImage, SparseInvertedIndex}; pub use search::SparseSearchResult; diff --git a/nodedb/src/event/wal_replay.rs b/nodedb/src/event/wal_replay.rs index 9e4d70f74..03ab26783 100644 --- a/nodedb/src/event/wal_replay.rs +++ b/nodedb/src/event/wal_replay.rs @@ -275,7 +275,9 @@ fn record_to_events(record: &WalRecord, sequence: &mut u64) -> Vec { // WriteAborted names a refused write; the record it names has already // been dropped from this stream by the replay-source filter (see // `WalManager::replay_from`). The marker itself is not a row write. - | RecordType::WriteAborted => Vec::new(), + | RecordType::WriteAborted + // ProposalApplied marks a Raft proposal as applied; it writes no row. + | RecordType::ProposalApplied => Vec::new(), } } diff --git a/nodedb/src/wal/manager/append.rs b/nodedb/src/wal/manager/append.rs index c53115ee0..7b6e59aa5 100644 --- a/nodedb/src/wal/manager/append.rs +++ b/nodedb/src/wal/manager/append.rs @@ -1,11 +1,52 @@ // SPDX-License-Identifier: BUSL-1.1 +use std::cell::Cell; + +use nodedb_wal::RecordTarget; use nodedb_wal::record::RecordType; use super::core::WalManager; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +thread_local! { + /// The proposal whose apply the current thread is appending records for, + /// `0` outside one. Set only by [`WalManager::with_apply_key`]. + static APPLY_KEY: Cell = const { Cell::new(0) }; +} + +/// Restores the thread's prior apply key when a keyed append scope ends, +/// panics included. +struct ApplyKeyScope { + prior: u64, +} + +impl Drop for ApplyKeyScope { + fn drop(&mut self) { + APPLY_KEY.with(|key| key.set(self.prior)); + } +} + +/// The apply key of the proposal the current thread is appending for, `0` +/// outside one. +pub(super) fn current_apply_key() -> u64 { + APPLY_KEY.with(Cell::get) +} + impl WalManager { + /// Run `append` with every record this thread appends inside it carrying + /// `apply_key` in its header: the idempotency key of the replicated + /// proposal being applied. The record and the key become durable in one + /// write, so the apply loop recovers which proposals applied from the + /// records themselves. + /// + /// `append` must not await: the key is scoped to the calling thread, and + /// a WAL append is synchronous. + pub fn with_apply_key(&self, apply_key: u64, append: impl FnOnce() -> R) -> R { + let prior = APPLY_KEY.with(|key| key.replace(apply_key)); + let _scope = ApplyKeyScope { prior }; + append() + } + /// Internal: append a record of the given type to the WAL. pub(super) fn append_record( &self, @@ -15,14 +56,18 @@ impl WalManager { database_id: DatabaseId, payload: &[u8], ) -> crate::Result { + let apply_key = current_apply_key(); let mut wal = self.wal.lock().unwrap_or_else(|p| p.into_inner()); let lsn = wal - .append( - record_type as u32, - tenant_id.as_u64(), - vshard_id.as_u32(), - database_id.as_u64(), + .append_keyed( + RecordTarget { + record_type: record_type as u32, + tenant_id: tenant_id.as_u64(), + vshard_id: vshard_id.as_u32(), + database_id: database_id.as_u64(), + }, payload, + apply_key, ) .map_err(crate::Error::Wal)?; Ok(Lsn::new(lsn)) diff --git a/nodedb/src/wal/manager/append_metadata.rs b/nodedb/src/wal/manager/append_metadata.rs index 4b0f07585..ce168e970 100644 --- a/nodedb/src/wal/manager/append_metadata.rs +++ b/nodedb/src/wal/manager/append_metadata.rs @@ -1,7 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 //! WAL appends for node-global metadata: temporal-purge audit, surrogate -//! allocation/binding, Calvin epoch tracking, and sync watermarks. +//! allocation/binding, Calvin epoch tracking, applied Raft proposals, and +//! sync watermarks. use nodedb_wal::record::RecordType; @@ -110,6 +111,31 @@ impl WalManager { ) } + /// Append a payload-free `ProposalApplied` marker for the replicated + /// proposal whose apply is in scope (see [`WalManager::with_apply_key`]). + /// An apply that writes no record of its own appends it in place of its + /// forward record, so the proposal's key still reaches the WAL. + /// + /// Returns `None` and appends nothing outside a proposal's apply. + pub fn append_proposal_applied( + &self, + tenant_id: TenantId, + vshard_id: VShardId, + database_id: DatabaseId, + ) -> crate::Result> { + if super::append::current_apply_key() == 0 { + return Ok(None); + } + self.append_record( + RecordType::ProposalApplied, + tenant_id, + vshard_id, + database_id, + &[], + ) + .map(Some) + } + /// Append a `SyncSeqAdvance` watermark record. Emitted by the Data Plane sync /// handler after durably applying an ingest message, to make the per-stream /// high-watermark crash-recoverable. diff --git a/nodedb/src/wal/manager/append_transaction.rs b/nodedb/src/wal/manager/append_transaction.rs index 2ddd74627..e8ae7e32f 100644 --- a/nodedb/src/wal/manager/append_transaction.rs +++ b/nodedb/src/wal/manager/append_transaction.rs @@ -35,6 +35,19 @@ impl WalManager { self.append_record(RecordType::TransactionRedo, tid, vs, db, &payload) } + /// Append a `TransactionRedo` record whose payload is an already-encoded + /// redo record. Used by the committed-redo apply path, which carries the + /// record encoded on the plan it dispatches. + pub fn append_transaction_redo_bytes( + &self, + tid: TenantId, + vs: VShardId, + db: DatabaseId, + payload: &[u8], + ) -> crate::Result { + self.append_record(RecordType::TransactionRedo, tid, vs, db, payload) + } + pub fn append_crdt_delta( &self, tid: TenantId, diff --git a/nodedb/src/wal/mod.rs b/nodedb/src/wal/mod.rs index cefff98e7..3f530e6a1 100644 --- a/nodedb/src/wal/mod.rs +++ b/nodedb/src/wal/mod.rs @@ -9,6 +9,7 @@ pub mod crdt_payload; pub mod manager; pub mod redo; pub mod replay; +pub mod timeseries_batch_payload; pub use audit_segment::AuditWalSegment; pub(crate) use crdt_doc_payload::CrdtDocOpWalRecord; @@ -20,3 +21,6 @@ pub use replay::SyncHwmReplayMaps; pub use replay::SyncHwmReplayStats; pub use replay::replay_surrogate_records; pub use replay::replay_sync_hwm_records; +pub(crate) use timeseries_batch_payload::{ + ColumnarConflictPolicy, DecodedBatchRecord, decode_batch_record, +}; diff --git a/nodedb/src/wal/redo/replay.rs b/nodedb/src/wal/redo/replay.rs index 3cfd5727e..bfafdd93f 100644 --- a/nodedb/src/wal/redo/replay.rs +++ b/nodedb/src/wal/redo/replay.rs @@ -40,8 +40,15 @@ //! mutually-exclusive tuple shapes — see the `replay_document_redo` and //! `replay_graph_redo` arms in `crate::data::executor`). //! -//! `calvin_stamp` is ignored here: it is read only by the Calvin recovery scan, -//! never gates engine replay. +//! `calvin_stamp` is ignored here: the Calvin recovery scan reads it, and it +//! does not gate engine replay. +//! +//! The committed-redo apply (`handlers::transaction::redo_apply`) drives this +//! same entry point with a one-record slice for every committed transaction, +//! so the live install and restart replay run the same arms. The arms settle +//! watermark skips and per-record errors by pass (`data::executor`'s +//! `replay_policy`): restart replay keeps its gates, the live install skips +//! none and fails on any error. use std::borrow::Cow; @@ -133,16 +140,13 @@ impl CoreLoop { /// `MultiVectorPut` / `MultiVectorDelete` sub-records the vector resolver /// emits — rebuilds any index from them. /// - /// Three arms take the STANDALONE slice rather than the merged one, at the + /// Two arms take the STANDALONE slice rather than the merged one, at the /// exact positions they have always occupied so their ordering against the /// merged arms is unchanged: /// /// * `replay_document_vector_wal` — redo document puts rebuild their /// secondary vector index inline inside `replay_document_redo`, so /// feeding it the merged stream would index them a second time. - /// * `replay_crdt_wal_ordered` — CRDT deltas ride their own `CrdtDelta` - /// records, never redo sub-records, and that arm is already globally - /// LSN-ordered on its own. /// * `replay_graph_node_label_wal` — its redo counterpart is /// `replay_graph_node_labels_redo` below; handing both the merged stream /// would apply every label delta twice. @@ -171,8 +175,9 @@ impl CoreLoop { self.replay_timeseries_wal(&ordered, num_cores, tombstones); self.replay_array_wal(&ordered, num_cores, tombstones); // CRDT deltas and document/list intents share Loro state, so replay - // their standalone WAL records together in global LSN order. - self.replay_crdt_wal_ordered(records, num_cores, tombstones); + // them together in global LSN order: the standalone records and the + // intents a committed transaction journalled as redo sub-records. + self.replay_crdt_wal_ordered(&ordered, num_cores, tombstones); self.replay_fts_wal(&ordered, num_cores, tombstones); self.replay_spatial_wal(&ordered, num_cores, tombstones); // Graph node labels have no redb-backed durability (unlike edges, diff --git a/nodedb/src/wal/replay/surrogate/dispatch.rs b/nodedb/src/wal/replay/surrogate/dispatch.rs index 53782d2f1..76c6896c8 100644 --- a/nodedb/src/wal/replay/surrogate/dispatch.rs +++ b/nodedb/src/wal/replay/surrogate/dispatch.rs @@ -115,7 +115,9 @@ pub fn replay_surrogate_records( // WriteAborted only names a refused write's LSN; the record it // names is already gone from this stream (the replay source drops // it), and the marker itself binds no surrogate. - | RecordType::WriteAborted => {} + | RecordType::WriteAborted + // ProposalApplied only names an applied Raft proposal. + | RecordType::ProposalApplied => {} } } Ok(stats) diff --git a/nodedb/src/wal/replay/sync_hwm/advance.rs b/nodedb/src/wal/replay/sync_hwm/advance.rs index ab3e7b33d..94481e722 100644 --- a/nodedb/src/wal/replay/sync_hwm/advance.rs +++ b/nodedb/src/wal/replay/sync_hwm/advance.rs @@ -127,7 +127,9 @@ pub fn replay_sync_hwm_records( | RecordType::GraphNodeLabelSet | RecordType::GraphNodeLabelRemove // WriteAborted carries only a refused write's LSN, no sync HWM. - | RecordType::WriteAborted => {} + | RecordType::WriteAborted + // ProposalApplied carries only an applied Raft proposal's identity. + | RecordType::ProposalApplied => {} } } diff --git a/nodedb/src/wal/timeseries_batch_payload.rs b/nodedb/src/wal/timeseries_batch_payload.rs new file mode 100644 index 000000000..2ee10f522 --- /dev/null +++ b/nodedb/src/wal/timeseries_batch_payload.rs @@ -0,0 +1,172 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Decoding of `TimeseriesBatch` WAL record payloads, shared by restart +//! replay and the Control Plane's WAL catch-up, and the columnar insert's +//! conflict policy those records carry. + +use nodedb_physical::physical_plan::{ColumnarInsertIntent, UpdateValue}; + +/// What a columnar insert does with a row whose primary key already exists: +/// the insert's intent and its `ON CONFLICT DO UPDATE` assignments. +#[derive(Debug, Clone, PartialEq)] +pub(crate) struct ColumnarConflictPolicy { + pub intent: ColumnarInsertIntent, + pub on_conflict_updates: Vec<(String, UpdateValue)>, +} + +impl ColumnarConflictPolicy { + /// A plain insert: an existing row is replaced. + pub(crate) fn replace() -> Self { + Self { + intent: ColumnarInsertIntent::Insert, + on_conflict_updates: Vec::new(), + } + } + + /// The record encoding. A plain insert encodes as empty bytes. + pub(crate) fn encode(&self) -> crate::Result> { + if *self == Self::replace() { + return Ok(Vec::new()); + } + zerompk::to_msgpack_vec(&(self.intent, &self.on_conflict_updates)).map_err(|e| { + crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("columnar conflict policy: {e}"), + } + }) + } + + /// Decode the record encoding. Empty bytes are a plain insert. + pub(crate) fn decode(bytes: &[u8]) -> crate::Result { + if bytes.is_empty() { + return Ok(Self::replace()); + } + let (intent, on_conflict_updates) = + zerompk::from_msgpack::<(ColumnarInsertIntent, Vec<(String, UpdateValue)>)>(bytes) + .map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("columnar conflict policy: {e}"), + })?; + Ok(Self { + intent, + on_conflict_updates, + }) + } +} + +/// Decoded fields of a `TimeseriesBatch` WAL record. +pub(crate) struct DecodedBatchRecord { + /// `Some("columnar")` / `Some("timeseries")` for tagged records, `None` + /// for the untagged 2-tuple shape. + pub kind: Option, + pub collection: String, + pub payload: Vec, + pub provenance: Option, + /// Present in the format-preserving timeseries tuples. Absent records + /// use the UTF-8 heuristic. + pub format: Option, + /// Non-empty only for map-shaped columnar records. + pub surrogates: Vec, + /// A columnar insert's encoded [`ColumnarConflictPolicy`]. Empty for a + /// plain insert and for every other record shape. + pub conflict_policy: Vec, + /// The timestamp every untimed row takes. Present only in the six-element + /// autocommit ingest tuple. + pub default_timestamp_ms: Option, +} + +impl DecodedBatchRecord { + fn tuple( + kind: Option, + collection: String, + payload: Vec, + provenance: Option, + format: Option, + default_timestamp_ms: Option, + ) -> Self { + Self { + kind, + collection, + payload, + provenance, + format, + surrogates: Vec::new(), + conflict_policy: Vec::new(), + default_timestamp_ms, + } + } +} + +type ProvenanceField = Option; + +/// Decode a `TimeseriesBatch` WAL payload into its logical fields. +/// +/// Tries the map form and the longest tuple forms first. The map form is +/// unambiguous from the tuple forms, and zerompk enforces tuple arity. +pub(crate) fn decode_batch_record(payload: &[u8]) -> Result { + if let Ok(rec) = zerompk::from_msgpack::(payload) { + return Ok(DecodedBatchRecord { + kind: Some(rec.kind), + collection: rec.collection, + payload: rec.payload, + provenance: rec.provenance, + format: None, + surrogates: rec.surrogates, + conflict_policy: rec.conflict_policy, + default_timestamp_ms: None, + }); + } + zerompk::from_msgpack::<(String, String, Vec, ProvenanceField, String, i64)>(payload) + .map( + |(kind, collection, payload, provenance, format, default_ms)| { + DecodedBatchRecord::tuple( + Some(kind), + collection, + payload, + provenance, + Some(format), + Some(default_ms), + ) + }, + ) + .or_else(|_| { + zerompk::from_msgpack::<(String, String, Vec, ProvenanceField, String)>(payload) + .map(|(kind, collection, payload, provenance, format)| { + DecodedBatchRecord::tuple( + Some(kind), + collection, + payload, + provenance, + Some(format), + None, + ) + }) + }) + .or_else(|_| { + zerompk::from_msgpack::<(String, String, Vec, ProvenanceField)>(payload).map( + |(kind, collection, payload, provenance)| { + DecodedBatchRecord::tuple( + Some(kind), + collection, + payload, + provenance, + None, + None, + ) + }, + ) + }) + .or_else(|_| { + zerompk::from_msgpack::<(String, String, Vec)>(payload).map( + |(kind, collection, payload)| { + DecodedBatchRecord::tuple(Some(kind), collection, payload, None, None, None) + }, + ) + }) + .or_else(|_| { + zerompk::from_msgpack::<(String, Vec)>(payload).map(|(collection, payload)| { + DecodedBatchRecord::tuple(None, collection, payload, None, None, None) + }) + }) + .map_err(|_| ()) +} diff --git a/nodedb/tests/native/cases/native_txn_commit_visibility.rs b/nodedb/tests/native/cases/native_txn_commit_visibility.rs index 33aa9ae51..298cd90bf 100644 --- a/nodedb/tests/native/cases/native_txn_commit_visibility.rs +++ b/nodedb/tests/native/cases/native_txn_commit_visibility.rs @@ -5,11 +5,11 @@ //! read path — PK point lookups and filtered aggregates, not just full scans — //! on the writing connection and on fresh connections. //! -//! Pre-fix, `NativeTxnDp::dispatch_no_wal` routed the commit's `MetaOp` -//! tasks through the gateway without `task.vshard_id`; the gateway's -//! `primary_vshard` fallback sent them to vShard 0, so the commit batch was -//! durably applied on the wrong core. The bug needs (a) the gateway wired, -//! as production boot does, and (b) more than one Data Plane core, so that +//! The statement's `StageWrite` and the commit's `MetaOp` tasks name no +//! collection. Routed through the gateway, its `primary_vshard` fallback +//! sends them to vShard 0: the write stages in core 0's overlay, and the +//! owning core resolves and installs nothing. The case needs the gateway +//! wired, as production boot does, and more than one Data Plane core, so //! vShard 0 and the collection's owning vShard live on different cores. use std::time::Duration; diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 9e3c58043..183763a8f 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -235,6 +235,7 @@ mod sql_transactions_columnar_engine_rollback; mod sql_transactions_columnar_overlay; mod sql_transactions_columnar_predicate_dml_overlay; mod sql_transactions_columnar_row_level_security; +mod sql_transactions_commit_point_visibility; mod sql_transactions_crdt_overlay; mod sql_transactions_crdt_overlay_lifecycle; mod sql_transactions_cross_shard_read_reject; diff --git a/nodedb/tests/wire/cases/sql_transactions_commit_point_visibility.rs b/nodedb/tests/wire/cases/sql_transactions_commit_point_visibility.rs new file mode 100644 index 000000000..aa006f9d6 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_transactions_commit_point_visibility.rs @@ -0,0 +1,77 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A row an explicit pgwire transaction commits is visible to a PK point +//! lookup and a filtered aggregate, on the writing connection. +//! +//! The collection is chosen so its vShard lives on a core other than core 0. +//! A write staged on the wrong core leaves the owning core's overlay empty, +//! so COMMIT would resolve and install nothing there. + +use nodedb_types::id::{DatabaseId, VShardId}; + +use crate::harness::TestServer; + +const NUM_CORES: usize = 4; + +fn collection_on_nonzero_core(prefix: &str) -> String { + (0..64u32) + .map(|i| format!("{prefix}_{i}")) + .find(|name| { + let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, name).as_u32(); + !(vshard as usize).is_multiple_of(NUM_CORES) + }) + .expect("a candidate collection hashes off core 0") +} + +async fn assert_committed_row_visible(server: &TestServer, coll: &str) { + server.exec("BEGIN").await.unwrap(); + server + .exec(&format!( + "INSERT INTO {coll} (id, name) VALUES ('a1', 'alpha')" + )) + .await + .unwrap(); + server.exec("COMMIT").await.unwrap(); + + let point = server + .query_text(&format!("SELECT id FROM {coll} WHERE id = 'a1'")) + .await + .unwrap(); + assert_eq!(point, vec!["a1"], "PK point lookup on {coll}"); + + let count = server + .query_text(&format!("SELECT count(*) FROM {coll} WHERE name = 'alpha'")) + .await + .unwrap(); + assert_eq!(count, vec!["1"], "filtered count on {coll}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn pgwire_committed_strict_row_is_visible_to_a_point_lookup_on_a_nonzero_core() { + let server = TestServer::start_multicores(NUM_CORES).await; + let coll = collection_on_nonzero_core("pg_commit_vis_strict"); + server + .exec(&format!( + "CREATE COLLECTION {coll} (id STRING PRIMARY KEY, name STRING) \ + WITH (engine='document_strict')" + )) + .await + .unwrap(); + + assert_committed_row_visible(&server, &coll).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn pgwire_committed_schemaless_row_is_visible_to_a_point_lookup_on_a_nonzero_core() { + let server = TestServer::start_multicores(NUM_CORES).await; + let coll = collection_on_nonzero_core("pg_commit_vis_doc"); + server + .exec(&format!( + "CREATE COLLECTION {coll} (id STRING PRIMARY KEY, name STRING) \ + WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + + assert_committed_row_visible(&server, &coll).await; +} diff --git a/nodedb/tests/wire/cases/transactional_ddl_compensation.rs b/nodedb/tests/wire/cases/transactional_ddl_compensation.rs index 1dc525caf..bda641f05 100644 --- a/nodedb/tests/wire/cases/transactional_ddl_compensation.rs +++ b/nodedb/tests/wire/cases/transactional_ddl_compensation.rs @@ -23,15 +23,15 @@ async fn collection_names(server: &TestServer) -> Vec { .collect() } -/// Forces `commit::single_shard_batch_dispatch` (see -/// `control::server::shared::session::commit::single_shard::dispatch_batch`) -/// to fail before it ever calls the real Data-Plane dispatch, so no WAL or -/// disk state changes from the injected failure itself. +/// Forces `commit::single_shard_redo_commit` (see +/// `control::server::shared::session::commit::single_shard::commit_redo`) +/// to fail before the redo record is committed, so no WAL or disk state +/// changes from the injected failure itself. #[cfg(feature = "failpoints")] #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn dispatch_failure_after_create_compensates_the_finalized_collection() { let server = TestServer::start_with_failpoints( - "commit::single_shard_batch_dispatch=fail(injected dispatch failure)", + "commit::single_shard_redo_commit=fail(injected dispatch failure)", ) .await; From 809292676a14dbb92fe012da0ddea3902b3c9561 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 04:58:46 +0800 Subject: [PATCH 15/64] feat(kv): fault KV counter atomics on non-numeric values with typed detail INCR/INCRBY/DECR/INCRBYFLOAT previously computed on whatever bytes were stored, silently producing garbage on a non-numeric or out-of-range value. CounterFault now names the four ways a counter atomic computes no value (not an integer, not a float, integer overflow, non-finite float), each mapped to the RESP client text and SQLSTATE a caller expects. A new float_text module adds INCRBYFLOAT's decimal exactly, matching Redis's trimmed-decimal output for values that fit a Decimal, falling back to f64 outside that range. KvCounterShape travels with a counter atomic from planning through every path that computes its value: the live handler, transaction staging, resolve, and WAL replay. It tells the Data Plane, which does not see the catalog, whether the key belongs to a raw single-value collection or a typed row (and if typed, which numeric column moves and what template the fresh row uses). Structured error details now travel across the wire: ErrorPayload and NativeResponse carry an optional ErrorDetails, and NodeDbError::from_wire_with_details rebuilds a typed error from a carried detail when it matches the error's category, falling back to the code-only reconstruction otherwise. NodeDbError::kv_counter_fault turns a CounterFault into the OVERFLOW or TYPE_MISMATCH error a caller already handles. Documentation and RESP/native tests cover the new fault text and SQLSTATEs, a fresh key's stored shape, and a bare raw value read through every read-modify-write. --- docs/kv.md | 15 +- .../src/native/connection/response.rs | 3 +- .../proposal_committed_twice_applies_once.rs | 3 + .../src/rpc_codec/data_plane_error.rs | 12 +- nodedb-cluster/src/rpc_codec/mod.rs | 2 +- .../src/physical_plan/kv/counter_shape.rs | 39 ++ nodedb-physical/src/physical_plan/kv/mod.rs | 2 + nodedb-physical/src/physical_plan/kv/op.rs | 20 +- nodedb-physical/src/physical_plan/mod.rs | 2 +- .../src/planner/dml_helpers/kv_counter.rs | 272 +++++++++++++ nodedb-sql/src/planner/dml_helpers/mod.rs | 3 + nodedb-types/src/error/ctors/from_wire.rs | 56 +++ nodedb-types/src/error/ctors/write_path.rs | 30 ++ nodedb-types/src/error/sqlstate.rs | 5 + nodedb-types/src/protocol/frames.rs | 28 ++ .../src/protocol/text_fields/decode.rs | 2 +- .../protocol/text_fields/types/text_fields.rs | 5 +- nodedb/src/bridge/envelope/counter_fault.rs | 60 +++ nodedb/src/bridge/envelope/error_code.rs | 9 +- nodedb/src/bridge/envelope/mod.rs | 2 + .../control/cluster/data_plane_error_wire.rs | 53 ++- nodedb/src/control/gateway/error_map/resp.rs | 34 ++ .../control/planner/calvin/tx_class/shared.rs | 1 + .../src/control/planner/calvin/write_class.rs | 4 +- .../planner/catalog_adapter/adapter.rs | 11 + .../src/control/planner/rls_injection/kv.rs | 1 + .../sql_plan_convert/kv_counter_shape.rs | 60 +++ .../control/planner/sql_plan_convert/mod.rs | 1 + .../server/dispatch_utils/write_abort.rs | 2 +- .../server/native/dispatch/conversion.rs | 10 +- .../server/native/dispatch/direct_ops.rs | 7 +- .../native/dispatch/plan_builder/dispatch.rs | 8 +- .../server/native/dispatch/plan_builder/kv.rs | 50 +-- .../dispatch/plan_builder/kv_counter.rs | 86 ++++ .../native/dispatch/plan_builder/mod.rs | 1 + .../src/control/server/pgwire/ddl_encode.rs | 1 + .../server/resp/handler_kv/counters.rs | 51 ++- .../shared/ddl/neutral/kv_atomic/dispatch.rs | 12 +- .../shared/ddl/neutral/kv_atomic/handlers.rs | 60 ++- .../server/shared/ddl/neutral/rate_gate.rs | 16 +- .../src/control/server/shared/ddl/result.rs | 22 +- .../src/control/server/shared/ddl/sqlstate.rs | 6 +- .../server/shared/sql/staging_predicates.rs | 8 +- .../shared/write_admission/lock_keys.rs | 4 +- .../predicate/txn_buffering/classify.rs | 4 +- .../control/server/wal_dispatch_kv/append.rs | 24 +- .../control/server/wal_dispatch_kv/encode.rs | 166 +++++--- .../wal_replication/decode/entry_kv.rs | 8 +- .../src/control/wal_replication/decode/kv.rs | 10 +- .../wal_replication/encode/entry_kv.rs | 5 +- .../src/control/wal_replication/encode/kv.rs | 10 +- .../wal_replication/types/replicated_write.rs | 9 +- .../src/data/executor/handlers/kv/atomic.rs | 276 ++++++++++++- .../src/data/executor/handlers/kv/dispatch.rs | 6 +- .../executor/handlers/kv/resolve/apply.rs | 4 +- .../handlers/kv/resolve/atomic_ops.rs | 24 +- .../executor/handlers/kv/resolve/dispatch.rs | 10 +- .../handlers/transaction/resolve/entry.rs | 5 +- .../stage_write/stage_kv_atomic.rs | 20 +- .../transaction/sub_plan_kv_atomics.rs | 6 +- .../src/data/executor/wal_replay_kv_atomic.rs | 43 +- .../src/data/executor/wal_replay_kv_incr.rs | 292 +++++++------ nodedb/src/engine/kv/engine_atomic.rs | 230 ++++++++--- nodedb/src/engine/kv/engine_atomic_compute.rs | 382 +++++++++++++----- nodedb/src/engine/kv/engine_sorted.rs | 7 +- nodedb/src/engine/kv/float_text.rs | 224 ++++++++++ nodedb/src/engine/kv/mod.rs | 4 +- nodedb/src/error/types.rs | 3 - nodedb/src/error_classify.rs | 3 - nodedb/src/error_from_data_plane.rs | 50 ++- nodedb/tests/crash_resp_kv_write.rs | 72 ++++ .../executor_tests/test_kv_ttl_overlay.rs | 1 + nodedb/tests/native/cases/mod.rs | 1 + .../native/cases/native_kv_counter_faults.rs | 115 ++++++ .../wire/cases/kv_bare_value_counters.rs | 253 ++++++++++++ .../tests/wire/cases/kv_counter_fresh_row.rs | 143 +++++++ nodedb/tests/wire/cases/mod.rs | 2 + .../sql_transactions_kv_atomic_overlay.rs | 64 ++- 78 files changed, 2960 insertions(+), 595 deletions(-) create mode 100644 nodedb-physical/src/physical_plan/kv/counter_shape.rs create mode 100644 nodedb-sql/src/planner/dml_helpers/kv_counter.rs create mode 100644 nodedb/src/bridge/envelope/counter_fault.rs create mode 100644 nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs create mode 100644 nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs create mode 100644 nodedb/src/engine/kv/float_text.rs create mode 100644 nodedb/tests/native/cases/native_kv_counter_faults.rs create mode 100644 nodedb/tests/wire/cases/kv_bare_value_counters.rs create mode 100644 nodedb/tests/wire/cases/kv_counter_fresh_row.rs diff --git a/docs/kv.md b/docs/kv.md index 18baafdf9..5d857ffc4 100644 --- a/docs/kv.md +++ b/docs/kv.md @@ -155,10 +155,21 @@ SELECT KV_GETSET('session_token', 'player-123', 'new-token-xyz'); **RESP (Redis) equivalents:** `INCR`, `DECR`, `INCRBY`, `DECRBY`, `INCRBYFLOAT`, `GETSET` — all work over the RESP protocol. +**Value shapes:** + +- Raw value (a single `value` column, or RESP `SET`): the value is a byte string. `INCR`/`INCRBY`/`DECR`/`DECRBY` read it as decimal integer text and store the result as decimal text. `INCRBYFLOAT` and `KV_INCR_FLOAT` read decimal text (plain or exponent form, such as `5.0e3`), add exactly, and store the trimmed decimal text: `"0.1"` plus `0.2` stores `"0.3"`, `"3.0"` plus `0` stores `"3"`. Exact addition covers 28 significant digits below 7.9e28. A value outside that range adds in 64-bit float. `SET k 5` then `INCR k` leaves `"6"`. +- Absent key: the counter starts at 0 and is stored as decimal text. +- Typed row (several columns): the first numeric column in key order moves. Every other column stays. + **Error handling:** -- `TYPE_MISMATCH` (SQLSTATE 42846) — INCR on a non-numeric value -- `OVERFLOW` (SQLSTATE 22003) — i64 overflow on INCR +| Condition | RESP reply | SQLSTATE | +|---|---|---| +| Raw value is not a decimal integer in the i64 range | `ERR value is not an integer or out of range` | `22P02` | +| Raw value is not a decimal float | `ERR value is not a valid float` | `22P02` | +| Integer result leaves the i64 range | `ERR increment or decrement would overflow` | `22003` | +| Float result is NaN or infinite | `ERR increment would produce NaN or Infinity` | `22003` | +| Typed row has no numeric column | `WRONGTYPE ...` | `42846` | ## Sorted Indexes (Leaderboards) diff --git a/nodedb-client/src/native/connection/response.rs b/nodedb-client/src/native/connection/response.rs index 894ae949e..d97e66c2c 100644 --- a/nodedb-client/src/native/connection/response.rs +++ b/nodedb-client/src/native/connection/response.rs @@ -40,9 +40,10 @@ fn error_frame_to_typed( if payload.ndb_code == 0 { return NodeDbError::internal(payload.message.clone()); } - NodeDbError::from_wire( + NodeDbError::from_wire_with_details( nodedb_types::error::ErrorCode(payload.ndb_code), payload.message.clone(), + payload.details.clone(), ) } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs index 1475731e1..bd42b9609 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs @@ -108,6 +108,9 @@ async fn a_proposal_committed_twice_moves_the_counter_once() { ttl_ms: 0, surrogate, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + // The key is seeded before this proposal, so the shape an + // absent key takes never applies. + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); let write = ReplicableWrite::decide_for_replication(&plan).expect("KV_INCR replicates"); diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index 59e59a02d..13756ee80 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -72,8 +72,9 @@ pub enum DataPlaneErrorCode { collection: String, detail: String, }, - OverflowError { + CounterFault { collection: String, + fault: DataPlaneCounterFault, }, InsufficientBalance { collection: String, @@ -127,3 +128,12 @@ pub enum DataPlaneErrorCode { reason: String, }, } + +/// Wire mirror of `nodedb::bridge::envelope::CounterFault`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum DataPlaneCounterFault { + NotAnInteger, + NotAFloat, + IntegerOverflow, + NonFinite, +} diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index 507561f44..5efaef1d5 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -37,7 +37,7 @@ pub use cluster_mgmt::{ JoinGroupInfo, JoinNodeInfo, JoinRequest, JoinResponse, LEADER_REDIRECT_PREFIX, PingRequest, PongResponse, TopologyAck, TopologyUpdate, }; -pub use data_plane_error::DataPlaneErrorCode; +pub use data_plane_error::{DataPlaneCounterFault, DataPlaneErrorCode}; pub use data_propose::{DataProposeRequest, DataProposeResponse, ProposeTarget}; pub use execute::{ DescriptorVersionEntry, ExecuteRequest, ExecuteResponse, ExecuteStreamChunk, ExecuteStreamEnd, diff --git a/nodedb-physical/src/physical_plan/kv/counter_shape.rs b/nodedb-physical/src/physical_plan/kv/counter_shape.rs new file mode 100644 index 000000000..cb06d641e --- /dev/null +++ b/nodedb-physical/src/physical_plan/kv/counter_shape.rs @@ -0,0 +1,39 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The row a KV counter atomic creates when its key is absent. + +/// What `INCR` / `INCRBYFLOAT` stores for a key that does not exist yet. +/// +/// The Data Plane does not know a collection's declared columns. The Control +/// Plane decides the shape from the catalog when it builds the op, and the +/// shape travels with the op to every path that computes the value: the live +/// handler, transaction staging, the resolve path, and WAL replay. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Default, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum KvCounterShape { + /// A raw collection (a single `value` column, or none declared), or a + /// RESP command, where every value is a byte string. The new value is + /// stored as its decimal text. + #[default] + Raw, + /// A typed collection. The new row is `template` with `column` set to the + /// new value: the row `INSERT (key, column) VALUES (key, delta)` stores, + /// with DEFAULTs materialized. + Typed { + /// The declared numeric column the counter moves. `None` when the + /// collection declares no column of the counter's type. + column: Option, + /// A msgpack map body: the columns the insert stores other than + /// `column`. + template: Vec, + }, +} diff --git a/nodedb-physical/src/physical_plan/kv/mod.rs b/nodedb-physical/src/physical_plan/kv/mod.rs index bd8c12cb9..7e18a7612 100644 --- a/nodedb-physical/src/physical_plan/kv/mod.rs +++ b/nodedb-physical/src/physical_plan/kv/mod.rs @@ -3,8 +3,10 @@ //! KV engine operations dispatched to the Data Plane. pub mod collection; +pub mod counter_shape; pub mod op; pub mod resolved_mutation; +pub use counter_shape::KvCounterShape; pub use op::KvOp; pub use resolved_mutation::{KvResolveOutcome, KvResolvedMutation}; diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index 1ccd43fee..1a73a5c99 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -4,6 +4,7 @@ use nodedb_types::{QualifiedCollection, RlsWriteCheck, Surrogate}; +use super::counter_shape::KvCounterShape; use super::resolved_mutation::KvResolvedMutation; use crate::physical_plan::document::ReturningSpec; @@ -305,8 +306,10 @@ pub enum KvOp { restart_identity: bool, }, - /// Atomic increment: init 0 if absent, `TypeMismatch` if not i64, - /// `OverflowError` on wrap. `ttl_ms > 0` sets/resets TTL; `0` preserves it. + /// Atomic increment: init 0 if absent. A raw body is decimal text in and + /// out, and a body that is not a decimal i64 is a counter fault. A typed + /// row moves its first integer column. Overflow is a counter fault, never + /// a wrap. `ttl_ms > 0` sets/resets TTL; `0` preserves it. Incr { collection: QualifiedCollection, key: Vec, @@ -318,21 +321,28 @@ pub enum KvOp { /// Write policy evaluated against the computed post-increment image /// inside the engine, not guessed by the handler. rls_write_check: RlsWriteCheck, + /// The row an absent key becomes. + shape: KvCounterShape, }, /// Atomic float increment on a numeric value. Returns new value. /// - /// Same semantics as `Incr` but for f64 values. - /// If value is not f64, returns `TypeMismatch`. + /// A raw body is decimal text in and out, added exactly. A typed row's + /// column adds in `f64`. A NaN or infinite result is a counter fault. IncrFloat { collection: QualifiedCollection, key: Vec, - delta: f64, + /// The increment as the client's decimal text, checked at the + /// protocol boundary. It is parsed once, where it is added, so no + /// digit is lost to an `f64` on the way. + delta: String, /// Stable cross-engine identity. `Surrogate::ZERO` only in tests. surrogate: Surrogate, /// Compiled row-level-security WRITE predicate — see `Incr`, whose /// engine-internal compute-and-persist this mirrors. rls_write_check: RlsWriteCheck, + /// The row an absent key becomes. + shape: KvCounterShape, }, /// Compare-and-swap: set value to `new_value` only if current equals `expected`. diff --git a/nodedb-physical/src/physical_plan/mod.rs b/nodedb-physical/src/physical_plan/mod.rs index 22b64d0fc..9a8dc4d03 100644 --- a/nodedb-physical/src/physical_plan/mod.rs +++ b/nodedb-physical/src/physical_plan/mod.rs @@ -48,7 +48,7 @@ pub use exchange::{ExchangeMode, ExchangeOp}; pub use graph::{ BatchEdge, BspSuperstepPlan, BspSuperstepResult, GraphOp, WccSuperstepPlan, WccSuperstepResult, }; -pub use kv::{KvOp, KvResolveOutcome, KvResolvedMutation}; +pub use kv::{KvCounterShape, KvOp, KvResolveOutcome, KvResolvedMutation}; pub use meta::{MetaOp, SAVEPOINT_MARKER_BYTES}; pub use plan::PhysicalPlan; pub use query::{AggregateSpec, GroupKeySpec, JoinProjection, QueryOp}; diff --git a/nodedb-sql/src/planner/dml_helpers/kv_counter.rs b/nodedb-sql/src/planner/dml_helpers/kv_counter.rs new file mode 100644 index 000000000..9498c7be6 --- /dev/null +++ b/nodedb-sql/src/planner/dml_helpers/kv_counter.rs @@ -0,0 +1,272 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The row a KV counter atomic (`KV_INCR`, `KV_INCR_FLOAT`) creates for an +//! absent key. +//! +//! The KV engine stores the bytes it is handed and does not know the declared +//! columns. So the fresh row is planned here, from the catalog, by the same +//! code a `VALUES` insert runs: `INSERT (key, column) VALUES (key, 0)`. The +//! Control Plane encodes the cells into the template the engine fills in. + +use sqlparser::ast::{self, Expr, Value, ValueWithSpan}; +use sqlparser::tokenizer::Span; + +use super::kv_insert::build_kv_insert_plan; +use super::params::KvInsertParams; +use crate::catalog::SqlCatalog; +use crate::error::{Result, SqlError}; +use crate::types::*; + +/// The kind of number a counter atomic moves. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum KvCounterKind { + /// `KV_INCR` / `KV_DECR`: an integer column. + Integer, + /// `KV_INCR_FLOAT`: an integer or float column, the same columns it moves + /// in an existing row. + Float, +} + +/// The row a counter atomic creates for an absent key. +#[derive(Debug, Clone, PartialEq)] +pub enum KvCounterFreshRow { + /// The collection holds a single `value` column, or declares none: the + /// new value is stored as raw decimal text. + Raw, + /// The collection declares typed columns. + Typed { + /// The declared column the counter moves: the first column of `kind` + /// in name order, never the primary key. An existing row picks its + /// column by the same rule. `None` when there is none. + column: Option, + /// The cells the insert stores other than `column`: DEFAULTs, and the + /// primary key when it is a named column. + cells: Vec<(String, SqlValue)>, + }, +} + +/// Plan the row `INSERT (key, column) VALUES (key, 0)` stores in the KV +/// collection `info`, for a counter of `kind`. +pub fn plan_kv_counter_fresh_row( + info: &CollectionInfo, + key: &str, + kind: KvCounterKind, + catalog: &dyn SqlCatalog, +) -> Result { + if info.engine != EngineType::KeyValue { + return Err(SqlError::Unsupported { + detail: format!( + "KV counter on '{}', which is not a key-value collection", + info.name + ), + }); + } + let pk_col = info.primary_key.as_deref().unwrap_or("key"); + let value_columns: Vec<&ColumnInfo> = info + .columns + .iter() + .filter(|c| c.name != pk_col && c.name != "key" && c.name != "ttl") + .collect(); + if value_columns.is_empty() || (value_columns.len() == 1 && value_columns[0].name == "value") { + return Ok(KvCounterFreshRow::Raw); + } + + let column = value_columns + .iter() + .filter(|c| moves_as(kind, &c.data_type)) + .map(|c| c.name.clone()) + .min(); + let Some(column) = column else { + return Ok(KvCounterFreshRow::Typed { + column: None, + cells: Vec::new(), + }); + }; + + let placeholder = match kind { + KvCounterKind::Integer => "0", + KvCounterKind::Float => "0.0", + }; + let columns = [pk_col.to_string(), column.clone()]; + let row = ast::Parens::with_empty_span(vec![ + literal(Value::SingleQuotedString(key.to_string())), + literal(Value::Number(placeholder.to_string(), false)), + ]); + let plans = build_kv_insert_plan(KvInsertParams { + collection: info.name.clone(), + columns: &columns, + rows_ast: std::slice::from_ref(&row), + intent: KvInsertIntent::Put, + on_conflict_updates: Vec::new(), + pk_col: info.primary_key.as_deref(), + declared_columns: &info.columns, + catalog, + })?; + let cells = match plans.into_iter().next() { + Some(SqlPlan::KvInsert { mut entries, .. }) if entries.len() == 1 => { + let (_, cells) = entries.remove(0); + cells + .into_iter() + .filter(|(name, _)| *name != column) + .collect() + } + _ => { + return Err(SqlError::Unsupported { + detail: "KV counter fresh row did not plan as one KV insert".into(), + }); + } + }; + Ok(KvCounterFreshRow::Typed { + column: Some(column), + cells, + }) +} + +/// Whether a declared column of `data_type` is one a counter of `kind` +/// moves. +fn moves_as(kind: KvCounterKind, data_type: &SqlDataType) -> bool { + match kind { + KvCounterKind::Integer => matches!(data_type, SqlDataType::Int64), + KvCounterKind::Float => matches!(data_type, SqlDataType::Int64 | SqlDataType::Float64), + } +} + +fn literal(value: Value) -> Expr { + Expr::Value(ValueWithSpan { + value, + span: Span::empty(), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A catalog with no collections and no sequence state. The DEFAULTs + /// these cases declare are literals, so no accessor is reached. + struct NoCatalog; + + impl SqlCatalog for NoCatalog { + fn get_collection( + &self, + _database_id: nodedb_types::DatabaseId, + _name: &str, + ) -> std::result::Result, crate::catalog::SqlCatalogError> { + Ok(None) + } + } + + fn column(name: &str, data_type: SqlDataType, default: Option<&str>) -> ColumnInfo { + ColumnInfo { + name: name.to_string(), + data_type, + nullable: true, + is_primary_key: false, + default: default.map(str::to_string), + raw_type: None, + int_width: None, + float_width: None, + } + } + + fn kv_info(pk: &str, columns: Vec) -> CollectionInfo { + let mut key = column(pk, SqlDataType::String, None); + key.is_primary_key = true; + let mut all = vec![key]; + all.extend(columns); + CollectionInfo { + name: "c".into(), + engine: EngineType::KeyValue, + columns: all, + primary_key: Some(pk.into()), + has_auto_tier: false, + indexes: Vec::new(), + bitemporal: false, + primary: nodedb_types::PrimaryEngine::Document, + vector_primary: None, + partition_strategy: nodedb_types::PartitionStrategy::CollectionHomed, + open_schema: CollectionInfo::open_schema_for(EngineType::KeyValue), + } + } + + #[test] + fn a_single_value_column_is_raw() { + let info = kv_info("key", vec![column("value", SqlDataType::String, None)]); + let row = plan_kv_counter_fresh_row(&info, "k", KvCounterKind::Integer, &NoCatalog) + .expect("plan"); + assert_eq!(row, KvCounterFreshRow::Raw); + } + + #[test] + fn a_typed_collection_plans_the_insert_row_with_defaults() { + let info = kv_info( + "key", + vec![ + column("n", SqlDataType::Int64, None), + column("status", SqlDataType::String, Some("'new'")), + column("score", SqlDataType::Float64, None), + ], + ); + let row = plan_kv_counter_fresh_row(&info, "k", KvCounterKind::Integer, &NoCatalog) + .expect("plan"); + let KvCounterFreshRow::Typed { column, cells } = row else { + panic!("expected a typed row"); + }; + assert_eq!(column.as_deref(), Some("n")); + assert!( + cells + .iter() + .any(|(name, value)| name == "status" && *value == SqlValue::String("new".into())), + "{cells:?}" + ); + assert!( + cells.iter().all(|(name, _)| name != "n" && name != "key"), + "{cells:?}" + ); + } + + #[test] + fn a_named_primary_key_is_kept_in_the_row() { + let info = kv_info("id", vec![column("n", SqlDataType::Int64, None)]); + let row = plan_kv_counter_fresh_row(&info, "k1", KvCounterKind::Integer, &NoCatalog) + .expect("plan"); + let KvCounterFreshRow::Typed { cells, .. } = row else { + panic!("expected a typed row"); + }; + assert!( + cells + .iter() + .any(|(name, value)| name == "id" && *value == SqlValue::String("k1".into())), + "{cells:?}" + ); + } + + #[test] + fn counters_move_the_first_column_of_their_kind_in_name_order() { + let info = kv_info( + "key", + vec![ + column("b", SqlDataType::Int64, None), + column("a", SqlDataType::Float64, None), + column("label", SqlDataType::String, None), + ], + ); + let target = |kind| match plan_kv_counter_fresh_row(&info, "k", kind, &NoCatalog) { + Ok(KvCounterFreshRow::Typed { column, .. }) => column, + other => panic!("expected a typed row, got {other:?}"), + }; + assert_eq!(target(KvCounterKind::Integer).as_deref(), Some("b")); + assert_eq!(target(KvCounterKind::Float).as_deref(), Some("a")); + + let text_only = kv_info("key", vec![column("label", SqlDataType::String, None)]); + let row = plan_kv_counter_fresh_row(&text_only, "k", KvCounterKind::Integer, &NoCatalog) + .expect("plan"); + assert_eq!( + row, + KvCounterFreshRow::Typed { + column: None, + cells: Vec::new(), + } + ); + } +} diff --git a/nodedb-sql/src/planner/dml_helpers/mod.rs b/nodedb-sql/src/planner/dml_helpers/mod.rs index 853b27b09..99a930cfa 100644 --- a/nodedb-sql/src/planner/dml_helpers/mod.rs +++ b/nodedb-sql/src/planner/dml_helpers/mod.rs @@ -9,6 +9,7 @@ //! - [`vector_primary_insert`] — vector-primary collection insert plans //! - [`vector_primary_dml`] — vector-primary collection update / delete / truncate plans //! - [`kv_insert`] — KV engine insert plans +//! - [`kv_counter`] — the row a KV counter atomic creates for an absent key //! - [`insert_select_bind`] — `INSERT ... SELECT` target-column binding //! - [`params`] — parameter structs for the helpers above @@ -16,6 +17,7 @@ mod ast_extract; mod declared_defaults; mod insert_columns; mod insert_select_bind; +mod kv_counter; mod kv_insert; mod params; mod range_check; @@ -28,6 +30,7 @@ pub(super) use ast_extract::extract_table_name_from_table_with_joins; pub(super) use declared_defaults::materialize_defaults_in_rows; pub(super) use insert_columns::resolve_insert_columns; pub(super) use insert_select_bind::bind_insert_select_columns; +pub use kv_counter::{KvCounterFreshRow, KvCounterKind, plan_kv_counter_fresh_row}; pub(super) use kv_insert::build_kv_insert_plan; pub(super) use params::KvInsertParams; pub(super) use range_check::{ diff --git a/nodedb-types/src/error/ctors/from_wire.rs b/nodedb-types/src/error/ctors/from_wire.rs index 2cc5b13b1..67badbe63 100644 --- a/nodedb-types/src/error/ctors/from_wire.rs +++ b/nodedb-types/src/error/ctors/from_wire.rs @@ -49,6 +49,28 @@ impl NodeDbError { cause: None, } } + + /// Rebuild a typed error from a wire frame that also carries the + /// structured details the server held. + /// + /// The carried details are used when they belong to `code`, so a client + /// reads the collection, gate, or document the server named. Details of + /// another category, or none, fall back to [`NodeDbError::from_wire`]. + pub fn from_wire_with_details( + code: ErrorCode, + message: impl Into, + details: Option, + ) -> Self { + match details { + Some(details) if details.code() == code => Self { + code, + message: message.into(), + details, + cause: None, + }, + _ => Self::from_wire(code, message), + } + } } /// Map a numeric code onto the details variant that carries its category, @@ -114,6 +136,40 @@ mod tests { } } + #[test] + fn carried_details_keep_the_collection() { + let e = NodeDbError::from_wire_with_details( + ErrorCode::OVERFLOW, + "increment or decrement would overflow on counters", + Some(ErrorDetails::Overflow { + collection: "counters".into(), + }), + ); + assert_eq!( + e.details(), + &ErrorDetails::Overflow { + collection: "counters".into() + } + ); + } + + #[test] + fn details_of_another_category_fall_back_to_the_code() { + let e = NodeDbError::from_wire_with_details( + ErrorCode::OVERFLOW, + "overflow", + Some(ErrorDetails::TypeMismatch { + collection: "counters".into(), + }), + ); + assert_eq!( + e.details(), + &ErrorDetails::Overflow { + collection: String::new() + } + ); + } + #[test] fn genuinely_unmapped_codes_still_fall_back_to_internal() { assert!(NodeDbError::from_wire(ErrorCode(65000), "x").is_internal()); diff --git a/nodedb-types/src/error/ctors/write_path.rs b/nodedb-types/src/error/ctors/write_path.rs index 0970ccacd..309b1e3cd 100644 --- a/nodedb-types/src/error/ctors/write_path.rs +++ b/nodedb-types/src/error/ctors/write_path.rs @@ -203,6 +203,36 @@ impl NodeDbError { } } + /// A KV counter atomic (`INCR`, `INCRBYFLOAT`) refused on `collection`. + /// + /// `fault` is the client text, e.g. `value is not an integer or out of + /// range`. The message is `"{fault} on {collection}"`, the text the SQL + /// surfaces send. `out_of_range` picks the class: `OVERFLOW` for a result + /// out of range, `TYPE_MISMATCH` for a stored value that does not parse. + pub fn kv_counter_fault( + collection: impl Into, + fault: impl fmt::Display, + out_of_range: bool, + ) -> Self { + let collection = collection.into(); + let message = format!("{fault} on {collection}"); + if out_of_range { + Self { + code: ErrorCode::OVERFLOW, + message, + details: ErrorDetails::Overflow { collection }, + cause: None, + } + } else { + Self { + code: ErrorCode::TYPE_MISMATCH, + message, + details: ErrorDetails::TypeMismatch { collection }, + cause: None, + } + } + } + pub fn insufficient_balance(collection: impl Into, detail: impl fmt::Display) -> Self { let collection = collection.into(); Self { diff --git a/nodedb-types/src/error/sqlstate.rs b/nodedb-types/src/error/sqlstate.rs index cc8a8899c..a3287e26b 100644 --- a/nodedb-types/src/error/sqlstate.rs +++ b/nodedb-types/src/error/sqlstate.rs @@ -60,6 +60,10 @@ pub const DIVISION_BY_ZERO: &str = "22012"; /// negative, fractional, non-numeric, or wider than `usize`) pub const INVALID_LIMIT_VALUE: &str = "2201W"; +/// `22P02` — `invalid_text_representation` (stored text that does not +/// parse as the number an operation reads, e.g. `INCR` on `"abc"`) +pub const INVALID_TEXT_REPRESENTATION: &str = "22P02"; + /// `22023` — `invalid_parameter_value` (a `SET` value the parameter's own /// grammar refuses, e.g. `SET statement_timeout = 'later'`) pub const INVALID_PARAMETER_VALUE: &str = "22023"; @@ -346,6 +350,7 @@ mod tests { NUMERIC_VALUE_OUT_OF_RANGE, DIVISION_BY_ZERO, INVALID_LIMIT_VALUE, + INVALID_TEXT_REPRESENTATION, INTEGRITY_CONSTRAINT_VIOLATION, NOT_NULL_VIOLATION, FOREIGN_KEY_VIOLATION, diff --git a/nodedb-types/src/protocol/frames.rs b/nodedb-types/src/protocol/frames.rs index 233cd1cae..2c89de52d 100644 --- a/nodedb-types/src/protocol/frames.rs +++ b/nodedb-types/src/protocol/frames.rs @@ -125,6 +125,11 @@ pub struct ErrorPayload { #[serde(default, skip_serializing_if = "is_zero")] #[msgpack(default)] pub ndb_code: u16, + /// The structured details the server held: the collection, gate, or + /// document the error names. `None` when the server had none to send. + #[serde(default, skip_serializing_if = "Option::is_none")] + #[msgpack(default)] + pub details: Option, } /// `skip_serializing_if` predicate for [`ErrorPayload::ndb_code`]: zero is the @@ -196,12 +201,22 @@ impl NativeResponse { code: code.into(), message: message.into(), ndb_code, + details: None, }), auth: None, warnings: Vec::new(), } } + /// Attach the structured error details to an error response. A response + /// with no error payload is returned unchanged. + pub fn with_error_details(mut self, details: crate::error::ErrorDetails) -> Self { + if let Some(payload) = self.error.as_mut() { + payload.details = Some(details); + } + self + } + /// Create an auth success response. pub fn auth_ok(seq: u64, username: String, tenant_id: u64) -> Self { Self { @@ -331,6 +346,19 @@ mod tests { assert_eq!(payload.ndb_code, 1000); } + #[test] + fn error_payload_round_trips_the_details() { + let details = crate::error::ErrorDetails::Overflow { + collection: "counters".into(), + }; + let frame = NativeResponse::error_with_code(7, "22003", "overflow on counters", 1021) + .with_error_details(details.clone()); + let bytes = zerompk::to_msgpack_vec(&frame).expect("encode"); + let decoded: NativeResponse = zerompk::from_msgpack(&bytes).expect("decode"); + let payload = decoded.error.expect("error payload survives the wire"); + assert_eq!(payload.details, Some(details)); + } + #[test] fn error_payload_without_numeric_code_still_decodes() { // Hand-rolled 2-key map: exactly what a peer built before the numeric diff --git a/nodedb-types/src/protocol/text_fields/decode.rs b/nodedb-types/src/protocol/text_fields/decode.rs index 3e56301bd..93d6a4b76 100644 --- a/nodedb-types/src/protocol/text_fields/decode.rs +++ b/nodedb-types/src/protocol/text_fields/decode.rs @@ -234,7 +234,7 @@ impl<'a> zerompk::FromMessagePack<'a> for TextFields { out.incr_delta = Some(reader.read_i64()?); } FID_INCR_FLOAT_DELTA => { - out.incr_float_delta = Some(reader.read_f64()?); + out.incr_float_delta = Some(reader.read_string()?.into_owned()); } FID_EXPECTED => { out.expected = Some(reader.read_binary()?.into_owned()); diff --git a/nodedb-types/src/protocol/text_fields/types/text_fields.rs b/nodedb-types/src/protocol/text_fields/types/text_fields.rs index e129e2550..c789e8c63 100644 --- a/nodedb-types/src/protocol/text_fields/types/text_fields.rs +++ b/nodedb-types/src/protocol/text_fields/types/text_fields.rs @@ -171,9 +171,10 @@ pub struct TextFields { /// Integer delta for KvIncr. #[serde(skip_serializing_if = "Option::is_none")] pub incr_delta: Option, - /// Float delta for KvIncrFloat. + /// Delta for KvIncrFloat, as the client's decimal text, so no digit is + /// lost to an `f64` on the wire. #[serde(skip_serializing_if = "Option::is_none")] - pub incr_float_delta: Option, + pub incr_float_delta: Option, /// Expected value for KvCas. #[serde(skip_serializing_if = "Option::is_none")] pub expected: Option>, diff --git a/nodedb/src/bridge/envelope/counter_fault.rs b/nodedb/src/bridge/envelope/counter_fault.rs new file mode 100644 index 000000000..3bf5a2d18 --- /dev/null +++ b/nodedb/src/bridge/envelope/counter_fault.rs @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Why a KV counter atomic (`INCR`, `INCRBY`, `DECR`, `INCRBYFLOAT`) +//! computed no value. + +/// Why a KV counter atomic computed no value. +/// +/// Each fault has one client message, the text Redis answers with for the +/// same condition. RESP sends it after `ERR`. The SQL surfaces send it with +/// the SQLSTATE [`CounterFault::sqlstate`] names. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CounterFault { + /// The stored value is not a decimal integer in the `i64` range. + NotAnInteger, + /// The stored value is not a decimal float. + NotAFloat, + /// The integer result is outside the `i64` range. + IntegerOverflow, + /// The float result is NaN or infinite. + NonFinite, +} + +impl CounterFault { + /// The client message, without the RESP `ERR` prefix. + pub fn message(self) -> &'static str { + match self { + Self::NotAnInteger => "value is not an integer or out of range", + Self::NotAFloat => "value is not a valid float", + Self::IntegerOverflow => "increment or decrement would overflow", + Self::NonFinite => "increment would produce NaN or Infinity", + } + } + + /// The SQLSTATE for the fault: `22P02` for a stored value that does not + /// parse, `22003` for a result out of range. + pub fn sqlstate(self) -> &'static str { + use nodedb_types::error::sqlstate; + match self { + Self::NotAnInteger | Self::NotAFloat => sqlstate::INVALID_TEXT_REPRESENTATION, + Self::IntegerOverflow | Self::NonFinite => sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parse_faults_answer_invalid_text_representation() { + assert_eq!(CounterFault::NotAnInteger.sqlstate(), "22P02"); + assert_eq!(CounterFault::NotAFloat.sqlstate(), "22P02"); + } + + #[test] + fn range_faults_answer_numeric_value_out_of_range() { + assert_eq!(CounterFault::IntegerOverflow.sqlstate(), "22003"); + assert_eq!(CounterFault::NonFinite.sqlstate(), "22003"); + } +} diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index 66f706df7..92b75f6c5 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -79,8 +79,12 @@ pub enum ErrorCode { TypeGuardViolation { collection: String, detail: String }, /// Value type does not match expected type for operation (e.g. INCR on a string). TypeMismatch { collection: String, detail: String }, - /// Arithmetic overflow (e.g. i64::MAX + 1 on INCR). - OverflowError { collection: String }, + /// A KV counter atomic read a stored value it cannot parse as a + /// number, or computed a result out of range. + CounterFault { + collection: String, + fault: super::CounterFault, + }, /// Insufficient balance for transfer (source lacks required amount). InsufficientBalance { collection: String, detail: String }, /// Rate limit exceeded for a rate gate / cooldown. @@ -210,7 +214,6 @@ impl From for ErrorCode { crate::Error::TypeMismatch { collection, detail, .. } => Self::TypeMismatch { collection, detail }, - crate::Error::OverflowError { collection, .. } => Self::OverflowError { collection }, crate::Error::InsufficientBalance { collection, detail, .. } => Self::InsufficientBalance { collection, detail }, diff --git a/nodedb/src/bridge/envelope/mod.rs b/nodedb/src/bridge/envelope/mod.rs index f8b35d6ac..ec09e4a9a 100644 --- a/nodedb/src/bridge/envelope/mod.rs +++ b/nodedb/src/bridge/envelope/mod.rs @@ -2,12 +2,14 @@ //! Request/response envelopes exchanged over the SPSC bridge. +pub mod counter_fault; pub mod error_code; pub mod payload; pub mod request; pub mod response; pub mod status; +pub use counter_fault::CounterFault; pub use error_code::ErrorCode; pub use nodedb_physical::physical_plan::PhysicalPlan; pub use payload::Payload; diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 85b577301..e29189546 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -8,9 +8,9 @@ //! fails to compile here until it is mirrored on the wire instead of silently //! degrading to `Internal` and losing its SQLSTATE at the coordinator. -use nodedb_cluster::rpc_codec::{DataPlaneErrorCode, TypedClusterError}; +use nodedb_cluster::rpc_codec::{DataPlaneCounterFault, DataPlaneErrorCode, TypedClusterError}; -use crate::bridge::envelope::ErrorCode; +use crate::bridge::envelope::{CounterFault, ErrorCode}; /// Map a local-execution [`crate::Error`] to the wire error a remote caller /// receives. @@ -121,7 +121,10 @@ impl From for DataPlaneErrorCode { ErrorCode::TypeMismatch { collection, detail } => { Self::TypeMismatch { collection, detail } } - ErrorCode::OverflowError { collection } => Self::OverflowError { collection }, + ErrorCode::CounterFault { collection, fault } => Self::CounterFault { + collection, + fault: fault.into(), + }, ErrorCode::InsufficientBalance { collection, detail } => { Self::InsufficientBalance { collection, detail } } @@ -219,7 +222,10 @@ impl From for ErrorCode { DataPlaneErrorCode::TypeMismatch { collection, detail } => { Self::TypeMismatch { collection, detail } } - DataPlaneErrorCode::OverflowError { collection } => Self::OverflowError { collection }, + DataPlaneErrorCode::CounterFault { collection, fault } => Self::CounterFault { + collection, + fault: fault.into(), + }, DataPlaneErrorCode::InsufficientBalance { collection, detail } => { Self::InsufficientBalance { collection, detail } } @@ -262,6 +268,28 @@ impl From for ErrorCode { } } +impl From for DataPlaneCounterFault { + fn from(fault: CounterFault) -> Self { + match fault { + CounterFault::NotAnInteger => Self::NotAnInteger, + CounterFault::NotAFloat => Self::NotAFloat, + CounterFault::IntegerOverflow => Self::IntegerOverflow, + CounterFault::NonFinite => Self::NonFinite, + } + } +} + +impl From for CounterFault { + fn from(fault: DataPlaneCounterFault) -> Self { + match fault { + DataPlaneCounterFault::NotAnInteger => Self::NotAnInteger, + DataPlaneCounterFault::NotAFloat => Self::NotAFloat, + DataPlaneCounterFault::IntegerOverflow => Self::IntegerOverflow, + DataPlaneCounterFault::NonFinite => Self::NonFinite, + } + } +} + #[cfg(test)] mod tests { use super::*; @@ -324,6 +352,23 @@ mod tests { assert_eq!(ErrorCode::from(wire), original); } + #[test] + fn counter_fault_roundtrips_verbatim() { + for fault in [ + CounterFault::NotAnInteger, + CounterFault::NotAFloat, + CounterFault::IntegerOverflow, + CounterFault::NonFinite, + ] { + let original = ErrorCode::CounterFault { + collection: "counters".into(), + fault, + }; + let wire = DataPlaneErrorCode::from(original.clone()); + assert_eq!(ErrorCode::from(wire), original); + } + } + #[test] fn counted_code_roundtrips_across_the_u64_wire_field() { let original = ErrorCode::RecursionDepthExceeded { diff --git a/nodedb/src/control/gateway/error_map/resp.rs b/nodedb/src/control/gateway/error_map/resp.rs index 1bb64df9d..5eb4ae095 100644 --- a/nodedb/src/control/gateway/error_map/resp.rs +++ b/nodedb/src/control/gateway/error_map/resp.rs @@ -31,6 +31,13 @@ impl GatewayErrorMap { Error::RemoteTyped { code, message } => { format!("{} {message}", remote_code_to_resp_prefix(*code)) } + // A counter fault answers with the exact reply Redis gives for + // the same condition. + Error::DataPlane(crate::bridge::envelope::ErrorCode::CounterFault { + fault, .. + }) => { + format!("ERR {}", fault.message()) + } Error::DataPlane(_) => { let public = crate::error_classify::classify(err); format!( @@ -90,6 +97,33 @@ mod tests { assert!(msg.starts_with("WRONGTYPE "), "{msg}"); } + #[test] + fn resp_counter_faults_use_the_redis_error_text() { + use crate::bridge::envelope::{CounterFault, ErrorCode}; + let cases = [ + ( + CounterFault::NotAnInteger, + "ERR value is not an integer or out of range", + ), + (CounterFault::NotAFloat, "ERR value is not a valid float"), + ( + CounterFault::IntegerOverflow, + "ERR increment or decrement would overflow", + ), + ( + CounterFault::NonFinite, + "ERR increment would produce NaN or Infinity", + ), + ]; + for (fault, expected) in cases { + let err = Error::DataPlane(ErrorCode::CounterFault { + collection: "counters".into(), + fault, + }); + assert_eq!(GatewayErrorMap::to_resp(&err), expected); + } + } + #[test] fn to_resp_remote_typed_is_wired_to_helper() { use nodedb_types::error::ErrorCode; diff --git a/nodedb/src/control/planner/calvin/tx_class/shared.rs b/nodedb/src/control/planner/calvin/tx_class/shared.rs index dcce00baf..52072e479 100644 --- a/nodedb/src/control/planner/calvin/tx_class/shared.rs +++ b/nodedb/src/control/planner/calvin/tx_class/shared.rs @@ -414,6 +414,7 @@ mod lockstep_tests { ttl_ms: 0, surrogate: Surrogate::new(3), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, })); } diff --git a/nodedb/src/control/planner/calvin/write_class.rs b/nodedb/src/control/planner/calvin/write_class.rs index 5aa7c3f65..d5a9940d8 100644 --- a/nodedb/src/control/planner/calvin/write_class.rs +++ b/nodedb/src/control/planner/calvin/write_class.rs @@ -481,6 +481,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); assert!(is_write_plan(&plan), "KvOp::Incr must be a write"); } @@ -490,9 +491,10 @@ mod tests { let plan = PhysicalPlan::Kv(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), key: b"k".to_vec(), - delta: 1.5, + delta: "1.5".into(), surrogate: Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); assert!(is_write_plan(&plan), "KvOp::IncrFloat must be a write"); } diff --git a/nodedb/src/control/planner/catalog_adapter/adapter.rs b/nodedb/src/control/planner/catalog_adapter/adapter.rs index 73f120e4b..6a80c1489 100644 --- a/nodedb/src/control/planner/catalog_adapter/adapter.rs +++ b/nodedb/src/control/planner/catalog_adapter/adapter.rs @@ -122,6 +122,17 @@ impl OriginCatalog { } } + /// Bind the node-wide sequence counters, so a column DEFAULT that calls + /// `nextval` can allocate. For a planner built with [`OriginCatalog::new`] + /// that plans rows outside a pgwire session. + pub fn with_sequence_registry( + mut self, + registry: Arc, + ) -> Self { + self.sequence_registry = Some(registry); + self + } + /// Bind the calling session's `currval` map to this adapter. pub fn with_session_sequences( mut self, diff --git a/nodedb/src/control/planner/rls_injection/kv.rs b/nodedb/src/control/planner/rls_injection/kv.rs index fdc89f98e..5091d6335 100644 --- a/nodedb/src/control/planner/rls_injection/kv.rs +++ b/nodedb/src/control/planner/rls_injection/kv.rs @@ -518,6 +518,7 @@ mod tests { ttl_ms: 0, surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); assert!(inject(&mut plan, &store).is_ok()); assert!(write_check(&plan).has_predicate()); diff --git a/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs b/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs new file mode 100644 index 000000000..4fe129c2b --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Resolve the [`KvCounterShape`] a KV counter op carries: the row an absent +//! key becomes. + +use std::sync::Arc; + +use nodedb_physical::physical_plan::KvCounterShape; +use nodedb_sql::SqlCatalog; +use nodedb_sql::planner::dml_helpers::{ + KvCounterFreshRow, KvCounterKind, plan_kv_counter_fresh_row, +}; + +use super::value::{write_msgpack_map_header, write_msgpack_str, write_msgpack_value}; +use crate::control::planner::catalog_adapter::OriginCatalog; +use crate::control::planner::plan_error_map::map_plan_error; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; + +/// The shape a SQL counter op on `collection` gives an absent `key`. +/// +/// A typed collection creates the row `INSERT (key, column) VALUES (key, n)` +/// stores, DEFAULTs included. A raw collection, or one the catalog does not +/// hold, stores decimal text. +pub(crate) fn kv_counter_shape( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + key: &str, + kind: KvCounterKind, +) -> crate::Result { + let catalog = OriginCatalog::new( + Arc::clone(&state.credentials), + tenant_id.as_u64(), + database_id, + Some(Arc::clone(&state.retention_policy_registry)), + ) + .with_sequence_registry(Arc::clone(&state.sequence_registry)); + let info = catalog + .get_collection(database_id, collection) + .map_err(|e| map_plan_error(e.into(), tenant_id))?; + let Some(info) = info else { + return Ok(KvCounterShape::Raw); + }; + let fresh = plan_kv_counter_fresh_row(&info, key, kind, &catalog) + .map_err(|e| map_plan_error(e, tenant_id))?; + Ok(match fresh { + KvCounterFreshRow::Raw => KvCounterShape::Raw, + KvCounterFreshRow::Typed { column, cells } => { + let mut template = Vec::with_capacity(cells.len() * 32); + write_msgpack_map_header(&mut template, cells.len()); + for (name, value) in &cells { + write_msgpack_str(&mut template, name); + write_msgpack_value(&mut template, value); + } + KvCounterShape::Typed { column, template } + } + }) +} diff --git a/nodedb/src/control/planner/sql_plan_convert/mod.rs b/nodedb/src/control/planner/sql_plan_convert/mod.rs index c08ece90b..db0b41a2a 100644 --- a/nodedb/src/control/planner/sql_plan_convert/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/mod.rs @@ -12,6 +12,7 @@ pub mod expr; pub mod filter; pub mod filter_scan_side; pub mod group_key_name; +pub mod kv_counter_shape; pub mod lateral; pub mod output_schema; pub mod output_schema_types; diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index dac86af8a..bea98634f 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -161,7 +161,7 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { | ErrorCode::TransitionCheckViolation { .. } | ErrorCode::TypeGuardViolation { .. } | ErrorCode::TypeMismatch { .. } - | ErrorCode::OverflowError { .. } + | ErrorCode::CounterFault { .. } | ErrorCode::InsufficientBalance { .. } // Admission verdicts: the request never reached the mutation at all. | ErrorCode::RateExceeded { .. } diff --git a/nodedb/src/control/server/native/dispatch/conversion.rs b/nodedb/src/control/server/native/dispatch/conversion.rs index 9e8e4649c..baa151254 100644 --- a/nodedb/src/control/server/native/dispatch/conversion.rs +++ b/nodedb/src/control/server/native/dispatch/conversion.rs @@ -148,6 +148,7 @@ pub(crate) fn error_code_to_native( let (_, sqlstate, message) = error_code_to_sqlstate(code); let public = nodedb_types::NodeDbError::from(crate::Error::DataPlane(code.clone())); NativeResponse::error_with_code(seq, sqlstate, message, public.code().0) + .with_error_details(public.details().clone()) } /// Encode a protocol-neutral DDL dispatch result into a single @@ -174,7 +175,14 @@ pub(crate) fn ddl_result_to_native( sqlstate, code, message, - }) => NativeResponse::error_with_code(seq, sqlstate, message, code.0), + details, + }) => { + let frame = NativeResponse::error_with_code(seq, sqlstate, message, code.0); + match details { + Some(details) => frame.with_error_details(*details), + None => frame, + } + } // Unknown pgwire response variants are dropped during translation, so // the first element is the first meaningful result — the bridge // returns on the first known variant. diff --git a/nodedb/src/control/server/native/dispatch/direct_ops.rs b/nodedb/src/control/server/native/dispatch/direct_ops.rs index 73d7d7e86..a8585200f 100644 --- a/nodedb/src/control/server/native/dispatch/direct_ops.rs +++ b/nodedb/src/control/server/native/dispatch/direct_ops.rs @@ -37,9 +37,10 @@ pub(crate) async fn handle_direct_op( let vshard_id = ctx.vshard_for_key(vshard_key); let tenant_id = ctx.tenant_id(); - // CRDT Apply allocates a surrogate while planning; authorize the exact - // collection first. - if matches!(op, OpCode::CrdtApply) { + // CRDT Apply allocates a surrogate while planning, and a KV counter plans + // its fresh row from the catalog, which can evaluate a DEFAULT. Authorize + // the exact collection first. + if matches!(op, OpCode::CrdtApply | OpCode::KvIncr | OpCode::KvIncrFloat) { let audit = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( &ctx.state.audit, )); diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs b/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs index fc2b5b2bb..bc5b4ecbf 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs @@ -8,7 +8,9 @@ use nodedb_types::protocol::{OpCode, TextFields}; use crate::bridge::envelope::PhysicalPlan; use super::super::DispatchCtx; -use super::{columnar, crdt, document, graph, kv, query, spatial, text, timeseries, vector}; +use super::{ + columnar, crdt, document, graph, kv, kv_counter, query, spatial, text, timeseries, vector, +}; /// Build a PhysicalPlan from an opcode and request fields. pub(crate) fn build_plan( @@ -84,8 +86,8 @@ pub(crate) fn build_plan( OpCode::KvDropIndex => kv::build_drop_index(ctx, fields, collection), OpCode::KvTruncate => kv::build_truncate(ctx, collection), // KV atomic operations. - OpCode::KvIncr => kv::build_incr(ctx, collection, fields), - OpCode::KvIncrFloat => kv::build_incr_float(ctx, collection, fields), + OpCode::KvIncr => kv_counter::build_incr(ctx, collection, fields), + OpCode::KvIncrFloat => kv_counter::build_incr_float(ctx, collection, fields), OpCode::KvCas => kv::build_cas(ctx, collection, fields), OpCode::KvGetSet => kv::build_getset(ctx, collection, fields), // KV sorted index operations. diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index c0c0282fb..720edea80 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -262,7 +262,7 @@ pub(crate) fn build_truncate( /// Resolve the stable cross-engine surrogate for a KV atomic op, content- /// addressed on `(collection, key)` — the same binding a normal insert of that /// key allocated, so an atomic op on an existing key keeps its identity. -fn assign_kv_surrogate( +pub(super) fn assign_kv_surrogate( ctx: &DispatchCtx<'_>, collection: &str, key: &[u8], @@ -272,54 +272,6 @@ fn assign_kv_surrogate( .assign(ctx.database_id(), ctx.tenant_id(), collection, key) } -pub(crate) fn build_incr( - ctx: &DispatchCtx<'_>, - collection: &str, - fields: &TextFields, -) -> crate::Result { - let key = fields - .key - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'key'".to_string(), - })?; - let delta = fields.incr_delta.unwrap_or(1); - let ttl_ms = fields.ttl_ms.unwrap_or(0); - let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; - - Ok(PhysicalPlan::Kv(KvOp::Incr { - collection: QualifiedCollection::new(ctx.database_id(), collection), - key: key.as_bytes().to_vec(), - delta, - ttl_ms, - surrogate, - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - })) -} - -pub(crate) fn build_incr_float( - ctx: &DispatchCtx<'_>, - collection: &str, - fields: &TextFields, -) -> crate::Result { - let key = fields - .key - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'key'".to_string(), - })?; - let delta = fields.incr_float_delta.unwrap_or(1.0); - let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; - - Ok(PhysicalPlan::Kv(KvOp::IncrFloat { - collection: QualifiedCollection::new(ctx.database_id(), collection), - key: key.as_bytes().to_vec(), - delta, - surrogate, - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - })) -} - pub(crate) fn build_cas( ctx: &DispatchCtx<'_>, collection: &str, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs new file mode 100644 index 000000000..dabf6960f --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs @@ -0,0 +1,86 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native `KvIncr` / `KvIncrFloat` plan builders. + +use nodedb_physical::physical_plan::KvOp; +use nodedb_sql::planner::dml_helpers::KvCounterKind; +use nodedb_types::QualifiedCollection; +use nodedb_types::protocol::TextFields; + +use super::kv::assign_kv_surrogate; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::planner::sql_plan_convert::kv_counter_shape::kv_counter_shape; +use crate::control::server::native::dispatch::DispatchCtx; + +fn required_key(fields: &TextFields) -> crate::Result<&str> { + fields + .key + .as_deref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'key'".to_string(), + }) +} + +pub(crate) fn build_incr( + ctx: &DispatchCtx<'_>, + collection: &str, + fields: &TextFields, +) -> crate::Result { + let key = required_key(fields)?; + let delta = fields.incr_delta.unwrap_or(1); + let ttl_ms = fields.ttl_ms.unwrap_or(0); + let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; + // An absent key takes the collection's shape, as a SQL `KV_INCR` does. + let shape = kv_counter_shape( + ctx.state, + ctx.tenant_id(), + ctx.database_id(), + collection, + key, + KvCounterKind::Integer, + )?; + + Ok(PhysicalPlan::Kv(KvOp::Incr { + collection: QualifiedCollection::new(ctx.database_id(), collection), + key: key.as_bytes().to_vec(), + delta, + ttl_ms, + surrogate, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape, + })) +} + +pub(crate) fn build_incr_float( + ctx: &DispatchCtx<'_>, + collection: &str, + fields: &TextFields, +) -> crate::Result { + let key = required_key(fields)?; + // The delta stays the client's decimal text, so the engine adds every + // digit the client sent. + let delta = fields.incr_float_delta.as_deref().unwrap_or("1"); + if !crate::engine::kv::float_text::is_decimal_number(delta) { + return Err(crate::Error::BadRequest { + detail: format!("KvIncrFloat: delta must be a decimal number, got '{delta}'"), + }); + } + let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; + let shape = kv_counter_shape( + ctx.state, + ctx.tenant_id(), + ctx.database_id(), + collection, + key, + KvCounterKind::Float, + )?; + + Ok(PhysicalPlan::Kv(KvOp::IncrFloat { + collection: QualifiedCollection::new(ctx.database_id(), collection), + key: key.as_bytes().to_vec(), + delta: delta.to_string(), + surrogate, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape, + })) +} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs b/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs index e1d342a31..61b4cf4e0 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs @@ -12,6 +12,7 @@ pub(crate) mod document; pub(crate) mod graph; mod helpers; pub(crate) mod kv; +pub(crate) mod kv_counter; pub(crate) mod query; pub(crate) mod spatial; pub(crate) mod text; diff --git a/nodedb/src/control/server/pgwire/ddl_encode.rs b/nodedb/src/control/server/pgwire/ddl_encode.rs index ea11f4194..29dd1992c 100644 --- a/nodedb/src/control/server/pgwire/ddl_encode.rs +++ b/nodedb/src/control/server/pgwire/ddl_encode.rs @@ -43,6 +43,7 @@ pub fn ddl_results_to_pgwire( sqlstate, code, message, + .. }) => { let mut info = ErrorInfo::new("ERROR".to_owned(), sqlstate, message); info.routine = Some(code.to_string()); diff --git a/nodedb/src/control/server/resp/handler_kv/counters.rs b/nodedb/src/control/server/resp/handler_kv/counters.rs index 45444f7c9..4ea2d7916 100644 --- a/nodedb/src/control/server/resp/handler_kv/counters.rs +++ b/nodedb/src/control/server/resp/handler_kv/counters.rs @@ -4,7 +4,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::state::SharedState; -use nodedb_physical::physical_plan::KvOp; +use nodedb_physical::physical_plan::{KvCounterShape, KvOp}; use nodedb_types::{DatabaseId, QualifiedCollection}; use super::super::codec::RespValue; @@ -105,9 +105,12 @@ async fn dispatch_incr( surrogate, // Filled by the RLS injection pass `dispatch_kv_write` runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // RESP treats every value as a byte string, as `SET` and `GET` do, so + // an absent key starts as decimal text in any collection. + shape: KvCounterShape::Raw, }); - match dispatch_kv_write(state, session, plan).await { + match dispatch_counter(state, session, plan).await { Ok(resp) => match payload_field_i64(&resp.payload, "value") { Some(new_val) => RespValue::integer(new_val), // The counter did change; a response we cannot read means we do @@ -130,13 +133,11 @@ pub(in crate::control::server::resp) async fn handle_incrbyfloat( } let key = cmd.args[0].clone(); - let delta_str = match cmd.arg_str(1) { - Some(s) => s, - None => return RespValue::err("ERR value is not a valid float"), - }; - let delta: f64 = match delta_str.parse() { - Ok(v) => v, - Err(_) => return RespValue::err("ERR value is not a valid float"), + // The delta stays the client's decimal text, so the engine adds every + // digit the client sent. + let delta = match cmd.arg_str(1) { + Some(s) if crate::engine::kv::float_text::is_decimal_number(s) => s.to_string(), + _ => return RespValue::err("ERR value is not a valid float"), }; if let Some(refusal) = refuse_if_counter_is_redacted(state, session) { @@ -154,16 +155,40 @@ pub(in crate::control::server::resp) async fn handle_incrbyfloat( surrogate, // Filled by the RLS injection pass `dispatch_kv_write` runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // See `dispatch_incr`. + shape: KvCounterShape::Raw, }); - match dispatch_kv_write(state, session, plan).await { + match dispatch_counter(state, session, plan).await { + // The reply is a bulk string. A raw body answers with the exact text + // it stored, the bytes `GET` returns. A typed row answers with the + // column's new number. Ok(resp) => { - // Return the new value as a bulk string (Redis convention). - match payload_json(&resp.payload).get("value") { - Some(v) => RespValue::bulk(v.to_string().into_bytes()), + let reply = payload_json(&resp.payload); + let text = match reply.get("text").and_then(serde_json::Value::as_str) { + Some(text) => Some(text.to_string()), + None => reply + .get("value") + .and_then(serde_json::Value::as_f64) + .map(|v| v.to_string()), + }; + match text { + Some(text) => RespValue::bulk(text.into_bytes()), None => RespValue::err("ERR counter response could not be decoded"), } } Err(e) => RespValue::from_error(&e), } } + +/// Dispatch a counter write and turn a Data-Plane error status into a typed +/// error, so a counter fault reaches the client as its own message. +async fn dispatch_counter( + state: &SharedState, + session: &RespSession, + plan: PhysicalPlan, +) -> crate::Result { + let resp = dispatch_kv_write(state, session, plan).await?; + crate::control::server::dispatch_utils::reject_data_plane_error(&resp)?; + Ok(resp) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs index 2316c3aa1..790b6b5a9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs @@ -167,11 +167,15 @@ pub(crate) async fn dispatch_and_respond( /// bug in whatever produced it; it is surfaced as an internal error rather than /// silently downgraded to success. fn data_plane_error(code: Option) -> DdlError { - let (_, sqlstate, message) = match code { - Some(code) => crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code), - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), + let Some(code) = code else { + return ddl_err("XX000", "unknown data plane error"); }; - ddl_err(sqlstate, message) + let (_, sqlstate, message) = + crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code); + // The public error carries the code a client classifies by and the + // details naming the collection, which the SQLSTATE alone cannot give. + let public = nodedb_types::NodeDbError::from(crate::Error::DataPlane(code)); + DdlError::from_public(sqlstate, message, &public) } /// Build a single-text-column row set carrying `text` under `col`. diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs index 46bea8487..eb449c407 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs @@ -11,7 +11,8 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; use crate::types::{DatabaseId, VShardId}; -use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; +use nodedb_physical::physical_plan::{KvCounterShape, KvOp, PhysicalPlan}; +use nodedb_sql::planner::dml_helpers::KvCounterKind; use super::super::super::result::{DdlError, DdlResult}; use super::dispatch::{ @@ -61,6 +62,7 @@ pub async fn kv_incr( key.as_bytes(), ) .map_err(|e| ddl_err("XX000", e.to_string()))?; + let shape = counter_shape(state, identity, &collection, &key, KvCounterKind::Integer)?; let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), @@ -70,6 +72,7 @@ pub async fn kv_incr( // Filled by `dispatch_and_respond`, which runs the same RLS injection // pass the planner-driven path runs. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape, }); dispatch_and_respond( @@ -104,12 +107,18 @@ pub async fn kv_incr_float( let collection = unquote(&args[0]).to_lowercase(); let key = unquote(&args[1]); - let delta: f64 = args[2].trim().parse().map_err(|_| { - ddl_err( + // The delta stays the client's decimal text, so the engine adds every + // digit the client wrote. + let delta = unquote(&args[2]).trim().to_string(); + if !crate::engine::kv::float_text::is_decimal_number(&delta) { + return Err(ddl_err( "42601", - format!("KV_INCR_FLOAT: delta must be a float, got '{}'", args[2]), - ) - })?; + format!( + "KV_INCR_FLOAT: delta must be a decimal number, got '{}'", + args[2] + ), + )); + } let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); let surrogate = state @@ -121,6 +130,7 @@ pub async fn kv_incr_float( key.as_bytes(), ) .map_err(|e| ddl_err("XX000", e.to_string()))?; + let shape = counter_shape(state, identity, &collection, &key, KvCounterKind::Float)?; let plan = PhysicalPlan::Kv(KvOp::IncrFloat { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), @@ -128,6 +138,7 @@ pub async fn kv_incr_float( surrogate, // Filled by `dispatch_and_respond` — see `kv_incr`. rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape, }); dispatch_and_respond( @@ -142,6 +153,43 @@ pub async fn kv_incr_float( .await } +/// The row an absent `key` in `collection` becomes, planned from the catalog. +/// +/// The caller's grants are checked first, the same pair `dispatch_and_respond` +/// checks: planning the row reads the catalog and can evaluate a DEFAULT, so +/// a caller refused the collection must be refused before either happens. +fn counter_shape( + state: &SharedState, + identity: &AuthenticatedIdentity, + collection: &str, + key: &str, + kind: KvCounterKind, +) -> Result { + let gate = super::super::read_gate::CollectionReadGate::for_request( + state, + identity, + DatabaseId::DEFAULT, + ); + gate.authorize(collection)?; + gate.authorize_permission( + collection, + crate::control::security::identity::Permission::Write, + )?; + crate::control::planner::sql_plan_convert::kv_counter_shape::kv_counter_shape( + state, + identity.tenant_id, + DatabaseId::DEFAULT, + collection, + key, + kind, + ) + .map_err(|error| { + let (_, sqlstate, message) = + crate::control::server::pgwire::types::error_to_sqlstate(&error); + DdlError::new(sqlstate, message) + }) +} + /// Handle `SELECT KV_CAS(collection, key, expected, new_value)` /// /// Returns `{"success": bool, "current_value": ""}` as a single text column. diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index 21d5ed7c4..6b8ec1e05 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -111,6 +111,9 @@ pub async fn rate_check( // dispatch bypasses the injection pass by design rather than pending // a decision that will never come. rls_write_check: nodedb_types::RlsWriteCheck::system_internal_collection(), + // The rate-gate collection declares no columns: a counter is decimal + // text, which `rate_remaining` reads back. + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); match crate::control::server::dispatch_utils::dispatch_to_data_plane( @@ -201,8 +204,17 @@ pub async fn rate_remaining( .await { Ok(resp) if resp.status == Status::Ok && !resp.payload.is_empty() => { - // Counter is stored as MessagePack i64. - zerompk::from_msgpack::(&resp.payload).unwrap_or(0) + // `KV_INCR` stores the counter as a raw body: its decimal text. + std::str::from_utf8(&resp.payload) + .ok() + .and_then(|text| text.parse::().ok()) + .ok_or(ddl_err( + "XX000", + format!( + "RATE_REMAINING: counter '{rate_key}' does not hold decimal text; \ + reset the gate with RATE_RESET" + ), + ))? } _ => 0, // Key doesn't exist yet — no usage. }; diff --git a/nodedb/src/control/server/shared/ddl/result.rs b/nodedb/src/control/server/shared/ddl/result.rs index 3a135037c..09a85d0c1 100644 --- a/nodedb/src/control/server/shared/ddl/result.rs +++ b/nodedb/src/control/server/shared/ddl/result.rs @@ -6,7 +6,7 @@ //! http, pgwire, RESP) can encode from them without depending on the pgwire //! `Response` representation. -use nodedb_types::error::{ErrorCode, sqlstate}; +use nodedb_types::error::{ErrorCode, ErrorDetails, sqlstate}; use crate::control::server::response_shape::types::ShapedRows; @@ -42,6 +42,9 @@ pub struct DdlError { pub sqlstate: String, pub code: ErrorCode, pub message: String, + /// The structured details of a typed verdict: the collection, gate, or + /// document it names. `None` for an error built from a SQLSTATE alone. + pub details: Option>, } impl DdlError { @@ -55,6 +58,22 @@ impl DdlError { sqlstate, code, message: message.into(), + details: None, + } + } + + /// Build a `DdlError` from a classified public error: its code and its + /// details travel with the SQLSTATE and message the SQL surfaces render. + pub fn from_public( + sqlstate: impl Into, + message: impl Into, + public: &nodedb_types::NodeDbError, + ) -> Self { + DdlError { + sqlstate: sqlstate.into(), + code: public.code(), + message: message.into(), + details: Some(Box::new(public.details().clone())), } } @@ -65,6 +84,7 @@ impl DdlError { sqlstate: sqlstate.to_string(), code, message: message.into(), + details: None, } } diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index a9e2e540f..30e3fddc2 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -139,10 +139,10 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, sqlstate::CANNOT_COERCE, format!("type mismatch on {collection}: {detail}"), ), - ErrorCode::OverflowError { collection } => ( + ErrorCode::CounterFault { collection, fault } => ( "ERROR", - sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, - format!("arithmetic overflow on {collection}"), + fault.sqlstate(), + format!("{} on {collection}", fault.message()), ), ErrorCode::InsufficientBalance { collection, detail } => ( "ERROR", diff --git a/nodedb/src/control/server/shared/sql/staging_predicates.rs b/nodedb/src/control/server/shared/sql/staging_predicates.rs index 99e586fdf..b2b763379 100644 --- a/nodedb/src/control/server/shared/sql/staging_predicates.rs +++ b/nodedb/src/control/server/shared/sql/staging_predicates.rs @@ -441,13 +441,15 @@ mod tests { ttl_ms: 0, surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }))); assert!(is_stageable_write(&kv_plan(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), key: b"k".to_vec(), - delta: 1.0, + delta: "1".into(), surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }))); assert!(is_stageable_write(&kv_plan(KvOp::Cas { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), @@ -486,13 +488,15 @@ mod tests { ttl_ms: 0, surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }, KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), key: b"k".to_vec(), - delta: 1.0, + delta: "1".into(), surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }, KvOp::Cas { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), diff --git a/nodedb/src/control/server/shared/write_admission/lock_keys.rs b/nodedb/src/control/server/shared/write_admission/lock_keys.rs index 6fc7b46ac..bf5b6f3d2 100644 --- a/nodedb/src/control/server/shared/write_admission/lock_keys.rs +++ b/nodedb/src/control/server/shared/write_admission/lock_keys.rs @@ -272,6 +272,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }), LockKey::Kv { collection: Arc::from("counters"), @@ -286,9 +287,10 @@ mod tests { kv_key(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), key: b"k1".to_vec(), - delta: 1.5, + delta: "1.5".into(), surrogate: Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }), LockKey::Kv { collection: Arc::from("counters"), diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 4d5b169c9..4cbf917ca 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -1319,13 +1319,15 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }), PhysicalPlan::Kv(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), key: Vec::new(), - delta: 0.0, + delta: "0".into(), surrogate: Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }), PhysicalPlan::Kv(KvOp::Cas { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), diff --git a/nodedb/src/control/server/wal_dispatch_kv/append.rs b/nodedb/src/control/server/wal_dispatch_kv/append.rs index 152e07823..4bfccc436 100644 --- a/nodedb/src/control/server/wal_dispatch_kv/append.rs +++ b/nodedb/src/control/server/wal_dispatch_kv/append.rs @@ -7,9 +7,9 @@ use crate::wal::manager::WalManager; use nodedb_physical::physical_plan::KvOp; use super::encode::{ - KvRegisterSortedIndexFields, KvTransferFields, encode_kv_batch_put, encode_kv_cas, - encode_kv_delete, encode_kv_drop_index, encode_kv_drop_sorted_index, encode_kv_expire, - encode_kv_field_set, encode_kv_getset, encode_kv_incr, encode_kv_incr_float, + KvIncrRecord, KvRegisterSortedIndexFields, KvTransferFields, encode_kv_batch_put, + encode_kv_cas, encode_kv_delete, encode_kv_drop_index, encode_kv_drop_sorted_index, + encode_kv_expire, encode_kv_field_set, encode_kv_getset, encode_kv_incr, encode_kv_incr_float, encode_kv_insert_on_conflict_update, encode_kv_persist, encode_kv_predicate_delete, encode_kv_predicate_update, encode_kv_put, encode_kv_register_index, encode_kv_register_sorted_index, encode_kv_transfer, encode_kv_transfer_item, @@ -192,18 +192,20 @@ pub fn wal_append_kv_op( delta, ttl_ms, surrogate, + shape, .. } => { let (now_ms, expire_at_ms) = resolve_expiry(*ttl_ms, now_override); resolved_now_ms = now_ms; - let entry = encode_kv_incr( - collection.as_str(), + let entry = encode_kv_incr(KvIncrRecord { + collection: collection.as_str(), key, - *delta, - *ttl_ms, - surrogate.as_u32(), + delta: *delta, + ttl_ms: *ttl_ms, + surrogate: surrogate.as_u32(), + shape, expire_at_ms, - )?; + })?; Some(wal.append_put(tenant_id, vshard_id, database_id, &entry)?) } KvOp::IncrFloat { @@ -211,9 +213,11 @@ pub fn wal_append_kv_op( key, delta, surrogate, + shape, .. } => { - let entry = encode_kv_incr_float(collection.as_str(), key, *delta, surrogate.as_u32())?; + let entry = + encode_kv_incr_float(collection.as_str(), key, delta, surrogate.as_u32(), shape)?; Some(wal.append_put(tenant_id, vshard_id, database_id, &entry)?) } KvOp::Cas { diff --git a/nodedb/src/control/server/wal_dispatch_kv/encode.rs b/nodedb/src/control/server/wal_dispatch_kv/encode.rs index 93d6aa0c5..fe0ac861e 100644 --- a/nodedb/src/control/server/wal_dispatch_kv/encode.rs +++ b/nodedb/src/control/server/wal_dispatch_kv/encode.rs @@ -2,6 +2,7 @@ //! Pure payload encoders for KV WAL records. +use nodedb_physical::physical_plan::KvCounterShape; use nodedb_physical::physical_plan::UpdateValue; /// Serialize `value` to a MessagePack WAL payload, wrapping any encode error @@ -147,16 +148,19 @@ pub(crate) fn encode_kv_cas( } /// Encode a `kv_incr_float` WAL payload: `("kv_incr_float", collection, key, -/// delta, surrogate)`. Delta record: replay re-runs `incr_float` on the present value. +/// delta, surrogate, shape)`. `delta` is the client's decimal text. Delta +/// record: replay re-runs `incr_float` on the present value, and an absent key +/// takes `shape`. pub(crate) fn encode_kv_incr_float( collection: &str, key: &[u8], - delta: f64, + delta: &str, surrogate: u32, + shape: &KvCounterShape, ) -> crate::Result> { encode( "incr_float", - &("kv_incr_float", collection, key, delta, surrogate), + &("kv_incr_float", collection, key, delta, surrogate, shape), ) } @@ -274,35 +278,46 @@ pub(crate) fn encode_kv_drop_index(collection: &str, field: &str) -> crate::Resu encode("drop index", &("kv_drop_index", collection, field)) } -/// Encode a `kv_incr` WAL payload. `expire_at_ms = None` produces the six-element -/// shape (`ttl_ms == 0` means "preserve existing TTL"); `Some(instant)` appends -/// a 7th element so replay installs the exact resolved expiry. -pub(crate) fn encode_kv_incr( - collection: &str, - key: &[u8], - delta: i64, - ttl_ms: u64, - surrogate: u32, - expire_at_ms: Option, -) -> crate::Result> { - match expire_at_ms { - None => encode( - "incr", - &("kv_incr", collection, key, delta, ttl_ms, surrogate), - ), - Some(expire_at_ms) => encode( - "incr", - &( - "kv_incr", - collection, - key, - delta, - ttl_ms, - surrogate, - expire_at_ms, - ), +/// Fields of a `kv_incr` WAL payload. +pub(crate) struct KvIncrRecord<'a> { + pub collection: &'a str, + pub key: &'a [u8], + pub delta: i64, + /// `0` preserves the existing TTL. + pub ttl_ms: u64, + pub surrogate: u32, + /// The row an absent key becomes. + pub shape: &'a KvCounterShape, + /// The absolute expiry the live write resolved. `Some` only when + /// `ttl_ms > 0`, so replay installs the exact instant. + pub expire_at_ms: Option, +} + +/// Encode a `kv_incr` WAL payload: `("kv_incr", collection, key, delta, +/// ttl_ms, surrogate, shape, expire_at_ms)`. +pub(crate) fn encode_kv_incr(record: KvIncrRecord<'_>) -> crate::Result> { + let KvIncrRecord { + collection, + key, + delta, + ttl_ms, + surrogate, + shape, + expire_at_ms, + } = record; + encode( + "incr", + &( + "kv_incr", + collection, + key, + delta, + ttl_ms, + surrogate, + shape, + expire_at_ms, ), - } + ) } /// Fields of a `kv_register_sorted_index` WAL payload, bundled so @@ -381,10 +396,10 @@ pub(crate) fn encode_kv_truncate(collection: &str) -> crate::Result> { #[cfg(test)] mod tests { - use nodedb_physical::physical_plan::UpdateValue; + use nodedb_physical::physical_plan::{KvCounterShape, UpdateValue}; use super::{ - KvTransferFields, encode_kv_batch_put, encode_kv_cas, encode_kv_expire, + KvIncrRecord, KvTransferFields, encode_kv_batch_put, encode_kv_cas, encode_kv_expire, encode_kv_field_set, encode_kv_getset, encode_kv_incr, encode_kv_incr_float, encode_kv_insert_on_conflict_update, encode_kv_put, encode_kv_register_index, encode_kv_transfer, encode_kv_transfer_item, @@ -570,16 +585,19 @@ mod tests { } #[test] - fn kv_incr_float_encodes_delta_with_surrogate() { - let entry = encode_kv_incr_float("scores", b"dmg", 3.125, 5).unwrap(); + fn kv_incr_float_encodes_decimal_delta_with_surrogate_and_shape() { + let entry = + encode_kv_incr_float("scores", b"dmg", "3.125", 5, &KvCounterShape::Raw).unwrap(); - let (disc, collection, key, delta, surrogate) = - zerompk::from_msgpack::<(&str, String, Vec, f64, u32)>(&entry).unwrap(); + let (disc, collection, key, delta, surrogate, shape) = + zerompk::from_msgpack::<(&str, String, Vec, String, u32, KvCounterShape)>(&entry) + .unwrap(); assert_eq!(disc, "kv_incr_float"); assert_eq!(collection, "scores"); assert_eq!(key, b"dmg"); - assert_eq!(delta, 3.125); + assert_eq!(delta, "3.125"); assert_eq!(surrogate, 5); + assert_eq!(shape, KvCounterShape::Raw); } #[test] @@ -725,53 +743,67 @@ mod tests { assert_eq!(expire_at_ms, 1_234); } - #[test] - fn kv_incr_without_expire_at_matches_historical_shape() { - let entry = encode_kv_incr("counters", b"hits", 3, 0, 7, None).unwrap(); + /// The decoded `kv_incr` tuple. + type IncrTuple = ( + String, + String, + Vec, + i64, + u64, + u32, + KvCounterShape, + Option, + ); - // Byte-identical to the historical six-element tuple encoding. - let expected = - zerompk::to_msgpack_vec(&("kv_incr", "counters", b"hits", 3i64, 0u64, 7u32)).unwrap(); - assert_eq!(entry, expected); + #[test] + fn kv_incr_carries_shape_and_no_expiry_when_ttl_is_preserved() { + let shape = KvCounterShape::Typed { + column: Some("n".into()), + template: vec![0x80], + }; + let entry = encode_kv_incr(KvIncrRecord { + collection: "counters", + key: b"hits", + delta: 3, + ttl_ms: 0, + surrogate: 7, + shape: &shape, + expire_at_ms: None, + }) + .unwrap(); - let (disc, collection, key, delta, ttl_ms, surrogate) = - zerompk::from_msgpack::<(&str, String, Vec, i64, u64, u32)>(&entry).unwrap(); + let (disc, collection, key, delta, ttl_ms, surrogate, decoded_shape, expire_at_ms) = + zerompk::from_msgpack::(&entry).unwrap(); assert_eq!(disc, "kv_incr"); assert_eq!(collection, "counters"); assert_eq!(key, b"hits"); assert_eq!(delta, 3); assert_eq!(ttl_ms, 0); assert_eq!(surrogate, 7); + assert_eq!(decoded_shape, shape); + assert_eq!(expire_at_ms, None); } #[test] fn kv_incr_with_expire_at_carries_absolute_instant() { - let entry = encode_kv_incr( - "counters", - b"daily", - 1, - 86_400_000, - 9, - Some(1_700_000_000_000), - ) + let entry = encode_kv_incr(KvIncrRecord { + collection: "counters", + key: b"daily", + delta: 1, + ttl_ms: 86_400_000, + surrogate: 9, + shape: &KvCounterShape::Raw, + expire_at_ms: Some(1_700_000_000_000), + }) .unwrap(); - let (disc, collection, key, delta, ttl_ms, surrogate, expire_at_ms) = - zerompk::from_msgpack::<(&str, String, Vec, i64, u64, u32, u64)>(&entry).unwrap(); + let (disc, _, _, delta, ttl_ms, surrogate, _, expire_at_ms) = + zerompk::from_msgpack::(&entry).unwrap(); assert_eq!(disc, "kv_incr"); - assert_eq!(collection, "counters"); - assert_eq!(key, b"daily"); assert_eq!(delta, 1); assert_eq!(ttl_ms, 86_400_000); assert_eq!(surrogate, 9); - assert_eq!(expire_at_ms, 1_700_000_000_000); - - // The historical six-element decode rejects the extended payload - // (strict array-length check), so the two shapes never alias. - assert!( - zerompk::from_msgpack::<(&str, String, Vec, i64, u64, u32)>(&entry).is_err(), - "extended payload must not decode as the six-element tuple" - ); + assert_eq!(expire_at_ms, Some(1_700_000_000_000)); } #[test] diff --git a/nodedb/src/control/wal_replication/decode/entry_kv.rs b/nodedb/src/control/wal_replication/decode/entry_kv.rs index 137d323b6..c8896d86f 100644 --- a/nodedb/src/control/wal_replication/decode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/decode/entry_kv.rs @@ -165,16 +165,18 @@ pub(super) fn decode_arm(write: &ReplicatedWrite) -> crate::Result<(PhysicalPlan ttl_ms, surrogate, resolved_now_ms: rn, + shape, } => { resolved_now_ms = *rn; - kv::incr(collection, key, *delta, *ttl_ms, *surrogate)? + kv::incr(collection, key, *delta, *ttl_ms, *surrogate, shape)? } ReplicatedWrite::KvIncrFloat { collection, key, delta, surrogate, - } => kv::incr_float(collection, key, *delta, *surrogate)?, + shape, + } => kv::incr_float(collection, key, delta, *surrogate, shape)?, ReplicatedWrite::KvCas { collection, key, @@ -472,6 +474,7 @@ mod tests { ttl_ms: 60_000, surrogate: 1, resolved_now_ms: Some(1_000), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }, ); let bytes = entry_with_ttl.to_bytes(); @@ -496,6 +499,7 @@ mod tests { ttl_ms: 0, surrogate: 1, resolved_now_ms: None, + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }, ); let bytes_no_ttl = entry_no_ttl.to_bytes(); diff --git a/nodedb/src/control/wal_replication/decode/kv.rs b/nodedb/src/control/wal_replication/decode/kv.rs index 0c4c51774..291f9e8d8 100644 --- a/nodedb/src/control/wal_replication/decode/kv.rs +++ b/nodedb/src/control/wal_replication/decode/kv.rs @@ -6,7 +6,7 @@ //! whole plan afterwards. use crate::bridge::envelope::PhysicalPlan; -use nodedb_physical::physical_plan::{KvOp, ReturningSpec}; +use nodedb_physical::physical_plan::{KvCounterShape, KvOp, ReturningSpec}; use nodedb_types::RlsWriteCheck; /// A decoded RETURNING projection spec plus the read filters gating it — see @@ -212,6 +212,7 @@ pub(super) fn incr( delta: i64, ttl_ms: u64, surrogate: u32, + shape: &KvCounterShape, ) -> crate::Result { let surrogate = nodedb_types::Surrogate::new(surrogate); Ok(PhysicalPlan::Kv(KvOp::Incr { @@ -221,22 +222,25 @@ pub(super) fn incr( ttl_ms, surrogate, rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: shape.clone(), })) } pub(super) fn incr_float( collection: &str, key: &[u8], - delta: f64, + delta: &str, surrogate: u32, + shape: &KvCounterShape, ) -> crate::Result { let surrogate = nodedb_types::Surrogate::new(surrogate); Ok(PhysicalPlan::Kv(KvOp::IncrFloat { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), key: key.to_vec(), - delta, + delta: delta.to_owned(), surrogate, rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: shape.clone(), })) } diff --git a/nodedb/src/control/wal_replication/encode/entry_kv.rs b/nodedb/src/control/wal_replication/encode/entry_kv.rs index 37d6470f2..a43398d8f 100644 --- a/nodedb/src/control/wal_replication/encode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/encode/entry_kv.rs @@ -152,12 +152,14 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { ttl_ms, surrogate, rls_write_check: _, + shape, } => kv::incr( collection.as_str(), key, *delta, *ttl_ms, surrogate.as_u32(), + shape, ), // A follower has no writing identity; decode stamps `already_decided_elsewhere()`. KvOp::IncrFloat { @@ -166,7 +168,8 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { delta, surrogate, rls_write_check: _, - } => kv::incr_float(collection.as_str(), key, *delta, surrogate.as_u32()), + shape, + } => kv::incr_float(collection.as_str(), key, delta, surrogate.as_u32(), shape), // A follower has no writing identity; decode stamps `already_decided_elsewhere()`. KvOp::Cas { collection, diff --git a/nodedb/src/control/wal_replication/encode/kv.rs b/nodedb/src/control/wal_replication/encode/kv.rs index e23f112ea..20558e899 100644 --- a/nodedb/src/control/wal_replication/encode/kv.rs +++ b/nodedb/src/control/wal_replication/encode/kv.rs @@ -4,7 +4,7 @@ use super::super::types::ReplicatedWrite; use super::entry::encode_returning; -use nodedb_physical::physical_plan::{ReturningSpec, UpdateValue}; +use nodedb_physical::physical_plan::{KvCounterShape, ReturningSpec, UpdateValue}; use nodedb_types::Surrogate; /// Resolve the wall-clock instant for a TTL-bearing write once, at proposal @@ -193,6 +193,7 @@ pub(super) fn incr( delta: i64, ttl_ms: u64, surrogate: u32, + shape: &KvCounterShape, ) -> ReplicatedWrite { ReplicatedWrite::KvIncr { collection: collection.to_owned(), @@ -201,20 +202,23 @@ pub(super) fn incr( ttl_ms, surrogate, resolved_now_ms: resolve_now_ms(ttl_ms), + shape: shape.clone(), } } pub(super) fn incr_float( collection: &str, key: &[u8], - delta: f64, + delta: &str, surrogate: u32, + shape: &KvCounterShape, ) -> ReplicatedWrite { ReplicatedWrite::KvIncrFloat { collection: collection.to_owned(), key: key.to_vec(), - delta, + delta: delta.to_owned(), surrogate, + shape: shape.clone(), } } diff --git a/nodedb/src/control/wal_replication/types/replicated_write.rs b/nodedb/src/control/wal_replication/types/replicated_write.rs index 9020f5236..1f5f10210 100644 --- a/nodedb/src/control/wal_replication/types/replicated_write.rs +++ b/nodedb/src/control/wal_replication/types/replicated_write.rs @@ -12,7 +12,7 @@ use super::wire_shapes::{ }; use nodedb_physical::physical_plan::document::MergeClauseOp; use nodedb_physical::physical_plan::{ - ColumnarInsertIntent, CrdtWriteVerb, UpdateValue, VectorDirectWriteIntent, + ColumnarInsertIntent, CrdtWriteVerb, KvCounterShape, UpdateValue, VectorDirectWriteIntent, VectorResolvedMutation, VectorWriteTargets, }; use nodedb_types::{PayloadIndexKind, VectorQuantization, VectorStorageDtype}; @@ -493,12 +493,17 @@ pub enum ReplicatedWrite { surrogate: u32, /// See `KvPut::resolved_now_ms`. resolved_now_ms: Option, + /// The row an absent key becomes. + shape: KvCounterShape, }, KvIncrFloat { collection: String, key: Vec, - delta: f64, + /// The client's decimal text. + delta: String, surrogate: u32, + /// The row an absent key becomes. + shape: KvCounterShape, }, KvCas { collection: String, diff --git a/nodedb/src/data/executor/handlers/kv/atomic.rs b/nodedb/src/data/executor/handlers/kv/atomic.rs index 50a705349..0102ba4e1 100644 --- a/nodedb/src/data/executor/handlers/kv/atomic.rs +++ b/nodedb/src/data/executor/handlers/kv/atomic.rs @@ -2,14 +2,16 @@ //! KV atomic operation handlers: Incr, IncrFloat, Cas, GetSet. +use nodedb_physical::physical_plan::KvCounterShape; +use nodedb_query::msgpack_scan::{KvBodyShape, kv_body_shape}; use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::AtomicError; use crate::engine::kv::current_ms; +use crate::engine::kv::{AtomicError, Incremented}; /// Shared identity context for a single-key KV atomic operation /// (INCR_FLOAT / GETSET) dispatched to this core. @@ -36,8 +38,9 @@ pub(in crate::data::executor) fn atomic_error_code( collection: collection.to_string(), detail, }, - AtomicError::Overflow => ErrorCode::OverflowError { + AtomicError::Counter(fault) => ErrorCode::CounterFault { collection: collection.to_string(), + fault, }, AtomicError::Encode { detail } => ErrorCode::Internal { detail }, // Nothing was written: the engine consults the gate before it @@ -46,12 +49,25 @@ pub(in crate::data::executor) fn atomic_error_code( } } +/// The reply to an `INCR_FLOAT`: the new value as a number, and for a raw +/// body also the stored text. RESP answers with the text, so its reply is +/// byte for byte what `GET` returns. +pub(in crate::data::executor) fn incr_float_reply(value: f64, written: &[u8]) -> serde_json::Value { + match std::str::from_utf8(written) { + Ok(text) if kv_body_shape(written) == KvBodyShape::Raw => { + serde_json::json!({ "value": value, "text": text }) + } + _ => serde_json::json!({ "value": value }), + } +} + impl CoreLoop { pub(in crate::data::executor) fn execute_kv_incr( &mut self, ctx: KvAtomicCtx<'_>, delta: i64, ttl_ms: u64, + shape: &KvCounterShape, ) -> Response { let KvAtomicCtx { task, @@ -89,26 +105,27 @@ impl CoreLoop { }, delta, ttl_ms, + shape, &admit, ) { - Ok(new_value) => { + Ok(Incremented { value, written }) => { if let Some(ref m) = self.metrics { m.record_kv_put(); } - let new_bytes = zerompk::to_msgpack_vec(&new_value).unwrap_or_default(); + // The event carries the bytes the engine stored: the whole row + // for a typed row, the decimal text for a raw body. let key_str = String::from_utf8_lossy(key); self.emit_write_event( task, collection, crate::event::WriteOp::Update, crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), - Some(&new_bytes), + Some(written.as_slice()), None, ); self.note_kv_write_lsn(task, did, tid, collection, key); - match response_codec::encode_json_as_msgpack( - &serde_json::json!({ "value": new_value }), - ) { + match response_codec::encode_json_as_msgpack(&serde_json::json!({ "value": value })) + { Ok(payload) => self.response_with_payload(task, payload), Err(e) => self.response_error( task, @@ -125,7 +142,8 @@ impl CoreLoop { pub(in crate::data::executor) fn execute_kv_incr_float( &mut self, ctx: KvAtomicCtx<'_>, - delta: f64, + delta: &str, + shape: &KvCounterShape, ) -> Response { let KvAtomicCtx { task, @@ -136,7 +154,7 @@ impl CoreLoop { surrogate, rls_write_check, } = ctx; - debug!(core = self.core_id, %collection, delta, "kv incr_float"); + debug!(core = self.core_id, %collection, %delta, "kv incr_float"); if self.kv_engine.is_over_budget() { return self.response_error(task, ErrorCode::ResourcesExhausted); @@ -159,26 +177,26 @@ impl CoreLoop { surrogate, }, delta, + shape, &admit, ) { - Ok(new_value) => { + Ok(Incremented { value, written }) => { if let Some(ref m) = self.metrics { m.record_kv_put(); } - let new_bytes = zerompk::to_msgpack_vec(&new_value).unwrap_or_default(); + // The event carries the bytes the engine stored: the whole row + // for a typed row, the decimal text for a raw body. let key_str = String::from_utf8_lossy(key); self.emit_write_event( task, collection, crate::event::WriteOp::Update, crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), - Some(&new_bytes), + Some(written.as_slice()), None, ); self.note_kv_write_lsn(task, did, tid, collection, key); - match response_codec::encode_json_as_msgpack( - &serde_json::json!({ "value": new_value }), - ) { + match response_codec::encode_json_as_msgpack(&incr_float_reply(value, &written)) { Ok(payload) => self.response_with_payload(task, payload), Err(e) => self.response_error( task, @@ -380,3 +398,229 @@ impl CoreLoop { self.response_error(task, atomic_error_code(error, collection)) } } + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + use std::sync::Arc; + + use nodedb_physical::physical_plan::{KvCounterShape, KvOp}; + use nodedb_types::{QualifiedCollection, RlsWriteCheck, Surrogate, Value}; + + use super::KvAtomicCtx; + use crate::bridge::envelope::{PhysicalPlan, Status}; + use crate::data::executor::core_loop::CoreLoop; + use crate::data::executor::task::ExecutionTask; + use crate::event::WriteOp; + use crate::event::bus::{EventConsumerRx, create_event_bus_with_capacity}; + use crate::types::{DatabaseId, TenantId, VShardId}; + + const TID: u64 = 1; + const COLLECTION: &str = "kv_counters"; + + struct CoreHarness { + core: CoreLoop, + events: EventConsumerRx, + _req_tx: nodedb_bridge::buffer::Producer, + _resp_rx: nodedb_bridge::buffer::Consumer, + _dir: tempfile::TempDir, + } + + /// A core whose write events land on `events`. + fn make_core() -> CoreHarness { + use crate::bridge::dispatch::{BridgeRequest, BridgeResponse}; + use nodedb_bridge::buffer::RingBuffer; + + let dir = tempfile::tempdir().expect("tempdir"); + let (req_tx, req_rx) = RingBuffer::channel::(64); + let (resp_tx, resp_rx) = RingBuffer::channel::(64); + let mut core = CoreLoop::open( + 0, + req_rx, + resp_tx, + dir.path(), + Arc::new(nodedb_types::OrdinalClock::new()), + crate::data::executor::core_loop::test_governor(), + ) + .expect("open core"); + let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 64); + core.set_event_producer(producers.pop().expect("producer")); + CoreHarness { + core, + events: consumers.pop().expect("consumer"), + _req_tx: req_tx, + _resp_rx: resp_rx, + _dir: dir, + } + } + + fn did() -> u64 { + DatabaseId::DEFAULT.as_u64() + } + + fn task() -> ExecutionTask { + CoreLoop::replay_task( + TenantId::new(TID), + DatabaseId::DEFAULT, + VShardId::new(0), + PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, COLLECTION), + key: b"seed".to_vec(), + rls_filters: Vec::new(), + surrogate_ceiling: None, + }), + None, + ) + } + + fn seed(core: &mut CoreLoop, key: &[u8], value: &[u8]) { + core.kv_engine.put(crate::engine::kv::KvPutParams { + database_id: did(), + tenant_id: TID, + collection: COLLECTION, + key, + value, + ttl_ms: 0, + now_ms: crate::engine::kv::current_ms(), + surrogate: Surrogate::new(1), + }); + } + + fn stored(core: &CoreLoop, key: &[u8]) -> Vec { + core.kv_engine + .get(did(), TID, COLLECTION, key, crate::engine::kv::current_ms()) + .expect("the key holds a value") + } + + fn typed_row(fields: &[(&str, Value)]) -> Vec { + let map: HashMap = fields + .iter() + .map(|(k, v)| ((*k).to_string(), v.clone())) + .collect(); + nodedb_types::value_to_msgpack(&Value::Object(map)).expect("encode row") + } + + fn columns(bytes: &[u8]) -> HashMap { + match nodedb_types::value_from_msgpack(bytes).expect("decode row") { + Value::Object(map) => map, + other => panic!("expected a typed row, got {other:?}"), + } + } + + fn ctx<'a>( + task: &'a ExecutionTask, + key: &'a [u8], + check: &'a RlsWriteCheck, + ) -> KvAtomicCtx<'a> { + KvAtomicCtx { + task, + did: did(), + tid: TID, + collection: COLLECTION, + key, + surrogate: Surrogate::new(1), + rls_write_check: check, + } + } + + #[test] + fn incr_on_a_typed_row_emits_the_whole_stored_row() { + let mut h = make_core(); + seed( + &mut h.core, + b"player", + &typed_row(&[ + ("label", Value::String("gold".into())), + ("n", Value::Integer(5)), + ]), + ); + + let t = task(); + let check = RlsWriteCheck::already_decided_elsewhere(); + let resp = h + .core + .execute_kv_incr(ctx(&t, b"player", &check), 3, 0, &KvCounterShape::Raw); + assert_eq!(resp.status, Status::Ok, "{:?}", resp.error_code); + + let event = h.events.try_recv().expect("INCR emits a write event"); + assert_eq!(event.op, WriteOp::Update); + let new_value = event.new_value.expect("the event carries the new row"); + assert_eq!( + new_value.as_ref(), + stored(&h.core, b"player").as_slice(), + "the event carries exactly the bytes the engine stored" + ); + let row = columns(&new_value); + assert_eq!(row.get("n"), Some(&Value::Integer(8))); + assert_eq!(row.get("label"), Some(&Value::String("gold".into()))); + } + + #[test] + fn incr_float_on_a_typed_row_emits_the_whole_stored_row() { + let mut h = make_core(); + seed( + &mut h.core, + b"player", + &typed_row(&[ + ("label", Value::String("gold".into())), + ("score", Value::Float(1.5)), + ]), + ); + + let t = task(); + let check = RlsWriteCheck::already_decided_elsewhere(); + let resp = + h.core + .execute_kv_incr_float(ctx(&t, b"player", &check), "1", &KvCounterShape::Raw); + assert_eq!(resp.status, Status::Ok, "{:?}", resp.error_code); + + let event = h.events.try_recv().expect("INCR_FLOAT emits a write event"); + let new_value = event.new_value.expect("the event carries the new row"); + assert_eq!(new_value.as_ref(), stored(&h.core, b"player").as_slice()); + let row = columns(&new_value); + assert_eq!(row.get("score"), Some(&Value::Float(2.5))); + assert_eq!(row.get("label"), Some(&Value::String("gold".into()))); + } + + #[test] + fn incr_on_a_raw_body_emits_the_stored_decimal_text() { + let mut h = make_core(); + seed(&mut h.core, b"hits", b"41"); + + let t = task(); + let check = RlsWriteCheck::already_decided_elsewhere(); + let resp = h + .core + .execute_kv_incr(ctx(&t, b"hits", &check), 1, 0, &KvCounterShape::Raw); + assert_eq!(resp.status, Status::Ok, "{:?}", resp.error_code); + + let event = h.events.try_recv().expect("INCR emits a write event"); + assert_eq!(event.new_value.as_deref(), Some(b"42".as_slice())); + assert_eq!(stored(&h.core, b"hits"), b"42".to_vec()); + } + + #[test] + fn incr_on_raw_text_that_is_not_an_integer_answers_the_counter_fault() { + let mut h = make_core(); + seed(&mut h.core, b"name", b"abc"); + + let t = task(); + let check = RlsWriteCheck::already_decided_elsewhere(); + let resp = h + .core + .execute_kv_incr(ctx(&t, b"name", &check), 1, 0, &KvCounterShape::Raw); + assert_eq!(resp.status, Status::Error); + assert_eq!( + resp.error_code.map(|code| *code), + Some(crate::bridge::envelope::ErrorCode::CounterFault { + collection: COLLECTION.into(), + fault: crate::bridge::envelope::CounterFault::NotAnInteger, + }) + ); + assert!( + h.events.try_recv().is_none(), + "a refused INCR emits no event" + ); + assert_eq!(stored(&h.core, b"name"), b"abc".to_vec()); + } +} diff --git a/nodedb/src/data/executor/handlers/kv/dispatch.rs b/nodedb/src/data/executor/handlers/kv/dispatch.rs index 195785a07..c7ad06c4d 100644 --- a/nodedb/src/data/executor/handlers/kv/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/dispatch.rs @@ -273,6 +273,7 @@ impl CoreLoop { ttl_ms, surrogate, rls_write_check, + shape, } => self.execute_kv_incr( super::atomic::KvAtomicCtx { task, @@ -285,6 +286,7 @@ impl CoreLoop { }, *delta, *ttl_ms, + shape, ), KvOp::IncrFloat { collection, @@ -292,6 +294,7 @@ impl CoreLoop { delta, surrogate, rls_write_check, + shape, } => self.execute_kv_incr_float( super::atomic::KvAtomicCtx { task, @@ -302,7 +305,8 @@ impl CoreLoop { surrogate: *surrogate, rls_write_check, }, - *delta, + delta, + shape, ), KvOp::Cas { collection, diff --git a/nodedb/src/data/executor/handlers/kv/resolve/apply.rs b/nodedb/src/data/executor/handlers/kv/resolve/apply.rs index 5d1b6b85e..faf014da2 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/apply.rs @@ -261,8 +261,9 @@ mod tests { .get(did(), TID, collection, key, crate::engine::kv::current_ms()) } + /// A raw counter body: the decimal text of `v`. fn i64_bytes(v: i64) -> Vec { - zerompk::to_msgpack_vec(&v).expect("encode i64") + v.to_string().into_bytes() } /// Run the resolve handler and decode its outcome. @@ -299,6 +300,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, } } diff --git a/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs b/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs index 885bde5ca..9c0e6ce05 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs @@ -5,12 +5,14 @@ //! `KvEngine::{incr, incr_float, cas, getset}` call — recomputing here would //! let resolve and apply disagree. -use nodedb_physical::physical_plan::KvResolveOutcome; +use nodedb_physical::physical_plan::{KvCounterShape, KvResolveOutcome}; use super::context::{ResolveResult, ResolvedPut, expiry_from_ttl, one, put_mutation}; use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::handlers::kv::atomic::{KvAtomicCtx, atomic_error_code}; +use crate::data::executor::handlers::kv::atomic::{ + KvAtomicCtx, atomic_error_code, incr_float_reply, +}; use crate::data::executor::handlers::kv::rls::admit_kv_row; use crate::data::executor::response_codec; use crate::engine::kv::current_ms; @@ -30,6 +32,7 @@ impl CoreLoop { ctx: KvAtomicCtx<'_>, delta: i64, ttl_ms: u64, + shape: &KvCounterShape, ) -> ResolveResult { let KvAtomicCtx { task, @@ -45,7 +48,7 @@ impl CoreLoop { } let now_ms = self.kv_ttl_now_ms(task); let current = self.kv_resolve_read(did, tid, collection, key, now_ms); - let (new_value, new_bytes) = compute::incr(current.as_deref(), delta) + let (new_value, new_bytes) = compute::incr(current.as_deref(), delta, shape) .map_err(|e| atomic_error_code(e, collection))?; admit_kv_row(rls_write_check, &new_bytes, key, tid, collection)?; @@ -72,7 +75,12 @@ impl CoreLoop { /// Resolve `INCR_FLOAT`. Always preserves the key's existing TTL, the /// same `ttl_ms = 0` call `KvEngine::incr_float` makes. - pub(super) fn resolve_kv_incr_float(&self, ctx: KvAtomicCtx<'_>, delta: f64) -> ResolveResult { + pub(super) fn resolve_kv_incr_float( + &self, + ctx: KvAtomicCtx<'_>, + delta: &str, + shape: &KvCounterShape, + ) -> ResolveResult { let KvAtomicCtx { did, tid, @@ -87,12 +95,12 @@ impl CoreLoop { } let now_ms = self.kv_atomic_now_ms(); let current = self.kv_resolve_read(did, tid, collection, key, now_ms); - let (new_value, new_bytes) = compute::incr_float(current.as_deref(), delta) + let (new_value, new_bytes) = compute::incr_float(current.as_deref(), delta, shape) .map_err(|e| atomic_error_code(e, collection))?; admit_kv_row(rls_write_check, &new_bytes, key, tid, collection)?; let response_payload = - response_codec::encode_json_as_msgpack(&serde_json::json!({ "value": new_value }))?; + response_codec::encode_json_as_msgpack(&incr_float_reply(new_value, &new_bytes))?; Ok(one( put_mutation(ResolvedPut { collection, @@ -312,8 +320,9 @@ mod tests { .get(did(), TID, collection, key, crate::engine::kv::current_ms()) } + /// A raw counter body: the decimal text of `v`. fn i64_bytes(v: i64) -> Vec { - zerompk::to_msgpack_vec(&v).expect("encode i64") + v.to_string().into_bytes() } /// Run the resolve handler and decode its outcome. @@ -353,6 +362,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, } } diff --git a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs index f7e18121d..4857078ea 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs @@ -135,6 +135,7 @@ impl CoreLoop { ttl_ms, surrogate, rls_write_check, + shape, } => self.resolve_kv_incr( KvAtomicCtx { task, @@ -147,6 +148,7 @@ impl CoreLoop { }, *delta, *ttl_ms, + shape, ), KvOp::IncrFloat { collection, @@ -154,6 +156,7 @@ impl CoreLoop { delta, surrogate, rls_write_check, + shape, } => self.resolve_kv_incr_float( KvAtomicCtx { task, @@ -164,7 +167,8 @@ impl CoreLoop { surrogate: *surrogate, rls_write_check, }, - *delta, + delta, + shape, ), KvOp::Cas { collection, @@ -424,8 +428,9 @@ mod tests { .get(did(), TID, collection, key, crate::engine::kv::current_ms()) } + /// A raw counter body: the decimal text of `v`. fn i64_bytes(v: i64) -> Vec { - zerompk::to_msgpack_vec(&v).expect("encode i64") + v.to_string().into_bytes() } /// Run the resolve handler and decode its outcome. @@ -449,6 +454,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, } } diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index 1b3c738be..614603353 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -566,6 +566,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }, ); assert_eq!(resp.status, Status::Ok, "stage incr: {resp:?}"); @@ -595,8 +596,8 @@ mod tests { // to 42 — not the last delta (2) nor the first (40). assert_eq!(value, overlay_bytes); assert_eq!( - zerompk::from_msgpack::(&value).expect("i64"), - 42, + value, + b"42".to_vec(), "resolve carries the absolute resolved value, not a delta" ); } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs index 38cfe7368..8436ae538 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs @@ -36,13 +36,14 @@ //! realistic transaction, and never persisted (COMMIT replay uses the real //! `KvEngine` atomic path, which ignores the overlay's surrogate entirely). -use nodedb_physical::physical_plan::KvOp; +use nodedb_physical::physical_plan::{KvCounterShape, KvOp}; use nodedb_types::Surrogate; use super::context::StageCtx; use super::stage_kv::kv_row_identity; use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::kv::atomic::incr_float_reply; use crate::data::executor::handlers::transaction::overlay::StagedTtl; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; @@ -85,10 +86,11 @@ impl CoreLoop { // overlay keys its own slots (see module doc) and ignores it. surrogate: _, rls_write_check, + shape, } => { let ctx = self.kv_atomic_stage_ctx(task, tid, txn_id, collection.as_str(), key); self.stage_kv_ttl_side_effect(&ctx, *ttl_ms); - self.stage_kv_incr(&ctx, key, *delta, rls_write_check) + self.stage_kv_incr(&ctx, key, *delta, shape, rls_write_check) } KvOp::IncrFloat { collection, @@ -96,9 +98,10 @@ impl CoreLoop { delta, surrogate: _, rls_write_check, + shape, } => { let ctx = self.kv_atomic_stage_ctx(task, tid, txn_id, collection.as_str(), key); - self.stage_kv_incr_float(&ctx, key, *delta, rls_write_check) + self.stage_kv_incr_float(&ctx, key, delta, shape, rls_write_check) } KvOp::Cas { collection, @@ -188,10 +191,11 @@ impl CoreLoop { ctx: &StageCtx<'_>, key: &[u8], delta: i64, + shape: &KvCounterShape, rls_write_check: &nodedb_types::RlsWriteCheck, ) -> Response { let current = self.resolve_kv_current(ctx, key); - match atomic_compute::incr(current.as_deref(), delta) { + match atomic_compute::incr(current.as_deref(), delta, shape) { Ok((new_i64, new_bytes)) => { if let Err(e) = self.stage_admit_kv_image(ctx, &new_bytes, rls_write_check) { return self.response_error(ctx.task, e); @@ -209,19 +213,21 @@ impl CoreLoop { &mut self, ctx: &StageCtx<'_>, key: &[u8], - delta: f64, + delta: &str, + shape: &KvCounterShape, rls_write_check: &nodedb_types::RlsWriteCheck, ) -> Response { let current = self.resolve_kv_current(ctx, key); - match atomic_compute::incr_float(current.as_deref(), delta) { + match atomic_compute::incr_float(current.as_deref(), delta, shape) { Ok((new_f64, new_bytes)) => { if let Err(e) = self.stage_admit_kv_image(ctx, &new_bytes, rls_write_check) { return self.response_error(ctx.task, e); } + let reply = incr_float_reply(new_f64, &new_bytes); if let Err(e) = self.stage_put_capped(ctx, new_bytes) { return self.response_error(ctx.task, e); } - self.kv_atomic_json_response(ctx.task, &serde_json::json!({ "value": new_f64 })) + self.kv_atomic_json_response(ctx.task, &reply) } Err(e) => self.response_atomic_error(ctx.task, ctx.collection, e), } diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs index 24f45fe33..7f47b89cf 100644 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs +++ b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs @@ -39,6 +39,7 @@ impl CoreLoop { ttl_ms, surrogate, rls_write_check, + shape, } => { let now_ms = current_ms(); let prior = self @@ -56,6 +57,7 @@ impl CoreLoop { }, *delta, *ttl_ms, + shape, ); if resp.status == Status::Error { return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { @@ -76,6 +78,7 @@ impl CoreLoop { delta, surrogate, rls_write_check, + shape, } => { let now_ms = current_ms(); let prior = self @@ -91,7 +94,8 @@ impl CoreLoop { surrogate: *surrogate, rls_write_check, }, - *delta, + delta, + shape, ); if resp.status == Status::Error { return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { diff --git a/nodedb/src/data/executor/wal_replay_kv_atomic.rs b/nodedb/src/data/executor/wal_replay_kv_atomic.rs index 149669608..a261c96d6 100644 --- a/nodedb/src/data/executor/wal_replay_kv_atomic.rs +++ b/nodedb/src/data/executor/wal_replay_kv_atomic.rs @@ -17,6 +17,7 @@ //! the same pre-state, fails identically, and mutates nothing — this //! converges rather than diverges, so no success-gate is applied here. +use nodedb_physical::physical_plan::KvCounterShape; use tracing::warn; use super::core_loop::CoreLoop; @@ -143,8 +144,9 @@ impl CoreLoop { record_lsn: u64, tombstones: &nodedb_wal::TombstoneSet, ) -> Option { - let (disc, collection, key, delta, surrogate) = - zerompk::from_msgpack::<(&str, String, Vec, f64, u32)>(payload).ok()?; + let (disc, collection, key, delta, surrogate, shape) = + zerompk::from_msgpack::<(&str, String, Vec, String, u32, KvCounterShape)>(payload) + .ok()?; if disc != "kv_incr_float" { return None; } @@ -161,7 +163,8 @@ impl CoreLoop { now_ms, surrogate: nodedb_types::Surrogate::new(surrogate), }, - delta, + &delta, + &shape, // Replay re-applies a write the policy already admitted when it was // first accepted; re-deciding it here would make recovery depend on // the policies of whoever happens to be connected. @@ -187,12 +190,13 @@ impl CoreLoop { ); Some(0) } - Err(AtomicError::Overflow) => { + Err(AtomicError::Counter(fault)) => { warn!( core = self.core_id, collection = %collection, key = %String::from_utf8_lossy(&key), - "WAL kv_incr_float replay: overflow (NaN/Inf), skipping record" + fault = fault.message(), + "WAL kv_incr_float replay: no value computed, skipping record" ); Some(0) } @@ -297,7 +301,7 @@ impl CoreLoop { ) { let detail = match error { AtomicError::TypeMismatch { detail } | AtomicError::Encode { detail } => detail, - AtomicError::Overflow => "overflow".to_string(), + AtomicError::Counter(fault) => fault.message().to_string(), AtomicError::Rejected(error) => abort_replay( "kv", "swap_admission", @@ -325,7 +329,7 @@ mod tests { use crate::control::server::wal_dispatch::wal_append_if_write; use crate::types::{DatabaseId, TenantId, VShardId}; use crate::wal::manager::WalManager; - use nodedb_physical::physical_plan::KvOp; + use nodedb_physical::physical_plan::{KvCounterShape, KvOp}; use nodedb_types::{QualifiedCollection, RlsWriteCheck, Surrogate}; use nodedb_wal::TombstoneSet; @@ -495,16 +499,18 @@ mod tests { let incr1 = PhysicalPlan::Kv(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "scores"), key: b"dmg".to_vec(), - delta: 3.0, + delta: "3.0".into(), surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let incr2 = PhysicalPlan::Kv(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "scores"), key: b"dmg".to_vec(), - delta: 1.5, + delta: "1.5".into(), surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let records = append_via_autocommit(&[incr1, incr2]); @@ -512,11 +518,10 @@ mod tests { let mut h = make_core(); h.core.replay_kv_wal(&records, 1, &TombstoneSet::new()); - let bytes = get_value(&h.core, "scores", b"dmg").expect("dmg survives replay"); - let value: f64 = zerompk::from_msgpack(&bytes).expect("decode f64"); - assert!( - (value - 4.5).abs() < f64::EPSILON, - "incr_float must replay both increments against the empty-start state, got {value}" + assert_eq!( + get_value(&h.core, "scores", b"dmg"), + Some(b"4.5".to_vec()), + "incr_float must replay both increments against the empty-start state" ); } @@ -525,7 +530,7 @@ mod tests { let put_str = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "scores"), key: b"str".to_vec(), - value: zerompk::to_msgpack_vec(&"hello").expect("encode"), + value: b"hello".to_vec(), ttl_ms: 0, surrogate: Surrogate::new(1), returning: None, @@ -534,9 +539,10 @@ mod tests { let incr = PhysicalPlan::Kv(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "scores"), key: b"str".to_vec(), - delta: 1.0, + delta: "1".into(), surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let records = append_via_autocommit(&[put_str, incr]); @@ -544,10 +550,9 @@ mod tests { let mut h = make_core(); h.core.replay_kv_wal(&records, 1, &TombstoneSet::new()); - let bytes = get_value(&h.core, "scores", b"str").expect("str survives replay"); - let value: String = zerompk::from_msgpack(&bytes).expect("decode string"); assert_eq!( - value, "hello", + get_value(&h.core, "scores", b"str"), + Some(b"hello".to_vec()), "incr_float over a non-numeric value must replay to a no-op, value unchanged" ); } diff --git a/nodedb/src/data/executor/wal_replay_kv_incr.rs b/nodedb/src/data/executor/wal_replay_kv_incr.rs index cf2dd403e..ebe28fc8e 100644 --- a/nodedb/src/data/executor/wal_replay_kv_incr.rs +++ b/nodedb/src/data/executor/wal_replay_kv_incr.rs @@ -10,130 +10,48 @@ //! diverging, so no success-gate is applied here (same rationale as //! `kv_incr_float` in `wal_replay_kv_atomic.rs`). //! -//! `kv_incr` optionally carries a Control-Plane-resolved absolute -//! `expire_at_ms` as a trailing seventh element, present only when the live -//! write's `ttl_ms > 0` (see `encode_kv_incr`'s doc comment — `ttl_ms == 0` -//! means "preserve whatever TTL the key already had", which has no instant -//! to carry). Both shapes are genuinely produced in production, so both must -//! be decoded; the seven-element shape is tried first because zerompk's -//! strict array-length check means it can never match the six-element -//! tuple, but skipping it would silently drop the recorded absolute instant. -//! When present, replay installs it verbatim via -//! `KvEngine::incr_with_absolute_expiry` instead of recomputing +//! The record is `("kv_incr", collection, key, delta, ttl_ms, surrogate, +//! shape, expire_at_ms)`. `shape` is the row an absent key becomes, so replay +//! creates the same row the live write did. `expire_at_ms` is `Some` only +//! when the live write's `ttl_ms > 0`: replay installs that instant verbatim +//! via `KvEngine::incr_with_absolute_expiry` instead of recomputing //! `now_ms + ttl_ms`, which would drift the expiry forward by the -//! crash-to-restart delay. +//! crash-to-restart delay. `ttl_ms == 0` preserves the key's existing TTL. //! //! Unlike the `Put` family, `kv_incr` carries its own surrogate in the //! record rather than relying on the separately-durable surrogate catalog, //! so replay reconstructs it from the payload's `u32` instead of using //! `Surrogate::ZERO`. +use nodedb_physical::physical_plan::KvCounterShape; use tracing::warn; use super::core_loop::CoreLoop; use crate::data::executor::core_loop::write_index::KeyRepr; use crate::data::executor::replay_abort::abort_replay; -use crate::engine::kv::{AtomicError, AtomicKeyCtx}; +use crate::engine::kv::{AtomicError, AtomicKeyCtx, IncrStep, Incremented}; + +/// The decoded `kv_incr` record. +type KvIncrRecord<'a> = ( + &'a str, + String, + Vec, + i64, + u64, + u32, + KvCounterShape, + Option, +); impl CoreLoop { - /// Try both `kv_incr` WAL payload shapes in turn, seven-element - /// (absolute expiry) before six-element (preserve). Returns `None` when - /// neither decodes (caller tries the next candidate arm in - /// `wal_replay/kv.rs`), otherwise `Some(puts)` from whichever shape - /// decoded. - pub(super) fn try_replay_kv_incr( - &mut self, - payload: &[u8], - tenant_id: u64, - database_id: u64, - now_ms: u64, - record_lsn: u64, - tombstones: &nodedb_wal::TombstoneSet, - ) -> Option { - if let Some(applied) = self.try_replay_kv_incr_with_expiry( - payload, - tenant_id, - database_id, - now_ms, - record_lsn, - tombstones, - ) { - return Some(applied); - } - self.try_replay_kv_incr_preserve( - payload, - tenant_id, - database_id, - now_ms, - record_lsn, - tombstones, - ) - } - - /// Seven-element shape: `("kv_incr", collection, key, delta, ttl_ms, - /// surrogate, expire_at_ms)` — recorded only when the live write's - /// `ttl_ms > 0`. + /// Decode + tombstone-gate + replay one `kv_incr` WAL record. /// - /// Returns `None` when `payload` does not match this shape, otherwise - /// `Some(puts)` — `1` if the increment applied, `0` if tombstoned or the - /// current value was not numeric (a type-mismatch or overflow replays to - /// the same no-op the live dispatch produced). - fn try_replay_kv_incr_with_expiry( - &mut self, - payload: &[u8], - tenant_id: u64, - database_id: u64, - now_ms: u64, - record_lsn: u64, - tombstones: &nodedb_wal::TombstoneSet, - ) -> Option { - let (disc, collection, key, delta, ttl_ms, surrogate, expire_at_ms) = - zerompk::from_msgpack::<(&str, String, Vec, i64, u64, u32, u64)>(payload).ok()?; - if disc != "kv_incr" { - return None; - } - let tombstones = &tombstones.for_database(database_id); - if self.skip_kv_replay_record(tombstones, tenant_id, &collection, record_lsn) { - return Some(0); - } - let result = self.kv_engine.incr_with_absolute_expiry( - AtomicKeyCtx { - database_id, - tenant_id, - collection: &collection, - key: &key, - now_ms, - surrogate: nodedb_types::Surrogate::new(surrogate), - }, - delta, - ttl_ms, - expire_at_ms, - // Replay re-applies a write the policy already admitted when it was - // first accepted; re-deciding it here would make recovery depend on - // the policies of whoever happens to be connected. - &crate::engine::kv::admit_any, - ); - let applied = self.log_kv_incr_result(&collection, &key, delta, record_lsn, result); - if applied > 0 { - self.note_replay_write_lsn( - database_id, - tenant_id, - &collection, - Some(KeyRepr::KvKey(Box::from(key.as_slice()))), - record_lsn, - ); - } - Some(applied) - } - - /// Six-element shape: `("kv_incr", collection, key, delta, ttl_ms, - /// surrogate)` — recorded when the live write's `ttl_ms == 0` (preserve - /// whatever TTL the key already had; no absolute instant to carry). - /// - /// Returns `None` when `payload` does not match this shape, otherwise - /// `Some(puts)` — `1` if the increment applied, `0` if tombstoned or the - /// current value was not numeric. - fn try_replay_kv_incr_preserve( + /// Returns `None` when `payload` is not a `kv_incr` record (caller tries + /// the next candidate arm in `wal_replay/kv.rs`), otherwise `Some(puts)`: + /// `1` if the increment applied, `0` if tombstoned or the live write + /// computed no value (a type mismatch or overflow replays to the same + /// no-op). + pub(super) fn try_replay_kv_incr( &mut self, payload: &[u8], tenant_id: u64, @@ -142,8 +60,8 @@ impl CoreLoop { record_lsn: u64, tombstones: &nodedb_wal::TombstoneSet, ) -> Option { - let (disc, collection, key, delta, ttl_ms, surrogate) = - zerompk::from_msgpack::<(&str, String, Vec, i64, u64, u32)>(payload).ok()?; + let (disc, collection, key, delta, ttl_ms, surrogate, shape, expire_at_ms) = + zerompk::from_msgpack::>(payload).ok()?; if disc != "kv_incr" { return None; } @@ -151,20 +69,31 @@ impl CoreLoop { if self.skip_kv_replay_record(tombstones, tenant_id, &collection, record_lsn) { return Some(0); } - let result = self.kv_engine.incr( - AtomicKeyCtx { - database_id, - tenant_id, - collection: &collection, - key: &key, - now_ms, - surrogate: nodedb_types::Surrogate::new(surrogate), - }, - delta, - ttl_ms, - // Already-admitted redo — see `incr_with_absolute_expiry` above. - &crate::engine::kv::admit_any, - ); + let ctx = AtomicKeyCtx { + database_id, + tenant_id, + collection: &collection, + key: &key, + now_ms, + surrogate: nodedb_types::Surrogate::new(surrogate), + }; + // Replay re-applies a write the policy already admitted when it was + // first accepted. Re-deciding it here would make recovery depend on + // the policies of whoever happens to be connected. + let admit = &crate::engine::kv::admit_any; + let result = match expire_at_ms { + Some(expire_at_ms) => self.kv_engine.incr_with_absolute_expiry( + ctx, + IncrStep { + delta, + ttl_ms, + shape: &shape, + }, + expire_at_ms, + admit, + ), + None => self.kv_engine.incr(ctx, delta, ttl_ms, &shape, admit), + }; let applied = self.log_kv_incr_result(&collection, &key, delta, record_lsn, result); if applied > 0 { self.note_replay_write_lsn( @@ -178,8 +107,8 @@ impl CoreLoop { Some(applied) } - /// Shared result handling for both `kv_incr` shapes: `Ok` counts as one - /// applied put; `TypeMismatch` / `Overflow` / `Encode` are + /// Shared result handling for a `kv_incr` replay: `Ok` counts as one + /// applied put; `TypeMismatch` / `Counter` / `Encode` are /// correctly-converging no-ops (the live dispatch would have failed /// identically), logged and skipped rather than treated as errors. /// @@ -192,7 +121,7 @@ impl CoreLoop { key: &[u8], delta: i64, record_lsn: u64, - result: Result, + result: Result, AtomicError>, ) -> usize { match result { Ok(_) => 1, @@ -207,13 +136,14 @@ impl CoreLoop { ); 0 } - Err(AtomicError::Overflow) => { + Err(AtomicError::Counter(fault)) => { warn!( core = self.core_id, collection = %collection, key = %String::from_utf8_lossy(key), delta, - "WAL kv_incr replay: overflow, skipping record" + fault = fault.message(), + "WAL kv_incr replay: no value computed, skipping record" ); 0 } @@ -258,10 +188,10 @@ mod tests { use crate::bridge::envelope::PhysicalPlan; use crate::control::server::wal_dispatch::wal_append_if_write; - use crate::control::server::wal_dispatch_kv::encode::encode_kv_incr; + use crate::control::server::wal_dispatch_kv::encode::{KvIncrRecord, encode_kv_incr}; use crate::types::{DatabaseId, TenantId, VShardId}; use crate::wal::manager::WalManager; - use nodedb_physical::physical_plan::KvOp; + use nodedb_physical::physical_plan::{KvCounterShape, KvOp}; use nodedb_types::{QualifiedCollection, RlsWriteCheck, Surrogate}; use nodedb_wal::TombstoneSet; @@ -327,7 +257,10 @@ mod tests { .kv_engine .get(DatabaseId::DEFAULT.as_u64(), TID, collection, key, now_ms) .expect("value present"); - zerompk::from_msgpack::(&bytes).expect("decode i64") + std::str::from_utf8(&bytes) + .expect("a raw counter is UTF-8 text") + .parse::() + .expect("a raw counter is decimal text") } fn ttl_ms(core: &CoreLoop, collection: &str, key: &[u8]) -> Option { @@ -340,7 +273,7 @@ mod tests { let put_p = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), key: b"hits".to_vec(), - value: zerompk::to_msgpack_vec(&5i64).expect("encode"), + value: b"5".to_vec(), ttl_ms: 0, surrogate: Surrogate::new(1), returning: None, @@ -353,6 +286,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let records = append_via_autocommit(&[put_p, incr]); @@ -372,7 +306,7 @@ mod tests { let put_p = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), key: b"hits".to_vec(), - value: zerompk::to_msgpack_vec(&5i64).expect("encode"), + value: b"5".to_vec(), ttl_ms: 0, surrogate: Surrogate::new(1), returning: None, @@ -385,6 +319,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let incr2 = PhysicalPlan::Kv(KvOp::Incr { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), @@ -393,6 +328,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let records = append_via_autocommit(&[put_p, incr1, incr2]); @@ -412,7 +348,7 @@ mod tests { let put_p = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), key: b"temp".to_vec(), - value: zerompk::to_msgpack_vec(&5i64).expect("encode"), + value: b"5".to_vec(), ttl_ms: 60_000, surrogate: Surrogate::new(1), returning: None, @@ -425,6 +361,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let records = append_via_autocommit(&[put_p, incr]); @@ -449,14 +386,22 @@ mod tests { let put_seed = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), key: b"daily".to_vec(), - value: zerompk::to_msgpack_vec(&0i64).expect("encode"), + value: b"0".to_vec(), ttl_ms: 0, surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), }); - let entry = encode_kv_incr("counters", b"daily", 1, 5_000, 1, Some(6_000)) - .expect("encode kv_incr with absolute expiry"); + let entry = encode_kv_incr(KvIncrRecord { + collection: "counters", + key: b"daily", + delta: 1, + ttl_ms: 5_000, + surrogate: 1, + shape: &KvCounterShape::Raw, + expire_at_ms: Some(6_000), + }) + .expect("encode kv_incr with absolute expiry"); let dir = tempfile::tempdir().expect("wal tempdir"); let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); @@ -494,7 +439,7 @@ mod tests { let put_str = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), key: b"str".to_vec(), - value: zerompk::to_msgpack_vec(&"hello").expect("encode"), + value: b"hello".to_vec(), ttl_ms: 0, surrogate: Surrogate::new(1), returning: None, @@ -507,6 +452,7 @@ mod tests { ttl_ms: 0, surrogate: Surrogate::new(1), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let records = append_via_autocommit(&[put_str, incr]); @@ -520,15 +466,66 @@ mod tests { .kv_engine .get(DatabaseId::DEFAULT.as_u64(), TID, "counters", b"str", bytes) .expect("str survives replay"); - let decoded: String = zerompk::from_msgpack(&value).expect("decode string"); assert_eq!( - decoded, "hello", + value, + b"hello".to_vec(), "incr over a non-numeric value must replay to a no-op, value unchanged" ); } #[test] - fn production_wal_append_emits_seven_element_shape_for_ttl_bearing_incr() { + fn kv_incr_on_an_absent_key_replays_the_typed_row_it_created() { + let mut template_row = std::collections::HashMap::new(); + template_row.insert( + "status".to_string(), + nodedb_types::Value::String("new".into()), + ); + let template = nodedb_types::value_to_msgpack(&nodedb_types::Value::Object(template_row)) + .expect("encode template"); + let incr = PhysicalPlan::Kv(KvOp::Incr { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), + key: b"fresh".to_vec(), + delta: 5, + ttl_ms: 0, + surrogate: Surrogate::new(1), + rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Typed { + column: Some("n".into()), + template, + }, + }); + + let records = append_via_autocommit(&[incr]); + + let mut h = make_core(); + h.core.replay_kv_wal(&records, 1, &TombstoneSet::new()); + + let now_ms = crate::engine::kv::current_ms(); + let bytes = h + .core + .kv_engine + .get( + DatabaseId::DEFAULT.as_u64(), + TID, + "counters", + b"fresh", + now_ms, + ) + .expect("the fresh row survives replay"); + let nodedb_types::Value::Object(row) = + nodedb_types::value_from_msgpack(&bytes).expect("decode row") + else { + panic!("replay must recreate a typed row"); + }; + assert_eq!(row.get("n"), Some(&nodedb_types::Value::Integer(5))); + assert_eq!( + row.get("status"), + Some(&nodedb_types::Value::String("new".into())) + ); + } + + #[test] + fn production_wal_append_records_the_resolved_expiry_for_ttl_bearing_incr() { let observed_now_ms = crate::engine::kv::current_ms(); let dir = tempfile::tempdir().expect("wal tempdir"); @@ -541,6 +538,7 @@ mod tests { ttl_ms: 86_400_000, surrogate: Surrogate::new(7), rls_write_check: RlsWriteCheck::already_decided_elsewhere(), + shape: KvCounterShape::Raw, }); let outcome = wal_append_if_write( &wal, @@ -563,14 +561,14 @@ mod tests { .find(|r| r.header.tenant_id == TID) .expect("incr record present"); - let (disc, _collection, _key, _delta, ttl_ms_field, _surrogate, expire_at_ms) = - zerompk::from_msgpack::<(&str, String, Vec, i64, u64, u32, u64)>(&record.payload) - .expect("seven-element kv_incr shape"); + let (disc, _collection, _key, _delta, ttl_ms_field, _surrogate, _shape, expire_at_ms) = + zerompk::from_msgpack::>(&record.payload) + .expect("kv_incr record"); assert_eq!(disc, "kv_incr"); assert_eq!(ttl_ms_field, 86_400_000); assert_eq!( expire_at_ms, - resolved + 86_400_000, + Some(resolved + 86_400_000), "the emitted record must carry the same instant wal_append_if_write resolved" ); } diff --git a/nodedb/src/engine/kv/engine_atomic.rs b/nodedb/src/engine/kv/engine_atomic.rs index becbb6164..e76dcbcd3 100644 --- a/nodedb/src/engine/kv/engine_atomic.rs +++ b/nodedb/src/engine/kv/engine_atomic.rs @@ -6,6 +6,8 @@ //! hash slot). No cross-core coordination is needed because each key maps //! to exactly one core. +use nodedb_physical::physical_plan::KvCounterShape; + use super::engine::KvEngine; use super::engine_atomic_compute as compute; use super::engine_helpers::{expiry_key, table_key}; @@ -38,13 +40,38 @@ pub struct GetSetResult { pub written: Vec, } +/// The value a counter atomic computed, and the bytes it stored. +/// +/// `written` is the whole stored body: the re-encoded row for a typed row, +/// the decimal text for a raw body. A write event carries these bytes. +#[derive(Debug, Clone, PartialEq)] +pub struct Incremented { + /// The new counter value. + pub value: T, + /// The bytes the increment stored. + pub written: Vec, +} + +/// One `INCR` step: the delta, the TTL request, and the row an absent key +/// becomes. +#[derive(Clone, Copy)] +pub struct IncrStep<'a> { + /// The signed increment. + pub delta: i64, + /// TTL in milliseconds. `0` preserves the existing TTL. + pub ttl_ms: u64, + /// The row an absent key becomes. + pub shape: &'a KvCounterShape, +} + /// Errors specific to atomic KV operations. #[derive(Debug)] pub enum AtomicError { - /// Value is not the expected numeric type for INCR/DECR. + /// A typed row has no column of the type the atomic reads. TypeMismatch { detail: String }, - /// Integer overflow on INCR/DECR. - Overflow, + /// A counter atomic read a stored value it cannot parse as a number, or + /// computed a result out of range. + Counter(crate::bridge::envelope::CounterFault), /// The computed new value failed to re-encode as MessagePack. Encode { detail: String }, /// The [`AtomicAdmission`] gate refused the computed post-image, so nothing @@ -91,11 +118,17 @@ pub struct AtomicKeyCtx<'a> { } impl KvEngine { - /// Atomically increment an i64 value by `delta`. Returns the new value. + /// Atomically increment an i64 value by `delta`. Returns the new value + /// and the bytes stored. /// - /// - If key doesn't exist: initializes to 0, adds delta, returns delta. - /// - If value is not a MessagePack integer: returns `TypeMismatch`. - /// - On i64 overflow: returns `Overflow` (never wraps silently). + /// - If key doesn't exist: initializes to 0, adds delta, and stores the + /// row `shape` names: decimal text, or a typed row. + /// - A raw body is read as decimal text and written back as decimal + /// text. A typed row moves its first integer column in key order. + /// - A raw body that is not a decimal i64: returns + /// `Counter(NotAnInteger)`. A typed row without an integer column: + /// returns `TypeMismatch`. + /// - On i64 overflow: returns `Counter(IntegerOverflow)`. It never wraps. /// - TTL behavior: if `ttl_ms > 0` and key is new, sets TTL. /// If key exists and `ttl_ms > 0`, resets TTL. If `ttl_ms == 0`, preserves. /// - If `admit` refuses the computed value: returns `Rejected` and writes @@ -105,9 +138,19 @@ impl KvEngine { ctx: AtomicKeyCtx<'_>, delta: i64, ttl_ms: u64, + shape: &KvCounterShape, admit: AtomicAdmission<'_>, - ) -> Result { - self.incr_resolved(ctx, delta, ttl_ms, None, admit) + ) -> Result, AtomicError> { + self.incr_resolved( + ctx, + IncrStep { + delta, + ttl_ms, + shape, + }, + None, + admit, + ) } /// Atomically increment an i64 value by `delta`, installing an @@ -125,12 +168,11 @@ impl KvEngine { pub fn incr_with_absolute_expiry( &mut self, ctx: AtomicKeyCtx<'_>, - delta: i64, - ttl_ms: u64, + step: IncrStep<'_>, expire_at_ms: u64, admit: AtomicAdmission<'_>, - ) -> Result { - self.incr_resolved(ctx, delta, ttl_ms, Some(expire_at_ms), admit) + ) -> Result, AtomicError> { + self.incr_resolved(ctx, step, Some(expire_at_ms), admit) } /// Shared INCR body: computes the new value, then installs it via @@ -139,56 +181,67 @@ impl KvEngine { fn incr_resolved( &mut self, ctx: AtomicKeyCtx<'_>, - delta: i64, - ttl_ms: u64, + step: IncrStep<'_>, expire_override: Option, admit: AtomicAdmission<'_>, - ) -> Result { + ) -> Result, AtomicError> { + let IncrStep { + delta, + ttl_ms, + shape, + } = step; let tkey = table_key(ctx.database_id, ctx.tenant_id, ctx.collection); let table = self.ensure_table(tkey, ctx.tenant_id, ctx.collection); let current = table.get(ctx.key, ctx.now_ms).map(|v| v.to_vec()); - let (new_i64, new_bytes) = compute::incr(current.as_deref(), delta)?; + let (value, written) = compute::incr(current.as_deref(), delta, shape)?; // Decided before `atomic_put`, so a refused image is never durable and // never reaches the expiry wheel or the secondary indexes. - admit(&new_bytes).map_err(|error| AtomicError::Rejected(Box::new(error)))?; + admit(&written).map_err(|error| AtomicError::Rejected(Box::new(error)))?; self.atomic_put( ctx, tkey, - &new_bytes, + &written, ttl_ms, current.is_none(), expire_override, ); - Ok(new_i64) + Ok(Incremented { value, written }) } - /// Atomically increment an f64 value by `delta`. Returns the new value. + /// Atomically increment an f64 value by `delta`. Returns the new value + /// and the bytes stored. /// - /// - If key doesn't exist: initializes to 0.0, adds delta, returns delta. - /// - If value is not a MessagePack float or integer: returns `TypeMismatch`. - /// - f64 does not overflow in the traditional sense (it goes to infinity), - /// but NaN/Infinity results are rejected as `Overflow`. + /// - `delta` is the client's decimal text. + /// - If key doesn't exist: initializes to 0, adds delta, and stores the + /// row `shape` names: decimal text, or a typed row. + /// - A raw body is read as decimal text and written back as decimal + /// text. A typed row moves its first numeric column in key order. + /// - A raw body that is not a decimal float: returns + /// `Counter(NotAFloat)`. A typed row without a numeric column: returns + /// `TypeMismatch`. + /// - A NaN or infinite result: returns `Counter(NonFinite)`. /// - If `admit` refuses the computed value: returns `Rejected` and writes /// nothing. pub fn incr_float( &mut self, ctx: AtomicKeyCtx<'_>, - delta: f64, + delta: &str, + shape: &KvCounterShape, admit: AtomicAdmission<'_>, - ) -> Result { + ) -> Result, AtomicError> { let tkey = table_key(ctx.database_id, ctx.tenant_id, ctx.collection); let table = self.ensure_table(tkey, ctx.tenant_id, ctx.collection); let current = table.get(ctx.key, ctx.now_ms).map(|v| v.to_vec()); - let (new_f64, new_bytes) = compute::incr_float(current.as_deref(), delta)?; + let (value, written) = compute::incr_float(current.as_deref(), delta, shape)?; // Decided before the value is installed — see `incr_resolved`. - admit(&new_bytes).map_err(|error| AtomicError::Rejected(Box::new(error)))?; + admit(&written).map_err(|error| AtomicError::Rejected(Box::new(error)))?; // incr_float always preserves existing TTL (ttl_ms = 0). - self.atomic_put(ctx, tkey, &new_bytes, 0, current.is_none(), None); + self.atomic_put(ctx, tkey, &written, 0, current.is_none(), None); - Ok(new_f64) + Ok(Incremented { value, written }) } /// Atomic compare-and-swap. @@ -397,6 +450,18 @@ mod tests { use super::super::engine_write::KvPutParams; use super::*; + use crate::bridge::envelope::CounterFault; + + static RAW: KvCounterShape = KvCounterShape::Raw; + + /// A raw-shaped `INCR` step. + fn step(delta: i64, ttl_ms: u64) -> IncrStep<'static> { + IncrStep { + delta, + ttl_ms, + shape: &RAW, + } + } fn make_engine() -> KvEngine { KvEngine::new(1000, 16, 0.75, 4, 64, 1000, 1024) @@ -417,28 +482,36 @@ mod tests { #[test] fn incr_new_key() { let mut engine = make_engine(); - let result = engine.incr(ctx("counters", b"hits"), 10, 0, &admit_any); - assert_eq!(result.unwrap(), 10); + let result = engine + .incr(ctx("counters", b"hits"), 10, 0, &RAW, &admit_any) + .expect("incr"); + assert_eq!(result.value, 10); + assert_eq!(result.written, b"10".to_vec()); + assert_eq!( + engine.get(0, 1, "counters", b"hits", 1000).as_deref(), + Some(b"10".as_slice()), + "the engine stores exactly the bytes it returns" + ); } #[test] fn incr_existing_key() { let mut engine = make_engine(); engine - .incr(ctx("counters", b"hits"), 10, 0, &admit_any) + .incr(ctx("counters", b"hits"), 10, 0, &RAW, &admit_any) .unwrap(); - let result = engine.incr(ctx("counters", b"hits"), 5, 0, &admit_any); - assert_eq!(result.unwrap(), 15); + let result = engine.incr(ctx("counters", b"hits"), 5, 0, &RAW, &admit_any); + assert_eq!(result.expect("incr").value, 15); } #[test] fn incr_negative_delta() { let mut engine = make_engine(); engine - .incr(ctx("counters", b"gold"), 100, 0, &admit_any) + .incr(ctx("counters", b"gold"), 100, 0, &RAW, &admit_any) .unwrap(); - let result = engine.incr(ctx("counters", b"gold"), -30, 0, &admit_any); - assert_eq!(result.unwrap(), 70); + let result = engine.incr(ctx("counters", b"gold"), -30, 0, &RAW, &admit_any); + assert_eq!(result.expect("incr").value, 70); } /// The increment is computed inside the engine, so the gate is the only @@ -448,7 +521,7 @@ mod tests { fn a_refused_increment_writes_nothing() { let mut engine = make_engine(); engine - .incr(ctx("counters", b"hits"), 7, 0, &admit_any) + .incr(ctx("counters", b"hits"), 7, 0, &RAW, &admit_any) .unwrap(); let deny = |_: &[u8]| { @@ -457,21 +530,24 @@ mod tests { resource: "test".into(), }) }; - let result = engine.incr(ctx("counters", b"hits"), 5, 0, &deny); + let result = engine.incr(ctx("counters", b"hits"), 5, 0, &RAW, &deny); assert!(matches!(result, Err(AtomicError::Rejected(_)))); let stored = engine .get(0, 1, "counters", b"hits", 1000) .expect("the refused increment must leave the prior row in place"); - let value: i64 = zerompk::from_msgpack(&stored).unwrap(); - assert_eq!(value, 7, "a refused increment must not be applied"); + assert_eq!( + stored, + b"7".to_vec(), + "a refused increment must not be applied" + ); } #[test] fn incr_overflow() { let mut engine = make_engine(); // Set to MAX. - let bytes = zerompk::to_msgpack_vec(&i64::MAX).unwrap(); + let bytes = i64::MAX.to_string().into_bytes(); engine.put(KvPutParams { database_id: 0, tenant_id: 1, @@ -482,14 +558,17 @@ mod tests { now_ms: 1000, surrogate: Surrogate::ZERO, }); - let result = engine.incr(ctx("counters", b"max"), 1, 0, &admit_any); - assert!(matches!(result, Err(AtomicError::Overflow))); + let result = engine.incr(ctx("counters", b"max"), 1, 0, &RAW, &admit_any); + assert!(matches!( + result, + Err(AtomicError::Counter(CounterFault::IntegerOverflow)) + )); } #[test] - fn incr_type_mismatch() { + fn incr_on_raw_text_that_is_not_an_integer_is_refused() { let mut engine = make_engine(); - let bytes = zerompk::to_msgpack_vec(&"hello").unwrap(); + let bytes = b"hello".to_vec(); engine.put(KvPutParams { database_id: 0, tenant_id: 1, @@ -500,15 +579,18 @@ mod tests { now_ms: 1000, surrogate: Surrogate::ZERO, }); - let result = engine.incr(ctx("counters", b"str"), 1, 0, &admit_any); - assert!(matches!(result, Err(AtomicError::TypeMismatch { .. }))); + let result = engine.incr(ctx("counters", b"str"), 1, 0, &RAW, &admit_any); + assert!(matches!( + result, + Err(AtomicError::Counter(CounterFault::NotAnInteger)) + )); } #[test] fn incr_with_ttl_new_key() { let mut engine = make_engine(); engine - .incr(ctx("counters", b"daily"), 1, 86_400_000, &admit_any) + .incr(ctx("counters", b"daily"), 1, 86_400_000, &RAW, &admit_any) .unwrap(); let ttl = engine.get_ttl_ms(0, 1, "counters", b"daily", 1000); assert!(ttl.is_some()); @@ -519,7 +601,7 @@ mod tests { fn incr_preserves_ttl_when_zero() { let mut engine = make_engine(); // Set key with TTL. - let bytes = zerompk::to_msgpack_vec(&50i64).unwrap(); + let bytes = b"50".to_vec(); engine.put(KvPutParams { database_id: 0, tenant_id: 1, @@ -532,7 +614,7 @@ mod tests { }); // Incr with ttl_ms=0 should preserve existing TTL. engine - .incr(ctx("counters", b"temp"), 10, 0, &admit_any) + .incr(ctx("counters", b"temp"), 10, 0, &RAW, &admit_any) .unwrap(); let ttl = engine.get_ttl_ms(0, 1, "counters", b"temp", 1000); assert!(ttl.is_some()); @@ -546,7 +628,12 @@ mod tests { // 1000 + 5000 = 6000. Passing an explicit absolute instant must // override that derivation entirely. engine - .incr_with_absolute_expiry(ctx("counters", b"daily"), 1, 5_000, 1_000_000, &admit_any) + .incr_with_absolute_expiry( + ctx("counters", b"daily"), + step(1, 5_000), + 1_000_000, + &admit_any, + ) .unwrap(); let ttl = engine.get_ttl_ms(0, 1, "counters", b"daily", 1000); assert_eq!( @@ -559,7 +646,7 @@ mod tests { #[test] fn incr_with_absolute_expiry_and_zero_ttl_still_preserves_existing_expiry() { let mut engine = make_engine(); - let bytes = zerompk::to_msgpack_vec(&50i64).unwrap(); + let bytes = b"50".to_vec(); engine.put(KvPutParams { database_id: 0, tenant_id: 1, @@ -575,7 +662,12 @@ mod tests { // ttl_ms == 0 must ignore the supplied absolute instant and preserve // the existing expiry exactly as `incr` does. engine - .incr_with_absolute_expiry(ctx("counters", b"temp"), 10, 0, 999_999_999, &admit_any) + .incr_with_absolute_expiry( + ctx("counters", b"temp"), + step(10, 0), + 999_999_999, + &admit_any, + ) .unwrap(); let ttl_after = engine.get_ttl_ms(0, 1, "counters", b"temp", 1000); assert_eq!( @@ -587,24 +679,31 @@ mod tests { #[test] fn incr_float_new_key() { let mut engine = make_engine(); - let result = engine.incr_float(ctx("scores", b"dmg"), 3.125, &admit_any); - assert!((result.unwrap() - 3.125).abs() < f64::EPSILON); + let result = engine + .incr_float(ctx("scores", b"dmg"), "3.125", &RAW, &admit_any) + .expect("incr_float"); + assert!((result.value - 3.125).abs() < f64::EPSILON); + assert_eq!(result.written, b"3.125".to_vec()); } #[test] fn incr_float_existing() { let mut engine = make_engine(); engine - .incr_float(ctx("scores", b"dmg"), 3.0, &admit_any) + .incr_float(ctx("scores", b"dmg"), "3.0", &RAW, &admit_any) .unwrap(); - let result = engine.incr_float(ctx("scores", b"dmg"), 1.5, &admit_any); - assert!((result.unwrap() - 4.5).abs() < f64::EPSILON); + let result = engine + .incr_float(ctx("scores", b"dmg"), "1.5", &RAW, &admit_any) + .expect("incr_float"); + assert!((result.value - 4.5).abs() < f64::EPSILON); + assert_eq!(result.written, b"4.5".to_vec()); } #[test] fn incr_float_infinity_rejected() { let mut engine = make_engine(); - let bytes = zerompk::to_msgpack_vec(&f64::MAX).unwrap(); + let bytes_text = f64::MAX.to_string(); + let bytes = bytes_text.clone().into_bytes(); engine.put(KvPutParams { database_id: 0, tenant_id: 1, @@ -615,8 +714,11 @@ mod tests { now_ms: 1000, surrogate: Surrogate::ZERO, }); - let result = engine.incr_float(ctx("scores", b"big"), f64::MAX, &admit_any); - assert!(matches!(result, Err(AtomicError::Overflow))); + let result = engine.incr_float(ctx("scores", b"big"), &bytes_text, &RAW, &admit_any); + assert!(matches!( + result, + Err(AtomicError::Counter(CounterFault::NonFinite)) + )); } #[test] diff --git a/nodedb/src/engine/kv/engine_atomic_compute.rs b/nodedb/src/engine/kv/engine_atomic_compute.rs index 846d8b243..49ab2c406 100644 --- a/nodedb/src/engine/kv/engine_atomic_compute.rs +++ b/nodedb/src/engine/kv/engine_atomic_compute.rs @@ -1,28 +1,44 @@ // SPDX-License-Identifier: BUSL-1.1 //! Pure value computation for `INCR`/`INCR_FLOAT`/`CAS`/`GETSET`, shared by -//! the autocommit `KvEngine` methods (`engine_atomic.rs`) and the -//! in-transaction staging handlers (`stage_kv_atomic.rs`), so a staged value -//! and its COMMIT-time durable replay are always computed by the exact same -//! code. Split out of `engine_atomic.rs` to keep that file under the -//! file-size limit. +//! the autocommit `KvEngine` methods (`engine_atomic.rs`), the in-transaction +//! staging handlers (`stage_kv_atomic.rs`), the resolve handlers, and WAL +//! replay. Every path computes a stored value with the same function, so all +//! of them store the same bytes. +//! +//! A body has one of two shapes ([`kv_body_shape`]). A typed row (a msgpack +//! map) keeps its typed column semantics. A raw body (the single-`value` SQL +//! form, RESP `SET`) is a byte string. `INCR` and `INCR_FLOAT` read it as +//! decimal text by the Redis rules and store the result as decimal text. use std::collections::HashMap; -use nodedb_query::msgpack_scan::{KvBodyShape, row_to_kv_body}; +use nodedb_query::msgpack_scan::{KvBodyShape, kv_body_shape, row_to_kv_body}; use nodedb_types::Value; +use nodedb_physical::physical_plan::KvCounterShape; + use super::engine_atomic::AtomicError; +use super::float_text; +use crate::bridge::envelope::CounterFault; /// The field of a typed row an atomic never targets. const KEY_FIELD: &str = "key"; -/// Decode a map-shaped body into its typed columns. Returns `None` for a -/// body of any other shape. -fn decode_map(bytes: &[u8]) -> Option> { +/// Decode a map-shaped body into its typed columns. Returns `Ok(None)` for a +/// raw body, and `TypeMismatch` for a map-shaped body that does not decode. +fn typed_row(bytes: &[u8]) -> Result>, AtomicError> { + if kv_body_shape(bytes) != KvBodyShape::Map { + return Ok(None); + } match nodedb_types::value_from_msgpack(bytes) { - Ok(Value::Object(map)) => Some(map), - _ => None, + Ok(Value::Object(map)) => Ok(Some(map)), + Ok(other) => Err(AtomicError::TypeMismatch { + detail: format!("stored row is {}, not an object", other.type_name()), + }), + Err(e) => Err(AtomicError::TypeMismatch { + detail: format!("stored row does not decode: {e}"), + }), } } @@ -91,111 +107,133 @@ fn integral_f64_to_i64(v: f64) -> Option { (v.fract() == 0.0 && v >= i64::MIN as f64 && v <= i64::MAX as f64).then_some(v as i64) } -fn not_an_integer() -> AtomicError { +fn not_an_integer_column() -> AtomicError { AtomicError::TypeMismatch { - detail: "value is not an integer".into(), + detail: "row has no integer column".into(), } } -fn not_numeric() -> AtomicError { +fn not_a_numeric_column() -> AtomicError { AtomicError::TypeMismatch { - detail: "value is not numeric".into(), + detail: "row has no numeric column".into(), } } -/// Decode a bare MessagePack scalar as i64. -fn decode_scalar_i64(bytes: &[u8]) -> Result { - // Try i64 first, then u64 (MessagePack encodes small positive as u64). - if let Ok(v) = zerompk::from_msgpack::(bytes) { - return Ok(v); - } - if let Ok(v) = zerompk::from_msgpack::(bytes) { - return i64::try_from(v).map_err(|_| AtomicError::Overflow); - } - // A float with no fractional part truncates to i64. - zerompk::from_msgpack::(bytes) +/// Read a raw body as a decimal i64 by the Redis rule. +fn parse_raw_i64(bytes: &[u8]) -> Result { + std::str::from_utf8(bytes) .ok() - .and_then(integral_f64_to_i64) - .ok_or(not_an_integer()) + .filter(|text| is_canonical_integer(text)) + .and_then(|text| text.parse::().ok()) + .ok_or(AtomicError::Counter(CounterFault::NotAnInteger)) } -/// Decode a bare MessagePack scalar as f64. -fn decode_scalar_f64(bytes: &[u8]) -> Result { - if let Ok(v) = zerompk::from_msgpack::(bytes) { - return Ok(v); - } - // Accept integer values promoted to float. - if let Ok(v) = zerompk::from_msgpack::(bytes) { - return Ok(v as f64); - } - if let Ok(v) = zerompk::from_msgpack::(bytes) { - return Ok(v as f64); - } - Err(not_numeric()) +/// The Redis integer grammar: `0`, or an optional `-` then digits with no +/// leading zero. A `+` sign, whitespace, and an empty body are refused. +fn is_canonical_integer(text: &str) -> bool { + let digits = text.strip_prefix('-').unwrap_or(text); + text == "0" + || (digits + .bytes() + .next() + .is_some_and(|b| (b'1'..=b'9').contains(&b)) + && digits.bytes().all(|b| b.is_ascii_digit())) } -/// Encode an `i64` as MessagePack, wrapping the (practically unreachable, but -/// not type-system-excluded) encode failure in [`AtomicError::Encode`] rather -/// than panicking. -fn encode_i64(v: i64) -> Result, AtomicError> { - zerompk::to_msgpack_vec(&v).map_err(|e| AtomicError::Encode { - detail: format!("i64 re-encode: {e}"), - }) +/// The raw body for an integer: its decimal text, the text +/// `scalar_to_raw_bytes` writes for the same value. +fn raw_decimal(v: i64) -> Vec { + v.to_string().into_bytes() } -/// Encode an `f64` as MessagePack, same rationale as [`encode_i64`]. -fn encode_f64(v: f64) -> Result, AtomicError> { - zerompk::to_msgpack_vec(&v).map_err(|e| AtomicError::Encode { - detail: format!("f64 re-encode: {e}"), - }) +/// The row an absent key becomes under a typed [`KvCounterShape`]: the +/// template with `column` set to `value`. +fn fresh_typed_row( + column: &Option, + template: &[u8], + value: Value, + missing_column: AtomicError, +) -> Result, AtomicError> { + let column = column.as_ref().ok_or(missing_column)?; + let mut map = typed_row(template)?.ok_or(AtomicError::TypeMismatch { + detail: "fresh row template is not a typed row".into(), + })?; + map.insert(column.clone(), value); + encode_map(map) } -/// Compute the new value for `INCR`, given the current raw bytes (if -/// any). Returns `(new_i64, new_bytes)`. +/// Compute the new value for `INCR`, given the current body (if any). +/// Returns `(new_i64, new_bytes)`. /// -/// A typed row keeps its shape: the numeric column [`target_field`] picks -/// moves, and every other column stays. A bare scalar stays a bare scalar. -pub fn incr(current: Option<&[u8]>, delta: i64) -> Result<(i64, Vec), AtomicError> { - if let Some(mut map) = current.and_then(decode_map) { - let (field, old_i64) = target_field(&map, column_i64).ok_or(not_an_integer())?; - let new_i64 = old_i64.checked_add(delta).ok_or(AtomicError::Overflow)?; +/// A typed row keeps its shape: the integer column [`target_field`] picks +/// moves, and every other column stays. A raw body is decimal text in and +/// decimal text out. An absent key starts at 0 and takes `shape`. +pub fn incr( + current: Option<&[u8]>, + delta: i64, + shape: &KvCounterShape, +) -> Result<(i64, Vec), AtomicError> { + let overflow = AtomicError::Counter(CounterFault::IntegerOverflow); + let Some(bytes) = current else { + let written = match shape { + KvCounterShape::Raw => raw_decimal(delta), + KvCounterShape::Typed { column, template } => fresh_typed_row( + column, + template, + Value::Integer(delta), + not_an_integer_column(), + )?, + }; + return Ok((delta, written)); + }; + if let Some(mut map) = typed_row(bytes)? { + let (field, old_i64) = target_field(&map, column_i64).ok_or(not_an_integer_column())?; + let new_i64 = old_i64.checked_add(delta).ok_or(overflow)?; map.insert(field, Value::Integer(new_i64)); return Ok((new_i64, encode_map(map)?)); } - let old_i64 = match current { - None => 0i64, - Some(bytes) => decode_scalar_i64(bytes)?, - }; - let new_i64 = old_i64.checked_add(delta).ok_or(AtomicError::Overflow)?; - Ok((new_i64, encode_i64(new_i64)?)) + let new_i64 = parse_raw_i64(bytes)?.checked_add(delta).ok_or(overflow)?; + Ok((new_i64, raw_decimal(new_i64))) } -/// Compute the new value for `INCR_FLOAT`. Returns `(new_f64, new_bytes)`. +/// Compute the new value for `INCR_FLOAT`. `delta` is the client's decimal +/// text. Returns `(new_f64, new_bytes)`. /// -/// A typed row keeps its shape, as in [`incr`]. A bare scalar is stored as -/// a bare f64. -pub fn incr_float(current: Option<&[u8]>, delta: f64) -> Result<(f64, Vec), AtomicError> { - let mut row = current.and_then(decode_map); - let (field, old_f64) = match (&row, current) { - (Some(map), _) => { - let (field, old) = target_field(map, column_f64).ok_or(not_numeric())?; - (Some(field), old) - } - (None, None) => (None, 0.0f64), - (None, Some(bytes)) => (None, decode_scalar_f64(bytes)?), +/// A typed row keeps its shape, as in [`incr`], and its column adds in +/// `f64`. A raw body is decimal text in and decimal text out, added exactly +/// by the Redis rules (see `float_text`). An absent key starts at 0 and takes +/// `shape`. +pub fn incr_float( + current: Option<&[u8]>, + delta: &str, + shape: &KvCounterShape, +) -> Result<(f64, Vec), AtomicError> { + let Some(bytes) = current else { + return match shape { + KvCounterShape::Raw => float_text::fresh(delta), + KvCounterShape::Typed { column, template } => { + let value = float_text::delta_to_f64(delta)?; + let written = fresh_typed_row( + column, + template, + Value::Float(value), + not_a_numeric_column(), + )?; + Ok((value, written)) + } + }; + }; + let Some(mut map) = typed_row(bytes)? else { + return float_text::add(bytes, delta); }; + let delta = float_text::delta_to_f64(delta)?; + let (field, old_f64) = target_field(&map, column_f64).ok_or(not_a_numeric_column())?; let new_f64 = old_f64 + delta; - if new_f64.is_nan() || new_f64.is_infinite() { - return Err(AtomicError::Overflow); + if !new_f64.is_finite() { + return Err(AtomicError::Counter(CounterFault::NonFinite)); } - let new_bytes = match (row.take(), field) { - (Some(mut map), Some(field)) => { - map.insert(field, Value::Float(new_f64)); - encode_map(map)? - } - _ => encode_f64(new_f64)?, - }; - Ok((new_f64, new_bytes)) + map.insert(field, Value::Float(new_f64)); + Ok((new_f64, encode_map(map)?)) } /// Write `new_value` into the string column of the typed row `row` and @@ -215,7 +253,7 @@ fn swap_string_column( /// A typed row and its string column, when `current` is a typed row with /// one. [`cas`] and [`getset`] address the same column. fn string_column(current: Option<&[u8]>) -> Option<(HashMap, String, String)> { - let row = current.and_then(decode_map)?; + let row = typed_row(current?).ok().flatten()?; let (column, text) = target_field(&row, column_string)?; Some((row, column, text)) } @@ -265,6 +303,16 @@ pub fn getset(current: Option<&[u8]>, new_value: &[u8]) -> Result, Atomi mod tests { use super::*; + static RAW: KvCounterShape = KvCounterShape::Raw; + + /// A typed shape moving `column`, with `rest` as the other stored columns. + fn typed_shape(column: Option<&str>, rest: &[(&str, Value)]) -> KvCounterShape { + KvCounterShape::Typed { + column: column.map(str::to_string), + template: row(rest), + } + } + fn row(fields: &[(&str, Value)]) -> Vec { let map: HashMap = fields .iter() @@ -274,13 +322,15 @@ mod tests { } fn columns(bytes: &[u8]) -> HashMap { - decode_map(bytes).expect("a typed row stays a typed row") + typed_row(bytes) + .expect("a typed row decodes") + .expect("a typed row stays a typed row") } #[test] fn incr_on_a_one_column_typed_row_keeps_the_row() { let current = row(&[("n", Value::Integer(5))]); - let (new_i64, bytes) = incr(Some(¤t), 3).expect("incr"); + let (new_i64, bytes) = incr(Some(¤t), 3, &RAW).expect("incr"); assert_eq!(new_i64, 8); assert_eq!(columns(&bytes).get("n"), Some(&Value::Integer(8))); } @@ -292,7 +342,7 @@ mod tests { ("a", Value::Integer(1)), ("label", Value::String("x".into())), ]); - let (new_i64, bytes) = incr(Some(¤t), 1).expect("incr"); + let (new_i64, bytes) = incr(Some(¤t), 1, &RAW).expect("incr"); assert_eq!(new_i64, 2); let cols = columns(&bytes); assert_eq!(cols.get("a"), Some(&Value::Integer(2))); @@ -307,9 +357,9 @@ mod tests { ("b", Value::Integer(2)), ("c", Value::Integer(3)), ]); - let (_, first) = incr(Some(¤t), 1).expect("incr"); + let (_, first) = incr(Some(¤t), 1, &RAW).expect("incr"); for _ in 0..16 { - let (_, again) = incr(Some(¤t), 1).expect("incr"); + let (_, again) = incr(Some(¤t), 1, &RAW).expect("incr"); assert_eq!(again, first); } } @@ -318,29 +368,155 @@ mod tests { fn incr_on_a_typed_row_without_a_numeric_column_is_a_type_mismatch() { let current = row(&[("label", Value::String("x".into()))]); assert!(matches!( - incr(Some(¤t), 1), + incr(Some(¤t), 1, &RAW), Err(AtomicError::TypeMismatch { .. }) )); } #[test] - fn incr_on_a_bare_scalar_stays_a_bare_scalar() { - let current = zerompk::to_msgpack_vec(&5i64).expect("encode"); - let (new_i64, bytes) = incr(Some(¤t), 3).expect("incr"); - assert_eq!(new_i64, 8); - assert_eq!(zerompk::from_msgpack::(&bytes).expect("decode"), 8); - let (fresh, _) = incr(None, 4).expect("incr"); + fn incr_on_a_raw_body_reads_and_writes_decimal_text() { + let (new_i64, bytes) = incr(Some(b"5"), 1, &RAW).expect("incr"); + assert_eq!(new_i64, 6); + assert_eq!(bytes, b"6".to_vec()); + + let (new_i64, bytes) = incr(Some(b"-10"), 3, &RAW).expect("incr"); + assert_eq!(new_i64, -7); + assert_eq!(bytes, b"-7".to_vec()); + + let (fresh, bytes) = incr(None, 4, &RAW).expect("incr"); assert_eq!(fresh, 4); + assert_eq!(bytes, b"4".to_vec()); + } + + #[test] + fn incr_on_non_integer_raw_text_is_not_an_integer() { + for body in [ + b"abc".as_slice(), + b"", + b"1.5", + b"+5", + b"05", + b"-0", + b" 5", + b"5 ", + b"99999999999999999999", + ] { + assert!( + matches!( + incr(Some(body), 1, &RAW), + Err(AtomicError::Counter(CounterFault::NotAnInteger)) + ), + "{:?}", + String::from_utf8_lossy(body) + ); + } + } + + #[test] + fn incr_past_the_i64_range_is_an_overflow() { + let max = i64::MAX.to_string(); + assert!(matches!( + incr(Some(max.as_bytes()), 1, &RAW), + Err(AtomicError::Counter(CounterFault::IntegerOverflow)) + )); + let min = i64::MIN.to_string(); + assert!(matches!( + incr(Some(min.as_bytes()), -1, &RAW), + Err(AtomicError::Counter(CounterFault::IntegerOverflow)) + )); + let (value, bytes) = incr(Some(min.as_bytes()), 0, &RAW).expect("i64::MIN parses"); + assert_eq!(value, i64::MIN); + assert_eq!(bytes, min.into_bytes()); + } + + #[test] + fn incr_float_on_a_raw_body_reads_and_writes_decimal_text() { + let (new_f64, bytes) = incr_float(Some(b"1.5"), "1", &RAW).expect("incr_float"); + assert_eq!(new_f64, 2.5); + assert_eq!(bytes, b"2.5".to_vec()); + + let (new_f64, bytes) = incr_float(Some(b"10.5"), "0.5", &RAW).expect("incr_float"); + assert_eq!(new_f64, 11.0); + assert_eq!(bytes, b"11".to_vec()); + + let (_, bytes) = incr_float(Some(b"5"), "0.25", &RAW).expect("incr_float"); + assert_eq!(bytes, b"5.25".to_vec()); + + for (stored, delta, expected) in [ + ("0.1", "0.2", "0.3"), + ("10.5", "0.1", "10.6"), + ("5.0e3", "200", "5200"), + ("3.0", "0", "3"), + ("-1.5", "1.5", "0"), + ("1", "0.12345678901234567891", "1.12345678901234567891"), + ] { + let (_, bytes) = incr_float(Some(stored.as_bytes()), delta, &RAW).expect("incr_float"); + assert_eq!(bytes, expected.as_bytes().to_vec(), "{stored} + {delta}"); + } + } + + #[test] + fn incr_float_on_non_numeric_raw_text_is_not_a_float() { + for body in [b"abc".as_slice(), b"", b"NaN", b" 1.5"] { + assert!( + matches!( + incr_float(Some(body), "1", &RAW), + Err(AtomicError::Counter(CounterFault::NotAFloat)) + ), + "{:?}", + String::from_utf8_lossy(body) + ); + } + } + + #[test] + fn incr_float_to_infinity_is_non_finite() { + let max = f64::MAX.to_string(); + assert!(matches!( + incr_float(Some(max.as_bytes()), &max, &RAW), + Err(AtomicError::Counter(CounterFault::NonFinite)) + )); } #[test] fn incr_float_on_a_one_column_typed_row_keeps_the_row() { let current = row(&[("score", Value::Float(1.5))]); - let (new_f64, bytes) = incr_float(Some(¤t), 1.0).expect("incr_float"); + let (new_f64, bytes) = incr_float(Some(¤t), "1", &RAW).expect("incr_float"); assert_eq!(new_f64, 2.5); assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); } + #[test] + fn incr_on_an_absent_key_under_a_typed_shape_creates_the_typed_row() { + let shape = typed_shape(Some("n"), &[("status", Value::String("new".into()))]); + let (value, bytes) = incr(None, 7, &shape).expect("incr"); + assert_eq!(value, 7); + let cols = columns(&bytes); + assert_eq!(cols.get("n"), Some(&Value::Integer(7))); + assert_eq!(cols.get("status"), Some(&Value::String("new".into()))); + } + + #[test] + fn incr_float_on_an_absent_key_under_a_typed_shape_creates_the_typed_row() { + let shape = typed_shape(Some("score"), &[]); + let (value, bytes) = incr_float(None, "2.5", &shape).expect("incr_float"); + assert_eq!(value, 2.5); + assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); + } + + #[test] + fn an_absent_key_under_a_typed_shape_without_a_column_is_a_type_mismatch() { + let shape = typed_shape(None, &[]); + assert!(matches!( + incr(None, 1, &shape), + Err(AtomicError::TypeMismatch { .. }) + )); + assert!(matches!( + incr_float(None, "1", &shape), + Err(AtomicError::TypeMismatch { .. }) + )); + } + #[test] fn cas_on_a_one_column_typed_row_swaps_the_column() { let current = row(&[("state", Value::String("idle".into()))]); diff --git a/nodedb/src/engine/kv/engine_sorted.rs b/nodedb/src/engine/kv/engine_sorted.rs index 48e52e4f5..973f9f1f2 100644 --- a/nodedb/src/engine/kv/engine_sorted.rs +++ b/nodedb/src/engine/kv/engine_sorted.rs @@ -348,9 +348,14 @@ mod tests { // `ttl_ms == 0` preserves whatever TTL the key already has, so the // increment under test is the only thing this write changes. 0, + &nodedb_physical::physical_plan::KvCounterShape::Raw, &admit_any, ); - assert_eq!(updated.ok(), Some(99), "p1's score must become 10 + 89"); + assert_eq!( + updated.ok().map(|result| result.value), + Some(99), + "p1's score must become 10 + 89" + ); assert_eq!( ranked_keys(e.sorted_index_top_k(0, 1, "lb", 10, n)), diff --git a/nodedb/src/engine/kv/float_text.rs b/nodedb/src/engine/kv/float_text.rs new file mode 100644 index 000000000..bb80a302a --- /dev/null +++ b/nodedb/src/engine/kv/float_text.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `INCRBYFLOAT` on a raw KV body: decimal text in, decimal text out. +//! +//! Redis adds in `long double` and prints the sum with 17 fractional digits, +//! trailing zeros trimmed, so `"0.1"` plus `0.2` stores `"0.3"`. Rust has no +//! `long double`. Exact decimal addition gives the same text for every sum +//! that fits a [`Decimal`]: 28 significant digits, magnitude below 7.9e28. +//! An operand or a sum outside that range is added in `f64` instead. + +use std::str::FromStr; + +use rust_decimal::Decimal; +use rust_decimal::prelude::ToPrimitive; + +use super::engine_atomic::AtomicError; +use crate::bridge::envelope::CounterFault; + +/// Add `delta` to the raw body `stored`. Returns the new value and the text +/// to store. +/// +/// `stored` and `delta` must both be decimal numbers (see +/// [`is_decimal_number`]). Anything else is `Counter(NotAFloat)`. A sum that +/// is not finite is `Counter(NonFinite)`. +pub(super) fn add(stored: &[u8], delta: &str) -> Result<(f64, Vec), AtomicError> { + let text = std::str::from_utf8(stored) + .ok() + .filter(|text| is_decimal_number(text)) + .ok_or(AtomicError::Counter(CounterFault::NotAFloat))?; + let delta_f64 = delta_to_f64(delta)?; + if let (Some(base), Some(step)) = (parse_decimal(text), parse_decimal(delta)) + && let Some(sum) = base.checked_add(step) + { + let sum = sum.normalize(); + let value = sum + .to_f64() + .ok_or(AtomicError::Counter(CounterFault::NonFinite))?; + return Ok((value, sum.to_string().into_bytes())); + } + let base: f64 = text + .parse() + .map_err(|_| AtomicError::Counter(CounterFault::NotAFloat))?; + let value = base + delta_f64; + if !value.is_finite() { + return Err(AtomicError::Counter(CounterFault::NonFinite)); + } + Ok((value, float_text(value))) +} + +/// The text for a fresh float counter: `0` plus `delta`. +pub(super) fn fresh(delta: &str) -> Result<(f64, Vec), AtomicError> { + add(b"0", delta) +} + +/// `delta` as a finite `f64`, for a typed column. A delta that is not a +/// decimal number is `Counter(NotAFloat)`. One outside the `f64` range is +/// `Counter(NonFinite)`. +pub fn delta_to_f64(delta: &str) -> Result { + if !is_decimal_number(delta) { + return Err(AtomicError::Counter(CounterFault::NotAFloat)); + } + let value: f64 = delta + .parse() + .map_err(|_| AtomicError::Counter(CounterFault::NotAFloat))?; + if value.is_finite() { + Ok(value) + } else { + Err(AtomicError::Counter(CounterFault::NonFinite)) + } +} + +/// The exact value of `text`, or `None` when it does not fit a [`Decimal`]. +fn parse_decimal(text: &str) -> Option { + if text.contains(['e', 'E']) { + Decimal::from_scientific(text).ok() + } else { + Decimal::from_str(text).ok() + } +} + +/// The text for an `f64` sum outside the [`Decimal`] range: plain decimal +/// digits with no exponent, the form Redis prints. +fn float_text(value: f64) -> Vec { + let text = value.to_string(); + if text == "-0" { + b"0".to_vec() + } else { + text.into_bytes() + } +} + +/// The number grammar `INCRBYFLOAT` accepts, for a stored body and for the +/// client's increment: an optional sign, then digits with an optional point +/// (at least one digit), then an optional exponent `e` or `E` with an +/// optional sign and at least one digit. No whitespace, digit separators, +/// `inf`, or `nan`. +pub fn is_decimal_number(text: &str) -> bool { + let bytes = text.as_bytes(); + let mut i = 0; + if matches!(bytes.first(), Some(b'+' | b'-')) { + i += 1; + } + let int_digits = count_digits(&bytes[i..]); + i += int_digits; + let mut frac_digits = 0; + if bytes.get(i) == Some(&b'.') { + i += 1; + frac_digits = count_digits(&bytes[i..]); + i += frac_digits; + } + if int_digits + frac_digits == 0 { + return false; + } + if matches!(bytes.get(i), Some(b'e' | b'E')) { + i += 1; + if matches!(bytes.get(i), Some(b'+' | b'-')) { + i += 1; + } + let exp_digits = count_digits(&bytes[i..]); + if exp_digits == 0 { + return false; + } + i += exp_digits; + } + i == bytes.len() +} + +fn count_digits(bytes: &[u8]) -> usize { + bytes.iter().take_while(|b| b.is_ascii_digit()).count() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn text_of(stored: &str, delta: &str) -> String { + let (_, bytes) = add(stored.as_bytes(), delta).expect("add"); + String::from_utf8(bytes).expect("UTF-8") + } + + #[test] + fn decimal_text_adds_exactly_like_redis() { + assert_eq!(text_of("0.1", "0.2"), "0.3"); + assert_eq!(text_of("10.5", "0.1"), "10.6"); + assert_eq!(text_of("5.0e3", "200"), "5200"); + assert_eq!(text_of("3.0", "0"), "3"); + assert_eq!(text_of("-1.5", "1.5"), "0"); + assert_eq!(text_of("1.5", "1"), "2.5"); + assert_eq!(text_of("+2", "-0.5"), "1.5"); + assert_eq!(text_of("1E-2", "0"), "0.01"); + assert_eq!(text_of("1", "1e1"), "11"); + } + + #[test] + fn a_twenty_digit_delta_adds_exactly() { + assert_eq!( + text_of("1", "0.12345678901234567891"), + "1.12345678901234567891" + ); + assert_eq!(text_of("10000000000000000000", "1"), "10000000000000000001"); + } + + #[test] + fn the_returned_value_matches_the_stored_text() { + let (value, bytes) = add(b"0.1", "0.2").expect("add"); + assert_eq!(value, 0.3); + assert_eq!(bytes, b"0.3".to_vec()); + } + + #[test] + fn a_fresh_counter_stores_the_delta_text() { + assert_eq!(fresh("2.5").expect("fresh").1, b"2.5".to_vec()); + assert_eq!(fresh("0").expect("fresh").1, b"0".to_vec()); + assert_eq!(fresh("-0.0").expect("fresh").1, b"0".to_vec()); + } + + #[test] + fn a_sum_outside_the_decimal_range_adds_in_f64() { + let (value, bytes) = add(b"1e300", "1").expect("add"); + assert_eq!(value, 1e300); + assert_eq!(bytes, 1e300f64.to_string().into_bytes()); + } + + #[test] + fn text_that_is_not_a_number_is_not_a_float() { + for stored in [ + "abc", "", "NaN", "inf", " 1.5", "1.5 ", "1_000", ".", "1e", "e5", "0x10", + ] { + assert!( + matches!( + add(stored.as_bytes(), "1"), + Err(AtomicError::Counter(CounterFault::NotAFloat)) + ), + "{stored:?}" + ); + } + } + + #[test] + fn a_non_finite_sum_is_refused() { + let max = f64::MAX.to_string(); + assert!(matches!( + add(max.as_bytes(), &max), + Err(AtomicError::Counter(CounterFault::NonFinite)) + )); + assert!(matches!( + add(b"1", "1e400"), + Err(AtomicError::Counter(CounterFault::NonFinite)) + )); + } + + #[test] + fn a_delta_that_is_not_a_number_is_not_a_float() { + for delta in ["abc", "", "inf", "NaN", " 1"] { + assert!( + matches!( + add(b"1", delta), + Err(AtomicError::Counter(CounterFault::NotAFloat)) + ), + "{delta:?}" + ); + } + } +} diff --git a/nodedb/src/engine/kv/mod.rs b/nodedb/src/engine/kv/mod.rs index 2b182f567..5af11ece0 100644 --- a/nodedb/src/engine/kv/mod.rs +++ b/nodedb/src/engine/kv/mod.rs @@ -13,6 +13,7 @@ mod engine_stats; mod engine_write; pub mod entry; pub mod expiry_wheel; +pub mod float_text; mod hash_helpers; pub mod hash_table; pub mod index; @@ -26,7 +27,8 @@ pub use engine::{ KvEngine, KvEntryImage, KvKeyRef, RestoreCompositeIndexParams, RestoreFieldIndexParams, }; pub use engine_atomic::{ - AtomicAdmission, AtomicError, AtomicKeyCtx, CasResult, GetSetResult, admit_any, + AtomicAdmission, AtomicError, AtomicKeyCtx, CasResult, GetSetResult, IncrStep, Incremented, + admit_any, }; pub use engine_atomic_compute as atomic_compute; pub use engine_index::RegisterIndexParams; diff --git a/nodedb/src/error/types.rs b/nodedb/src/error/types.rs index 45f820daf..58c686df5 100644 --- a/nodedb/src/error/types.rs +++ b/nodedb/src/error/types.rs @@ -148,9 +148,6 @@ pub enum Error { detail: String, }, - #[error("arithmetic overflow on {collection} key {key}")] - OverflowError { collection: String, key: String }, - #[error("insufficient balance on {collection} key {key}: {detail}")] InsufficientBalance { collection: String, diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index 66ddc30ab..48da6ab87 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -103,9 +103,6 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { Error::TypeMismatch { collection, detail, .. } => NodeDbError::type_mismatch(collection.clone(), detail), - Error::OverflowError { collection, key } => { - NodeDbError::overflow(collection.clone(), format!("key {key}")) - } Error::InsufficientBalance { collection, key, diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index dfeab5e45..c16acb085 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -11,7 +11,7 @@ use nodedb_types::error::{ErrorCode as PublicCode, NodeDbError}; -use crate::bridge::envelope::ErrorCode; +use crate::bridge::envelope::{CounterFault, ErrorCode}; /// Convert a deterministic Data-Plane code into the public error a client /// can classify. @@ -107,9 +107,16 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { ErrorCode::TypeMismatch { collection, detail } => { NodeDbError::type_mismatch(collection, detail) } - ErrorCode::OverflowError { collection } => { - NodeDbError::overflow(collection, "arithmetic overflow") - } + // The same text the SQL surfaces send, with the collection in the + // details. RESP renders the bare Redis text from the code itself. + ErrorCode::CounterFault { collection, fault } => NodeDbError::kv_counter_fault( + collection, + fault.message(), + matches!( + fault, + CounterFault::IntegerOverflow | CounterFault::NonFinite + ), + ), ErrorCode::InsufficientBalance { collection, detail } => { NodeDbError::insufficient_balance(collection, detail) } @@ -190,6 +197,41 @@ mod tests { ); } + #[test] + fn counter_fault_carries_the_collection() { + let e = data_plane_code_to_public(ErrorCode::CounterFault { + collection: "counters".into(), + fault: CounterFault::NotAnInteger, + }); + assert_eq!(e.code(), PublicCode::TYPE_MISMATCH); + assert_eq!( + e.message(), + "value is not an integer or out of range on counters" + ); + assert_eq!( + e.details(), + &nodedb_types::error::ErrorDetails::TypeMismatch { + collection: "counters".into() + } + ); + + let e = data_plane_code_to_public(ErrorCode::CounterFault { + collection: "counters".into(), + fault: CounterFault::IntegerOverflow, + }); + assert_eq!(e.code(), PublicCode::OVERFLOW); + assert_eq!( + e.message(), + "increment or decrement would overflow on counters" + ); + assert_eq!( + e.details(), + &nodedb_types::error::ErrorDetails::Overflow { + collection: "counters".into() + } + ); + } + #[test] fn internal_stays_internal() { let e = data_plane_code_to_public(ErrorCode::Internal { diff --git a/nodedb/tests/crash_resp_kv_write.rs b/nodedb/tests/crash_resp_kv_write.rs index 140ddcc76..ef68b549a 100644 --- a/nodedb/tests/crash_resp_kv_write.rs +++ b/nodedb/tests/crash_resp_kv_write.rs @@ -104,6 +104,78 @@ async fn resp_kv_set_survives_kill_9() { ); } +/// WAL replay recomputes each counter increment from the replayed body, so +/// replay must read and write decimal text the way the live write did. +#[tokio::test(flavor = "multi_thread")] +async fn resp_kv_counters_replay_to_the_acknowledged_values_after_kill_9() { + let mut h = no_incidental_checkpoint(); + let spawned_at = Instant::now(); + h.spawn(); + h.wait_ready(); + + h.exec( + "CREATE COLLECTION resp_kv_counters (key TEXT PRIMARY KEY, value TEXT) \ + WITH (engine='kv')", + ) + .await; + let password = resp_test_password(); + h.exec(&format!( + "CREATE USER resp_kv_counter_user PASSWORD '{password}'" + )) + .await; + h.exec("GRANT ROLE readwrite TO resp_kv_counter_user").await; + + let mut client = resp_client::session( + resp_addr(h.resp_port), + "resp_kv_counter_user", + &password, + "resp_kv_counters", + ) + .await; + + assert_eq!( + client.cmd(&["SET", "hits", "41"]).await, + Reply::Simple("OK".to_string()) + ); + assert_eq!( + client.cmd(&["INCRBY", "hits", "1"]).await, + Reply::Integer(42) + ); + assert_eq!(client.cmd(&["INCR", "fresh"]).await, Reply::Integer(1)); + assert_eq!( + client.cmd(&["SET", "score", "1.5"]).await, + Reply::Simple("OK".to_string()) + ); + assert_eq!( + client.cmd(&["INCRBYFLOAT", "score", "1"]).await, + Reply::Bulk(Some("2.5".to_string())) + ); + + assert!( + spawned_at.elapsed() < MAX_TEST_WALL_CLOCK, + "test ran long enough that an incidental checkpoint cycle becomes possible \ + even with the interval pushed out; tighten the test or the bound" + ); + h.kill_9(); + h.reopen(); + + let mut post_crash_client = resp_client::session( + resp_addr(h.resp_port), + "resp_kv_counter_user", + &password, + "resp_kv_counters", + ) + .await; + for (key, expected) in [("hits", "42"), ("fresh", "1"), ("score", "2.5")] { + let got = post_crash_client.cmd(&["GET", key]).await; + assert_eq!( + got, + Reply::Bulk(Some(expected.to_string())), + "counter {key} must replay to the value acknowledged before kill -9" + ); + } +} + #[tokio::test(flavor = "multi_thread")] async fn http_query_kv_write_survives_kill_9() { let mut h = no_incidental_checkpoint(); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs index 4102c4a63..0ca7b6cd1 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs @@ -374,6 +374,7 @@ fn staged_incr_with_ttl_is_observed_by_in_tx_get_ttl() { ttl_ms: 30_000, surrogate: nodedb_types::Surrogate::ZERO, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, })), }); let resp = send_txn(&mut core, &mut tx, &mut rx, txn_id, stage_incr); diff --git a/nodedb/tests/native/cases/mod.rs b/nodedb/tests/native/cases/mod.rs index 703daefc9..4a5882c28 100644 --- a/nodedb/tests/native/cases/mod.rs +++ b/nodedb/tests/native/cases/mod.rs @@ -8,6 +8,7 @@ mod native_dml_affected_counts; mod native_dml_outcome_conformance; mod native_error_code_classification; mod native_gateway_txn_overlay; +mod native_kv_counter_faults; mod native_primary_key_nullability; mod native_protocol; mod native_result_projection; diff --git a/nodedb/tests/native/cases/native_kv_counter_faults.rs b/nodedb/tests/native/cases/native_kv_counter_faults.rs new file mode 100644 index 000000000..fea37ce41 --- /dev/null +++ b/nodedb/tests/native/cases/native_kv_counter_faults.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A KV counter fault over the native protocol keeps its collection. +//! +//! `KV_INCR` on a raw value that is not a decimal integer, or past the i64 +//! range, is refused by the Data Plane with the collection it ran on. The +//! native frame carries the data-exception SQLSTATE pgwire sends, the same +//! message text, the numeric code, and structured details that name the +//! collection. + +use nodedb_test_support::native_harness::{do_handshake, send_sql}; +use nodedb_test_support::pgwire_harness::TestServer; + +use nodedb_types::error::{ErrorCode, ErrorDetails, NodeDbError, sqlstate}; +use nodedb_types::protocol::opcodes::ResponseStatus; +use nodedb_types::protocol::{ErrorPayload, HelloFrame}; +use tokio::net::TcpStream; + +const COLLECTION: &str = "native_counters"; + +async fn native_session(srv: &TestServer) -> TcpStream { + let addr = format!("127.0.0.1:{}", srv.native_port) + .parse() + .expect("native addr"); + let (stream, _ack) = do_handshake(addr, &HelloFrame::current()) + .await + .expect("native handshake"); + stream +} + +async fn seeded_server() -> TestServer { + let server = TestServer::start().await; + server + .exec(&format!( + "CREATE COLLECTION {COLLECTION} (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')" + )) + .await + .unwrap(); + server + .exec(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('name', 'abc')" + )) + .await + .unwrap(); + server + .exec(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('max', '{}')", + i64::MAX + )) + .await + .unwrap(); + server +} + +async fn refusal(stream: &mut TcpStream, seq: u64, sql: &str) -> ErrorPayload { + let resp = send_sql(stream, seq, sql).await; + assert_eq!(resp.status, ResponseStatus::Error, "{sql} must be refused"); + resp.error.expect("error payload expected") +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn native_counter_parse_fault_names_the_collection() { + let server = seeded_server().await; + let mut stream = native_session(&server).await; + + let err = refusal( + &mut stream, + 1, + &format!("SELECT KV_INCR('{COLLECTION}', 'name', 1)"), + ) + .await; + assert_eq!(err.code, sqlstate::INVALID_TEXT_REPRESENTATION); + assert_eq!(err.ndb_code, ErrorCode::TYPE_MISMATCH.0); + assert_eq!( + err.message, + format!("value is not an integer or out of range on {COLLECTION}") + ); + let expected = ErrorDetails::TypeMismatch { + collection: COLLECTION.into(), + }; + assert_eq!(err.details.as_ref(), Some(&expected)); + + let typed = NodeDbError::from_wire_with_details( + ErrorCode(err.ndb_code), + err.message.clone(), + err.details.clone(), + ); + assert_eq!(typed.details(), &expected); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn native_counter_overflow_names_the_collection() { + let server = seeded_server().await; + let mut stream = native_session(&server).await; + + let err = refusal( + &mut stream, + 1, + &format!("SELECT KV_INCR('{COLLECTION}', 'max', 1)"), + ) + .await; + assert_eq!(err.code, sqlstate::NUMERIC_VALUE_OUT_OF_RANGE); + assert_eq!(err.ndb_code, ErrorCode::OVERFLOW.0); + assert_eq!( + err.message, + format!("increment or decrement would overflow on {COLLECTION}") + ); + assert_eq!( + err.details, + Some(ErrorDetails::Overflow { + collection: COLLECTION.into(), + }) + ); +} diff --git a/nodedb/tests/wire/cases/kv_bare_value_counters.rs b/nodedb/tests/wire/cases/kv_bare_value_counters.rs new file mode 100644 index 000000000..a878c2fe7 --- /dev/null +++ b/nodedb/tests/wire/cases/kv_bare_value_counters.rs @@ -0,0 +1,253 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Counter atomics on a KV row stored as raw bytes. +//! +//! A single-`value` SQL insert and RESP `SET` store the value as a byte +//! string: an integer is its decimal text. `INCR`, `INCRBY`, `DECR`, and +//! `INCRBYFLOAT` read that text by the Redis rules and store the result as +//! decimal text, so RESP `GET` and a SQL `SELECT value` read the new number +//! back. A value that does not parse, and a result out of range, answer the +//! Redis error text over RESP and a data-exception SQLSTATE over pgwire. + +use crate::harness::TestServer; +use crate::harness::resp_client::Reply; + +const COLLECTION: &str = "kvcount"; + +async fn create_bare_value_collection(server: &TestServer) { + server + .exec(&format!( + "CREATE COLLECTION {COLLECTION} (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')" + )) + .await + .unwrap(); +} + +async fn value_of(server: &TestServer, key: &str) -> String { + let rows = server + .query_text(&format!( + "SELECT value FROM {COLLECTION} WHERE key = '{key}'" + )) + .await + .unwrap(); + assert_eq!( + rows.len(), + 1, + "expected exactly one row for {key}, got {rows:?}" + ); + rows[0].clone() +} + +/// Parse the JSON payload `SELECT KV_*(...)` returns as its single text column. +fn json_of(rows: &[String]) -> serde_json::Value { + serde_json::from_str(&rows[0]).expect("KV_* result must be JSON") +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn resp_incr_on_set_decimal_text_counts_from_the_stored_number() { + let server = TestServer::start().await; + create_bare_value_collection(&server).await; + let mut resp = server.resp_session("kvcount_incr_user", COLLECTION).await; + + assert_eq!( + resp.cmd(&["SET", "k", "5"]).await, + Reply::Simple("OK".into()) + ); + assert_eq!( + resp.cmd(&["INCR", "k"]).await, + Reply::Integer(6), + "INCR reads the stored text \"5\" as the number 5" + ); + assert_eq!( + resp.cmd(&["GET", "k"]).await, + Reply::Bulk(Some("6".into())), + "INCR stores the result as decimal text" + ); + + assert_eq!( + resp.cmd(&["SET", "big", "41"]).await, + Reply::Simple("OK".into()) + ); + assert_eq!(resp.cmd(&["INCRBY", "big", "1"]).await, Reply::Integer(42)); + assert_eq!(resp.cmd(&["DECR", "big"]).await, Reply::Integer(41)); + assert_eq!( + resp.cmd(&["GET", "big"]).await, + Reply::Bulk(Some("41".into())) + ); + assert_eq!(value_of(&server, "big").await, "41"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn resp_incr_on_text_that_is_not_an_integer_answers_the_redis_error() { + let server = TestServer::start().await; + create_bare_value_collection(&server).await; + let mut resp = server.resp_session("kvcount_nan_user", COLLECTION).await; + + assert_eq!( + resp.cmd(&["SET", "k", "abc"]).await, + Reply::Simple("OK".into()) + ); + assert_eq!( + resp.cmd(&["INCR", "k"]).await, + Reply::Error("ERR value is not an integer or out of range".into()) + ); + assert_eq!( + resp.cmd(&["INCRBYFLOAT", "k", "1"]).await, + Reply::Error("ERR value is not a valid float".into()) + ); + assert_eq!( + resp.cmd(&["GET", "k"]).await, + Reply::Bulk(Some("abc".into())), + "a refused increment leaves the value unchanged" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn resp_incrby_past_the_i64_range_answers_the_overflow_error() { + let server = TestServer::start().await; + create_bare_value_collection(&server).await; + let mut resp = server.resp_session("kvcount_ovf_user", COLLECTION).await; + + let max = i64::MAX.to_string(); + assert_eq!( + resp.cmd(&["SET", "k", &max]).await, + Reply::Simple("OK".into()) + ); + assert_eq!( + resp.cmd(&["INCRBY", "k", "1"]).await, + Reply::Error("ERR increment or decrement would overflow".into()) + ); + assert_eq!(resp.cmd(&["GET", "k"]).await, Reply::Bulk(Some(max))); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn resp_incrbyfloat_on_decimal_text_stores_decimal_text() { + let server = TestServer::start().await; + create_bare_value_collection(&server).await; + let mut resp = server.resp_session("kvcount_float_user", COLLECTION).await; + + assert_eq!( + resp.cmd(&["SET", "k", "1.5"]).await, + Reply::Simple("OK".into()) + ); + assert_eq!( + resp.cmd(&["INCRBYFLOAT", "k", "1"]).await, + Reply::Bulk(Some("2.5".into())) + ); + assert_eq!( + resp.cmd(&["GET", "k"]).await, + Reply::Bulk(Some("2.5".into())) + ); + assert_eq!(value_of(&server, "k").await, "2.5"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn resp_incrbyfloat_adds_decimal_text_exactly_like_redis() { + let server = TestServer::start().await; + create_bare_value_collection(&server).await; + let mut resp = server.resp_session("kvcount_exact_user", COLLECTION).await; + + for (key, stored, delta, expected) in [ + ("a", "0.1", "0.2", "0.3"), + ("b", "10.5", "0.1", "10.6"), + ("c", "5.0e3", "200", "5200"), + ("d", "3.0", "0", "3"), + ("e", "-1.5", "1.5", "0"), + ] { + assert_eq!( + resp.cmd(&["SET", key, stored]).await, + Reply::Simple("OK".into()) + ); + assert_eq!( + resp.cmd(&["INCRBYFLOAT", key, delta]).await, + Reply::Bulk(Some(expected.into())), + "{stored} + {delta}" + ); + assert_eq!( + resp.cmd(&["GET", key]).await, + Reply::Bulk(Some(expected.into())), + "{stored} + {delta} is stored as the reply text" + ); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sql_kv_incr_on_a_single_value_row_counts_from_the_stored_number() { + let server = TestServer::start().await; + create_bare_value_collection(&server).await; + + server + .exec(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('k', '5')" + )) + .await + .unwrap(); + let rows = server + .query_text(&format!("SELECT KV_INCR('{COLLECTION}', 'k', 1)")) + .await + .unwrap(); + assert_eq!(json_of(&rows)["value"], 6); + assert_eq!(value_of(&server, "k").await, "6"); + + let mut resp = server.resp_session("kvcount_sql_user", COLLECTION).await; + assert_eq!( + resp.cmd(&["GET", "k"]).await, + Reply::Bulk(Some("6".into())), + "KV_INCR stores decimal text, the same bytes RESP INCR stores" + ); + + server + .exec(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('f', '1.5')" + )) + .await + .unwrap(); + let rows = server + .query_text(&format!("SELECT KV_INCR_FLOAT('{COLLECTION}', 'f', 1)")) + .await + .unwrap(); + assert_eq!(json_of(&rows)["value"], 2.5); + assert_eq!(json_of(&rows)["text"], "2.5"); + assert_eq!(value_of(&server, "f").await, "2.5"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sql_kv_incr_faults_answer_data_exception_sqlstates() { + let server = TestServer::start().await; + create_bare_value_collection(&server).await; + + server + .exec(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('name', 'abc')" + )) + .await + .unwrap(); + server + .expect_error( + &format!("SELECT KV_INCR('{COLLECTION}', 'name', 1)"), + "SQLSTATE 22P02", + ) + .await; + server + .expect_error( + &format!("SELECT KV_INCR_FLOAT('{COLLECTION}', 'name', 1)"), + "SQLSTATE 22P02", + ) + .await; + + server + .exec(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('max', '{}')", + i64::MAX + )) + .await + .unwrap(); + server + .expect_error( + &format!("SELECT KV_INCR('{COLLECTION}', 'max', 1)"), + "SQLSTATE 22003", + ) + .await; + assert_eq!(value_of(&server, "max").await, i64::MAX.to_string()); +} diff --git a/nodedb/tests/wire/cases/kv_counter_fresh_row.rs b/nodedb/tests/wire/cases/kv_counter_fresh_row.rs new file mode 100644 index 000000000..3bf5163b6 --- /dev/null +++ b/nodedb/tests/wire/cases/kv_counter_fresh_row.rs @@ -0,0 +1,143 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A KV counter on an absent key creates a row of the collection's shape. +//! +//! In a typed collection, `KV_INCR` / `KV_INCR_FLOAT` on a key that does not +//! exist stores the row `INSERT (key, column) VALUES (key, delta)` stores, +//! DEFAULTs included, so a `SELECT` of the column reads the value. RESP treats +//! every value as a byte string, as `SET` and `GET` do, so a RESP `INCR` +//! stores decimal text in any collection. + +use crate::harness::TestServer; +use crate::harness::resp_client::Reply; + +async fn create_typed(server: &TestServer) { + server + .exec( + "CREATE COLLECTION kvfresh (key TEXT PRIMARY KEY, n INT, \ + status TEXT DEFAULT 'new') WITH (engine='kv')", + ) + .await + .unwrap(); +} + +async fn row_of(server: &TestServer, key: &str) -> std::collections::HashMap { + let rows = server + .query_named_rows(&format!( + "SELECT n, status FROM kvfresh WHERE key = '{key}'" + )) + .await + .unwrap(); + assert_eq!(rows.len(), 1, "one row for {key}: {rows:?}"); + rows.into_iter().next().unwrap_or_default() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_incr_on_an_absent_key_creates_a_typed_row() { + let server = TestServer::start().await; + create_typed(&server).await; + + server + .query_text("SELECT KV_INCR('kvfresh', 'a', 5)") + .await + .unwrap(); + let row = row_of(&server, "a").await; + assert_eq!(row.get("n").map(String::as_str), Some("5"), "{row:?}"); + assert_eq!( + row.get("status").map(String::as_str), + Some("new"), + "the fresh row carries the DEFAULT an insert stores: {row:?}" + ); + + server + .query_text("SELECT KV_INCR('kvfresh', 'a', 2)") + .await + .unwrap(); + assert_eq!( + row_of(&server, "a").await.get("n").map(String::as_str), + Some("7") + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_incr_float_on_an_absent_key_creates_a_typed_row() { + let server = TestServer::start().await; + server + .exec( + "CREATE COLLECTION kvscore (key TEXT PRIMARY KEY, score FLOAT, \ + status TEXT DEFAULT 'new') WITH (engine='kv')", + ) + .await + .unwrap(); + + server + .query_text("SELECT KV_INCR_FLOAT('kvscore', 'b', 2.5)") + .await + .unwrap(); + let rows = server + .query_named_rows("SELECT score, status FROM kvscore WHERE key = 'b'") + .await + .unwrap(); + assert_eq!(rows.len(), 1, "{rows:?}"); + assert_eq!(rows[0].get("score").map(String::as_str), Some("2.5")); + assert_eq!(rows[0].get("status").map(String::as_str), Some("new")); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_typed_fresh_row_survives_restart() { + let server = TestServer::start().await; + create_typed(&server).await; + server + .query_text("SELECT KV_INCR('kvfresh', 'c', 9)") + .await + .unwrap(); + + let (server, dir) = server.take_dir(); + server.graceful_shutdown().await; + let (server, _dir) = TestServer::open_on_path(dir).await; + + let row = row_of(&server, "c").await; + assert_eq!(row.get("n").map(String::as_str), Some("9"), "{row:?}"); + assert_eq!(row.get("status").map(String::as_str), Some("new")); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn resp_incr_on_an_absent_key_stores_decimal_text_in_a_typed_collection() { + let server = TestServer::start().await; + create_typed(&server).await; + let mut resp = server.resp_session("kvfresh_resp_user", "kvfresh").await; + + assert_eq!(resp.cmd(&["INCR", "r"]).await, Reply::Integer(1)); + assert_eq!( + resp.cmd(&["GET", "r"]).await, + Reply::Bulk(Some("1".into())), + "RESP GET returns the byte string RESP INCR stored" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn resp_incrbyfloat_adds_a_twenty_digit_delta_exactly() { + let server = TestServer::start().await; + server + .exec( + "CREATE COLLECTION kvexact (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')", + ) + .await + .unwrap(); + let mut resp = server.resp_session("kvexact_user", "kvexact").await; + + assert_eq!( + resp.cmd(&["SET", "k", "1"]).await, + Reply::Simple("OK".into()) + ); + assert_eq!( + resp.cmd(&["INCRBYFLOAT", "k", "0.12345678901234567891"]) + .await, + Reply::Bulk(Some("1.12345678901234567891".into())) + ); + assert_eq!( + resp.cmd(&["GET", "k"]).await, + Reply::Bulk(Some("1.12345678901234567891".into())) + ); +} diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 183763a8f..3841777e5 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -95,7 +95,9 @@ mod group_by_unaliased_aggregate; mod healthz_calvin_readiness; mod http_result_projection; mod insert_select_cross_engine; +mod kv_bare_value_counters; mod kv_column_defaults; +mod kv_counter_fresh_row; mod kv_predicate_dml; mod kv_sql_select; mod kv_write_row_level_security; diff --git a/nodedb/tests/wire/cases/sql_transactions_kv_atomic_overlay.rs b/nodedb/tests/wire/cases/sql_transactions_kv_atomic_overlay.rs index 138742f13..a3aeb388f 100644 --- a/nodedb/tests/wire/cases/sql_transactions_kv_atomic_overlay.rs +++ b/nodedb/tests/wire/cases/sql_transactions_kv_atomic_overlay.rs @@ -9,19 +9,12 @@ //! within the same transaction, and are discarded on `ROLLBACK`. COMMIT's //! durable replay is unchanged. //! -//! RYOW / persistence is asserted via a follow-up `SELECT KV_INCR(k, 0)` (a -//! true no-op add) rather than `SELECT n FROM c WHERE key = ...`, because the -//! ordinary SQL projection read is unreliable for a row any atomic op has -//! ever touched: the base `KvEngine`'s `atomic_put` (`engine/kv/ -//! engine_atomic.rs`) unconditionally writes with `Surrogate::ZERO`, -//! discarding a real surrogate a prior `INSERT`/`UPSERT` assigned that row -//! for SQL/surrogate-indexed access. This reproduces identically in -//! autocommit (no transaction involved) -- confirmed by direct probe -- so -//! it is a pre-existing base-engine gap, not something introduced by the -//! staging work this suite covers, and is out of scope to fix here. -//! `SELECT KV_INCR(k, 0)` sidesteps it: it reads the same way every `KV_*` -//! call does (`resolve_kv_current` / the base engine's own `table.get(key)` -//! by raw key bytes), which is unaffected by the surrogate reset. +//! Read-your-own-writes and persistence are asserted two ways: a follow-up +//! `SELECT KV_INCR(k, 0)` (a no-op add that reads through the atomic path), +//! and a plain `SELECT n FROM c WHERE key = ...`. A KV point `SELECT` reads +//! the row by its raw key, consulting the transaction's staging overlay +//! first, and an atomic write stores the plan's surrogate, so both reads see +//! the value the atomic computed. use crate::harness::TestServer; @@ -37,6 +30,14 @@ fn json_of(rows: &[String]) -> serde_json::Value { serde_json::from_str(&rows[0]).expect("KV_* result must be JSON") } +/// The `n` column of `key`, read by a plain point `SELECT`. +async fn n_of(server: &TestServer, key: &str) -> Vec { + server + .query_text(&format!("SELECT n FROM c WHERE key = '{key}'")) + .await + .unwrap() +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn incr_in_tx_returns_computed_value_and_chains() { let server = TestServer::start().await; @@ -81,6 +82,12 @@ async fn incr_in_tx_returns_computed_value_and_chains() { "second in-tx INCR must chain off the first staged value" ); + assert_eq!( + n_of(&server, "ctr").await, + vec!["10".to_string()], + "an in-tx SELECT must observe the chained staged INCR" + ); + server.exec("COMMIT").await.unwrap(); let committed = server @@ -92,6 +99,32 @@ async fn incr_in_tx_returns_computed_value_and_chains() { 10, "COMMIT must persist the chained INCR" ); + assert_eq!( + n_of(&server, "ctr").await, + vec!["10".to_string()], + "a SELECT after COMMIT must read the committed INCR" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn autocommit_incr_is_read_back_by_a_point_select() { + let server = TestServer::start().await; + setup(&server).await; + + server + .exec("INSERT INTO c (key, n) VALUES ('ctr', 5)") + .await + .unwrap(); + let rows = server + .query_text("SELECT KV_INCR('c', 'ctr', 3)") + .await + .unwrap(); + assert_eq!(json_of(&rows)["value"], 8); + assert_eq!( + n_of(&server, "ctr").await, + vec!["8".to_string()], + "a point SELECT must read the value an autocommit INCR stored" + ); } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] @@ -121,6 +154,11 @@ async fn incr_in_tx_rollback_reverts_to_base_value() { 5, "ROLLBACK must discard the staged INCR" ); + assert_eq!( + n_of(&server, "ctr").await, + vec!["5".to_string()], + "a SELECT after ROLLBACK must read the base value" + ); } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] From 02f085110004d6ad1cb44f46f7ea80ba279933a7 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 06:54:53 +0800 Subject: [PATCH 16/64] refactor(wal): replace the apply-key thread-local with an explicit appender WalManager::with_apply_key scoped a proposal's idempotency key to the calling thread for the duration of an append closure, so a reader of an append call site could not see which key its records carried, and an append inside the wrong scope silently picked up the ambient key. WalManager::appender(apply_key) now returns a WalAppender handle whose append_* methods carry that key explicitly. Every WAL append site, across WAL dispatch, the write funnel, write abort, Calvin recovery and commit resolution, the surrogate appender, checkpoint and collection-tombstone paths, and bitemporal purge, takes an appender up front instead of pairing a WalManager reference with a closure. NO_APPLY_KEY names the key of a record no replicated proposal owns. --- .../post_apply/async_dispatch/collection.rs | 1 + nodedb/src/control/checkpoint_manager.rs | 14 +-- .../scheduler/driver/core/commit_redo.rs | 16 +-- .../driver/core/commit_resolve/apply_tail.rs | 14 ++- .../cluster/calvin/scheduler/recovery.rs | 48 +++++---- .../distributed_applier/proposal_ledger.rs | 35 +++--- nodedb/src/control/orchestrated_write.rs | 2 +- .../submit_write/funnel/response.rs | 20 ++-- .../submit_write/funnel/wal_append.rs | 18 ++-- .../server/dispatch_utils/write_abort.rs | 14 ++- .../control/server/sync/columnar_handler.rs | 3 +- nodedb/src/control/server/sync/fts_handler.rs | 5 +- .../raft_dispatch/durability_test_support.rs | 2 +- .../control/server/sync/spatial_handler.rs | 5 +- .../control/server/sync/timeseries_handler.rs | 3 +- .../src/control/server/sync/vector_handler.rs | 5 +- .../src/control/server/wal_dispatch/array.rs | 5 +- .../control/server/wal_dispatch/columnar.rs | 5 +- .../src/control/server/wal_dispatch/core.rs | 9 +- .../src/control/server/wal_dispatch/crdt.rs | 5 +- .../control/server/wal_dispatch/document.rs | 5 +- .../src/control/server/wal_dispatch/graph.rs | 17 +-- .../control/server/wal_dispatch/spatial.rs | 4 +- .../src/control/server/wal_dispatch/text.rs | 4 +- .../control/server/wal_dispatch/timeseries.rs | 71 ++++++++++++- .../server/wal_dispatch/vector/append.rs | 8 +- .../server/wal_dispatch/write_set_redo.rs | 15 +-- .../server/wal_dispatch_fts_spatial.rs | 12 +-- .../control/server/wal_dispatch_kv/append.rs | 6 +- nodedb/src/control/surrogate/wal_appender.rs | 17 ++- nodedb/src/engine/bitemporal/enforcement.rs | 17 +-- nodedb/src/wal/manager/append.rs | 93 +++------------- nodedb/src/wal/manager/append_batch.rs | 4 +- nodedb/src/wal/manager/append_index.rs | 4 +- nodedb/src/wal/manager/append_metadata.rs | 16 +-- nodedb/src/wal/manager/append_transaction.rs | 4 +- nodedb/src/wal/manager/append_truncate.rs | 4 +- nodedb/src/wal/manager/append_vector.rs | 4 +- nodedb/src/wal/manager/appender.rs | 100 ++++++++++++++++++ nodedb/src/wal/manager/durable_commit.rs | 18 ++-- nodedb/src/wal/manager/encryption.rs | 61 ++++++----- nodedb/src/wal/manager/mod.rs | 2 + nodedb/src/wal/manager/ops.rs | 50 +++++---- .../inproc/cases/calvin_scheduler_restart.rs | 35 ++++-- nodedb/tests/inproc/cases/wal_catchup.rs | 18 ++-- 45 files changed, 500 insertions(+), 318 deletions(-) create mode 100644 nodedb/src/wal/manager/appender.rs diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs index 4786c6155..c3b9ace3b 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs @@ -102,6 +102,7 @@ pub(crate) async fn reclaim_collection_storage( // predecessor writes after a same-name CREATE. shared .wal + .appender(crate::wal::manager::NO_APPLY_KEY) .append_collection_tombstone( TenantId::new(tenant_id), DatabaseId::new(database_id), diff --git a/nodedb/src/control/checkpoint_manager.rs b/nodedb/src/control/checkpoint_manager.rs index dbbe5df59..38856fcb5 100644 --- a/nodedb/src/control/checkpoint_manager.rs +++ b/nodedb/src/control/checkpoint_manager.rs @@ -294,12 +294,14 @@ pub async fn run_checkpoint_cycle(inputs: CheckpointCycleInputs<'_>) -> Option { debug!( marker_lsn = marker_lsn.as_u64(), diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index 6a673422d..6124816a7 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -77,12 +77,16 @@ impl Scheduler { let redo_lsn = if redo.ops.is_empty() { None } else { - match self.shared.wal.append_transaction_redo( - tenant_id, - VShardId::new(self.vshard_id), - database_id, - &redo, - ) { + match self + .shared + .wal + .appender(crate::wal::manager::NO_APPLY_KEY) + .append_transaction_redo( + tenant_id, + VShardId::new(self.vshard_id), + database_id, + &redo, + ) { Ok(lsn) => Some(lsn), Err(e) => { self.halt_apply( diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs index 7ccb67e49..33ea34afa 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -189,11 +189,15 @@ impl Scheduler { self.record_calvin_write_versions(txn_id, lsn); Some(lsn) } - None => match self.shared.wal.append_calvin_applied( - crate::types::VShardId::new(self.vshard_id), - txn_id.epoch, - txn_id.position, - ) { + None => match self + .shared + .wal + .appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied( + crate::types::VShardId::new(self.vshard_id), + txn_id.epoch, + txn_id.position, + ) { // The CalvinApplied WAL LSN is the committed write-LSN for this // apply — the SAME shard-local WAL-LSN space fast-path writes and // read watermarks use. Record the apply's per-key write versions diff --git a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs index ebecaf198..c2d2ee2aa 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs @@ -173,10 +173,16 @@ mod tests { use crate::types::VShardId; // Epoch 5: position 0 applied, position 1 NOT applied. Epoch 2: pos 0. - wal.append_calvin_applied(VShardId::new(1), 2, 0).unwrap(); - wal.append_calvin_applied(VShardId::new(1), 5, 0).unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(1), 2, 0) + .unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(1), 5, 0) + .unwrap(); // A different vshard (must be ignored). - wal.append_calvin_applied(VShardId::new(2), 99, 0).unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(2), 99, 0) + .unwrap(); wal.sync().unwrap(); let rec = read_applied_recovery(&wal, 1).unwrap(); @@ -205,7 +211,8 @@ mod tests { // Epoch 7 carries two independent positions on this vShard; only // position 0 committed before the crash. - wal.append_calvin_applied(VShardId::new(vshard), 7, 0) + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard), 7, 0) .unwrap(); wal.sync().unwrap(); @@ -229,7 +236,8 @@ mod tests { // A pure-read/empty-ops txn still writes a standalone CalvinApplied // marker at (epoch 1, position 0). - wal.append_calvin_applied(VShardId::new(vshard), 1, 0) + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard), 1, 0) .unwrap(); // A write-bearing Calvin txn journals its applied-marker as a @@ -246,13 +254,14 @@ mod tests { vshard_id: vshard, }), }; - wal.append_transaction_redo( - TenantId::new(0), - VShardId::new(vshard), - DatabaseId::DEFAULT, - &write_bearing, - ) - .unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_transaction_redo( + TenantId::new(0), + VShardId::new(vshard), + DatabaseId::DEFAULT, + &write_bearing, + ) + .unwrap(); // A single-shard TransactionRedo (calvin_stamp: None) must be ignored // by Calvin recovery. @@ -264,13 +273,14 @@ mod tests { }], calvin_stamp: None, }; - wal.append_transaction_redo( - TenantId::new(0), - VShardId::new(vshard), - DatabaseId::DEFAULT, - &single_shard, - ) - .unwrap(); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_transaction_redo( + TenantId::new(0), + VShardId::new(vshard), + DatabaseId::DEFAULT, + &single_shard, + ) + .unwrap(); wal.sync().unwrap(); diff --git a/nodedb/src/control/distributed_applier/proposal_ledger.rs b/nodedb/src/control/distributed_applier/proposal_ledger.rs index eb0e47bbc..5d9311937 100644 --- a/nodedb/src/control/distributed_applier/proposal_ledger.rs +++ b/nodedb/src/control/distributed_applier/proposal_ledger.rs @@ -130,6 +130,7 @@ impl ProposalLedger { mod tests { use super::*; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; fn applied(payload: &[u8]) -> AppliedWrite { AppliedWrite { @@ -187,9 +188,11 @@ mod tests { let dir = tempfile::tempdir().expect("tempdir"); let wal = open_wal(&dir); let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); - wal.with_apply_key(0xAB, || wal.append_put(tid, vs, db, b"keyed")) + wal.appender(0xAB) + .append_put(tid, vs, db, b"keyed") .expect("append keyed put"); - wal.append_put(tid, vs, db, b"unkeyed") + wal.appender(NO_APPLY_KEY) + .append_put(tid, vs, db, b"unkeyed") .expect("append unkeyed put"); wal.sync().expect("sync wal"); @@ -207,17 +210,19 @@ mod tests { let wal = open_wal(&dir); let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); let forward = wal - .with_apply_key(0xAB, || wal.append_put(tid, vs, db, b"refused")) + .appender(0xAB) + .append_put(tid, vs, db, b"refused") .expect("append keyed put"); - wal.append_write_aborted(tid, vs, db, forward) + wal.appender(NO_APPLY_KEY) + .append_write_aborted(tid, vs, db, forward) .expect("append unkeyed abort"); let final_forward = wal - .with_apply_key(0xCD, || wal.append_put(tid, vs, db, b"refused for good")) + .appender(0xCD) + .append_put(tid, vs, db, b"refused for good") .expect("append keyed put"); - wal.with_apply_key(0xCD, || { - wal.append_write_aborted(tid, vs, db, final_forward) - }) - .expect("append keyed abort"); + wal.appender(0xCD) + .append_write_aborted(tid, vs, db, final_forward) + .expect("append keyed abort"); wal.sync().expect("sync wal"); let ledger = ProposalLedger::from_records( @@ -235,18 +240,20 @@ mod tests { } #[test] - fn a_proposal_applied_marker_is_appended_only_inside_an_apply() { + fn a_proposal_applied_marker_is_appended_only_under_an_apply_key() { let dir = tempfile::tempdir().expect("tempdir"); let wal = open_wal(&dir); let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); assert!( - wal.append_proposal_applied(tid, vs, db) - .expect("append outside an apply") + wal.appender(NO_APPLY_KEY) + .append_proposal_applied(tid, vs, db) + .expect("append with no apply key") .is_none() ); assert!( - wal.with_apply_key(0xEF, || wal.append_proposal_applied(tid, vs, db)) - .expect("append inside an apply") + wal.appender(0xEF) + .append_proposal_applied(tid, vs, db) + .expect("append under an apply key") .is_some() ); wal.sync().expect("sync wal"); diff --git a/nodedb/src/control/orchestrated_write.rs b/nodedb/src/control/orchestrated_write.rs index bd3987632..a8acb1917 100644 --- a/nodedb/src/control/orchestrated_write.rs +++ b/nodedb/src/control/orchestrated_write.rs @@ -43,7 +43,7 @@ pub(crate) async fn apply_orchestrated_write( // WAL-only restart rebuilds the index from pre-write records. No-op // on a target with no write-set. crate::control::server::wal_dispatch::mint_dispatch_local_redo( - &state.wal, + state.wal.appender(crate::wal::manager::NO_APPLY_KEY), tenant_id, database_id, collection, diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs index c4cb0921f..8c10d1aa3 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -179,16 +179,14 @@ pub(super) async fn collect_classify_and_finish( rollback_on_err( shared, &ddl_transition, - shared.wal.with_apply_key(apply_key, || { - wal_dispatch::append_write_set_redo( - &shared.wal, - tenant_id, - vshard_id, - database_id, - collection, - &response.write_set, - ) - }), + wal_dispatch::append_write_set_redo( + shared.wal.appender(apply_key), + tenant_id, + vshard_id, + database_id, + collection, + &response.write_set, + ), )? } else { None @@ -196,7 +194,7 @@ pub(super) async fn collect_classify_and_finish( drop(deferred_guards); // Durable-at-ack barrier: an acknowledged write must be WAL-fsync-durable - // before this response (the client ack) returns. `WalManager::append_*` only + // before this response (the client ack) returns. `WalAppender::append_*` only // buffers the record and mints its `Lsn`; without this barrier a `kill -9` // loses the buffered bytes, which is invisible for engines whose rows are // committed durably by redb but silently destroys every engine whose only diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs index 307baa6e1..efc382bc6 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs @@ -82,16 +82,14 @@ pub(super) fn authorize_and_append( let outcome = rollback_on_err( shared, &ddl_transition, - shared.wal.with_apply_key(apply_key, || { - wal_dispatch::wal_append(WalAppendRequest { - wal: &shared.wal, - tenant_id, - vshard_id, - database_id, - plan: &plan, - credentials: None, - now_override, - }) + wal_dispatch::wal_append(WalAppendRequest { + wal: shared.wal.appender(apply_key), + tenant_id, + vshard_id, + database_id, + plan: &plan, + credentials: None, + now_override, }), )?; (outcome.lsn, outcome.resolved_now_ms) diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index bea98634f..ddbf57f81 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -99,14 +99,12 @@ pub(crate) async fn abort_refused_write( } else { 0 }; - let abort_lsn = shared.wal.with_apply_key(marker_key, || { - shared.wal.append_write_aborted( - target.tenant_id, - target.vshard_id, - target.database_id, - wal_lsn, - ) - })?; + let abort_lsn = shared.wal.appender(marker_key).append_write_aborted( + target.tenant_id, + target.vshard_id, + target.database_id, + wal_lsn, + )?; shared.wal.wait_durable(abort_lsn).await?; tracing::debug!( aborted_lsn = wal_lsn.as_u64(), diff --git a/nodedb/src/control/server/sync/columnar_handler.rs b/nodedb/src/control/server/sync/columnar_handler.rs index 3747d2368..3a0c687f2 100644 --- a/nodedb/src/control/server/sync/columnar_handler.rs +++ b/nodedb/src/control/server/sync/columnar_handler.rs @@ -19,6 +19,7 @@ use nodedb_types::value::Value; use super::session::SyncSession; use super::wire::*; use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::wal::manager::NO_APPLY_KEY; // ── PK extraction helper ───────────────────────────────────────────────────── @@ -193,7 +194,7 @@ impl<'a> ColumnarDispatcher for SharedStateColumnarDispatcher<'a> { // WAL append — surrogates are persisted so followers never mint their // own divergent ids. let appended_lsn = wal_append_columnar( - &self.shared.wal, + self.shared.wal.appender(NO_APPLY_KEY), tenant_id, vshard, database_id, diff --git a/nodedb/src/control/server/sync/fts_handler.rs b/nodedb/src/control/server/sync/fts_handler.rs index e8dc1187e..96d5c7d81 100644 --- a/nodedb/src/control/server/sync/fts_handler.rs +++ b/nodedb/src/control/server/sync/fts_handler.rs @@ -18,6 +18,7 @@ use async_trait::async_trait; use nodedb_types::Surrogate; use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::wal::manager::NO_APPLY_KEY; // ── Dispatcher trait ───────────────────────────────────────────────────────── @@ -106,7 +107,7 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { &text, ); let wal_lsn = wal_append_fts_index( - &self.shared.wal, + self.shared.wal.appender(NO_APPLY_KEY), tenant_id, vshard, database_id, @@ -158,7 +159,7 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { let fts_delete_payload = nodedb_wal::record::FtsDeletePayload::new(prov.clone(), &collection, &surrogate_hex); let wal_lsn = wal_append_fts_delete( - &self.shared.wal, + self.shared.wal.appender(NO_APPLY_KEY), tenant_id, vshard, database_id, diff --git a/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs b/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs index 79a227d29..2f817b2aa 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs @@ -60,7 +60,7 @@ pub(super) fn append_buffered_record(state: &SharedState) -> Lsn { "00000001", ); crate::control::server::wal_dispatch::wal_append_fts_delete( - &state.wal, + state.wal.appender(crate::wal::manager::NO_APPLY_KEY), tenant(), vshard(), DatabaseId::DEFAULT, diff --git a/nodedb/src/control/server/sync/spatial_handler.rs b/nodedb/src/control/server/sync/spatial_handler.rs index 22bebf0d9..144254b47 100644 --- a/nodedb/src/control/server/sync/spatial_handler.rs +++ b/nodedb/src/control/server/sync/spatial_handler.rs @@ -19,6 +19,7 @@ use nodedb_types::Surrogate; use nodedb_types::geometry::Geometry; use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::wal::manager::NO_APPLY_KEY; // ── Dispatcher trait ───────────────────────────────────────────────────────── @@ -113,7 +114,7 @@ impl<'a> SpatialDispatcher for SharedStateSpatialDispatcher<'a> { let spatial_put_payload = encode_spatial_put_payload(&collection, &field, surrogate, &geometry, &prov)?; let wal_lsn = wal_append_spatial_put( - &self.shared.wal, + self.shared.wal.appender(NO_APPLY_KEY), tenant_id, vshard, database_id, @@ -166,7 +167,7 @@ impl<'a> SpatialDispatcher for SharedStateSpatialDispatcher<'a> { let spatial_delete_payload = encode_spatial_delete_payload(&collection, &field, surrogate, &prov); let wal_lsn = wal_append_spatial_delete( - &self.shared.wal, + self.shared.wal.appender(NO_APPLY_KEY), tenant_id, vshard, database_id, diff --git a/nodedb/src/control/server/sync/timeseries_handler.rs b/nodedb/src/control/server/sync/timeseries_handler.rs index 7084853d9..7830824bf 100644 --- a/nodedb/src/control/server/sync/timeseries_handler.rs +++ b/nodedb/src/control/server/sync/timeseries_handler.rs @@ -16,6 +16,7 @@ use tracing::{debug, error}; use super::session::SyncSession; use super::wire::*; use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::wal::manager::NO_APPLY_KEY; // ── Dispatcher trait ───────────────────────────────────────────────────────── @@ -93,7 +94,7 @@ impl<'a> TimeseriesDispatcher for SharedStateTimeseriesDispatcher<'a> { // Allocate a WAL LSN on the Control Plane before dispatching to the // Data Plane. This is the canonical LSN for dedup tracking. let appended_lsn = wal_append_timeseries( - &self.shared.wal, + self.shared.wal.appender(NO_APPLY_KEY), TimeseriesWalAppendContext { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/server/sync/vector_handler.rs b/nodedb/src/control/server/sync/vector_handler.rs index c5f644fe0..e54a39e15 100644 --- a/nodedb/src/control/server/sync/vector_handler.rs +++ b/nodedb/src/control/server/sync/vector_handler.rs @@ -16,6 +16,7 @@ use async_trait::async_trait; use nodedb_types::Surrogate; use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::wal::manager::NO_APPLY_KEY; // ── Dispatcher trait ───────────────────────────────────────────────────────── @@ -106,7 +107,7 @@ impl<'a> VectorDispatcher for SharedStateVectorDispatcher<'a> { // Data Plane. Sync path MUST write to WAL; non-sync path already does // this via `wal_append_if_write_with_creds` in the main dispatch. let wal_lsn = wal_append_vector_put( - &self.shared.wal, + self.shared.wal.appender(NO_APPLY_KEY), tenant_id, vshard, database_id, @@ -169,7 +170,7 @@ impl<'a> VectorDispatcher for SharedStateVectorDispatcher<'a> { // Allocate WAL LSN on the Control Plane before dispatching to the // Data Plane. let wal_lsn = wal_append_vector_delete_by_surrogate( - &self.shared.wal, + self.shared.wal.appender(NO_APPLY_KEY), tenant_id, vshard, database_id, diff --git a/nodedb/src/control/server/wal_dispatch/array.rs b/nodedb/src/control/server/wal_dispatch/array.rs index 4af38b8bf..d8f006d53 100644 --- a/nodedb/src/control/server/wal_dispatch/array.rs +++ b/nodedb/src/control/server/wal_dispatch/array.rs @@ -11,7 +11,7 @@ use crate::engine::array::wal::{ encode_put_with_version, }; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; /// Append the WAL record for a single `ArrayOp`, returning the allocated LSN /// for the cell write variants (`Some`) or `None` for every read / slice / @@ -20,7 +20,7 @@ use crate::wal::manager::WalManager; /// The match over [`ArrayOp`] is **exhaustive** (`wildcard_enum_match_arm` is /// denied), so a future write variant cannot silently become non-durable. pub(super) fn wal_append_array_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -100,6 +100,7 @@ pub(super) fn wal_append_array_op( #[cfg(test)] mod tests { use super::*; + use crate::wal::manager::WalManager; use nodedb_array::types::ArrayId; use nodedb_physical::physical_plan::PhysicalPlan; diff --git a/nodedb/src/control/server/wal_dispatch/columnar.rs b/nodedb/src/control/server/wal_dispatch/columnar.rs index a4f74505b..361a656d3 100644 --- a/nodedb/src/control/server/wal_dispatch/columnar.rs +++ b/nodedb/src/control/server/wal_dispatch/columnar.rs @@ -7,7 +7,7 @@ use nodedb_physical::physical_plan::ColumnarOp; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; /// Append the WAL record for a single `ColumnarOp`, returning the allocated LSN /// for the write variants (`Some`) or `None` for the scan variants, which carry @@ -16,7 +16,7 @@ use crate::wal::manager::WalManager; /// The match over [`ColumnarOp`] is **exhaustive** (`wildcard_enum_match_arm` /// is denied), so a future write variant cannot silently become non-durable. pub(super) fn wal_append_columnar_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -155,6 +155,7 @@ pub(super) fn wal_append_columnar_op( #[cfg(test)] mod tests { use super::*; + use crate::wal::manager::WalManager; use nodedb_physical::physical_plan::{ColumnarInsertIntent, PhysicalPlan}; fn open_wal(dir: &std::path::Path) -> WalManager { diff --git a/nodedb/src/control/server/wal_dispatch/core.rs b/nodedb/src/control/server/wal_dispatch/core.rs index c6566851f..667a5bacf 100644 --- a/nodedb/src/control/server/wal_dispatch/core.rs +++ b/nodedb/src/control/server/wal_dispatch/core.rs @@ -5,7 +5,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::credential::CredentialStore; use crate::types::{DatabaseId, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::{NO_APPLY_KEY, WalAppender, WalManager}; use super::super::wal_dispatch_kv; @@ -39,7 +39,8 @@ pub struct WalAppendOutcome { /// redo record is to be encoded, and the two optional knobs only some callers /// need. pub struct WalAppendRequest<'a> { - pub wal: &'a WalManager, + /// The appender, which names the apply key every appended record carries. + pub wal: WalAppender<'a>, pub tenant_id: TenantId, pub vshard_id: VShardId, pub database_id: DatabaseId, @@ -61,6 +62,8 @@ pub struct WalAppendRequest<'a> { /// /// Serializes the write as MessagePack and appends to the appropriate /// WAL record type. Read operations are no-ops (return Ok immediately). +/// The records carry no apply key: no replicated proposal owns them. A +/// proposal's apply goes through [`wal_append`] with a keyed appender. /// /// Returns the WAL LSN allocated for writes it appended (`Some`), or `None` /// for reads / control ops that need no WAL record. The caller stamps the @@ -89,7 +92,7 @@ pub fn wal_append_if_write_with_creds( credentials: Option<&CredentialStore>, ) -> crate::Result { wal_append(WalAppendRequest { - wal, + wal: wal.appender(NO_APPLY_KEY), tenant_id, vshard_id, database_id, diff --git a/nodedb/src/control/server/wal_dispatch/crdt.rs b/nodedb/src/control/server/wal_dispatch/crdt.rs index 73743e2c6..a2ad505a0 100644 --- a/nodedb/src/control/server/wal_dispatch/crdt.rs +++ b/nodedb/src/control/server/wal_dispatch/crdt.rs @@ -8,7 +8,7 @@ use nodedb_physical::physical_plan::CrdtOp; use nodedb_wal::record::RecordType; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; /// Which CRDT WAL record class a `CrdtOp` write journals as. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -37,7 +37,7 @@ impl CrdtRecordKind { /// for every read / constraint / policy variant that carries no durable /// per-write effect on THIS path. pub(super) fn wal_append_crdt_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -296,6 +296,7 @@ fn encode_crdt_doc_op_payload(payload: crate::wal::CrdtDocOpWalRecord) -> crate: #[cfg(test)] mod tests { use super::*; + use crate::wal::manager::WalManager; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_types::{QualifiedCollection, Surrogate}; diff --git a/nodedb/src/control/server/wal_dispatch/document.rs b/nodedb/src/control/server/wal_dispatch/document.rs index b5d070cea..10611ad5c 100644 --- a/nodedb/src/control/server/wal_dispatch/document.rs +++ b/nodedb/src/control/server/wal_dispatch/document.rs @@ -7,7 +7,7 @@ use nodedb_physical::physical_plan::DocumentOp; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; /// Encode a document PUT redo record: `(collection, document_id, value, /// Option, surrogate)`. Must match `wal_replay_redo_document`'s decode. @@ -50,7 +50,7 @@ pub(crate) fn encode_document_delete_record( /// Append the WAL record for a `DocumentOp`: the allocated LSN for point-write /// variants, `None` otherwise. Exhaustive so a new variant can't silently skip durability. pub(super) fn wal_append_document_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -149,6 +149,7 @@ pub(super) fn wal_append_document_op( #[cfg(test)] mod tests { use super::*; + use crate::wal::manager::WalManager; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_types::{QualifiedCollection, Surrogate}; diff --git a/nodedb/src/control/server/wal_dispatch/graph.rs b/nodedb/src/control/server/wal_dispatch/graph.rs index d725b4ad6..569eb7430 100644 --- a/nodedb/src/control/server/wal_dispatch/graph.rs +++ b/nodedb/src/control/server/wal_dispatch/graph.rs @@ -11,13 +11,13 @@ use nodedb_physical::physical_plan::{BatchEdge, GraphOp}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; /// Append the WAL record for a single `GraphOp`, returning the allocated LSN /// for edge/node-label writes or `None` for traversal/algorithm/read variants. /// Exhaustive match so a future write variant can't silently become non-durable. pub(super) fn wal_append_graph_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -99,7 +99,7 @@ pub(super) fn wal_append_graph_op( /// Append one `Put` WAL record per edge in a batched edge insert. Returns the /// last record's LSN as a "durable through here" watermark. Empty batch → `Ok(None)`. pub(crate) fn wal_append_graph_edge_put_batch( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -127,7 +127,7 @@ pub(crate) fn wal_append_graph_edge_put_batch( /// Append one `Delete` WAL record per edge in a batched edge delete (`CREATE /// GRAPH INDEX` rollback). Same last-LSN-as-watermark contract as [`wal_append_graph_edge_put_batch`]. pub(crate) fn wal_append_graph_edge_delete_batch( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -149,6 +149,7 @@ pub(crate) fn wal_append_graph_edge_delete_batch( #[cfg(test)] mod tests { use super::*; + use crate::wal::manager::{NO_APPLY_KEY, WalManager}; use nodedb_types::Surrogate; fn edge(collection: &str, src: &str, label: &str, dst: &str) -> BatchEdge { @@ -177,7 +178,7 @@ mod tests { ]; let lsn = wal_append_graph_edge_put_batch( - &wal, + wal.appender(NO_APPLY_KEY), TenantId::new(7), VShardId::new(0), DatabaseId::DEFAULT, @@ -223,7 +224,7 @@ mod tests { ]; let lsn = wal_append_graph_edge_delete_batch( - &wal, + wal.appender(NO_APPLY_KEY), TenantId::new(7), VShardId::new(0), DatabaseId::DEFAULT, @@ -263,7 +264,7 @@ mod tests { let wal = open_wal(dir.path()); let lsn = wal_append_graph_edge_put_batch( - &wal, + wal.appender(NO_APPLY_KEY), TenantId::new(7), VShardId::new(0), DatabaseId::DEFAULT, @@ -280,7 +281,7 @@ mod tests { let wal = open_wal(dir.path()); let lsn = wal_append_graph_edge_delete_batch( - &wal, + wal.appender(NO_APPLY_KEY), TenantId::new(7), VShardId::new(0), DatabaseId::DEFAULT, diff --git a/nodedb/src/control/server/wal_dispatch/spatial.rs b/nodedb/src/control/server/wal_dispatch/spatial.rs index 2a7d26c0f..25e2f9ecf 100644 --- a/nodedb/src/control/server/wal_dispatch/spatial.rs +++ b/nodedb/src/control/server/wal_dispatch/spatial.rs @@ -5,7 +5,7 @@ use nodedb_physical::physical_plan::SpatialOp; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; use super::super::wal_dispatch_fts_spatial; @@ -23,7 +23,7 @@ use super::super::wal_dispatch_fts_spatial; /// `VectorOp::DeleteBySurrogate`'s identical "sync path bypasses it, but log /// here too" reasoning in `wal_dispatch/vector.rs`). pub(crate) fn wal_append_spatial_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, diff --git a/nodedb/src/control/server/wal_dispatch/text.rs b/nodedb/src/control/server/wal_dispatch/text.rs index 9af27dc41..57c91b116 100644 --- a/nodedb/src/control/server/wal_dispatch/text.rs +++ b/nodedb/src/control/server/wal_dispatch/text.rs @@ -6,7 +6,7 @@ use nodedb_physical::physical_plan::TextOp; use nodedb_wal::record::RecordType; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; /// Append the WAL record for a single `TextOp`, returning the allocated LSN /// for the FTS write variants (`Some`) or `None` for every read/search @@ -22,7 +22,7 @@ use crate::wal::manager::WalManager; /// `VectorOp::DeleteBySurrogate`'s identical "sync path bypasses it, but log /// here too" reasoning in `wal_dispatch/vector.rs`). pub(crate) fn wal_append_text_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, diff --git a/nodedb/src/control/server/wal_dispatch/timeseries.rs b/nodedb/src/control/server/wal_dispatch/timeseries.rs index 31401c139..48fe4b47b 100644 --- a/nodedb/src/control/server/wal_dispatch/timeseries.rs +++ b/nodedb/src/control/server/wal_dispatch/timeseries.rs @@ -9,11 +9,11 @@ use nodedb_physical::physical_plan::TimeseriesOp; use crate::control::security::credential::CredentialStore; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; /// Inputs of [`wal_append_timeseries_op`]. pub(super) struct TimeseriesAppend<'a> { - pub wal: &'a WalManager, + pub wal: WalAppender<'a>, pub tenant_id: TenantId, pub vshard_id: VShardId, pub database_id: DatabaseId, @@ -254,7 +254,7 @@ pub(crate) struct TimeseriesWalAppendContext<'a> { /// listener and sync handler for dedup tracking and `flush_wal_lsn`. /// Returns `None` if WAL is bypassed. pub(crate) fn wal_append_timeseries( - wal: &WalManager, + wal: WalAppender<'_>, context: TimeseriesWalAppendContext<'_>, payload: &[u8], provenance: Option<&nodedb_types::sync::wire::SyncProvenance>, @@ -373,7 +373,7 @@ pub struct ColumnarWalAppendArgs<'a> { /// `wal_append_timeseries` but encodes `ColumnarWalRecord` so replay restores /// per-row surrogates. Always returns `Some` — columnar has no `wal=false`. pub fn wal_append_columnar( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -400,6 +400,7 @@ pub fn wal_append_columnar( #[cfg(test)] mod tests { use super::*; + use crate::wal::manager::{NO_APPLY_KEY, WalManager}; use nodedb_physical::physical_plan::PhysicalPlan; fn open_wal(dir: &std::path::Path) -> WalManager { @@ -462,7 +463,7 @@ mod tests { }); let outcome = super::super::wal_append(super::super::WalAppendRequest { - wal: &wal, + wal: wal.appender(NO_APPLY_KEY), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), database_id: DatabaseId::DEFAULT, @@ -488,6 +489,66 @@ mod tests { assert_eq!(decoded.format.as_deref(), Some("ilp")); } + /// A `wal=false` ingest appends no batch record. Under a proposal's apply + /// key it appends a `ProposalApplied` marker instead and returns the + /// marker's LSN as the write's LSN, so the funnel's durability barrier, + /// which waits on that LSN, makes the marker durable before the ack. + #[test] + fn a_wal_bypassed_ingest_under_an_apply_key_returns_its_marker_lsn() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(dir.path()); + let credentials = CredentialStore::new().expect("in-memory credential store"); + let mut collection = + crate::control::security::catalog::StoredCollection::new(1, "metrics", "owner"); + collection.timeseries_config = Some(r#"{"wal":"false"}"#.to_string()); + credentials + .catalog() + .put_collection(DatabaseId::DEFAULT, &collection) + .expect("store collection"); + let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), + payload: b"metrics value=1".to_vec(), + format: "ilp".to_string(), + wal_lsn: None, + surrogates: vec![], + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: vec![], + }); + let append = |apply_key: u64| { + super::super::wal_append(super::super::WalAppendRequest { + wal: wal.appender(apply_key), + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + database_id: DatabaseId::DEFAULT, + plan: &plan, + credentials: Some(&credentials), + now_override: None, + }) + .expect("append") + }; + + assert_eq!( + append(NO_APPLY_KEY).lsn, + None, + "outside a proposal's apply nothing is appended" + ); + let marker_lsn = append(0xAB) + .lsn + .expect("the marker's LSN is the write's LSN"); + + wal.sync().expect("sync wal"); + let records = wal.replay().expect("read wal"); + assert_eq!(records.len(), 1, "only the marker reaches the WAL"); + assert_eq!(records[0].header.lsn, marker_lsn.as_u64()); + assert_eq!( + nodedb_wal::record::RecordType::from_raw(records[0].logical_record_type()), + Some(nodedb_wal::record::RecordType::ProposalApplied) + ); + assert_eq!(records[0].apply_key(), 0xAB); + } + #[test] fn truncate_appends_timeseries_truncate_record() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/nodedb/src/control/server/wal_dispatch/vector/append.rs b/nodedb/src/control/server/wal_dispatch/vector/append.rs index 9bbd6e6f7..36ec66e42 100644 --- a/nodedb/src/control/server/wal_dispatch/vector/append.rs +++ b/nodedb/src/control/server/wal_dispatch/vector/append.rs @@ -10,7 +10,7 @@ use nodedb_physical::physical_plan::VectorOp; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; use super::encode::{ VectorDirectUpdatePayload, VectorDirectUpsertPayload, VectorResolvedDirectWritePayload, @@ -54,7 +54,7 @@ pub struct VectorDeleteWalArgs<'a> { /// exactly as the non-sync `VectorOp::Insert` arm in `wal_append_if_write_with_creds` does, /// so replay decodes both paths with the same 7-element shape. pub fn wal_append_vector_put( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -83,7 +83,7 @@ pub fn wal_append_vector_put( /// silently become non-durable (the class of bug this function was hardened /// against). Read and maintenance ops map to `None` explicitly, by name. pub(crate) fn wal_append_vector_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -427,7 +427,7 @@ pub(crate) fn wal_append_vector_op( /// back to `execute_vector_delete_by_surrogate`; the legacy 2-element and 3-element /// delete arms fall through to direct node-id deletion and remain backward-compatible. pub fn wal_append_vector_delete_by_surrogate( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, diff --git a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs index bb63d05f9..2da8ea86e 100644 --- a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs +++ b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs @@ -9,7 +9,7 @@ use crate::bridge::envelope::{PhysicalPlan, Response, Status, WriteSetEntry}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; use nodedb_physical::physical_plan::{DocumentOp, MetaOp}; use super::document::{encode_document_delete_record, encode_document_put_record}; @@ -63,7 +63,7 @@ pub fn plan_post_apply_redo(plan: &PhysicalPlan) -> Option { /// the row the way a live event does and the Data Plane replay keys on the /// surrogate. Called under the write-admission guard. pub fn append_write_set_redo( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -103,7 +103,7 @@ pub fn append_write_set_redo( /// Mint the post-apply redo for a `dispatch_local` response built outside the /// autocommit funnel's own redo minting. No-op when not `Ok` or write-set is empty. pub fn mint_dispatch_local_redo( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, database_id: DatabaseId, collection: &str, @@ -127,6 +127,7 @@ pub fn mint_dispatch_local_redo( #[cfg(test)] mod tests { use super::*; + use crate::wal::manager::{NO_APPLY_KEY, WalManager}; use nodedb_physical::physical_plan::ReturningSpec; use nodedb_types::sync::wire::SyncProvenance; use nodedb_types::{QualifiedCollection, RowIdentity, Surrogate}; @@ -261,7 +262,7 @@ mod tests { }]; let lsn = append_write_set_redo( - &wal, + wal.appender(NO_APPLY_KEY), TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT, @@ -303,7 +304,7 @@ mod tests { }]; append_write_set_redo( - &wal, + wal.appender(NO_APPLY_KEY), TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT, @@ -337,7 +338,7 @@ mod tests { }]; append_write_set_redo( - &wal, + wal.appender(NO_APPLY_KEY), TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT, @@ -364,7 +365,7 @@ mod tests { let dir = tempfile::tempdir().expect("tempdir"); let wal = open_wal(dir.path()); let lsn = append_write_set_redo( - &wal, + wal.appender(NO_APPLY_KEY), TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT, diff --git a/nodedb/src/control/server/wal_dispatch_fts_spatial.rs b/nodedb/src/control/server/wal_dispatch_fts_spatial.rs index a277a2c21..de48224d8 100644 --- a/nodedb/src/control/server/wal_dispatch_fts_spatial.rs +++ b/nodedb/src/control/server/wal_dispatch_fts_spatial.rs @@ -3,7 +3,7 @@ //! WAL append helpers for FTS and Spatial sync ingest paths. //! //! Each helper accepts a prebuilt payload struct, serializes it, and appends -//! to the WAL via `WalManager`. The CP allocates the LSN here; the gate runs +//! to the WAL through the caller's `WalAppender`. The CP allocates the LSN here; the gate runs //! Data-Plane-side at the apply handler. //! //! Callers are responsible for constructing the payload (which bundles @@ -14,7 +14,7 @@ use nodedb_types::geometry::Geometry; use nodedb_types::sync::wire::SyncProvenance; use crate::types::{DatabaseId, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; /// Build a `SpatialPutPayload` from raw op fields, msgpack-encoding the /// geometry the same way the sync ingest path (`spatial_handler.rs`) does. @@ -63,7 +63,7 @@ pub(crate) fn encode_spatial_delete_payload( /// The `payload` already carries provenance so replay routes through /// `execute_fts_index_doc` and the idempotency gate fires on replay. pub fn wal_append_fts_index( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -79,7 +79,7 @@ pub fn wal_append_fts_index( /// The `payload` already carries provenance so replay routes through /// `execute_fts_delete_doc` and the idempotency gate fires on replay. pub fn wal_append_fts_delete( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -95,7 +95,7 @@ pub fn wal_append_fts_delete( /// The `payload` carries provenance and the msgpack-encoded `Geometry` /// (identical to what `SpatialInsertMsg.geometry_bytes` carries). pub fn wal_append_spatial_put( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -108,7 +108,7 @@ pub fn wal_append_spatial_put( /// Append a spatial delete to the WAL and return the assigned LSN. pub fn wal_append_spatial_delete( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, diff --git a/nodedb/src/control/server/wal_dispatch_kv/append.rs b/nodedb/src/control/server/wal_dispatch_kv/append.rs index 4bfccc436..80e1090f0 100644 --- a/nodedb/src/control/server/wal_dispatch_kv/append.rs +++ b/nodedb/src/control/server/wal_dispatch_kv/append.rs @@ -3,7 +3,7 @@ //! Dispatch of `KvOp` variants to WAL append calls. use crate::types::{DatabaseId, TenantId, VShardId}; -use crate::wal::manager::WalManager; +use crate::wal::manager::WalAppender; use nodedb_physical::physical_plan::KvOp; use super::encode::{ @@ -45,7 +45,7 @@ fn resolve_expiry(ttl_ms: u64, now_override: Option) -> (Option, Optio /// `now_override` pins `expire_at_ms` to an instant decided elsewhere (e.g. a /// Raft-committed entry), so every replica's redo installs it verbatim. pub fn wal_append_kv_op( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, @@ -383,7 +383,7 @@ pub fn wal_append_kv_op( /// Append one mutation of a resolved KV write and return its LSN. Uses the /// absolute expiry already resolved — no clock read here, so redo matches apply. fn append_kv_resolved_mutation( - wal: &WalManager, + wal: WalAppender<'_>, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, diff --git a/nodedb/src/control/surrogate/wal_appender.rs b/nodedb/src/control/surrogate/wal_appender.rs index ac9c0b345..b4920836d 100644 --- a/nodedb/src/control/surrogate/wal_appender.rs +++ b/nodedb/src/control/surrogate/wal_appender.rs @@ -18,6 +18,7 @@ use std::sync::Arc; use nodedb_types::{DatabaseId, TenantId}; use crate::wal::WalManager; +use crate::wal::manager::NO_APPLY_KEY; /// Pluggable WAL appender. Tests substitute `NoopWalAppender`; /// production wires [`WalSurrogateAppender`] (a thin wrapper over @@ -45,7 +46,7 @@ pub trait SurrogateWalAppender: Send + Sync { } /// Production appender — wraps `Arc` and forwards to -/// `WalManager::append_surrogate_alloc`. +/// `WalAppender::append_surrogate_alloc`. pub struct WalSurrogateAppender { wal: Arc, } @@ -58,7 +59,10 @@ impl WalSurrogateAppender { impl SurrogateWalAppender for WalSurrogateAppender { fn record_alloc_to_wal(&self, hi: u32) -> crate::Result<()> { - self.wal.append_surrogate_alloc(hi).map(|_| ()) + self.wal + .appender(NO_APPLY_KEY) + .append_surrogate_alloc(hi) + .map(|_| ()) } fn record_bind_to_wal( @@ -69,8 +73,13 @@ impl SurrogateWalAppender for WalSurrogateAppender { collection: &str, pk_bytes: &[u8], ) -> crate::Result<()> { - self.wal - .append_surrogate_bind(database_id, tenant_id, surrogate, collection, pk_bytes)?; + self.wal.appender(NO_APPLY_KEY).append_surrogate_bind( + database_id, + tenant_id, + surrogate, + collection, + pk_bytes, + )?; // Force the record to disk before the assigner releases its // write-lock. A crash after `assign` returns must always see // the binding on replay; group-commit batching alone does not diff --git a/nodedb/src/engine/bitemporal/enforcement.rs b/nodedb/src/engine/bitemporal/enforcement.rs index 4e254b4db..6599043d3 100644 --- a/nodedb/src/engine/bitemporal/enforcement.rs +++ b/nodedb/src/engine/bitemporal/enforcement.rs @@ -156,13 +156,16 @@ async fn run_one(state: &Arc, entry: &Entry) { Ok(payload) => { let purged = parse_count_from_payload(entry.engine, &payload); if purged > 0 - && let Err(e) = state.wal.append_temporal_purge( - tenant_id, - entry.engine.wire_tag(), - &entry.collection, - cutoff_system_ms, - purged, - ) + && let Err(e) = state + .wal + .appender(crate::wal::manager::NO_APPLY_KEY) + .append_temporal_purge( + tenant_id, + entry.engine.wire_tag(), + &entry.collection, + cutoff_system_ms, + purged, + ) { warn!( tenant = tenant_id.as_u64(), diff --git a/nodedb/src/wal/manager/append.rs b/nodedb/src/wal/manager/append.rs index 7b6e59aa5..4877d35f2 100644 --- a/nodedb/src/wal/manager/append.rs +++ b/nodedb/src/wal/manager/append.rs @@ -1,78 +1,11 @@ // SPDX-License-Identifier: BUSL-1.1 -use std::cell::Cell; - -use nodedb_wal::RecordTarget; use nodedb_wal::record::RecordType; -use super::core::WalManager; +use super::appender::WalAppender; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -thread_local! { - /// The proposal whose apply the current thread is appending records for, - /// `0` outside one. Set only by [`WalManager::with_apply_key`]. - static APPLY_KEY: Cell = const { Cell::new(0) }; -} - -/// Restores the thread's prior apply key when a keyed append scope ends, -/// panics included. -struct ApplyKeyScope { - prior: u64, -} - -impl Drop for ApplyKeyScope { - fn drop(&mut self) { - APPLY_KEY.with(|key| key.set(self.prior)); - } -} - -/// The apply key of the proposal the current thread is appending for, `0` -/// outside one. -pub(super) fn current_apply_key() -> u64 { - APPLY_KEY.with(Cell::get) -} - -impl WalManager { - /// Run `append` with every record this thread appends inside it carrying - /// `apply_key` in its header: the idempotency key of the replicated - /// proposal being applied. The record and the key become durable in one - /// write, so the apply loop recovers which proposals applied from the - /// records themselves. - /// - /// `append` must not await: the key is scoped to the calling thread, and - /// a WAL append is synchronous. - pub fn with_apply_key(&self, apply_key: u64, append: impl FnOnce() -> R) -> R { - let prior = APPLY_KEY.with(|key| key.replace(apply_key)); - let _scope = ApplyKeyScope { prior }; - append() - } - - /// Internal: append a record of the given type to the WAL. - pub(super) fn append_record( - &self, - record_type: RecordType, - tenant_id: TenantId, - vshard_id: VShardId, - database_id: DatabaseId, - payload: &[u8], - ) -> crate::Result { - let apply_key = current_apply_key(); - let mut wal = self.wal.lock().unwrap_or_else(|p| p.into_inner()); - let lsn = wal - .append_keyed( - RecordTarget { - record_type: record_type as u32, - tenant_id: tenant_id.as_u64(), - vshard_id: vshard_id.as_u32(), - database_id: database_id.as_u64(), - }, - payload, - apply_key, - ) - .map_err(crate::Error::Wal)?; - Ok(Lsn::new(lsn)) - } - +impl WalAppender<'_> { pub fn append_put( &self, tid: TenantId, @@ -97,6 +30,7 @@ impl WalManager { #[cfg(test)] mod tests { use super::*; + use crate::wal::manager::{NO_APPLY_KEY, WalManager}; use nodedb_wal::record::{FtsIndexPayload, SyncSeqAdvancePayload}; #[test] @@ -107,6 +41,7 @@ mod tests { let wal = WalManager::open_for_testing(&path).unwrap(); let lsn = wal + .appender(NO_APPLY_KEY) .append_sync_seq_advance(0xCAFE_BABE_DEAD_BEEF, 7, 42, 1_000_000) .unwrap(); assert_eq!(lsn, Lsn::new(1)); @@ -137,9 +72,10 @@ mod tests { let v = VShardId::new(0); let db = DatabaseId::DEFAULT; - let lsn1 = wal.append_put(t, v, db, b"key1=value1").unwrap(); - let lsn2 = wal.append_put(t, v, db, b"key2=value2").unwrap(); - let lsn3 = wal.append_delete(t, v, db, b"key1").unwrap(); + let appender = wal.appender(NO_APPLY_KEY); + let lsn1 = appender.append_put(t, v, db, b"key1=value1").unwrap(); + let lsn2 = appender.append_put(t, v, db, b"key2=value2").unwrap(); + let lsn3 = appender.append_delete(t, v, db, b"key1").unwrap(); assert_eq!(lsn1, Lsn::new(1)); assert_eq!(lsn2, Lsn::new(2)); @@ -175,9 +111,10 @@ mod tests { calvin_stamp: None, }; - let lsn1 = wal.append_transaction_redo(t, v, db, &record).unwrap(); - let lsn2 = wal.append_transaction_redo(t, v, db, &record).unwrap(); - let lsn3 = wal.append_transaction_redo(t, v, db, &record).unwrap(); + let appender = wal.appender(NO_APPLY_KEY); + let lsn1 = appender.append_transaction_redo(t, v, db, &record).unwrap(); + let lsn2 = appender.append_transaction_redo(t, v, db, &record).unwrap(); + let lsn3 = appender.append_transaction_redo(t, v, db, &record).unwrap(); assert_eq!(lsn1, Lsn::new(1)); assert_eq!(lsn2, Lsn::new(2)); @@ -209,6 +146,7 @@ mod tests { let db = DatabaseId::DEFAULT; let lsn = wal + .appender(NO_APPLY_KEY) .append_crdt_delta(t, v, db, b"loro-delta-bytes") .unwrap(); assert_eq!(lsn, Lsn::new(1)); @@ -247,7 +185,10 @@ mod tests { let v = VShardId::new(7); let db = DatabaseId::DEFAULT; - let lsn = wal.append_fts_index(t, v, db, &bytes).unwrap(); + let lsn = wal + .appender(NO_APPLY_KEY) + .append_fts_index(t, v, db, &bytes) + .unwrap(); assert_eq!(lsn, Lsn::new(1)); wal.sync().unwrap(); diff --git a/nodedb/src/wal/manager/append_batch.rs b/nodedb/src/wal/manager/append_batch.rs index 549eecfca..ae987006b 100644 --- a/nodedb/src/wal/manager/append_batch.rs +++ b/nodedb/src/wal/manager/append_batch.rs @@ -4,10 +4,10 @@ use nodedb_wal::record::RecordType; -use super::core::WalManager; +use super::appender::WalAppender; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -impl WalManager { +impl WalAppender<'_> { /// Append a checkpoint marker. Serializes the LSN before writing. pub fn append_checkpoint( &self, diff --git a/nodedb/src/wal/manager/append_index.rs b/nodedb/src/wal/manager/append_index.rs index 92f0579e8..4271beb3a 100644 --- a/nodedb/src/wal/manager/append_index.rs +++ b/nodedb/src/wal/manager/append_index.rs @@ -5,10 +5,10 @@ use nodedb_wal::record::RecordType; -use super::core::WalManager; +use super::appender::WalAppender; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -impl WalManager { +impl WalAppender<'_> { /// Append an `FtsIndex` record. Payload is a length-prefixed `FtsIndexPayload` /// produced by `nodedb_wal::record::FtsIndexPayload::to_bytes()`. pub fn append_fts_index( diff --git a/nodedb/src/wal/manager/append_metadata.rs b/nodedb/src/wal/manager/append_metadata.rs index ce168e970..7de585629 100644 --- a/nodedb/src/wal/manager/append_metadata.rs +++ b/nodedb/src/wal/manager/append_metadata.rs @@ -6,10 +6,10 @@ use nodedb_wal::record::RecordType; -use super::core::WalManager; +use super::appender::WalAppender; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -impl WalManager { +impl WalAppender<'_> { /// Append a `TemporalPurge` audit record. Emitted by the /// Control Plane's bitemporal-retention scheduler after a successful /// dispatch of `MetaOp::TemporalPurge*` to the Data Plane, providing @@ -111,19 +111,19 @@ impl WalManager { ) } - /// Append a payload-free `ProposalApplied` marker for the replicated - /// proposal whose apply is in scope (see [`WalManager::with_apply_key`]). - /// An apply that writes no record of its own appends it in place of its - /// forward record, so the proposal's key still reaches the WAL. + /// Append a payload-free `ProposalApplied` marker carrying this + /// appender's apply key. An apply that writes no record of its own + /// appends it in place of its forward record, so the proposal's key still + /// reaches the WAL. /// - /// Returns `None` and appends nothing outside a proposal's apply. + /// Returns `None` and appends nothing for an appender with no apply key. pub fn append_proposal_applied( &self, tenant_id: TenantId, vshard_id: VShardId, database_id: DatabaseId, ) -> crate::Result> { - if super::append::current_apply_key() == 0 { + if self.apply_key() == super::appender::NO_APPLY_KEY { return Ok(None); } self.append_record( diff --git a/nodedb/src/wal/manager/append_transaction.rs b/nodedb/src/wal/manager/append_transaction.rs index e8ae7e32f..0be102734 100644 --- a/nodedb/src/wal/manager/append_transaction.rs +++ b/nodedb/src/wal/manager/append_transaction.rs @@ -4,10 +4,10 @@ use nodedb_wal::record::RecordType; -use super::core::WalManager; +use super::appender::WalAppender; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -impl WalManager { +impl WalAppender<'_> { pub fn append_transaction( &self, tid: TenantId, diff --git a/nodedb/src/wal/manager/append_truncate.rs b/nodedb/src/wal/manager/append_truncate.rs index e8446a6a3..a277ed32d 100644 --- a/nodedb/src/wal/manager/append_truncate.rs +++ b/nodedb/src/wal/manager/append_truncate.rs @@ -4,10 +4,10 @@ use nodedb_wal::record::RecordType; -use super::core::WalManager; +use super::appender::WalAppender; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -impl WalManager { +impl WalAppender<'_> { /// Append a `ColumnarTruncate` record for a columnar or spatial truncate. /// Payload is produced by `encode_columnar_truncate_payload`. pub fn append_columnar_truncate( diff --git a/nodedb/src/wal/manager/append_vector.rs b/nodedb/src/wal/manager/append_vector.rs index cb5f95d23..d6dd70cde 100644 --- a/nodedb/src/wal/manager/append_vector.rs +++ b/nodedb/src/wal/manager/append_vector.rs @@ -4,10 +4,10 @@ use nodedb_wal::record::RecordType; -use super::core::WalManager; +use super::appender::WalAppender; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; -impl WalManager { +impl WalAppender<'_> { pub fn append_vector_put( &self, tid: TenantId, diff --git a/nodedb/src/wal/manager/appender.rs b/nodedb/src/wal/manager/appender.rs new file mode 100644 index 000000000..95259b5de --- /dev/null +++ b/nodedb/src/wal/manager/appender.rs @@ -0,0 +1,100 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The handle every WAL append goes through. +//! +//! A record's header carries an apply key: the idempotency key of the +//! replicated proposal whose apply appended it, [`NO_APPLY_KEY`] for a record +//! no proposal owns. The apply loop rebuilds which proposals applied from +//! these keys. The key is part of the append call: a caller names it when it +//! takes an appender, so no append can read a key another caller set and no +//! caller has a key to clear. + +use nodedb_wal::RecordTarget; +use nodedb_wal::record::RecordType; + +use super::core::WalManager; +use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + +/// The apply key of a record no replicated proposal owns. +pub const NO_APPLY_KEY: u64 = 0; + +/// Appends WAL records that all carry one apply key. +#[derive(Clone, Copy)] +pub struct WalAppender<'a> { + wal: &'a WalManager, + apply_key: u64, +} + +impl WalManager { + /// An appender whose records carry `apply_key`. A replicated proposal's + /// apply passes the proposal's idempotency key. Every other append passes + /// [`NO_APPLY_KEY`]. + pub fn appender(&self, apply_key: u64) -> WalAppender<'_> { + WalAppender { + wal: self, + apply_key, + } + } +} + +impl WalAppender<'_> { + /// The apply key every record this appender writes carries. + pub fn apply_key(&self) -> u64 { + self.apply_key + } + + /// Append one record of `record_type`. + pub(super) fn append_record( + &self, + record_type: RecordType, + tenant_id: TenantId, + vshard_id: VShardId, + database_id: DatabaseId, + payload: &[u8], + ) -> crate::Result { + let mut wal = self.wal.wal.lock().unwrap_or_else(|p| p.into_inner()); + let lsn = wal + .append_keyed( + RecordTarget { + record_type: record_type as u32, + tenant_id: tenant_id.as_u64(), + vshard_id: vshard_id.as_u32(), + database_id: database_id.as_u64(), + }, + payload, + self.apply_key, + ) + .map_err(crate::Error::Wal)?; + Ok(Lsn::new(lsn)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn each_record_carries_the_key_of_the_appender_that_wrote_it() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); + let (t, v, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + + let keyed = wal.appender(0xAB); + keyed.append_put(t, v, db, b"keyed").expect("keyed append"); + wal.appender(NO_APPLY_KEY) + .append_put(t, v, db, b"unkeyed") + .expect("unkeyed append"); + keyed + .append_put(t, v, db, b"keyed again") + .expect("keyed append"); + wal.sync().expect("sync"); + + let keys: Vec = wal + .replay() + .expect("replay") + .iter() + .map(|record| record.apply_key()) + .collect(); + assert_eq!(keys, vec![0xAB, NO_APPLY_KEY, 0xAB]); + } +} diff --git a/nodedb/src/wal/manager/durable_commit.rs b/nodedb/src/wal/manager/durable_commit.rs index 452a339b0..6f84c08ce 100644 --- a/nodedb/src/wal/manager/durable_commit.rs +++ b/nodedb/src/wal/manager/durable_commit.rs @@ -112,6 +112,7 @@ impl WalManager { mod tests { use super::*; use crate::types::{DatabaseId, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; fn open_wal(dir: &std::path::Path) -> WalManager { WalManager::open_for_testing(&dir.join("test.wal")).expect("open wal") @@ -122,6 +123,7 @@ mod tests { let dir = tempfile::tempdir().expect("tempdir"); let wal = open_wal(dir.path()); let lsn = wal + .appender(NO_APPLY_KEY) .append_put( TenantId::new(1), VShardId::new(0), @@ -138,6 +140,7 @@ mod tests { let dir = tempfile::tempdir().expect("tempdir"); let wal = open_wal(dir.path()); let lsn = wal + .appender(NO_APPLY_KEY) .append_put( TenantId::new(1), VShardId::new(0), @@ -157,13 +160,14 @@ mod tests { let mut lsns = Vec::new(); for _ in 0..16 { lsns.push( - wal.append_put( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - b"payload", - ) - .expect("append"), + wal.appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"payload", + ) + .expect("append"), ); } let max = *lsns.iter().max().expect("nonempty"); diff --git a/nodedb/src/wal/manager/encryption.rs b/nodedb/src/wal/manager/encryption.rs index 73ec59036..62bf9440b 100644 --- a/nodedb/src/wal/manager/encryption.rs +++ b/nodedb/src/wal/manager/encryption.rs @@ -264,6 +264,7 @@ fn write_signing_root( mod tests { use super::*; use crate::types::{DatabaseId, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; #[test] fn crdt_signing_root_survives_chained_runtime_rotation_and_restart() { @@ -287,25 +288,27 @@ mod tests { let mut wal = WalManager::open_encrypted(&wal_dir, false, &key_a).unwrap(); let stable_root = wal.crdt_signing_root().unwrap().unwrap(); - wal.append_put( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - b"a", - ) - .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"a", + ) + .unwrap(); wal.rotate_key(&key_b).unwrap(); drop(wal); let mut wal = WalManager::open_encrypted_rotating(&wal_dir, false, &key_b, &key_a).unwrap(); assert_eq!(wal.crdt_signing_root().unwrap(), Some(stable_root)); - wal.append_put( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - b"b", - ) - .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"b", + ) + .unwrap(); wal.rotate_key(&key_c).unwrap(); drop(wal); @@ -340,26 +343,28 @@ mod tests { wal.set_encryption_ring(nodedb_wal::crypto::KeyRing::new(key_a.clone())) .unwrap(); let root = wal.crdt_signing_root().unwrap(); - wal.append_put( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - b"a", - ) - .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"a", + ) + .unwrap(); wal.set_encryption_ring(nodedb_wal::crypto::KeyRing::with_previous( key_b.clone(), key_a, )) .unwrap(); assert_eq!(wal.crdt_signing_root().unwrap(), root); - wal.append_put( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - b"b", - ) - .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"b", + ) + .unwrap(); wal.set_encryption_ring(nodedb_wal::crypto::KeyRing::with_previous(key_c, key_b)) .unwrap(); assert_eq!(wal.crdt_signing_root().unwrap(), root); diff --git a/nodedb/src/wal/manager/mod.rs b/nodedb/src/wal/manager/mod.rs index 0d5d60ff0..7d7ef79e3 100644 --- a/nodedb/src/wal/manager/mod.rs +++ b/nodedb/src/wal/manager/mod.rs @@ -7,6 +7,7 @@ pub mod append_metadata; pub mod append_transaction; pub mod append_truncate; pub mod append_vector; +pub mod appender; pub mod audit; pub mod core; pub mod durable_commit; @@ -14,4 +15,5 @@ pub mod encryption; pub mod ops; pub mod replay; +pub use appender::{NO_APPLY_KEY, WalAppender}; pub use core::WalManager; diff --git a/nodedb/src/wal/manager/ops.rs b/nodedb/src/wal/manager/ops.rs index 978f5afd8..f07c1557c 100644 --- a/nodedb/src/wal/manager/ops.rs +++ b/nodedb/src/wal/manager/ops.rs @@ -61,6 +61,7 @@ impl WalManager { mod tests { use super::*; use crate::types::{DatabaseId, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; #[test] fn next_lsn_continues_after_reopen() { @@ -69,20 +70,22 @@ mod tests { { let wal = WalManager::open_for_testing(&path).unwrap(); - wal.append_put( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - b"a", - ) - .unwrap(); - wal.append_put( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - b"b", - ) - .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"a", + ) + .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"b", + ) + .unwrap(); wal.sync().unwrap(); } @@ -90,6 +93,7 @@ mod tests { assert_eq!(wal.next_lsn(), Lsn::new(3)); let lsn = wal + .appender(NO_APPLY_KEY) .append_put( TenantId::new(1), VShardId::new(0), @@ -112,7 +116,8 @@ mod tests { let db = DatabaseId::DEFAULT; for i in 0..10u32 { - wal.append_put(t, v, db, format!("val-{i}").as_bytes()) + wal.appender(NO_APPLY_KEY) + .append_put(t, v, db, format!("val-{i}").as_bytes()) .unwrap(); } wal.sync().unwrap(); @@ -130,13 +135,14 @@ mod tests { let path = dir.path().join("wal_dir"); let wal = WalManager::open_for_testing(&path).unwrap(); - wal.append_put( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - b"data", - ) - .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"data", + ) + .unwrap(); wal.sync().unwrap(); let size = wal.total_size_bytes().unwrap(); diff --git a/nodedb/tests/inproc/cases/calvin_scheduler_restart.rs b/nodedb/tests/inproc/cases/calvin_scheduler_restart.rs index 2593b4a2d..e745f51c4 100644 --- a/nodedb/tests/inproc/cases/calvin_scheduler_restart.rs +++ b/nodedb/tests/inproc/cases/calvin_scheduler_restart.rs @@ -16,7 +16,7 @@ use tempfile::TempDir; use nodedb::control::cluster::calvin::scheduler::{NOT_YET_APPLIED_EPOCH, read_applied_recovery}; use nodedb::types::VShardId; -use nodedb::wal::manager::WalManager; +use nodedb::wal::manager::{NO_APPLY_KEY, WalManager}; // ── Helper ──────────────────────────────────────────────────────────────────── @@ -37,7 +37,8 @@ fn scheduler_restart_reads_applied_markers_after_five_epochs() { { let wal = open_wal(&dir); for epoch in 1u64..=5 { - wal.append_calvin_applied(VShardId::new(vshard_id), epoch, 0) + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard_id), epoch, 0) .unwrap(); } wal.sync().unwrap(); @@ -65,13 +66,17 @@ fn scheduler_restart_reports_max_and_all_positions_regardless_of_order() { { let wal = open_wal(&dir); - wal.append_calvin_applied(VShardId::new(vshard_id), 3, 0) + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard_id), 3, 0) .unwrap(); - wal.append_calvin_applied(VShardId::new(vshard_id), 1, 0) + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard_id), 1, 0) .unwrap(); - wal.append_calvin_applied(VShardId::new(vshard_id), 5, 0) + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard_id), 5, 0) .unwrap(); - wal.append_calvin_applied(VShardId::new(vshard_id), 2, 0) + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard_id), 2, 0) .unwrap(); wal.sync().unwrap(); } @@ -107,10 +112,12 @@ fn scheduler_restart_multi_position_epoch_does_not_lose_uncommitted_position() { { let wal = open_wal(&dir); // Prior epoch fully applied. - wal.append_calvin_applied(VShardId::new(vshard_id), 8, 0) + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard_id), 8, 0) .unwrap(); // Torn epoch: position 0 committed, position 1 did NOT (crash between). - wal.append_calvin_applied(VShardId::new(vshard_id), torn_epoch, 0) + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(vshard_id), torn_epoch, 0) .unwrap(); wal.sync().unwrap(); } @@ -158,9 +165,15 @@ fn scheduler_restart_vshard_isolation() { { let wal = open_wal(&dir); - wal.append_calvin_applied(VShardId::new(1), 10, 0).unwrap(); - wal.append_calvin_applied(VShardId::new(2), 99, 0).unwrap(); - wal.append_calvin_applied(VShardId::new(1), 20, 0).unwrap(); + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(1), 10, 0) + .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(2), 99, 0) + .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(1), 20, 0) + .unwrap(); wal.sync().unwrap(); } diff --git a/nodedb/tests/inproc/cases/wal_catchup.rs b/nodedb/tests/inproc/cases/wal_catchup.rs index 7f988989b..68b9e2a57 100644 --- a/nodedb/tests/inproc/cases/wal_catchup.rs +++ b/nodedb/tests/inproc/cases/wal_catchup.rs @@ -12,7 +12,7 @@ use nodedb::control::security::audit::NoopAuditEmitter; use nodedb::control::state::SharedState; use nodedb::data::executor::core_loop::CoreLoop; use nodedb::types::*; -use nodedb::wal::manager::WalManager; +use nodedb::wal::manager::{NO_APPLY_KEY, WalManager}; use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -162,6 +162,7 @@ impl TestStack { fn write_to_wal(&self, collection: &str, payload: Vec) { let wal_payload = zerompk::to_msgpack_vec(&(collection.to_string(), payload)).unwrap(); self.wal + .appender(NO_APPLY_KEY) .append_timeseries_batch( TenantId::new(1), VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection), @@ -513,13 +514,14 @@ fn startup_replay_recovers_all_wal_data() { 1_700_000_000_000_000_000i64 + batch as i64 * rows_per_batch as i64 * 1_000_000; let payload = ilp_payload(collection, rows_per_batch, start_ts); let wal_payload = zerompk::to_msgpack_vec(&(collection.to_string(), payload)).unwrap(); - wal.append_timeseries_batch( - TenantId::new(1), - VShardId::new(0), - DatabaseId::DEFAULT, - &wal_payload, - ) - .unwrap(); + wal.appender(NO_APPLY_KEY) + .append_timeseries_batch( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + &wal_payload, + ) + .unwrap(); } wal.sync().unwrap(); From 2ac08414ce7f6c94de76e237219db573803b32f4 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 06:55:15 +0800 Subject: [PATCH 17/64] feat(executor): fail-stop a core whose rollback state is unknown A rollback that fails part way, or a committed write whose post-install work (a memtable flush, an artifact republish) fails afterward, leaves a core holding state that neither matches its WAL nor can be undone. Continuing to serve reads and writes from that core would answer against state no replica or restart reproduces. CoreFailStop latches a core the first time either cause is observed: it logs an ERROR, files a diagnostic report, and refuses every queued and future request with RetryableRefusal until a restart rebuilds the core from the WAL. The latch is exposed as the nodedb_data_plane_core_fail_stopped gauge, folded into the /healthz readiness body and the native STATUS command, and gates opportunistic checkpointing and maintenance so a stopped core publishes no artifact of its unknown state. Reaching that guarantee required each engine's rollback to reverse exactly what it wrote instead of approximating it: - The graph CSR index gains an interning-aware restore module: exact edge and weight reversal, and withdrawal of an interned node or label only when it is the newest entry and nothing still refers to it, refused otherwise via a new GraphError::WithdrawRefused. - The edge store's temporal writer gains a revert module mirroring each bitemporal write with its exact undo, and node-edge cascade delete now runs through one cascade helper shared by point-delete and periodic sweep so a failed store write leaves the CSR and edge store still agreeing. - The IVF vector index gains roll_back_to, withdrawing every vector added after a mark and dropping training state the mark predates. - The transaction undo pipeline is restructured accordingly: undo/ apply.rs shrinks to dispatch, with edge and vector undo split into their own modules, and each engine's undo entry now carries the detail a failed reversal reports. The redo-apply and WAL-replay paths pass through the exact-restore data these rollbacks need, and SeriesCatalog gains Clone/PartialEq so a timeseries rollback can snapshot and compare its catalog state. --- nodedb-graph/src/csr/index/mod.rs | 2 + nodedb-graph/src/csr/index/mutation.rs | 8 +- nodedb-graph/src/csr/index/restore.rs | 380 +++++++++++++++++ nodedb-graph/src/error.rs | 6 + nodedb-types/src/timeseries/series.rs | 2 +- nodedb-vector/src/ivf.rs | 53 +++ nodedb/src/control/metrics/mod.rs | 2 +- .../control/metrics/system/core_fail_stop.rs | 136 ++++++ nodedb/src/control/metrics/system/fields.rs | 4 + nodedb/src/control/metrics/system/mod.rs | 2 + nodedb/src/control/metrics/system/render.rs | 1 + .../src/control/server/http/routes/health.rs | 17 + .../control/server/native/session/request.rs | 7 + .../src/data/executor/core_loop/fail_stop.rs | 196 +++++++++ .../executor/core_loop/graph_partition.rs | 27 ++ .../data/executor/core_loop/maintenance.rs | 32 +- nodedb/src/data/executor/core_loop/mod.rs | 1 + nodedb/src/data/executor/core_loop/open.rs | 1 + nodedb/src/data/executor/core_loop/state.rs | 8 +- nodedb/src/data/executor/core_loop/tick.rs | 46 +++ .../handlers/bulk_dml/delete_cascade.rs | 15 +- .../handlers/graph_edge_write/delete.rs | 47 ++- .../executor/handlers/graph_edge_write/put.rs | 76 ++-- .../executor/handlers/point/apply_delete.rs | 46 +-- .../executor/handlers/transaction/batch.rs | 7 +- .../handlers/transaction/batch_crdt.rs | 7 +- .../transaction/index_write_values.rs | 3 +- .../handlers/transaction/redo_apply/cover.rs | 46 +++ .../handlers/transaction/redo_apply/entry.rs | 141 ++++++- .../handlers/transaction/redo_apply/passes.rs | 13 +- .../handlers/transaction/redo_apply/settle.rs | 14 +- .../handlers/transaction/undo/apply.rs | 391 ++---------------- .../transaction/undo/document_outcome.rs | 28 +- .../handlers/transaction/undo/edge_write.rs | 382 +++++++++++++++++ .../handlers/transaction/undo/entry.rs | 30 +- .../handlers/transaction/undo/graph_node.rs | 86 +++- .../executor/handlers/transaction/undo/mod.rs | 1 + .../handlers/transaction/undo/rollback.rs | 50 +-- .../handlers/transaction/undo/timeseries.rs | 1 + .../handlers/transaction/undo/vector_write.rs | 104 +++++ nodedb/src/data/executor/handlers/truncate.rs | 15 +- .../src/data/executor/wal_replay/crdt_list.rs | 6 +- nodedb/src/data/executor/wal_replay/kv_put.rs | 15 +- .../data/executor/wal_replay_graph_labels.rs | 33 +- .../src/data/executor/wal_replay_kv_expiry.rs | 30 +- .../src/data/executor/wal_replay_kv_incr.rs | 15 +- .../executor/wal_replay_kv_insert_conflict.rs | 15 +- nodedb/src/data/executor/wal_replay_kv_ttl.rs | 30 +- nodedb/src/data/runtime/event_loop.rs | 10 +- nodedb/src/diag/context/data_plane.rs | 36 ++ nodedb/src/diag/context/mod.rs | 2 +- nodedb/src/diag/mod.rs | 8 +- nodedb/src/diag/recording/data_plane.rs | 17 + nodedb/src/diag/recording/mod.rs | 4 +- nodedb/src/engine/graph/edge_store/cascade.rs | 142 ++++--- nodedb/src/engine/graph/edge_store/mod.rs | 6 +- .../engine/graph/edge_store/node_identity.rs | 27 ++ .../engine/graph/edge_store/stats/update.rs | 36 +- .../engine/graph/edge_store/temporal/mod.rs | 2 + .../graph/edge_store/temporal/revert.rs | 343 +++++++++++++++ .../engine/graph/edge_store/temporal/write.rs | 169 +++++--- 61 files changed, 2569 insertions(+), 811 deletions(-) create mode 100644 nodedb-graph/src/csr/index/restore.rs create mode 100644 nodedb/src/control/metrics/system/core_fail_stop.rs create mode 100644 nodedb/src/data/executor/core_loop/fail_stop.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/edge_write.rs create mode 100644 nodedb/src/engine/graph/edge_store/temporal/revert.rs diff --git a/nodedb-graph/src/csr/index/mod.rs b/nodedb-graph/src/csr/index/mod.rs index bbd2ac74c..5afdfebfb 100644 --- a/nodedb-graph/src/csr/index/mod.rs +++ b/nodedb-graph/src/csr/index/mod.rs @@ -8,10 +8,12 @@ //! - `mutation` — `add_edge`, `remove_edge`, `remove_node_edges` //! - `lookup` — neighbor queries, accessors, degree, iterators //! - `scoped` — collection-scoped read paths (MATCH / RAG) +//! - `restore` — exact edge writes and the reversals a rollback uses pub mod interning; pub mod lookup; pub mod mutation; +pub mod restore; pub mod scoped; pub mod types; diff --git a/nodedb-graph/src/csr/index/mutation.rs b/nodedb-graph/src/csr/index/mutation.rs index d2e39d913..5b2b3853c 100644 --- a/nodedb-graph/src/csr/index/mutation.rs +++ b/nodedb-graph/src/csr/index/mutation.rs @@ -87,8 +87,10 @@ impl CsrIndex { { return Ok(()); } - // Check for duplicates in dense CSR (collection-aware). + // A dense copy is the edge itself: a deleted one comes back. if self.dense_has_edge(src_id, label_id, dst_id, collection_id) { + self.deleted_edges + .remove(&(src_id, label_id, dst_id, collection_id)); return Ok(()); } @@ -107,8 +109,8 @@ impl CsrIndex { self.buffer_in_weights[dst_id as usize].push(weight); } - // If this exact `(src, label, dst, collection)` copy was previously - // deleted, un-delete it. + // A node-edge removal marks buffered edges deleted too. With no dense + // copy that mark names this edge only, so it goes. self.deleted_edges .remove(&(src_id, label_id, dst_id, collection_id)); Ok(()) diff --git a/nodedb-graph/src/csr/index/restore.rs b/nodedb-graph/src/csr/index/restore.rs new file mode 100644 index 000000000..3173a871d --- /dev/null +++ b/nodedb-graph/src/csr/index/restore.rs @@ -0,0 +1,380 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Exact edge writes, and the reversal primitives a rollback of a graph +//! write uses. +//! +//! A rollback puts the index back to what it held before the write: the +//! edge's presence and weight, each node surrogate the write rebound, and +//! each node or node label the write interned. Interning appends, so a +//! rollback withdraws the newest entry first and refuses any other. + +use super::types::CsrIndex; +use crate::GraphError; + +impl CsrIndex { + /// Weight of the live `(src, label, dst)` edge in `collection`, `None` + /// when that edge is not live. + pub fn edge_weight_in_collection( + &self, + src: &str, + label: &str, + dst: &str, + collection: &str, + ) -> Option { + let src_id = *self.node_to_id.get(src)?; + let dst_id = *self.node_to_id.get(dst)?; + let label_id = *self.label_to_id.get(label)?; + let coll_id = *self.collection_to_id.get(collection)?; + let idx = src_id as usize; + + if let (Some(edges), Some(colls)) = ( + self.buffer_out.get(idx), + self.buffer_out_collections.get(idx), + ) { + let position = edges + .iter() + .zip(colls.iter()) + .position(|(&(l, d), &c)| l == label_id && d == dst_id && c == coll_id); + if let Some(k) = position { + let weight = if self.has_weights { + self.buffer_out_weights + .get(idx) + .and_then(|weights| weights.get(k)) + .copied() + .unwrap_or(1.0) + } else { + 1.0 + }; + return Some(weight); + } + } + + if self + .deleted_edges + .contains(&(src_id, label_id, dst_id, coll_id)) + || idx + 1 >= self.out_offsets.len() + { + return None; + } + let start = self.out_offsets[idx] as usize; + let end = self.out_offsets[idx + 1] as usize; + (start..end) + .find(|&i| { + self.out_labels[i] == label_id + && self.out_targets[i] == dst_id + && self.out_collections.get(i).copied().unwrap_or(0) == coll_id + }) + .map(|i| self.out_edge_weight(i)) + } + + /// Make the `(src, label, dst)` edge in `collection` live with `weight`. + /// + /// A live edge with another weight takes the new one. Returns the weight + /// the edge had while live before, `None` when it was not live, so a + /// rollback can put it back. + pub fn put_edge_in_collection( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, + weight: f64, + ) -> Result, GraphError> { + let prior = self.edge_weight_in_collection(src, label, dst, collection); + match prior { + Some(current) if current == weight => return Ok(prior), + Some(_) => self.remove_edge_in_collection(src, label, dst, collection), + None => {} + } + let src_id = self.ensure_node(src)?; + let dst_id = self.ensure_node(dst)?; + let label_id = self.ensure_label(label)?; + let coll_id = self.ensure_collection(collection); + if weight != 1.0 && !self.has_weights { + self.enable_weights(); + } + // A deleted dense copy stays deleted: the live copy is the buffer one. + // With no dense copy, a deletion mark names this edge only, so it goes. + if !self.dense_has_edge(src_id, label_id, dst_id, coll_id) { + self.deleted_edges + .remove(&(src_id, label_id, dst_id, coll_id)); + } + self.buffer_out[src_id as usize].push((label_id, dst_id)); + self.buffer_in[dst_id as usize].push((label_id, src_id)); + self.buffer_out_collections[src_id as usize].push(coll_id); + self.buffer_in_collections[dst_id as usize].push(coll_id); + if self.has_weights { + self.buffer_out_weights[src_id as usize].push(weight); + self.buffer_in_weights[dst_id as usize].push(weight); + } + Ok(prior) + } + + /// Put the edge back to `prior`: live with that weight, or absent when + /// `prior` is `None`. + pub fn restore_edge_in_collection( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, + prior: Option, + ) -> Result<(), GraphError> { + match prior { + Some(weight) => self + .put_edge_in_collection(src, label, dst, collection, weight) + .map(drop), + None => { + self.remove_edge_in_collection(src, label, dst, collection); + Ok(()) + } + } + } + + /// Put `node`'s surrogate back to `prior`, `0` for none. + pub fn restore_node_surrogate(&mut self, node: &str, prior: u32) { + let Some(&id) = self.node_to_id.get(node) else { + return; + }; + let Some(slot) = self.node_surrogates.get_mut(id as usize) else { + return; + }; + let current = *slot; + if current == prior { + return; + } + *slot = prior; + if current != 0 && self.surrogate_to_local.get(¤t) == Some(&id) { + self.surrogate_to_local.remove(¤t); + } + if prior != 0 { + self.surrogate_to_local.insert(prior, id); + } + } + + /// Whether the node label `label` is interned. + pub fn has_node_label_name(&self, label: &str) -> bool { + self.node_label_to_id.contains_key(label) + } + + /// Withdraw `node`, which a rolled-back write interned. + /// + /// An absent node is a no-op. The node must be the newest one, with no + /// edge and no label: withdrawing any other would renumber or orphan live + /// state, so that is refused. + pub fn withdraw_newest_node(&mut self, node: &str) -> Result<(), GraphError> { + let Some(&id) = self.node_to_id.get(node) else { + return Ok(()); + }; + let idx = id as usize; + let newest = idx + 1 == self.id_to_node.len(); + let no_buffered_edge = self.buffer_out.get(idx).is_none_or(Vec::is_empty) + && self.buffer_in.get(idx).is_none_or(Vec::is_empty); + let empty_range = |offsets: &[u32]| match (offsets.get(idx), offsets.get(idx + 1)) { + (Some(start), Some(end)) => start == end, + _ => true, + }; + let no_dense_edge = + empty_range(self.out_offsets.as_slice()) && empty_range(self.in_offsets.as_slice()); + let unlabeled = self.node_label_bits.get(idx).copied().unwrap_or(0) == 0; + if !(newest && no_buffered_edge && no_dense_edge && unlabeled) { + return Err(GraphError::WithdrawRefused { + kind: "node", + name: node.to_string(), + }); + } + + self.node_to_id.remove(node); + self.id_to_node.pop(); + // Offsets hold one entry more than there are nodes. + if self.out_offsets.len() > idx + 1 { + self.out_offsets.pop(); + } + if self.in_offsets.len() > idx + 1 { + self.in_offsets.pop(); + } + self.buffer_out.truncate(idx); + self.buffer_in.truncate(idx); + self.buffer_out_weights.truncate(idx); + self.buffer_in_weights.truncate(idx); + self.buffer_out_collections.truncate(idx); + self.buffer_in_collections.truncate(idx); + self.node_label_bits.truncate(idx); + self.node_surrogates.truncate(idx); + self.surrogate_to_local.retain(|_, local| *local != id); + // The id goes back to the pool: no deletion mark may name it. + self.deleted_edges + .retain(|&(src, _, dst, _)| src != id && dst != id); + self.access_counts.truncate(idx); + Ok(()) + } + + /// Withdraw the node label `label`, which a rolled-back write interned. + /// + /// An absent label is a no-op. The label must be the newest one and no + /// node may carry it, or the withdraw is refused. + pub fn withdraw_newest_node_label(&mut self, label: &str) -> Result<(), GraphError> { + let Some(&id) = self.node_label_to_id.get(label) else { + return Ok(()); + }; + let newest = usize::from(id) + 1 == self.node_label_names.len(); + let bit = 1u64 << id; + let carried = self.node_label_bits.iter().any(|bits| bits & bit != 0); + if !newest || carried { + return Err(GraphError::WithdrawRefused { + kind: "node label", + name: label.to_string(), + }); + } + self.node_label_to_id.remove(label); + self.node_label_names.pop(); + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::csr::index::types::Direction; + use crate::test_support::test_memory; + + #[test] + fn put_edge_replaces_the_weight_of_a_live_edge_and_reports_the_old_one() { + let mut csr = CsrIndex::new(test_memory()); + assert_eq!( + csr.put_edge_in_collection("a", "L", "b", "c", 2.5) + .expect("first put"), + None + ); + assert_eq!( + csr.put_edge_in_collection("a", "L", "b", "c", 9.0) + .expect("second put"), + Some(2.5) + ); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), Some(9.0)); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + } + + #[test] + fn put_edge_replaces_the_weight_of_a_compacted_edge() { + let mut csr = CsrIndex::new(test_memory()); + csr.put_edge_in_collection("a", "L", "b", "c", 2.5) + .expect("put"); + csr.compact().expect("compact"); + assert_eq!( + csr.put_edge_in_collection("a", "L", "b", "c", 9.0) + .expect("reweigh"), + Some(2.5) + ); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), Some(9.0)); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + csr.compact().expect("compact again"); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), Some(9.0)); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + } + + #[test] + fn restoring_an_edge_brings_back_a_deleted_compacted_edge() { + let mut csr = CsrIndex::new(test_memory()); + csr.put_edge_in_collection("a", "L", "b", "c", 1.0) + .expect("put"); + csr.compact().expect("compact"); + csr.remove_edge_in_collection("a", "L", "b", "c"); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), None); + + csr.restore_edge_in_collection("a", "L", "b", "c", Some(1.0)) + .expect("restore"); + assert_eq!(csr.edge_weight_in_collection("a", "L", "b", "c"), Some(1.0)); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + } + + #[test] + fn re_adding_a_deleted_compacted_edge_brings_it_back() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge_in_collection("a", "L", "b", "c") + .expect("edge"); + csr.compact().expect("compact"); + csr.remove_edge_in_collection("a", "L", "b", "c"); + csr.add_edge_in_collection("a", "L", "b", "c") + .expect("re-add"); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + } + + #[test] + fn withdrawing_the_newest_isolated_node_removes_it() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge("a", "L", "b").expect("edge"); + csr.put_edge_in_collection("b", "L", "z", "c", 1.0) + .expect("edge to z"); + csr.remove_edge_in_collection("b", "L", "z", "c"); + csr.set_node_surrogate("z", nodedb_types::Surrogate::new(42)); + + csr.withdraw_newest_node("z").expect("withdraw"); + + assert!(!csr.contains_node("z")); + assert_eq!(csr.node_count(), 2); + assert_eq!( + csr.node_id_for_surrogate(nodedb_types::Surrogate::new(42)), + None + ); + csr.add_edge("a", "L", "y") + .expect("a later node takes the freed id"); + csr.compact().expect("compact"); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 2); + } + + #[test] + fn withdrawing_a_node_that_is_not_the_newest_is_refused() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_node_label("x", "Person").expect("label x"); + csr.remove_node_label("x", "Person"); + csr.add_edge("a", "L", "b").expect("edge"); + assert!(matches!( + csr.withdraw_newest_node("x"), + Err(GraphError::WithdrawRefused { .. }) + )); + assert!(csr.contains_node("x")); + } + + #[test] + fn withdrawing_a_node_with_an_edge_is_refused() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge("a", "L", "b").expect("edge"); + assert!(matches!( + csr.withdraw_newest_node("b"), + Err(GraphError::WithdrawRefused { .. }) + )); + } + + #[test] + fn withdrawing_the_newest_unused_node_label_frees_its_slot() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_node_label("x", "Person").expect("label"); + csr.remove_node_label("x", "Person"); + csr.withdraw_newest_node_label("Person").expect("withdraw"); + assert!(!csr.has_node_label_name("Person")); + } + + #[test] + fn withdrawing_a_carried_node_label_is_refused() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_node_label("x", "Person").expect("label"); + assert!(matches!( + csr.withdraw_newest_node_label("Person"), + Err(GraphError::WithdrawRefused { .. }) + )); + } + + #[test] + fn restoring_a_surrogate_unbinds_the_one_a_write_set() { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge("a", "L", "b").expect("edge"); + csr.set_node_surrogate("a", nodedb_types::Surrogate::new(5)); + csr.restore_node_surrogate("a", 0); + assert_eq!(csr.node_surrogate("a"), None); + assert_eq!( + csr.node_id_for_surrogate(nodedb_types::Surrogate::new(5)), + None + ); + } +} diff --git a/nodedb-graph/src/error.rs b/nodedb-graph/src/error.rs index 7168c820e..d37f9b1f6 100644 --- a/nodedb-graph/src/error.rs +++ b/nodedb-graph/src/error.rs @@ -55,4 +55,10 @@ pub enum GraphError { /// Callers should apply backpressure and retry after memory is released. #[error("graph memory budget rejected: {0}")] MemoryBudget(#[from] MemError), + + /// A rollback asked to withdraw an interned node or node label that is + /// not the newest one, or that something still refers to. Withdrawing it + /// would renumber or orphan live state. + #[error("cannot withdraw {kind} '{name}': it is not the newest {kind} or it is still in use")] + WithdrawRefused { kind: &'static str, name: String }, } diff --git a/nodedb-types/src/timeseries/series.rs b/nodedb-types/src/timeseries/series.rs index 40937353c..6c7c21303 100644 --- a/nodedb-types/src/timeseries/series.rs +++ b/nodedb-types/src/timeseries/series.rs @@ -52,7 +52,7 @@ impl SeriesKey { /// On insert, if the SeriesId already maps to a *different* SeriesKey, the /// catalog rehashes with an incrementing attempt counter until it finds a free /// slot. This is one lookup per new series (not per row). -#[derive(Debug, Default, Serialize, Deserialize)] +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] pub struct SeriesCatalog { /// SeriesId → (SeriesKey, rehash attempt that produced this ID). entries: HashMap, diff --git a/nodedb-vector/src/ivf.rs b/nodedb-vector/src/ivf.rs index faba2c0df..26993fd28 100644 --- a/nodedb-vector/src/ivf.rs +++ b/nodedb-vector/src/ivf.rs @@ -371,6 +371,59 @@ mod tests { ); } + fn small_params() -> IvfPqParams { + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 4, + metric: DistanceMetric::L2, + } + } + + #[test] + fn rolling_back_withdraws_every_vector_added_after_the_mark() { + let vecs = make_vectors(64, 8); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new(8, small_params()); + idx.train(&refs, test_memory()); + idx.add_batch(&refs[..40]); + let mark = idx.len() as u32; + let before: Vec = idx.search(&vecs[5], 40).iter().map(|r| r.id).collect(); + + idx.add_batch(&refs[40..]); + idx.roll_back_to(mark, true); + + assert_eq!(idx.len(), 40); + assert!(idx.is_trained(), "the training the index held stays"); + let after = idx.search(&vecs[5], 64); + assert!( + after.iter().all(|r| r.id < mark), + "no vector added after the mark is found" + ); + let after_ids: Vec = after.iter().map(|r| r.id).collect(); + assert_eq!(after_ids, before, "the search reads as before the adds"); + + // The next add takes the first id past the mark again. + assert_eq!(idx.add(&vecs[63]), mark); + } + + #[test] + fn rolling_back_to_an_untrained_mark_drops_the_training() { + let vecs = make_vectors(16, 8); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new(8, small_params()); + idx.train(&refs, test_memory()); + idx.add_batch(&refs); + + idx.roll_back_to(0, false); + + assert!(idx.is_empty()); + assert!(!idx.is_trained()); + assert_eq!(idx.n_cells(), 0); + assert!(idx.search(&vecs[0], 5).is_empty()); + } + #[test] fn empty_index() { let idx = IvfPqIndex::new(8, IvfPqParams::default()); diff --git a/nodedb/src/control/metrics/mod.rs b/nodedb/src/control/metrics/mod.rs index d09e47642..5bd4a42b4 100644 --- a/nodedb/src/control/metrics/mod.rs +++ b/nodedb/src/control/metrics/mod.rs @@ -12,5 +12,5 @@ pub use database::{DatabaseCounters, DatabaseMetricsRegistry, DatabaseQuotaMetri pub use histogram::AtomicHistogram; pub use per_vshard::{PerVShardMetrics, PerVShardMetricsRegistry, VShardStatsSnapshot}; pub use purge::PurgeMetrics; -pub use system::{CoreHeartbeats, SystemMetrics}; +pub use system::{CoreFailStopReport, CoreFailStops, CoreHeartbeats, SystemMetrics}; pub use tenant::TenantQuotaMetrics; diff --git a/nodedb/src/control/metrics/system/core_fail_stop.rs b/nodedb/src/control/metrics/system/core_fail_stop.rs new file mode 100644 index 000000000..2c3d19039 --- /dev/null +++ b/nodedb/src/control/metrics/system/core_fail_stop.rs @@ -0,0 +1,136 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Node-wide record of Data Plane cores that fail-stopped. +//! +//! A core fail-stops when its state is unknown: a rollback failed part way, +//! or the work owed after a committed record's install failed. Such a core +//! refuses every request until restart. Its latch lives on the core. This +//! record is the node-wide view `/healthz`, the native `STATUS` and the +//! `nodedb_data_plane_core_fail_stopped` gauge read. +//! +//! The first report wins, like the Calvin halt and metadata-apply wedge +//! markers: a fail-stopped core stays stopped until restart, so a later +//! report cannot replace the cause an operator must act on. The gauge counts +//! every stopped core. + +use std::sync::OnceLock; +use std::sync::atomic::{AtomicU64, Ordering}; + +/// Why the first Data Plane core fail-stopped. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CoreFailStopReport { + pub core_id: usize, + /// The cause label, as the core's ERROR line names it. + pub cause: &'static str, + pub detail: String, +} + +/// First-report-wins record of fail-stopped Data Plane cores. Never clears. +#[derive(Debug, Default)] +pub struct CoreFailStops { + first: OnceLock, + stopped: AtomicU64, +} + +impl CoreFailStops { + /// Record one core that fail-stopped. A core reports once. + pub fn record(&self, report: CoreFailStopReport) { + self.stopped.fetch_add(1, Ordering::Relaxed); + let _ = self.first.set(report); + } + + /// The first core that fail-stopped, if one did. + pub fn report(&self) -> Option<&CoreFailStopReport> { + self.first.get() + } + + pub fn is_stopped(&self) -> bool { + self.first.get().is_some() + } + + /// Number of cores that fail-stopped. + pub fn stopped_cores(&self) -> u64 { + self.stopped.load(Ordering::Relaxed) + } + + /// Append the `nodedb_data_plane_core_fail_stopped` gauge. + pub fn write_prometheus(&self, out: &mut String) { + use std::fmt::Write as _; + let _ = writeln!( + out, + "# HELP nodedb_data_plane_core_fail_stopped Data Plane cores that stopped \ + serving because their state is unknown\n\ + # TYPE nodedb_data_plane_core_fail_stopped gauge\n\ + nodedb_data_plane_core_fail_stopped {}", + self.stopped_cores() + ); + } +} + +/// Readiness-probe rendering for a node with a fail-stopped core: `503`, +/// degraded. The other cores keep serving. +pub fn to_http_response( + report: &CoreFailStopReport, + stopped_cores: u64, +) -> (axum::http::StatusCode, serde_json::Value) { + ( + axum::http::StatusCode::SERVICE_UNAVAILABLE, + serde_json::json!({ + "status": "degraded", + "reason": "data_plane_core_fail_stopped", + "core_id": report.core_id, + "cause": report.cause, + "error": report.detail, + "stopped_cores": stopped_cores, + }), + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn report(core_id: usize) -> CoreFailStopReport { + CoreFailStopReport { + core_id, + cause: "rollback_failed", + detail: "undo entry 3 failed".into(), + } + } + + #[test] + fn a_fresh_record_reports_no_stopped_core() { + let stops = CoreFailStops::default(); + assert!(!stops.is_stopped()); + assert_eq!(stops.stopped_cores(), 0); + } + + #[test] + fn the_first_report_wins_and_every_report_is_counted() { + let stops = CoreFailStops::default(); + stops.record(report(2)); + stops.record(report(5)); + assert_eq!(stops.report().map(|r| r.core_id), Some(2)); + assert_eq!(stops.stopped_cores(), 2); + } + + #[test] + fn the_gauge_counts_stopped_cores() { + let stops = CoreFailStops::default(); + stops.record(report(1)); + let mut out = String::new(); + stops.write_prometheus(&mut out); + assert!( + out.contains("nodedb_data_plane_core_fail_stopped 1"), + "{out}" + ); + } + + #[test] + fn the_readiness_body_names_the_core_and_cause() { + let (status, body) = to_http_response(&report(4), 1); + assert_eq!(status, axum::http::StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(body["core_id"], serde_json::json!(4)); + assert_eq!(body["cause"], serde_json::json!("rollback_failed")); + } +} diff --git a/nodedb/src/control/metrics/system/fields.rs b/nodedb/src/control/metrics/system/fields.rs index 32f854f0d..c39299cda 100644 --- a/nodedb/src/control/metrics/system/fields.rs +++ b/nodedb/src/control/metrics/system/fields.rs @@ -8,6 +8,7 @@ use std::sync::{Arc, RwLock}; use super::super::histogram::{AtomicHistogram, WAL_FSYNC_BUCKETS_US}; use super::super::purge::PurgeMetrics; +use super::core_fail_stop::CoreFailStops; use super::heartbeat::CoreHeartbeats; use crate::data::executor::core_loop::pressure::ThrottleMetrics; use crate::data::io::IoMetrics; @@ -219,6 +220,9 @@ pub struct SystemMetrics { /// that stops advancing is the only evidence a core has stopped /// completing iterations. pub core_heartbeats: CoreHeartbeats, + /// Cores that fail-stopped because their state is unknown. A core records + /// itself here once, as it stops. + pub core_fail_stops: CoreFailStops, } impl SystemMetrics { diff --git a/nodedb/src/control/metrics/system/mod.rs b/nodedb/src/control/metrics/system/mod.rs index e40533ba8..2fabc642f 100644 --- a/nodedb/src/control/metrics/system/mod.rs +++ b/nodedb/src/control/metrics/system/mod.rs @@ -1,9 +1,11 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod core_fail_stop; mod fields; mod heartbeat; mod record; mod render; +pub use core_fail_stop::{CoreFailStopReport, CoreFailStops}; pub use fields::SystemMetrics; pub use heartbeat::CoreHeartbeats; diff --git a/nodedb/src/control/metrics/system/render.rs b/nodedb/src/control/metrics/system/render.rs index 0c233fd73..8a413e13f 100644 --- a/nodedb/src/control/metrics/system/render.rs +++ b/nodedb/src/control/metrics/system/render.rs @@ -18,6 +18,7 @@ impl SystemMetrics { self.purge.write_prometheus(&mut out); self.io_metrics.write_prometheus(&mut out); self.spsc_throttle.write_prometheus(&mut out); + self.core_fail_stops.write_prometheus(&mut out); out } diff --git a/nodedb/src/control/server/http/routes/health.rs b/nodedb/src/control/server/http/routes/health.rs index e747def5b..31e406c3a 100644 --- a/nodedb/src/control/server/http/routes/health.rs +++ b/nodedb/src/control/server/http/routes/health.rs @@ -127,6 +127,23 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); } + // A fail-stopped core refuses every request routed to it: its state is + // unknown until restart. The other cores serve, so the node is degraded. + if let Some(stops) = state + .shared + .system_metrics + .as_ref() + .map(|metrics| &metrics.core_fail_stops) + && let Some(report) = stops.report() + { + let (status, mut body) = crate::control::metrics::system::core_fail_stop::to_http_response( + report, + stops.stopped_cores(), + ); + body["node_id"] = json!(state.shared.node_id); + return (status, axum::Json(body)); + } + // A core that stops completing event-loop iterations panics nothing, so // the per-core panic watchdog stays quiet and every other check above // still passes. Fail readiness and name the cores: work routed to a diff --git a/nodedb/src/control/server/native/session/request.rs b/nodedb/src/control/server/native/session/request.rs index a81b253d2..5c4f3eefe 100644 --- a/nodedb/src/control/server/native/session/request.rs +++ b/nodedb/src/control/server/native/session/request.rs @@ -57,10 +57,17 @@ impl NativeSession { // A stalled Data Plane core is a third after-boot degradation with // the same consequence: the gate reads Ok while work sent to that // core never completes. One atomic load, so it stays on this path. + // A fail-stopped core refuses its work outright, with the same + // consequence. let native_status = if self.state.metadata_apply_wedge.is_wedged() || self.state.sequencer_halt.is_halted() || self.state.sequencer_halt.apply_halt().is_halted() || self.state.core_stall.is_stalled() + || self + .state + .system_metrics + .as_ref() + .is_some_and(|metrics| metrics.core_fail_stops.is_stopped()) { crate::control::startup::health::NativeStatus::Failed } else { diff --git a/nodedb/src/data/executor/core_loop/fail_stop.rs b/nodedb/src/data/executor/core_loop/fail_stop.rs new file mode 100644 index 000000000..c097782ab --- /dev/null +++ b/nodedb/src/data/executor/core_loop/fail_stop.rs @@ -0,0 +1,196 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Fail-stop of a core whose state is unknown. +//! +//! A core's state is unknown when a rollback fails part way, or when a +//! committed record installed and the work after its install failed: a +//! memtable flush that drained rows, or the republish of an artifact that +//! must hold the record. Restart replay rebuilds the state from the WAL. +//! Until then the core must not serve the state it holds. +//! +//! The first cause wins. It logs one ERROR, files one recorder report, +//! records the node-wide marker that +//! `/healthz`, the native `STATUS` and the `nodedb_data_plane_core_fail_stopped` +//! gauge read, and latches the core. A latched core refuses every request it +//! dequeues with `RetryableRefusal`: it applied nothing, so the funnel +//! cancels the request's record and another replica or a restart serves it. +//! The latch never clears. + +use tracing::{error, warn}; + +use crate::bridge::dispatch::BridgeResponse; +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::control::metrics::CoreFailStopReport; + +use super::CoreLoop; + +/// Why a core fail-stopped. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::data::executor) enum FailStopCause { + /// A rollback of an undo log failed part way. + RollbackFailed, + /// A committed record installed, and a flush or artifact republish owed + /// after its install failed. Neither can be rolled back. + PostInstallFailed, +} + +impl FailStopCause { + fn label(self) -> &'static str { + match self { + Self::RollbackFailed => "rollback_failed", + Self::PostInstallFailed => "post_install_failed", + } + } +} + +/// The fail-stop latch of one core. Never clears. +#[derive(Debug, Default)] +pub(in crate::data::executor) struct CoreFailStop { + cause: Option<(FailStopCause, String)>, +} + +impl CoreFailStop { + pub(in crate::data::executor) fn is_stopped(&self) -> bool { + self.cause.is_some() + } + + /// The cause and detail of the stop, if the core stopped. + pub(in crate::data::executor) fn cause(&self) -> Option<(FailStopCause, &str)> { + self.cause + .as_ref() + .map(|(cause, detail)| (*cause, detail.as_str())) + } +} + +impl CoreLoop { + /// Whether this core fail-stopped. A stopped core also publishes no + /// checkpoint and runs no maintenance: an artifact of its state would + /// carry that state past the restart that must rebuild it. + pub(crate) fn is_fail_stopped(&self) -> bool { + self.fail_stop.is_stopped() + } + + /// Stop this core from serving: its state is unknown. Only the first + /// cause is kept and reported. + pub(in crate::data::executor) fn fail_stop_core(&mut self, cause: FailStopCause, detail: &str) { + if self.fail_stop.is_stopped() { + return; + } + error!( + core = self.core_id, + cause = cause.label(), + detail, + "data plane core fail-stopped: its state is unknown, and it refuses every \ + request until a restart rebuilds it from the WAL" + ); + crate::diag::data_plane_core_fail_stopped(self.core_id, cause.label(), detail); + if let Some(metrics) = &self.metrics { + metrics.core_fail_stops.record(CoreFailStopReport { + core_id: self.core_id, + cause: cause.label(), + detail: detail.to_string(), + }); + } + self.fail_stop.cause = Some((cause, detail.to_string())); + } + + /// Fail-stop the core when `response` reports a failed rollback. + pub(in crate::data::executor) fn fail_stop_on_rollback_failure(&mut self, response: &Response) { + if let Some(ErrorCode::RollbackFailed { + entry_index, + detail, + }) = response.error_code.as_deref() + { + let detail = format!("undo entry {entry_index}: {detail}"); + self.fail_stop_core(FailStopCause::RollbackFailed, &detail); + } + } + + /// Refuse every queued request of a fail-stopped core. Returns how many + /// were refused. + pub(in crate::data::executor) fn refuse_queued_while_stopped(&mut self) -> usize { + let Some((cause, _)) = self.fail_stop.cause() else { + return 0; + }; + let reason = format!( + "core {} is fail-stopped ({}): its state is unknown until restart", + self.core_id, + cause.label() + ); + let mut refused = 0; + while let Some(task) = self.task_queue.pop_front() { + let response = self.response_error( + &task, + ErrorCode::RetryableRefusal { + reason: reason.clone(), + }, + ); + if let Err(e) = self + .response_tx + .try_push(BridgeResponse { inner: response }) + { + warn!(core = self.core_id, error = %e, "failed to send a fail-stop refusal"); + } + refused += 1; + } + refused + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::data::executor::core_loop::tests::make_core_with_dir; + + #[test] + fn the_first_cause_wins() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + assert!(!core.fail_stop.is_stopped()); + + core.fail_stop_core(FailStopCause::PostInstallFailed, "first"); + core.fail_stop_core(FailStopCause::RollbackFailed, "second"); + + assert_eq!( + core.fail_stop.cause(), + Some((FailStopCause::PostInstallFailed, "first")) + ); + } + + #[test] + fn a_rollback_failure_response_stops_the_core_and_records_the_marker() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let metrics = std::sync::Arc::new(crate::control::metrics::SystemMetrics::new()); + core.set_metrics(std::sync::Arc::clone(&metrics)); + let task = crate::data::executor::core_loop::tests::make_default_task(); + let response = core.response_error( + &task, + ErrorCode::RollbackFailed { + entry_index: 2, + detail: "restore failed".into(), + }, + ); + + core.fail_stop_on_rollback_failure(&response); + + assert!(core.fail_stop.is_stopped()); + assert_eq!(metrics.core_fail_stops.stopped_cores(), 1); + assert_eq!( + metrics.core_fail_stops.report().map(|r| r.cause), + Some("rollback_failed") + ); + } + + #[test] + fn a_response_without_a_rollback_failure_leaves_the_core_serving() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let task = crate::data::executor::core_loop::tests::make_default_task(); + let response = core.response_error(&task, ErrorCode::ConflictRetry); + + core.fail_stop_on_rollback_failure(&response); + + assert!(!core.fail_stop.is_stopped()); + } +} diff --git a/nodedb/src/data/executor/core_loop/graph_partition.rs b/nodedb/src/data/executor/core_loop/graph_partition.rs index d041f339f..fefda4559 100644 --- a/nodedb/src/data/executor/core_loop/graph_partition.rs +++ b/nodedb/src/data/executor/core_loop/graph_partition.rs @@ -39,6 +39,33 @@ impl CoreLoop { self.csr.get_or_create(db, tenant, memory) } + /// Remove every edge of `node`: tombstone them in the edge store, then + /// drop them from the CSR. The store cascade is one transaction, so on + /// its error neither store changed and the two still agree. + /// + /// Returns the tombstoned edges, for a caller that keeps an undo log. + pub(in crate::data::executor) fn cascade_node_edges( + &mut self, + database_id: u64, + tid: u64, + node: &str, + ) -> crate::Result> { + let has_edges = self.csr_partition(database_id, tid).is_some_and(|p| { + p.node_id_raw(node) + .is_some_and(|id| p.out_degree_raw(id) + p.in_degree_raw(id) > 0) + }); + if !has_edges { + return Ok(Vec::new()); + } + let ord = self.hlc.next_ordinal(); + let removed = + self.edge_store + .delete_edges_for_node(database_id, TenantId::new(tid), node, ord)?; + self.csr_partition_mut(database_id, tid) + .remove_node_edges(node); + Ok(removed) + } + /// Mark `node_id` as deleted within the caller's `(database, tenant)`. /// Used by PointDelete cascade so subsequent `EdgePut` to the same node /// is rejected as dangling. diff --git a/nodedb/src/data/executor/core_loop/maintenance.rs b/nodedb/src/data/executor/core_loop/maintenance.rs index 6a83fe662..39060c111 100644 --- a/nodedb/src/data/executor/core_loop/maintenance.rs +++ b/nodedb/src/data/executor/core_loop/maintenance.rs @@ -245,26 +245,18 @@ impl CoreLoop { .collect(); let swept_nodes = work.len(); for (db, tid, node) in &work { - let edges = match self.csr.partition_mut(*db, *tid) { - Some(partition) => partition.remove_node_edges(node), - None => 0, - }; - if edges > 0 { - let ord = self.hlc.next_ordinal(); - if let Err(e) = self - .edge_store - .delete_edges_for_node(db.as_u64(), *tid, node, ord) - { - tracing::warn!( - core = self.core_id, - db = db.as_u64(), - tid = tid.as_u64(), - node = %node, - error = %e, - "sweep: failed to delete edges from store" - ); - } - removed += edges; + // On an error neither store changed, so the edges stay in both + // and the next sweep retries them. + match self.cascade_node_edges(db.as_u64(), tid.as_u64(), node) { + Ok(edges) => removed += edges.len(), + Err(e) => tracing::warn!( + core = self.core_id, + db = db.as_u64(), + tid = tid.as_u64(), + node = %node, + error = %e, + "sweep: failed to delete edges from store" + ), } } if removed > 0 { diff --git a/nodedb/src/data/executor/core_loop/mod.rs b/nodedb/src/data/executor/core_loop/mod.rs index 55e9e6211..7e7bba79c 100644 --- a/nodedb/src/data/executor/core_loop/mod.rs +++ b/nodedb/src/data/executor/core_loop/mod.rs @@ -9,6 +9,7 @@ mod decode_stored; pub(in crate::data::executor) mod deferred; mod doc_config_seed; pub(in crate::data::executor) mod event_emit; +pub(in crate::data::executor) mod fail_stop; pub(in crate::data::executor) mod filter_match; mod graph_partition; pub(in crate::data::executor) mod index_value_versions; diff --git a/nodedb/src/data/executor/core_loop/open.rs b/nodedb/src/data/executor/core_loop/open.rs index 47b013ebb..45b6366c3 100644 --- a/nodedb/src/data/executor/core_loop/open.rs +++ b/nodedb/src/data/executor/core_loop/open.rs @@ -212,6 +212,7 @@ impl CoreLoop { balanced_txn_entries: None, redo_apply: crate::data::executor::handlers::transaction::redo_apply::RedoApplyState::new(), + fail_stop: super::fail_stop::CoreFailStop::default(), }) } } diff --git a/nodedb/src/data/executor/core_loop/state.rs b/nodedb/src/data/executor/core_loop/state.rs index 59d92273c..f8713a2a1 100644 --- a/nodedb/src/data/executor/core_loop/state.rs +++ b/nodedb/src/data/executor/core_loop/state.rs @@ -433,10 +433,8 @@ pub struct CoreLoop { /// `wait_until_drained` on the same registry, so the unlink pass /// only runs once every in-flight scan has released. /// - /// `None` in test / no-cluster bringup paths: callers then skip - /// the gate and scan unconditionally (matching pre-quiesce - /// behavior). In the server bootstrap path `main.rs` wires the - /// shared registry via `set_quiesce` after `SharedState::open`. + /// `None` in test / no-cluster bringup: scans skip the gate. Boot wires + /// the shared registry via `set_quiesce` after `SharedState::open`. pub(in crate::data::executor) quiesce: Option>, @@ -582,4 +580,6 @@ pub struct CoreLoop { /// Core count and per-record scratch of the committed-redo apply. pub(in crate::data::executor) redo_apply: crate::data::executor::handlers::transaction::redo_apply::RedoApplyState, + /// Set once this core's state is unknown. It then refuses every request. + pub(in crate::data::executor) fail_stop: super::fail_stop::CoreFailStop, } diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index 7bf187d88..b89d7bf00 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -91,6 +91,8 @@ impl CoreLoop { task.state = TaskState::Running; let resp = self.execute(&task); task.state = TaskState::Completed; + // A failed rollback leaves this core's state unknown. + self.fail_stop_on_rollback_failure(&resp); resp }; @@ -137,6 +139,12 @@ impl CoreLoop { self.drain_requests(); let mut processed = 0; while !self.task_queue.is_empty() { + // A fail-stopped core serves nothing, including the rest of the + // queue behind the request that stopped it. + if self.fail_stop.is_stopped() { + processed += self.refuse_queued_while_stopped(); + break; + } let batched = self.poll_write_batch(); if batched > 0 { processed += batched; @@ -252,6 +260,44 @@ mod tests { ); } + #[test] + fn a_fail_stopped_core_refuses_every_queued_request() { + let (mut core, mut req_tx, mut resp_rx, _dir) = make_core(); + core.fail_stop_core( + crate::data::executor::core_loop::fail_stop::FailStopCause::RollbackFailed, + "undo entry 0: restore failed", + ); + for _ in 0..2 { + req_tx + .try_push(BridgeRequest { + inner: make_request(PhysicalPlan::Document(DocumentOp::PointGet { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), + document_id: "y".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + })), + }) + .expect("queue request"); + } + + assert_eq!(core.tick(), 2); + for _ in 0..2 { + let resp = resp_rx.try_pop().expect("refusal"); + assert_eq!(resp.inner.status, Status::Error); + assert!( + matches!( + resp.inner.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ), + "{:?}", + resp.inner.error_code + ); + } + } + #[test] fn watermark_in_response() { let (mut core, mut req_tx, mut resp_rx, _dir) = make_core(); diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs index 19f851396..65ae806d4 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs @@ -104,18 +104,9 @@ impl CoreLoop { warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: secondary index cascade failed"); } // Cascade: graph edges. - let edges_removed = self - .csr_partition_mut(database_id, tid) - .remove_node_edges(doc_id); - let cascade_ord = self.hlc.next_ordinal(); - if edges_removed > 0 - && let Err(e) = self.edge_store.delete_edges_for_node( - database_id, - nodedb_types::TenantId::new(tid), - doc_id, - cascade_ord, - ) - { + // On an error neither edge store changed: the edges stay in both, + // and the dangling-edge sweep retries them. + if let Err(e) = self.cascade_node_edges(database_id, tid, doc_id) { crate::diag::orphaned_index_entry_after_delete(&e, collection, "graph_edge"); warn!(core = self.core_id, %doc_id, error = %e, "bulk delete: edge cascade failed"); } diff --git a/nodedb/src/data/executor/handlers/graph_edge_write/delete.rs b/nodedb/src/data/executor/handlers/graph_edge_write/delete.rs index 5079b9ff2..355a122fe 100644 --- a/nodedb/src/data/executor/handlers/graph_edge_write/delete.rs +++ b/nodedb/src/data/executor/handlers/graph_edge_write/delete.rs @@ -9,6 +9,9 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; use crate::types::TenantId; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::handlers::transaction::undo::edge_write::EdgeTarget; + use super::shared::{EdgeDeleteParams, owns_logical_edge_stats}; impl CoreLoop { @@ -22,10 +25,9 @@ impl CoreLoop { /// Edge delete with optional transactional compensation. /// - /// The `UndoEntry::DeleteEdge` is recorded only when a live pre-image - /// existed *and* the tombstone was durably written — never speculatively - /// before the write. A phantom entry would otherwise re-insert an edge that - /// was never deleted when the surrounding transaction rolls back. + /// The `UndoEntry::EdgeWrite` is recorded once the tombstone is written, + /// never before: it names the tombstone version, and a rollback removes + /// exactly that version. /// /// The RLS write policy is decided against that same pre-image and BEFORE /// the tombstone: the row a policy governs is the edge that exists now, and @@ -39,7 +41,7 @@ impl CoreLoop { &mut self, task: &ExecutionTask, params: EdgeDeleteParams<'_>, - undo: Option<&mut Vec>, + undo: Option<&mut Vec>, ) -> Response { let EdgeDeleteParams { tid, @@ -52,9 +54,9 @@ impl CoreLoop { debug!(core = self.core_id, tid, %collection, %src_id, %label, %dst_id, "edge delete"); let database_id = task.request.database_id.as_u64(); - // The pre-image is always read now: the RLS write gate needs it for - // any non-admit-all policy, the undo log needs it for compensation, - // and the response needs it to report a truthful affected count. + // The pre-image is always read: the RLS write gate needs it for any + // non-admit-all policy, and the response needs it to report a + // truthful affected count. let old_properties = self .edge_store .get_edge( @@ -81,8 +83,18 @@ impl CoreLoop { let ord = self .active_graph_system_from .unwrap_or_else(|| self.hlc.next_ordinal()); + let target = EdgeTarget { + database_id, + tid, + collection, + src_id, + label, + dst_id, + }; + // The CSR state the undo puts back, read only when an undo is kept. + let csr_prior = undo.is_some().then(|| self.capture_edge_csr(&target)); use crate::engine::graph::edge_store::EdgeRef; - match self.edge_store.soft_delete_edge_with_stats( + match self.edge_store.soft_delete_edge_recorded( EdgeRef::new( task.request.database_id, TenantId::new(tid), @@ -94,18 +106,11 @@ impl CoreLoop { ord, owns_logical_edge_stats(task, src_id), ) { - Ok(_) => { - // Tombstone is durable; record the compensation for a rollback. - if let (Some(undo), Some(props)) = (undo, old_properties) { - undo.push( - crate::data::executor::handlers::transaction::undo::UndoEntry::DeleteEdge { - collection: collection.to_string(), - src_id: src_id.to_string(), - label: label.to_string(), - dst_id: dst_id.to_string(), - old_properties: props, - }, - ); + Ok(tombstone) => { + // The tombstone is written whether or not the edge was live, + // so the undo that removes it is recorded either way. + if let (Some(undo), Some(csr)) = (undo, csr_prior) { + undo.push(UndoEntry::EdgeWrite(Box::new(target.undo(tombstone, csr)))); } let partition = self.csr_partition_mut(database_id, tid); partition.remove_edge_in_collection(src_id, label, dst_id, collection); diff --git a/nodedb/src/data/executor/handlers/graph_edge_write/put.rs b/nodedb/src/data/executor/handlers/graph_edge_write/put.rs index 7226d9f23..6aaaf6702 100644 --- a/nodedb/src/data/executor/handlers/graph_edge_write/put.rs +++ b/nodedb/src/data/executor/handlers/graph_edge_write/put.rs @@ -9,6 +9,9 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; use crate::types::TenantId; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::executor::handlers::transaction::undo::edge_write::EdgeTarget; + use super::shared::{EdgePutParams, owns_logical_edge_stats}; impl CoreLoop { @@ -22,22 +25,20 @@ impl CoreLoop { /// Edge upsert with optional transactional compensation. /// - /// When `undo` is `Some`, the `UndoEntry::PutEdge` is recorded at the one - /// correct point: *after* the edge-store version is durably written and - /// *before* the fallible CSR mutation. Recording it earlier (before the - /// dangling-endpoint validation or the edge-store write) would leave a - /// compensation entry for an operation that never touched storage — on - /// rollback that entry would soft-delete or re-insert a version that never - /// existed, corrupting bitemporal edge history. + /// When `undo` is `Some`, the `UndoEntry::EdgeWrite` is recorded after + /// the edge-store version is written and before the fallible CSR + /// mutation. It names the version the put added, so a rollback removes + /// exactly that version. An entry recorded before the store write would + /// name a version that does not exist. /// - /// A put unconditionally writes a new edge-store version and CSR entry — - /// there is no "already identical" no-op path — so a successful put - /// always reports exactly one edge affected. + /// A put writes a new edge-store version and makes the CSR edge live with + /// the weight in `properties`, so a successful put always reports exactly + /// one edge affected. pub(in crate::data::executor) fn execute_edge_put_with_undo( &mut self, task: &ExecutionTask, params: EdgePutParams<'_>, - undo: Option<&mut Vec>, + undo: Option<&mut Vec>, ) -> Response { let EdgePutParams { tid, @@ -69,23 +70,6 @@ impl CoreLoop { ); } - // Capture the pre-image only when a compensation record is requested. - let old_properties = if undo.is_some() { - self.edge_store - .get_edge( - database_id, - TenantId::new(tid), - collection, - src_id, - label, - dst_id, - ) - .ok() - .flatten() - } else { - None - }; - let ord = self .active_graph_system_from .unwrap_or_else(|| self.hlc.next_ordinal()); @@ -97,8 +81,18 @@ impl CoreLoop { Some(ms) => ms, None => nodedb_types::ordinal_to_ms(ord), }; + let target = EdgeTarget { + database_id, + tid, + collection, + src_id, + label, + dst_id, + }; + // The CSR state the undo puts back, read only when an undo is kept. + let csr_prior = undo.is_some().then(|| self.capture_edge_csr(&target)); use crate::engine::graph::edge_store::EdgeRef; - match self.edge_store.put_edge_versioned_with_stats( + match self.edge_store.put_edge_version_recorded( EdgeRef::new( task.request.database_id, TenantId::new(tid), @@ -114,30 +108,18 @@ impl CoreLoop { i64::MAX, owns_logical_edge_stats(task, src_id), ) { - Ok(()) => { + Ok(version) => { // Edge-store version is now durable; the compensation entry is // valid from here on even if the CSR mutation below fails. - if let Some(undo) = undo { - undo.push( - crate::data::executor::handlers::transaction::undo::UndoEntry::PutEdge { - collection: collection.to_string(), - src_id: src_id.to_string(), - label: label.to_string(), - dst_id: dst_id.to_string(), - old_properties, - }, - ); + if let (Some(undo), Some(csr)) = (undo, csr_prior) { + undo.push(UndoEntry::EdgeWrite(Box::new(target.undo(version, csr)))); } let weight = crate::engine::graph::csr::extract_weight_from_properties(properties); let partition = self.csr_partition_mut(database_id, tid); - let csr_result = if weight != 1.0 { - partition - .add_edge_weighted_in_collection(src_id, label, dst_id, collection, weight) - } else { - partition.add_edge_in_collection(src_id, label, dst_id, collection) - }; + let csr_result = + partition.put_edge_in_collection(src_id, label, dst_id, collection, weight); match csr_result { - Ok(()) => { + Ok(_) => { // Populate the per-node surrogates so future bitmap-gated // traversals can check membership without a separate lookup. partition.set_node_surrogate(src_id, src_surrogate); diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index bfb050d9b..03222dd32 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -81,11 +81,12 @@ pub(in crate::data::executor) struct PointDeleteOutcome { /// does not reverse in-memory spatial writes). pub spatial_deletes: Vec<(SpatialIndexKey, u64, nodedb_types::BoundingBox, String)>, /// Graph edges the unconditional graph-edge cascade removed from BOTH the - /// in-memory CSR partition AND the persistent edge store — each captured as - /// `(collection, src, label, dst, old_properties)`. Populated regardless of - /// caller (the cascade is unconditional), so a transactional caller pushes - /// one `UndoEntry::DeleteEdge` per entry and a rolled-back delete restores - /// every cascaded edge into both stores. Autocommit callers ignore it. + /// in-memory CSR partition AND the persistent edge store, each with its + /// prior properties and the tombstone version the cascade added. + /// Populated regardless of caller (the cascade is unconditional), so a + /// transactional caller pushes one `UndoEntry::EdgeWrite` per entry and a + /// rolled-back delete restores every cascaded edge into both stores. + /// Autocommit callers ignore it. pub edge_deletes: Vec, /// The node id this delete NEWLY marked deleted in the in-memory /// `deleted_nodes` edge referential-integrity tracker, if any. `Some(id)` @@ -343,29 +344,20 @@ impl CoreLoop { // Cascade 3: Remove graph edges where this document is src or dst. // Captured unconditionally (the cascade runs for both autocommit and // transactional callers) so a transactional caller can restore every - // removed edge on rollback via `UndoEntry::DeleteEdge`, which re-inserts - // into BOTH the CSR partition and the persistent edge store — matching - // the two stores this cascade removes from. - let mut edge_deletes: Vec = Vec::new(); - let edges_removed = self - .csr_partition_mut(database_id, tid) - .remove_node_edges(document_id); - if edges_removed > 0 { - // Also tombstone in persistent edge store, capturing each removed - // edge (with its pre-delete properties) for rollback restore. - let cascade_ord = self.hlc.next_ordinal(); - match self.edge_store.delete_edges_for_node( - database_id, - nodedb_types::TenantId::new(tid), - document_id, - cascade_ord, - ) { - Ok(removed) => edge_deletes = removed, - Err(e) => { - warn!(core = self.core_id, %document_id, error = %e, "edge cascade failed"); - } + // removed edge on rollback via `UndoEntry::EdgeWrite`, which restores + // BOTH the CSR partition and the persistent edge store — matching the + // two stores this cascade removes from. + // The store cascade runs first and is one transaction: its error + // refuses the delete with neither edge store nor CSR changed. + let edge_deletes = match self.cascade_node_edges(database_id, tid, document_id) { + Ok(removed) => removed, + Err(e) => { + warn!(core = self.core_id, %document_id, error = %e, "edge cascade failed; rejecting the delete"); + return Err(e); } - tracing::trace!(core = self.core_id, %document_id, edges_removed, "EDGE_CASCADE_DELETE"); + }; + if !edge_deletes.is_empty() { + tracing::trace!(core = self.core_id, %document_id, edges_removed = edge_deletes.len(), "EDGE_CASCADE_DELETE"); } // Cascade 4: Remove from spatial R-tree indexes + reverse map, and diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs index 14d2dc94e..c7d5baeeb 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch.rs @@ -281,12 +281,7 @@ impl CoreLoop { // unwinding past a half-restored transaction. let undo_len = undo_log.len(); let rollback_error_code = match catch_unwind(AssertUnwindSafe(|| { - self.rollback_undo_log_at( - task.request.database_id.as_u64(), - tid, - task.request.vshard_id, - undo_log, - ) + self.rollback_undo_log(task.request.database_id.as_u64(), tid, undo_log) })) { Ok(Ok(())) => error_code, Ok(Err((entry_index, detail))) => { diff --git a/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs b/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs index 4a5c06ed8..5bf629900 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs @@ -79,12 +79,7 @@ impl CoreLoop { ) -> Response { let undo_len = undo_log.len(); let rollback = catch_unwind(AssertUnwindSafe(|| { - self.rollback_undo_log_at( - task.request.database_id.as_u64(), - tid, - task.request.vshard_id, - undo_log, - ) + self.rollback_undo_log(task.request.database_id.as_u64(), tid, undo_log) })); let failure = match rollback { Ok(Ok(())) => None, diff --git a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs index aa9750140..e80c99476 100644 --- a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs +++ b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs @@ -61,8 +61,7 @@ fn entry_index_tuples(entry: &UndoEntry) -> Option<(String, Vec<(String, String) | UndoEntry::DeleteVector { .. } | UndoEntry::SpatialInsert { .. } | UndoEntry::SpatialDelete { .. } - | UndoEntry::PutEdge { .. } - | UndoEntry::DeleteEdge { .. } + | UndoEntry::EdgeWrite(_) | UndoEntry::KvPut { .. } | UndoEntry::KvDelete { .. } | UndoEntry::KvBatchPut { .. } diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs index 918fe71b1..e75f8356b 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs @@ -290,6 +290,52 @@ mod tests { ); } + /// A record applied below a published checkpoint that cannot be published + /// again leaves the checkpoint's claim false for it. The work cannot be + /// rolled back, so the core fail-stops and the record's events stay unsent. + #[test] + fn a_failed_republish_fail_stops_the_core() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let (mut producers, mut consumers) = + crate::event::bus::create_event_bus_with_capacity(1, 64); + core.set_event_producer(producers.pop().expect("producer")); + core.floors.kv_published_lsn = Lsn::new(100); + // A file where the checkpoint directory belongs makes the publish fail. + let ckpt_dir = core + .data_dir + .join("kv-ckpt") + .join(format!("core-{}", core.core_id)); + std::fs::create_dir_all(ckpt_dir.parent().expect("parent")).expect("kv-ckpt dir"); + std::fs::write(&ckpt_dir, b"not a directory").expect("block the checkpoint dir"); + + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(50)); + let redo = RedoRecord { + version: 1, + ops: vec![kv_put("cache", b"b", b"2", 2)], + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let response = core.execute_apply_transaction_redo( + &task, + TID, + CommittedRedo { + redo: &redo, + collections: &["cache".to_string()], + sum_targets: &[], + }, + ); + + assert_eq!(response.status, Status::Error); + assert!(core.fail_stop.is_stopped(), "the core fail-stops"); + assert!( + consumers[0].try_recv().is_none(), + "no event leaves for a record whose post-install work failed" + ); + } + #[test] fn a_columnar_record_applied_below_a_published_checkpoint_is_published_again() { use nodedb_types::Value; diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs index 5138662e4..76c84b8c0 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs @@ -26,7 +26,8 @@ //! the open [`RedoApplyScope`]; //! 3. a collection-floor write version for every collection written, and the //! index-value versions of every document row; -//! 4. the record's events (see `events`); +//! 4. the record's events, sent once the post-install work succeeded (see +//! `events`); //! 5. the fold target rows in `Response::write_set`, so the funnel journals //! them. @@ -36,6 +37,7 @@ use nodedb_wal::record::{RecordType, WalRecordArgs}; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::fail_stop::FailStopCause; use crate::data::executor::enforcement::write_hook::target_write_set; use crate::data::executor::task::ExecutionTask; use crate::types::TenantId; @@ -111,7 +113,6 @@ impl CoreLoop { sub_records: redo.ops.len(), database_id, tid, - vshard_id: task.request.vshard_id, }; if let Err(refusal) = self.validate_redo_pass(&target, committed.sum_targets) { return self.response_error(task, refusal.into_code()); @@ -120,17 +121,28 @@ impl CoreLoop { Ok(scope) => scope, Err(refusal) => return self.response_error(task, refusal.into_code()), }; - if let Err(error) = self.settle_redo_install(task, &mut scope) { - return self.response_error(task, error); - } - - // A record applied below a published engine watermark is published - // again, so restart replay does not skip it. - if let Err(error) = + // Settle, then publish again every artifact whose watermark covers + // the record, so restart replay does not skip it. Neither step can be + // rolled back once it started, so a failure leaves live state restart + // replay does not rebuild: the core fail-stops. The funnel keeps the + // record for restart replay. + let settled = self.settle_redo_install(task, &mut scope).and_then(|()| { self.cover_applied_record(lsn, &WrittenEngines::of(&redo), &scope.arrays_written) - { + }); + if let Err(error) = settled { + self.fail_stop_core( + FailStopCause::PostInstallFailed, + &format!( + "committed redo record at lsn {} installed, then failed: {error:?}", + lsn.as_u64() + ), + ); return self.response_error(task, error); } + // Events leave only once the record is settled and covered. + for event in std::mem::take(&mut scope.pending_events) { + self.send_write_event(event); + } let tenant = TenantId::new(tid); for collection in committed.collections { @@ -550,6 +562,115 @@ mod tests { assert!(!core.vector_collections.contains_key(&key)); } + /// The edge an install wrote leaves no version behind at any system time, + /// and the nodes it created leave the CSR: the core reads as restart + /// replay of the cancelled record would. + #[test] + fn an_install_failure_leaves_no_trace_of_the_edge_it_wrote() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let edge_put = RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(&crate::wal::EdgePutRedo { + collection: "knows".into(), + src_id: "alice".into(), + label: "KNOWS".into(), + dst_id: "bob".into(), + properties: Vec::new(), + src_surrogate: 31, + dst_surrogate: 32, + system_from: Some(500), + }) + .expect("encode edge put"), + }; + let redo = redo_bytes(vec![edge_put, mismatched_ingest()]); + + let response = core.execute_apply_transaction_redo( + &task_at(Some(95)), + TID, + CommittedRedo { + redo: &redo, + collections: &[], + sum_targets: &[], + }, + ); + + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ), + "{:?}", + response.error_code + ); + let edge = crate::engine::graph::edge_store::EdgeRef::new( + crate::types::DatabaseId::DEFAULT, + TenantId::new(TID), + "knows", + "alice", + "KNOWS", + "bob", + ); + for as_of in [500, 501, i64::MAX] { + assert_eq!( + core.edge_store + .ceiling_resolve_edge(edge, as_of, None) + .expect("resolve edge"), + None, + "no version of the rolled-back edge is visible at system time {as_of}" + ); + } + assert!( + core.edge_store + .scan_all_node_surrogates() + .expect("scan bindings") + .is_empty() + ); + assert!( + core.csr_partition(0, TID) + .is_none_or(|p| !p.contains_node("alice") && !p.contains_node("bob")), + "the nodes the edge created are gone from the CSR" + ); + } + + /// A label write on a node the CSR did not hold creates the node and + /// interns the label. A rolled-back install withdraws both. + #[test] + fn an_install_failure_withdraws_the_node_a_label_write_created() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let label_set = RedoSubRecord { + record_type: RecordType::GraphNodeLabelSet as u32, + payload: zerompk::to_msgpack_vec(&("carol".to_string(), vec!["Person".to_string()])) + .expect("encode label set"), + }; + let redo = redo_bytes(vec![label_set, mismatched_ingest()]); + + let response = core.execute_apply_transaction_redo( + &task_at(Some(96)), + TID, + CommittedRedo { + redo: &redo, + collections: &[], + sum_targets: &[], + }, + ); + + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ), + "{:?}", + response.error_code + ); + assert!( + core.csr_partition(0, TID) + .is_none_or(|p| !p.contains_node("carol") && !p.has_node_label_name("Person")), + "the node and the label name the write created are gone" + ); + } + /// A non-document sub-record that cannot be applied fails the online /// apply response instead of being logged and skipped. #[test] diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs index 1e3bd4570..db7724fb0 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs @@ -17,7 +17,6 @@ use nodedb_wal::WalRecord; use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::CoreLoop; -use crate::types::VShardId; use super::state::{RedoApplyPass, RedoApplyScope}; @@ -28,7 +27,7 @@ pub(super) enum PassRefusal { /// The install pass failed. Every write was rolled back. RolledBack(ErrorCode), /// The install pass failed and its rollback failed too. The core's state - /// is unknown. + /// is unknown, and the `RollbackFailed` code fail-stops the core. RollbackFailed(ErrorCode), } @@ -54,7 +53,6 @@ pub(super) struct RedoTarget<'a> { pub sub_records: usize, pub database_id: u64, pub tid: u64, - pub vshard_id: VShardId, } impl CoreLoop { @@ -99,7 +97,7 @@ impl CoreLoop { return Ok(scope); }; let undo = std::mem::take(&mut scope.undo); - match self.rollback_undo_log_at(target.database_id, target.tid, target.vshard_id, undo) { + match self.rollback_undo_log(target.database_id, target.tid, undo) { Ok(()) => Err(PassRefusal::RolledBack(cause)), Err((entry_index, detail)) => { Err(PassRefusal::RollbackFailed(ErrorCode::RollbackFailed { @@ -133,8 +131,11 @@ impl CoreLoop { .map_err(ErrorCode::from); match self.redo_apply.scope.take() { Some(scope) => Ok((applied, scope)), - None => Err(PassRefusal::RollbackFailed(ErrorCode::Internal { - detail: "committed transaction redo lost its apply scope".into(), + // The scope held the undo log. Without it nothing can be rolled + // back, so the core's state is unknown. + None => Err(PassRefusal::RollbackFailed(ErrorCode::RollbackFailed { + entry_index: 0, + detail: "committed transaction redo lost its apply scope and its undo log".into(), })), } } diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs index c1e97c29a..98754086d 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs @@ -4,12 +4,13 @@ //! //! A write the undo cannot reverse waits here: a memtable flush drains rows //! the undo would restore in memory, a vector seal moves inserted nodes out of -//! the growing segment, a truncate that removes files cannot be renamed back, -//! and an event the Event Plane consumed cannot be withdrawn. Once the install -//! succeeded, this runs them in one step. +//! the growing segment, and a truncate that removes files cannot be renamed +//! back. Once the install succeeded, this runs them in one step. The record's +//! events wait until this step and the cover step both succeeded. //! -//! A failure here comes after every sub-record landed, so it is not rolled -//! back. It fails the response as an ambiguous error, and the funnel keeps +//! A failure here comes after every sub-record landed. The undo cannot +//! reverse a partial flush, and a flush failure is an I/O failure no validate +//! pass can predict. So the apply fail-stops the core, and the funnel keeps //! the record for restart replay. use crate::bridge::envelope::ErrorCode; @@ -53,9 +54,6 @@ impl CoreLoop { task: &ExecutionTask, scope: &mut RedoApplyScope, ) -> Result<(), ErrorCode> { - for event in std::mem::take(&mut scope.pending_events) { - self.send_write_event(event); - } self.finalize_timeseries_truncates(&scope.undo); self.finalize_vector_truncates(&mut scope.undo); self.seal_full_vector_collections(); diff --git a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs index 5425640a9..a509d51ce 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs @@ -6,7 +6,6 @@ //! All methods return `Err((entry_index, detail))` on fatal failure so the //! caller can escalate to a typed `RollbackFailed` response. -use nodedb_types::Surrogate; use tracing::error; use crate::data::executor::core_loop::CoreLoop; @@ -110,189 +109,6 @@ impl CoreLoop { } } - // ── Graph ──────────────────────────────────────────────────────────────── - - #[cfg(test)] - pub(super) fn apply_undo_edge( - &mut self, - did: u64, - tid: u64, - entry_index: usize, - entry: UndoEntry, - ) -> Result<(), (usize, String)> { - self.apply_undo_edge_with_stats(did, tid, entry_index, entry, true) - } - - pub(super) fn apply_undo_edge_with_stats( - &mut self, - did: u64, - tid: u64, - entry_index: usize, - entry: UndoEntry, - account_stats: bool, - ) -> Result<(), (usize, String)> { - use crate::engine::graph::edge_store::EdgeRef; - let database = nodedb_types::DatabaseId::new(did); - match entry { - UndoEntry::PutEdge { - collection, - src_id, - label, - dst_id, - old_properties, - } => { - let tenant = nodedb_types::TenantId::new(tid); - let ord = self.hlc.next_ordinal(); - let edge_ref = - EdgeRef::new(database, tenant, &collection, &src_id, &label, &dst_id); - if let Some(old_props) = old_properties { - let valid_from_ms = nodedb_types::ordinal_to_ms(ord); - self.edge_store - .put_edge_versioned_with_stats( - edge_ref, - &old_props, - ord, - valid_from_ms, - i64::MAX, - account_stats, - ) - .map_err(|e| { - let detail = format!( - "edge restore {collection} {src_id}-[{label}]->{dst_id}: {e}" - ); - error!( - core = self.core_id, entry_index, - error = %detail, - "transaction undo: edge restore failed; shard state unknown" - ); - (entry_index, detail) - })?; - let weight = - crate::engine::graph::csr::extract_weight_from_properties(&old_props); - let partition = self.csr_partition_mut(did, tid); - partition.remove_edge_in_collection(&src_id, &label, &dst_id, &collection); - let csr_res = if weight != 1.0 { - partition.add_edge_weighted_in_collection( - &src_id, - &label, - &dst_id, - &collection, - weight, - ) - } else { - partition.add_edge_in_collection(&src_id, &label, &dst_id, &collection) - }; - csr_res.map_err(|e| { - let detail = - format!("CSR restore {collection} {src_id}-[{label}]->{dst_id}: {e}"); - error!( - core = self.core_id, entry_index, - error = %detail, - "transaction undo: CSR restore failed after edge_store restore; \ - shard state unknown" - ); - (entry_index, detail) - })?; - } else { - self.edge_store - .soft_delete_edge_with_stats(edge_ref, ord, account_stats) - .map_err(|e| { - let detail = format!( - "edge tombstone {collection} {src_id}-[{label}]->{dst_id}: {e}" - ); - error!( - core = self.core_id, entry_index, - error = %detail, - "transaction undo: edge tombstone failed; shard state unknown" - ); - (entry_index, detail) - })?; - self.csr_partition_mut(did, tid).remove_edge_in_collection( - &src_id, - &label, - &dst_id, - &collection, - ); - } - Ok(()) - } - UndoEntry::DeleteEdge { - collection, - src_id, - label, - dst_id, - old_properties, - } => { - let tenant = nodedb_types::TenantId::new(tid); - let ord = self.hlc.next_ordinal(); - let valid_from_ms = nodedb_types::ordinal_to_ms(ord); - // The cascade that produced this entry dropped the endpoints' - // durable identity bindings. The in-memory CSR still holds - // them, so restoring the edge restores the binding with it — - // otherwise a rolled-back delete would leave the graph intact - // but invisible to every cross-engine read after a restart. - let (src_surrogate, dst_surrogate) = self - .csr_partition(did, tid) - .map(|p| { - ( - p.node_surrogate(&src_id).unwrap_or(Surrogate::ZERO), - p.node_surrogate(&dst_id).unwrap_or(Surrogate::ZERO), - ) - }) - .unwrap_or((Surrogate::ZERO, Surrogate::ZERO)); - self.edge_store - .put_edge_versioned_with_stats( - EdgeRef::new(database, tenant, &collection, &src_id, &label, &dst_id) - .with_surrogates(src_surrogate, dst_surrogate), - &old_properties, - ord, - valid_from_ms, - i64::MAX, - account_stats, - ) - .map_err(|e| { - let detail = format!( - "edge re-insert {collection} {src_id}-[{label}]->{dst_id}: {e}" - ); - error!( - core = self.core_id, entry_index, - error = %detail, - "transaction undo: edge re-insert failed; shard state unknown" - ); - (entry_index, detail) - })?; - let weight = - crate::engine::graph::csr::extract_weight_from_properties(&old_properties); - let partition = self.csr_partition_mut(did, tid); - let csr_res = if weight != 1.0 { - partition.add_edge_weighted_in_collection( - &src_id, - &label, - &dst_id, - &collection, - weight, - ) - } else { - partition.add_edge_in_collection(&src_id, &label, &dst_id, &collection) - }; - csr_res.map_err(|e| { - let detail = format!("CSR re-insert {src_id}-[{label}]->{dst_id}: {e}"); - error!( - core = self.core_id, entry_index, - error = %detail, - "transaction undo: CSR re-insert failed after edge_store restore; \ - shard state unknown" - ); - (entry_index, detail) - }) - } - _ => Err(( - entry_index, - "apply_undo_edge called with non-edge entry".to_string(), - )), - } - } - // ── Columnar ───────────────────────────────────────────────────────────── pub(super) fn apply_undo_columnar( @@ -385,6 +201,7 @@ impl CoreLoop { memtable_config_before, memtable_memory_bytes_before, last_value_cache_before, + series_catalog_before, max_ingested_lsn_before, last_ts_ingest_before, reservation_bytes_before, @@ -447,6 +264,15 @@ impl CoreLoop { self.ts_last_value_caches.remove(&collection_key); } } + match series_catalog_before { + Some(catalog) => { + self.ts_series_catalogs + .insert(collection_key.clone(), catalog); + } + None => { + self.ts_series_catalogs.remove(&collection_key); + } + } match max_ingested_lsn_before { Some(lsn) => { self.ts_max_ingested_lsn.insert(collection_key, lsn); @@ -525,6 +351,12 @@ mod tests { cache.update(1, 10, 1.0); core.ts_last_value_caches.insert(key.clone(), cache.clone()); core.ts_max_ingested_lsn.insert(key.clone(), 7); + let mut catalog = nodedb_types::timeseries::SeriesCatalog::new(); + catalog.resolve(&nodedb_types::timeseries::SeriesKey::new( + "cpu", + vec![("host".into(), "old-host".into())], + )); + core.ts_series_catalogs.insert(key.clone(), catalog.clone()); let prior_timer = std::time::Instant::now(); core.last_ts_ingest = Some(prior_timer); @@ -534,6 +366,7 @@ mod tests { memtable_config_before: Some(config), memtable_memory_bytes_before: Some(memory_bytes), last_value_cache_before: Some(cache), + series_catalog_before: Some(catalog.clone()), max_ingested_lsn_before: Some(7), last_ts_ingest_before: Some(prior_timer), reservation_bytes_before: None, @@ -556,10 +389,22 @@ mod tests { .expect("cache") .update(1, 20, 2.0); core.ts_max_ingested_lsn.insert(key.clone(), 99); + core.ts_series_catalogs + .get_mut(&key) + .expect("catalog") + .resolve(&nodedb_types::timeseries::SeriesKey::new( + "cpu", + vec![("host".into(), "new-host".into())], + )); core.last_ts_ingest = Some(std::time::Instant::now()); core.apply_undo_timeseries(0, UndoEntry::TimeseriesIngest(token)) .expect("undo"); + assert_eq!( + core.ts_series_catalogs.get(&key), + Some(&catalog), + "the series the ingest registered are forgotten" + ); let restored = core .columnar_memtables .get(&key) @@ -597,12 +442,15 @@ mod tests { memtable_config_before: None, memtable_memory_bytes_before: None, last_value_cache_before: None, + series_catalog_before: None, max_ingested_lsn_before: None, last_ts_ingest_before: None, reservation_bytes_before: None, }; core.columnar_memtables .insert(key.clone(), timeseries_memtable()); + core.ts_series_catalogs + .insert(key.clone(), nodedb_types::timeseries::SeriesCatalog::new()); core.ts_last_value_caches .insert(key.clone(), LastValueCache::new()); core.ts_max_ingested_lsn.insert(key.clone(), 1); @@ -613,6 +461,10 @@ mod tests { assert!(!core.columnar_memtables.contains_key(&key)); assert!(!core.ts_last_value_caches.contains_key(&key)); assert!(!core.ts_max_ingested_lsn.contains_key(&key)); + assert!( + !core.ts_series_catalogs.contains_key(&key), + "the catalog the ingest created is gone" + ); assert!(core.last_ts_ingest.is_none()); } @@ -668,6 +520,14 @@ mod tests { )), "the last-value cache must follow the same initial pre-image" ); + assert!( + !core.ts_series_catalogs.contains_key(&( + crate::types::DatabaseId::DEFAULT, + TenantId::new(TID), + "metrics".to_string(), + )), + "the series catalog the first ingest created is rolled back" + ); } #[test] @@ -1004,167 +864,4 @@ mod tests { "vector_doc_map entry must be restored so a later delete can find the vector again" ); } - - // ── Graph edge-cascade undo ────────────────────────────────────────────── - - /// A rolled-back transactional document DELETE must restore every edge the - /// unconditional graph-edge cascade removed — into BOTH the persistent edge - /// store (`get_edge`) AND the in-memory CSR partition (`neighbors`), with the - /// original edge properties intact. This exercises the full capture→restore - /// path: `delete_edges_for_node` returns the removed edges, and - /// `apply_undo_edge` re-inserts each via a `DeleteEdge` undo entry. - #[test] - fn edge_cascade_delete_rollback_restores_csr_and_edge_store() { - use crate::engine::graph::csr::Direction; - use crate::engine::graph::edge_store::EdgeRef; - - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let tenant = TenantId::new(TID); - - // Seed alice-[KNOWS]->bob in BOTH stores, as a forward EdgePut would. - let seed_ord = core.hlc.next_ordinal(); - core.edge_store - .put_edge_versioned( - EdgeRef::new( - nodedb_types::DatabaseId::new(DB), - tenant, - "c", - "alice", - "KNOWS", - "bob", - ), - b"p1", - seed_ord, - nodedb_types::ordinal_to_ms(seed_ord), - i64::MAX, - ) - .unwrap(); - core.csr_partition_mut(DB, TID) - .add_edge("alice", "KNOWS", "bob") - .unwrap(); - - // Sanity: edge present in both stores. - assert_eq!( - core.edge_store - .get_edge(DB, tenant, "c", "alice", "KNOWS", "bob") - .unwrap(), - Some(b"p1".to_vec()) - ); - assert_eq!( - core.csr_partition_mut(DB, TID) - .neighbors("alice", None, Direction::Out), - vec![("KNOWS".to_string(), "bob".to_string())] - ); - - // Forward document-delete cascade (Cascade 3): remove from CSR + edge store, - // capturing the removed edges for rollback. - core.csr_partition_mut(DB, TID).remove_node_edges("alice"); - let cascade_ord = core.hlc.next_ordinal(); - let removed = core - .edge_store - .delete_edges_for_node(DB, tenant, "alice", cascade_ord) - .unwrap(); - assert_eq!(removed.len(), 1); - assert_eq!( - removed[0], - ( - "c".to_string(), - "alice".to_string(), - "KNOWS".to_string(), - "bob".to_string(), - b"p1".to_vec() - ) - ); - - // Both stores now show the edge gone. - assert!( - core.edge_store - .get_edge(DB, tenant, "c", "alice", "KNOWS", "bob") - .unwrap() - .is_none() - ); - assert!( - core.csr_partition_mut(DB, TID) - .neighbors("alice", None, Direction::Out) - .is_empty() - ); - - // Rollback: push one DeleteEdge undo per captured edge and apply it. - for (idx, (collection, src_id, label, dst_id, old_properties)) in - removed.into_iter().enumerate() - { - let undo = UndoEntry::DeleteEdge { - collection, - src_id, - label, - dst_id, - old_properties, - }; - core.apply_undo_edge(DB, TID, idx, undo).unwrap(); - } - - // Both stores fully restored, properties intact. - assert_eq!( - core.edge_store - .get_edge(DB, tenant, "c", "alice", "KNOWS", "bob") - .unwrap(), - Some(b"p1".to_vec()), - "edge store must be restored with original properties" - ); - assert_eq!( - core.csr_partition_mut(DB, TID) - .neighbors("alice", None, Direction::Out), - vec![("KNOWS".to_string(), "bob".to_string())], - "CSR adjacency must be restored" - ); - } - - #[test] - fn graph_edge_update_undo_restores_csr_weight() { - use crate::engine::graph::edge_store::EdgeRef; - - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let tenant = TenantId::new(TID); - let old_properties = nodedb_types::json_to_msgpack(&serde_json::json!({ "weight": 2.5 })) - .expect("encode old edge properties"); - let new_properties = nodedb_types::json_to_msgpack(&serde_json::json!({ "weight": 9.0 })) - .expect("encode new edge properties"); - let edge = EdgeRef::new( - crate::types::DatabaseId::new(DB), - tenant, - "c", - "alice", - "KNOWS", - "bob", - ); - core.edge_store - .put_edge_versioned(edge, &new_properties, 10, 10, i64::MAX) - .expect("seed updated edge"); - core.csr_partition_mut(DB, TID) - .add_edge_weighted_in_collection("alice", "KNOWS", "bob", "c", 9.0) - .expect("seed updated CSR edge"); - - core.apply_undo_edge( - DB, - TID, - 0, - UndoEntry::PutEdge { - collection: "c".into(), - src_id: "alice".into(), - label: "KNOWS".into(), - dst_id: "bob".into(), - old_properties: Some(old_properties), - }, - ) - .expect("undo edge update"); - - assert_eq!( - core.csr_partition_mut(DB, TID) - .edge_weight("alice", "KNOWS", "bob"), - Some(2.5), - "rollback must restore the committed CSR traversal weight" - ); - } } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs b/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs index 71949b057..78316ff42 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs @@ -19,6 +19,7 @@ use crate::data::executor::handlers::point::apply_delete::PointDeleteOutcome; use crate::data::executor::handlers::point::apply_put::PointPutOutcome; use super::UndoEntry; +use super::edge_write::{EdgeCsrPrior, EdgeWriteUndo}; /// The document row one write touched. pub(in crate::data::executor::handlers) struct DocumentRow<'a> { @@ -133,14 +134,25 @@ pub(in crate::data::executor::handlers) fn push_delete_undo( node_id, }); } - for (collection, src_id, label, dst_id, old_properties) in outcome.edge_deletes { - undo_log.push(UndoEntry::DeleteEdge { - collection, - src_id, - label, - dst_id, - old_properties, - }); + // The cascade dropped the deleted node's identity binding with its + // edges, so the undo binds the endpoints again. + for cascaded in outcome.edge_deletes { + let weight = + crate::engine::graph::csr::extract_weight_from_properties(&cascaded.old_properties); + undo_log.push(UndoEntry::EdgeWrite(Box::new(EdgeWriteUndo { + database_id: row.database_id, + tid: row.tid, + collection: cascaded.collection, + src_id: cascaded.src, + label: cascaded.label, + dst_id: cascaded.dst, + version: cascaded.tombstone, + csr: EdgeCsrPrior { + weight: Some(weight), + ..EdgeCsrPrior::default() + }, + rebind_endpoints: true, + }))); } } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/edge_write.rs b/nodedb/src/data/executor/handlers/transaction/undo/edge_write.rs new file mode 100644 index 000000000..482bd7cf8 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/edge_write.rs @@ -0,0 +1,382 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Undo of one graph edge write: a put, a delete, or one edge of a node-delete +//! cascade. +//! +//! The write adds one version to the bitemporal edge store. The undo removes +//! that version, so no read at any system time sees the rolled-back write, as +//! if it never ran. It also puts the CSR back: the edge's presence and +//! weight, each endpoint surrogate the write rebound, and each node the write +//! created. + +use tracing::error; + +use crate::data::executor::core_loop::CoreLoop; +use crate::engine::graph::edge_store::{EdgeRef, EdgeVersionWrite}; +use crate::types::{DatabaseId, TenantId}; + +/// The pre-image of one edge write. +pub(in crate::data::executor) struct EdgeWriteUndo { + pub database_id: u64, + pub tid: u64, + pub collection: String, + pub src_id: String, + pub label: String, + pub dst_id: String, + /// The version the write added to the edge store. + pub version: EdgeVersionWrite, + /// The CSR state the write found. + pub csr: EdgeCsrPrior, + /// Whether the write dropped the endpoints' durable identity bindings. A + /// node-delete cascade does. The undo binds them again from the CSR. + pub rebind_endpoints: bool, +} + +/// The CSR state one edge write found. +#[derive(Debug, Clone, Default, PartialEq)] +pub(in crate::data::executor) struct EdgeCsrPrior { + /// The edge's weight, `None` when the edge was not live. + pub weight: Option, + /// The surrogate each existing endpoint held, `0` for none. + pub surrogates: Vec<(String, u32)>, + /// The endpoints the CSR did not hold, in the order a write interns them. + pub created_nodes: Vec, +} + +/// One edge a write targets. +pub(in crate::data::executor) struct EdgeTarget<'a> { + pub database_id: u64, + pub tid: u64, + pub collection: &'a str, + pub src_id: &'a str, + pub label: &'a str, + pub dst_id: &'a str, +} + +impl EdgeTarget<'_> { + /// The undo of a write to this edge that added `version`. + pub(in crate::data::executor) fn undo( + &self, + version: EdgeVersionWrite, + csr: EdgeCsrPrior, + ) -> EdgeWriteUndo { + EdgeWriteUndo { + database_id: self.database_id, + tid: self.tid, + collection: self.collection.to_string(), + src_id: self.src_id.to_string(), + label: self.label.to_string(), + dst_id: self.dst_id.to_string(), + version, + csr, + rebind_endpoints: false, + } + } +} + +impl CoreLoop { + /// The CSR state a write to `target` finds now. + pub(in crate::data::executor) fn capture_edge_csr( + &self, + target: &EdgeTarget<'_>, + ) -> EdgeCsrPrior { + let partition = self.csr_partition(target.database_id, target.tid); + let mut prior = EdgeCsrPrior { + weight: partition.and_then(|p| { + p.edge_weight_in_collection( + target.src_id, + target.label, + target.dst_id, + target.collection, + ) + }), + ..EdgeCsrPrior::default() + }; + let endpoints = if target.src_id == target.dst_id { + vec![target.src_id] + } else { + vec![target.src_id, target.dst_id] + }; + for node in endpoints { + match partition.filter(|p| p.contains_node(node)) { + Some(p) => prior.surrogates.push(( + node.to_string(), + p.node_surrogate(node).map_or(0, |s| s.as_u32()), + )), + None => prior.created_nodes.push(node.to_string()), + } + } + prior + } + + /// Reverse one edge write. + pub(super) fn apply_undo_edge_write( + &mut self, + entry_index: usize, + undo: EdgeWriteUndo, + ) -> Result<(), (usize, String)> { + let EdgeWriteUndo { + database_id, + tid, + collection, + src_id, + label, + dst_id, + version, + csr, + rebind_endpoints, + } = undo; + let core = self.core_id; + let fail = |detail: String| { + error!( + core, + entry_index, + error = %detail, + "transaction undo: edge rollback failed; shard state unknown" + ); + (entry_index, detail) + }; + let edge_name = format!("{collection} {src_id}-[{label}]->{dst_id}"); + let database = DatabaseId::new(database_id); + let tenant = TenantId::new(tid); + + self.edge_store + .remove_edge_version( + EdgeRef::new(database, tenant, &collection, &src_id, &label, &dst_id), + &version, + ) + .map_err(|e| fail(format!("removing the version of {edge_name}: {e}")))?; + + if rebind_endpoints { + let bindings: Vec<(&str, u32)> = match self.csr_partition(database_id, tid) { + Some(p) => [src_id.as_str(), dst_id.as_str()] + .into_iter() + .filter_map(|node| p.node_surrogate(node).map(|s| (node, s.as_u32()))) + .collect(), + None => Vec::new(), + }; + for (node, raw) in bindings { + self.edge_store + .bind_node_surrogate(database, tenant, node, raw) + .map_err(|e| fail(format!("binding node '{node}' again: {e}")))?; + } + } + + let partition = self.csr_partition_mut(database_id, tid); + partition + .restore_edge_in_collection(&src_id, &label, &dst_id, &collection, csr.weight) + .map_err(|e| fail(format!("restoring the CSR edge {edge_name}: {e}")))?; + for (node, prior) in &csr.surrogates { + partition.restore_node_surrogate(node, *prior); + } + for node in csr.created_nodes.iter().rev() { + partition + .withdraw_newest_node(node) + .map_err(|e| fail(format!("withdrawing node '{node}': {e}")))?; + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::engine::graph::csr::Direction; + use crate::engine::graph::csr::extract_weight_from_properties; + + const DB: u64 = 0; + const TID: u64 = 1; + + fn tenant() -> TenantId { + TenantId::new(TID) + } + + fn edge<'a>(src: &'a str, dst: &'a str) -> EdgeRef<'a> { + EdgeRef::new(DatabaseId::new(DB), tenant(), "c", src, "KNOWS", dst) + } + + fn target<'a>(src: &'a str, dst: &'a str) -> EdgeTarget<'a> { + EdgeTarget { + database_id: DB, + tid: TID, + collection: "c", + src_id: src, + label: "KNOWS", + dst_id: dst, + } + } + + fn weighted(weight: f64) -> Vec { + nodedb_types::json_to_msgpack(&serde_json::json!({ "weight": weight })) + .expect("encode edge properties") + } + + /// Put an edge the way the put handler does, and return its undo. + fn put(core: &mut CoreLoop, src: &str, dst: &str, props: &[u8], ord: i64) -> EdgeWriteUndo { + let target = target(src, dst); + let csr = core.capture_edge_csr(&target); + let version = core + .edge_store + .put_edge_version_recorded( + edge(src, dst).with_surrogates(Surrogate::new(10), Surrogate::new(20)), + props, + ord, + ord, + i64::MAX, + true, + ) + .expect("put edge version"); + let partition = core.csr_partition_mut(DB, TID); + partition + .put_edge_in_collection( + src, + "KNOWS", + dst, + "c", + extract_weight_from_properties(props), + ) + .expect("put CSR edge"); + partition.set_node_surrogate(src, Surrogate::new(10)); + partition.set_node_surrogate(dst, Surrogate::new(20)); + target.undo(version, csr) + } + + fn resolve(core: &CoreLoop, src: &str, dst: &str, as_of: i64) -> Option> { + core.edge_store + .ceiling_resolve_edge(edge(src, dst), as_of, None) + .expect("resolve edge") + } + + /// An `AS OF` read at a system time between the write and its rollback + /// sees what it saw before the write: the rolled-back version is gone, + /// not shadowed by a newer compensating one. + #[test] + fn a_rolled_back_edge_update_leaves_no_version_at_any_system_time() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let _seed = put(&mut core, "alice", "bob", &weighted(2.5), 100); + + let undo = put(&mut core, "alice", "bob", &weighted(9.0), 200); + core.apply_undo_edge_write(0, undo).expect("undo update"); + + for as_of in [150, 200, 250, i64::MAX] { + assert_eq!( + resolve(&core, "alice", "bob", as_of), + Some(weighted(2.5)), + "system time {as_of} reads the edge as it was before the update" + ); + } + assert_eq!( + core.csr_partition(DB, TID) + .and_then(|p| p.edge_weight_in_collection("alice", "KNOWS", "bob", "c")), + Some(2.5), + "the CSR keeps the committed weight" + ); + } + + #[test] + fn a_rolled_back_edge_insert_withdraws_the_edge_and_the_nodes_it_created() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + + let undo = put(&mut core, "alice", "bob", b"", 100); + core.apply_undo_edge_write(0, undo).expect("undo insert"); + + assert_eq!(resolve(&core, "alice", "bob", 100), None); + assert_eq!(resolve(&core, "alice", "bob", i64::MAX), None); + let partition = core.csr_partition(DB, TID).expect("partition"); + assert!(!partition.contains_node("alice")); + assert!(!partition.contains_node("bob")); + assert_eq!(partition.node_count(), 0); + assert!( + core.edge_store + .scan_all_node_surrogates() + .expect("scan bindings") + .is_empty(), + "the identity bindings the insert made are gone" + ); + } + + #[test] + fn a_rolled_back_delete_removes_its_tombstone() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let _seed = put(&mut core, "alice", "bob", &weighted(2.5), 100); + + let target = target("alice", "bob"); + let csr = core.capture_edge_csr(&target); + let tombstone = core + .edge_store + .soft_delete_edge_recorded(edge("alice", "bob"), 200, true) + .expect("tombstone"); + core.csr_partition_mut(DB, TID) + .remove_edge_in_collection("alice", "KNOWS", "bob", "c"); + core.apply_undo_edge_write(0, target.undo(tombstone, csr)) + .expect("undo delete"); + + assert_eq!(resolve(&core, "alice", "bob", 250), Some(weighted(2.5))); + assert_eq!( + core.csr_partition(DB, TID) + .map(|p| p.neighbors("alice", None, Direction::Out)), + Some(vec![("KNOWS".to_string(), "bob".to_string())]) + ); + } + + /// A rolled-back node delete puts back every edge the cascade tombstoned + /// and the binding it dropped, in the edge store and in the CSR. + #[test] + fn a_rolled_back_node_delete_cascade_restores_edges_and_bindings() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let _seed = put(&mut core, "alice", "bob", &weighted(2.5), 100); + + core.csr_partition_mut(DB, TID).remove_node_edges("alice"); + let removed = core + .edge_store + .delete_edges_for_node(DB, tenant(), "alice", 200) + .expect("cascade"); + assert_eq!(removed.len(), 1); + for (idx, restore) in removed.into_iter().enumerate() { + let weight = extract_weight_from_properties(&restore.old_properties); + let mut undo = EdgeTarget { + database_id: DB, + tid: TID, + collection: &restore.collection, + src_id: &restore.src, + label: &restore.label, + dst_id: &restore.dst, + } + .undo( + restore.tombstone.clone(), + EdgeCsrPrior { + weight: Some(weight), + ..EdgeCsrPrior::default() + }, + ); + undo.rebind_endpoints = true; + core.apply_undo_edge_write(idx, undo).expect("undo cascade"); + } + + assert_eq!(resolve(&core, "alice", "bob", 250), Some(weighted(2.5))); + assert_eq!( + core.csr_partition(DB, TID) + .and_then(|p| p.edge_weight_in_collection("alice", "KNOWS", "bob", "c")), + Some(2.5) + ); + let mut bindings: Vec<(String, u32)> = core + .edge_store + .scan_all_node_surrogates() + .expect("scan bindings") + .into_iter() + .map(|record| (record.2, record.3)) + .collect(); + bindings.sort(); + assert_eq!( + bindings, + vec![("alice".to_string(), 10), ("bob".to_string(), 20)] + ); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs index 6a263e1eb..21f84a5c8 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs @@ -24,6 +24,9 @@ pub(in crate::data::executor) struct TimeseriesIngestUndo { /// accounting after rollback. pub memtable_memory_bytes_before: Option, pub last_value_cache_before: Option, + /// The collection's series catalog. Ingest registers each new series in + /// it. + pub series_catalog_before: Option, pub max_ingested_lsn_before: Option, pub last_ts_ingest_before: Option, pub reservation_bytes_before: Option, @@ -186,23 +189,9 @@ pub(in crate::data::executor) enum UndoEntry { bbox: nodedb_types::BoundingBox, document_id: String, }, - /// Undo an EdgePut by deleting the edge (or restoring old properties). - PutEdge { - collection: String, - src_id: String, - label: String, - dst_id: String, - /// `None` if edge didn't exist before (inserted); `Some(bytes)` if overwritten. - old_properties: Option>, - }, - /// Undo an EdgeDelete by re-inserting the edge with its old properties. - DeleteEdge { - collection: String, - src_id: String, - label: String, - dst_id: String, - old_properties: Vec, - }, + /// Undo a graph edge write: remove the version it added and put the CSR + /// back. + EdgeWrite(Box), /// Undo a KV write (Put / Insert / InsertIfAbsent / InsertOnConflictUpdate / /// FieldSet / Incr / IncrFloat / Cas / GetSet) by reinstating the key's /// prior state. @@ -343,12 +332,17 @@ pub(in crate::data::executor) enum UndoEntry { prior: Option, }, /// Undo a node-label set or removal: each label the op touched goes back - /// to whether the node carried it before (`true` = it did). + /// to whether the node carried it before (`true` = it did). The label + /// names and the node the op interned are withdrawn. NodeLabels { database_id: u64, tid: u64, node_id: String, prior: Vec<(String, bool)>, + /// Label names the op interned, in interning order. + interned_labels: Vec, + /// Whether the op created the node in the CSR. + created_node: bool, }, /// Undo a columnar insert by rolling back in-memory state. /// diff --git a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs index 5e0bf276b..c43efe33e 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs @@ -13,6 +13,9 @@ //! newly inserted the node, so this un-mark never resurrects a tombstone a //! prior committed op created. //! +//! The node-label undo puts each label back, then withdraws the label names +//! and the node the op interned, so the CSR holds what it held before. +//! //! Returns `Err((entry_index, detail))` on fatal failure so the caller can //! escalate to a typed `RollbackFailed` response. @@ -39,32 +42,96 @@ impl CoreLoop { } } - /// Put every label a node-label op touched back to its prior state. - pub(super) fn apply_undo_node_labels( - &mut self, - entry_index: usize, + /// The undo of a node-label op on `node_id` setting or removing `labels`, + /// captured before the op runs. + pub(in crate::data::executor) fn capture_node_labels_undo( + &self, database_id: u64, tid: u64, node_id: &str, - prior: Vec<(String, bool)>, + labels: &[String], + ) -> UndoEntry { + let partition = self.csr_partition(database_id, tid); + let local = partition.and_then(|p| p.node_id_raw(node_id)); + let prior = labels + .iter() + .map(|label| { + let carried = match (partition, local) { + (Some(p), Some(id)) => p.node_has_label(id, label), + _ => false, + }; + (label.clone(), carried) + }) + .collect(); + let mut interned_labels: Vec = Vec::new(); + for label in labels { + let known = partition.is_some_and(|p| p.has_node_label_name(label)); + if !known && !interned_labels.contains(label) { + interned_labels.push(label.clone()); + } + } + UndoEntry::NodeLabels { + database_id, + tid, + node_id: node_id.to_string(), + prior, + interned_labels, + created_node: local.is_none(), + } + } + + /// Put every label a node-label op touched back to its prior state, then + /// withdraw the label names and the node the op interned. + pub(super) fn apply_undo_node_labels( + &mut self, + entry_index: usize, + undo: NodeLabelsUndo, ) -> Result<(), (usize, String)> { + let NodeLabelsUndo { + database_id, + tid, + node_id, + prior, + interned_labels, + created_node, + } = undo; let partition = self.csr_partition_mut(database_id, tid); for (label, carried) in prior { if carried { - partition.add_node_label(node_id, &label).map_err(|e| { + partition.add_node_label(&node_id, &label).map_err(|e| { ( entry_index, format!("restoring label '{label}' on node '{node_id}': {e}"), ) })?; } else { - partition.remove_node_label(node_id, &label); + partition.remove_node_label(&node_id, &label); } } + for label in interned_labels.iter().rev() { + partition + .withdraw_newest_node_label(label) + .map_err(|e| (entry_index, format!("withdrawing label '{label}': {e}")))?; + } + if created_node { + partition + .withdraw_newest_node(&node_id) + .map_err(|e| (entry_index, format!("withdrawing node '{node_id}': {e}")))?; + } Ok(()) } } +/// The fields of an `UndoEntry::NodeLabels`. +pub(super) struct NodeLabelsUndo { + pub database_id: u64, + pub tid: u64, + pub node_id: String, + pub prior: Vec<(String, bool)>, + pub interned_labels: Vec, + pub created_node: bool, +} + #[cfg(test)] mod tests { use std::time::{Duration, Instant}; @@ -267,9 +334,8 @@ mod tests { ); assert!( undo_log.is_empty(), - "a rejected insert must record NO compensation entry; a phantom PutEdge \ - undo would soft-delete a never-written edge on rollback, corrupting \ - bitemporal history" + "a rejected insert must record no undo entry: it wrote no edge version \ + for a rollback to remove" ); assert!( core.edge_store diff --git a/nodedb/src/data/executor/handlers/transaction/undo/mod.rs b/nodedb/src/data/executor/handlers/transaction/undo/mod.rs index 591dea3b7..387c2d86a 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/mod.rs @@ -8,6 +8,7 @@ pub(in crate::data::executor) mod crdt_collection; pub(super) mod document; pub(super) mod document_fts; pub(in crate::data::executor::handlers) mod document_outcome; +pub(in crate::data::executor) mod edge_write; pub(super) mod entry; pub(in crate::data::executor) mod fts_doc; pub(super) mod graph_node; diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index 5fe856d60..2c2ffcdce 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -3,6 +3,7 @@ //! Rollback driver for the undo log. use super::UndoEntry; +use super::graph_node::NodeLabelsUndo; use crate::data::executor::core_loop::CoreLoop; impl CoreLoop { @@ -13,33 +14,14 @@ impl CoreLoop { /// Returns `Err((entry_index, detail))` on the first undo failure — /// the entry index is the original forward-order position of the failed /// entry (before reversal). On failure the caller **must** return a - /// `RollbackFailed` error to the client; the shard state is unknown - /// and requires a restart to restore consistency via WAL replay. + /// `RollbackFailed` error. The core's state is then unknown: the core + /// fail-stops when that response leaves it, and a restart rebuilds the + /// state through WAL replay. pub(in crate::data::executor::handlers) fn rollback_undo_log( &mut self, did: u64, tid: u64, undo_log: Vec, - ) -> Result<(), (usize, String)> { - self.rollback_undo_log_inner(did, tid, None, undo_log) - } - - pub(in crate::data::executor::handlers) fn rollback_undo_log_at( - &mut self, - did: u64, - tid: u64, - vshard_id: crate::types::VShardId, - undo_log: Vec, - ) -> Result<(), (usize, String)> { - self.rollback_undo_log_inner(did, tid, Some(vshard_id), undo_log) - } - - fn rollback_undo_log_inner( - &mut self, - did: u64, - tid: u64, - vshard_id: Option, - undo_log: Vec, ) -> Result<(), (usize, String)> { let total = undo_log.len(); for (rev_idx, entry) in undo_log.into_iter().rev().enumerate() { @@ -47,7 +29,7 @@ impl CoreLoop { // diagnostics (makes it easier to correlate with the sub-plan that // produced this undo entry). let original_idx = total.saturating_sub(1 + rev_idx); - self.apply_undo_entry(did, tid, vshard_id, original_idx, entry)?; + self.apply_undo_entry(did, tid, original_idx, entry)?; } Ok(()) } @@ -59,7 +41,6 @@ impl CoreLoop { &mut self, did: u64, tid: u64, - vshard_id: Option, entry_index: usize, entry: UndoEntry, ) -> Result<(), (usize, String)> { @@ -73,12 +54,7 @@ impl CoreLoop { UndoEntry::SpatialInsert { .. } | UndoEntry::SpatialDelete { .. } => { self.apply_undo_spatial(entry_index, entry) } - UndoEntry::PutEdge { ref src_id, .. } | UndoEntry::DeleteEdge { ref src_id, .. } => { - let account_stats = vshard_id.is_none_or(|vshard_id| { - vshard_id == crate::types::VShardId::from_key(src_id.as_bytes()) - }); - self.apply_undo_edge_with_stats(did, tid, entry_index, entry, account_stats) - } + UndoEntry::EdgeWrite(undo) => self.apply_undo_edge_write(entry_index, *undo), UndoEntry::KvPut { .. } | UndoEntry::KvDelete { .. } | UndoEntry::KvBatchPut { .. } @@ -153,7 +129,19 @@ impl CoreLoop { tid: label_tid, node_id, prior, - } => self.apply_undo_node_labels(entry_index, database_id, label_tid, &node_id, prior), + interned_labels, + created_node, + } => self.apply_undo_node_labels( + entry_index, + NodeLabelsUndo { + database_id, + tid: label_tid, + node_id, + prior, + interned_labels, + created_node, + }, + ), } } } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs b/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs index 06c0aa907..0c9c8d74e 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs @@ -31,6 +31,7 @@ impl CoreLoop { memtable_config_before: memtable.map(|memtable| memtable.config()), memtable_memory_bytes_before: memtable.map(|memtable| memtable.memory_bytes()), last_value_cache_before: self.ts_last_value_caches.get(collection_key).cloned(), + series_catalog_before: self.ts_series_catalogs.get(collection_key).cloned(), max_ingested_lsn_before: self.ts_max_ingested_lsn.get(collection_key).copied(), last_ts_ingest_before: self.last_ts_ingest, reservation_bytes_before: self diff --git a/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs b/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs index 17df75ad9..058b3e58b 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs @@ -222,3 +222,107 @@ impl CoreLoop { Ok(()) } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::engine::vector::ivf::{IvfPqIndex, IvfPqParams}; + use crate::types::{DatabaseId, TenantId}; + + const TID: u64 = 1; + + fn params() -> IvfPqParams { + IvfPqParams { + n_cells: 2, + pq_m: 2, + pq_k: 4, + nprobe: 2, + metric: nodedb_vector::DistanceMetric::L2, + } + } + + fn vectors() -> Vec> { + (0..8) + .map(|i| vec![i as f32, (i * 2) as f32, 1.0, 0.5]) + .collect() + } + + /// The IVF-PQ adds of a rolled-back write leave the index: it holds the + /// vectors and the training it held before the write. + #[test] + fn a_rolled_back_write_withdraws_its_ivf_adds() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let key: VectorIndexKey = (DatabaseId::DEFAULT, TenantId::new(TID), "docs:".into()); + let vecs = vectors(); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut index = IvfPqIndex::new(4, params()); + index.train( + &refs, + nodedb_mem::ScopedMemory::new( + crate::data::executor::core_loop::test_governor(), + DatabaseId::DEFAULT, + TenantId::new(TID), + nodedb_mem::EngineId::Vector, + ), + ); + index.add_batch(&refs[..4]); + core.ivf_indexes.insert(key.clone(), index); + + let undo = core + .capture_vector_write_undo(VectorWriteTarget { + index_key: &key, + tid: TID, + collection: "docs", + surrogates: &[], + ids: &[], + sidecars: false, + }) + .expect("capture undo"); + if let Some(index) = core.ivf_indexes.get_mut(&key) { + index.add_batch(&refs[4..]); + } + let UndoEntry::VectorWrite(undo) = undo else { + panic!("a vector write captures a VectorWrite undo"); + }; + core.apply_undo_vector_write(0, *undo) + .expect("undo vector write"); + + let index = core.ivf_indexes.get(&key).expect("index stays"); + assert_eq!(index.len(), 4); + assert!(index.is_trained()); + assert!( + index.search(&vecs[6], 8).iter().all(|r| r.id < 4), + "no vector the write added is found" + ); + } + + /// An IVF-PQ index the rolled-back write created is removed. + #[test] + fn a_rolled_back_write_removes_the_ivf_index_it_created() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let key: VectorIndexKey = (DatabaseId::DEFAULT, TenantId::new(TID), "docs:".into()); + + let undo = core + .capture_vector_write_undo(VectorWriteTarget { + index_key: &key, + tid: TID, + collection: "docs", + surrogates: &[], + ids: &[], + sidecars: false, + }) + .expect("capture undo"); + core.ivf_indexes + .insert(key.clone(), IvfPqIndex::new(4, params())); + let UndoEntry::VectorWrite(undo) = undo else { + panic!("a vector write captures a VectorWrite undo"); + }; + core.apply_undo_vector_write(0, *undo) + .expect("undo vector write"); + + assert!(!core.ivf_indexes.contains_key(&key)); + } +} diff --git a/nodedb/src/data/executor/handlers/truncate.rs b/nodedb/src/data/executor/handlers/truncate.rs index 745be0df3..6ca60567f 100644 --- a/nodedb/src/data/executor/handlers/truncate.rs +++ b/nodedb/src/data/executor/handlers/truncate.rs @@ -229,18 +229,9 @@ impl CoreLoop { collection: None, }); } - let edges = self - .csr_partition_mut(database_id, tid) - .remove_node_edges(&doc_id); - let cascade_ord = self.hlc.next_ordinal(); - if edges > 0 - && let Err(e) = self.edge_store.delete_edges_for_node( - database_id, - nodedb_types::TenantId::new(tid), - &doc_id, - cascade_ord, - ) - { + // On an error neither edge store changed: the edges stay in + // both, and the dangling-edge sweep retries them. + if let Err(e) = self.cascade_node_edges(database_id, tid, &doc_id) { warn!(core = self.core_id, %doc_id, error = %e, "truncate: edge cascade failed"); } self.doc_cache.invalidate( diff --git a/nodedb/src/data/executor/wal_replay/crdt_list.rs b/nodedb/src/data/executor/wal_replay/crdt_list.rs index 9ab9f902a..3b8a40c03 100644 --- a/nodedb/src/data/executor/wal_replay/crdt_list.rs +++ b/nodedb/src/data/executor/wal_replay/crdt_list.rs @@ -412,7 +412,8 @@ mod tests { None, ); let seed_bytes = seed_payload.encode().expect("encode seed"); - wal.append_crdt_delta(tid, vs, db, &seed_bytes) + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_crdt_delta(tid, vs, db, &seed_bytes) .expect("append seed"); // blocks: [] -> [blk-0] -> [blk-0, blk-1] -> [blk-1, blk-0] -> [blk-0] @@ -489,7 +490,8 @@ mod tests { None, ); let seed_bytes = seed_payload.encode().expect("encode seed"); - wal.append_crdt_delta(tid, vs, db, &seed_bytes) + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_crdt_delta(tid, vs, db, &seed_bytes) .expect("append seed"); // blocks: [] -> [blk-0, blk-1, blk-2, blk-3] -> move(3, 1) diff --git a/nodedb/src/data/executor/wal_replay/kv_put.rs b/nodedb/src/data/executor/wal_replay/kv_put.rs index 37cac8eac..fb86829da 100644 --- a/nodedb/src/data/executor/wal_replay/kv_put.rs +++ b/nodedb/src/data/executor/wal_replay/kv_put.rs @@ -398,13 +398,14 @@ mod tests { let dir = tempfile::tempdir().expect("wal tempdir"); let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); for payload in payloads { - wal.append_put( - TenantId::new(TID), - VShardId::new(0), - DatabaseId::DEFAULT, - payload, - ) - .expect("append"); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(TID), + VShardId::new(0), + DatabaseId::DEFAULT, + payload, + ) + .expect("append"); } wal.sync().expect("sync"); let records = wal.replay().expect("replay read"); diff --git a/nodedb/src/data/executor/wal_replay_graph_labels.rs b/nodedb/src/data/executor/wal_replay_graph_labels.rs index 086bb3b7b..5ae6d0669 100644 --- a/nodedb/src/data/executor/wal_replay_graph_labels.rs +++ b/nodedb/src/data/executor/wal_replay_graph_labels.rs @@ -40,7 +40,6 @@ use nodedb_wal::WalRecord; use nodedb_wal::record::RecordType; use super::core_loop::CoreLoop; -use super::handlers::transaction::undo::UndoEntry; use crate::types::DatabaseId; impl CoreLoop { @@ -80,13 +79,9 @@ impl CoreLoop { return Some(0); } if self.recording_redo_undo() { - let prior = self.node_label_prior(database_id.as_u64(), tenant_id, &node_id, &labels); - self.record_redo_undo([UndoEntry::NodeLabels { - database_id: database_id.as_u64(), - tid: tenant_id, - node_id: node_id.clone(), - prior, - }]); + let undo = + self.capture_node_labels_undo(database_id.as_u64(), tenant_id, &node_id, &labels); + self.record_redo_undo([undo]); } if is_set { @@ -121,28 +116,6 @@ impl CoreLoop { Some(1) } - /// Whether `node_id` carries each of `labels` now. - fn node_label_prior( - &self, - database_id: u64, - tenant_id: u64, - node_id: &str, - labels: &[String], - ) -> Vec<(String, bool)> { - let partition = self.csr_partition(database_id, tenant_id); - let local = partition.and_then(|p| p.node_id_raw(node_id)); - labels - .iter() - .map(|label| { - let carried = match (partition, local) { - (Some(p), Some(id)) => p.node_has_label(id, label), - _ => false, - }; - (label.clone(), carried) - }) - .collect() - } - /// Replay every `GraphNodeLabelSet` / `GraphNodeLabelRemove` record in /// `records`, routing each through [`CoreLoop::try_replay_graph_node_label`]. /// diff --git a/nodedb/src/data/executor/wal_replay_kv_expiry.rs b/nodedb/src/data/executor/wal_replay_kv_expiry.rs index 7c5ee2b8d..38d8f200a 100644 --- a/nodedb/src/data/executor/wal_replay_kv_expiry.rs +++ b/nodedb/src/data/executor/wal_replay_kv_expiry.rs @@ -246,13 +246,14 @@ mod tests { &put_seed, ) .expect("wal append seed put"); - wal.append_put( - TenantId::new(TID), - VShardId::new(0), - DatabaseId::DEFAULT, - &entry, - ) - .expect("append raw kv_expire record"); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(TID), + VShardId::new(0), + DatabaseId::DEFAULT, + &entry, + ) + .expect("append raw kv_expire record"); wal.sync().expect("wal sync"); let records = wal.replay().expect("wal replay read"); @@ -380,13 +381,14 @@ mod tests { let dir = tempfile::tempdir().expect("wal tempdir"); let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); - wal.append_put( - TenantId::new(TID), - VShardId::new(0), - DatabaseId::DEFAULT, - &entry, - ) - .expect("append raw kv_expire record"); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(TID), + VShardId::new(0), + DatabaseId::DEFAULT, + &entry, + ) + .expect("append raw kv_expire record"); wal.sync().expect("wal sync"); let records = wal.replay().expect("wal replay read"); diff --git a/nodedb/src/data/executor/wal_replay_kv_incr.rs b/nodedb/src/data/executor/wal_replay_kv_incr.rs index ebe28fc8e..a40887d5f 100644 --- a/nodedb/src/data/executor/wal_replay_kv_incr.rs +++ b/nodedb/src/data/executor/wal_replay_kv_incr.rs @@ -413,13 +413,14 @@ mod tests { &put_seed, ) .expect("wal append seed put"); - wal.append_put( - TenantId::new(TID), - VShardId::new(0), - DatabaseId::DEFAULT, - &entry, - ) - .expect("append raw kv_incr record"); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(TID), + VShardId::new(0), + DatabaseId::DEFAULT, + &entry, + ) + .expect("append raw kv_incr record"); wal.sync().expect("wal sync"); let records = wal.replay().expect("wal replay read"); diff --git a/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs b/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs index bc559dbf2..d077ee72c 100644 --- a/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs +++ b/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs @@ -491,13 +491,14 @@ mod tests { &put_p1, ) .expect("wal append seed put"); - wal.append_put( - TenantId::new(TID), - VShardId::new(0), - DatabaseId::DEFAULT, - &entry, - ) - .expect("append raw kv_insert_on_conflict_update record"); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(TID), + VShardId::new(0), + DatabaseId::DEFAULT, + &entry, + ) + .expect("append raw kv_insert_on_conflict_update record"); wal.sync().expect("wal sync"); let records = wal.replay().expect("wal replay read"); diff --git a/nodedb/src/data/executor/wal_replay_kv_ttl.rs b/nodedb/src/data/executor/wal_replay_kv_ttl.rs index bfaca2937..19642397f 100644 --- a/nodedb/src/data/executor/wal_replay_kv_ttl.rs +++ b/nodedb/src/data/executor/wal_replay_kv_ttl.rs @@ -92,13 +92,14 @@ mod tests { let dir = tempfile::tempdir().expect("wal tempdir"); let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); - wal.append_put( - TenantId::new(TID), - VShardId::new(0), - DatabaseId::DEFAULT, - &entry, - ) - .expect("append raw kv_put record"); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(TID), + VShardId::new(0), + DatabaseId::DEFAULT, + &entry, + ) + .expect("append raw kv_put record"); wal.sync().expect("wal sync"); let records = wal.replay().expect("wal replay read"); @@ -130,13 +131,14 @@ mod tests { let dir = tempfile::tempdir().expect("wal tempdir"); let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); - wal.append_put( - TenantId::new(TID), - VShardId::new(0), - DatabaseId::DEFAULT, - &entry, - ) - .expect("append raw kv_batch_put record"); + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(TID), + VShardId::new(0), + DatabaseId::DEFAULT, + &entry, + ) + .expect("append raw kv_batch_put record"); wal.sync().expect("wal sync"); let records = wal.replay().expect("wal replay read"); diff --git a/nodedb/src/data/runtime/event_loop.rs b/nodedb/src/data/runtime/event_loop.rs index c021680dd..c88338a49 100644 --- a/nodedb/src/data/runtime/event_loop.rs +++ b/nodedb/src/data/runtime/event_loop.rs @@ -148,7 +148,11 @@ pub(super) fn run_event_loop( // coordinated checkpoint clamps to. This one is an // opportunistic head start, so its only obligations are to make // the bytes durable and to say so when it cannot. - if last_checkpoint.elapsed() >= checkpoint_interval { + // A fail-stopped core publishes nothing: restart rebuilds its state + // from the WAL, and a checkpoint of the unknown state would stand in + // for that rebuild. + let stopped = core.is_fail_stopped(); + if !stopped && last_checkpoint.elapsed() >= checkpoint_interval { if let Err(e) = core.checkpoint_vector_indexes() { warn!( core = core_id, @@ -161,7 +165,9 @@ pub(super) fn run_event_loop( } // Periodic compaction + maintenance (tombstone cleanup, CSR compact, edge sweep). - core.maybe_run_maintenance(); + if !stopped { + core.maybe_run_maintenance(); + } // Heartbeat: if no user writes for ~1 second (±100ms jitter), // emit a heartbeat to advance the Event Plane's partition diff --git a/nodedb/src/diag/context/data_plane.rs b/nodedb/src/diag/context/data_plane.rs index 24c2e42bb..f370bd151 100644 --- a/nodedb/src/diag/context/data_plane.rs +++ b/nodedb/src/diag/context/data_plane.rs @@ -146,3 +146,39 @@ impl DomainContext for CalvinApplyHalted<'_> { }) } } + +/// A Data-Plane core fail-stopped because its state is unknown: a rollback +/// failed part way, or the work owed after a committed record's install +/// failed. +pub(in crate::diag) struct CoreFailStopped<'a> { + pub core_id: usize, + /// Cause label (`rollback_failed`, `post_install_failed`). + pub cause: &'a str, + pub detail: &'a str, +} + +impl DomainContext for CoreFailStopped<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb.data_plane_core_fail_stopped" + } + + fn grouping_key(&self) -> String { + // The cause names the bug; the core is the occurrence. + format!("cause={}", self.cause) + } + + fn to_json(&self) -> Value { + json!({ + "core_id": self.core_id, + "cause": self.cause, + "detail": self.detail, + "why_fatal": "the core's live state no longer matches what restart replay \ + rebuilds from the WAL and its published artifacts. Serving it \ + would answer reads and take writes against state no replica \ + and no restart reproduces, so the core refuses every request", + "operator_action": "fix the named cause (the failing undo entry, or the disk \ + behind the flush or artifact that failed), then restart \ + the node. Restart replay rebuilds the core from the WAL", + }) + } +} diff --git a/nodedb/src/diag/context/mod.rs b/nodedb/src/diag/context/mod.rs index 73b5d60fa..664fd0962 100644 --- a/nodedb/src/diag/context/mod.rs +++ b/nodedb/src/diag/context/mod.rs @@ -23,7 +23,7 @@ pub(in crate::diag) use catalog::{ pub(in crate::diag) use crdt::HistoryCompactionNotApplied; pub use data_plane::LostResponseWrite; pub(in crate::diag) use data_plane::{ - CalvinApplyHalted, CalvinCompletionTimeout, DataPlaneResponseLost, + CalvinApplyHalted, CalvinCompletionTimeout, CoreFailStopped, DataPlaneResponseLost, }; pub(in crate::diag) use ingest::IlpAcceptedLinesDropped; pub use ingest::IlpFlushOutcome; diff --git a/nodedb/src/diag/mod.rs b/nodedb/src/diag/mod.rs index 20c0239c1..2eeb445f2 100644 --- a/nodedb/src/diag/mod.rs +++ b/nodedb/src/diag/mod.rs @@ -12,10 +12,10 @@ pub use context::{DATABASE_SCOPE, IlpFlushOutcome, LostResponseWrite, TENANT_SCO pub use recording::{ batch_insert_without_surrogates, calvin_apply_halted, calvin_completion_timeout, catalog_apply_orphan_row, collection_purge_row_missing, consumer_group_offsets_retained, - data_plane_response_lost, data_plane_responses_lost, entry_kind, fts_index_update_failed, - history_compaction_not_applied, ilp_invalid_utf8_drop, ilp_line_read_drop, - metadata_apply_wedged, orphaned_index_entry_after_delete, quota_row_invalid, - quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, + data_plane_core_fail_stopped, data_plane_response_lost, data_plane_responses_lost, entry_kind, + fts_index_update_failed, history_compaction_not_applied, ilp_invalid_utf8_drop, + ilp_line_read_drop, metadata_apply_wedged, orphaned_index_entry_after_delete, + quota_row_invalid, quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, quota_scope_replay_aborted, replay_record_unapplied, retention_autowire_orphaned, scope_quota_not_installed, strict_row_undecodable, synonym_group_not_applied, vector_index_not_applied, wal_archival_failed_truncation_held, write_acked_without_durability, diff --git a/nodedb/src/diag/recording/data_plane.rs b/nodedb/src/diag/recording/data_plane.rs index 12b0d650c..d7c128c82 100644 --- a/nodedb/src/diag/recording/data_plane.rs +++ b/nodedb/src/diag/recording/data_plane.rs @@ -88,3 +88,20 @@ pub fn calvin_apply_halted( .with_backtrace() .emit(); } + +/// Report a Data-Plane core that fail-stopped because its state is unknown. +/// Called once per core, from its fail-stop latch on the first cause. +pub fn data_plane_core_fail_stopped(core_id: usize, cause: &str, detail: &str) { + let ctx = context::CoreFailStopped { + core_id, + cause, + detail, + }; + let _ = Capture::new( + EventKind::InvariantViolation, + "Data-Plane core fail-stopped: its state is unknown until restart", + ) + .domain(&ctx) + .with_backtrace() + .emit(); +} diff --git a/nodedb/src/diag/recording/mod.rs b/nodedb/src/diag/recording/mod.rs index b6a6db5d0..5590b940e 100644 --- a/nodedb/src/diag/recording/mod.rs +++ b/nodedb/src/diag/recording/mod.rs @@ -24,8 +24,8 @@ pub use catalog::{ }; pub use crdt::history_compaction_not_applied; pub use data_plane::{ - calvin_apply_halted, calvin_completion_timeout, data_plane_response_lost, - data_plane_responses_lost, + calvin_apply_halted, calvin_completion_timeout, data_plane_core_fail_stopped, + data_plane_response_lost, data_plane_responses_lost, }; pub use ingest::{ilp_invalid_utf8_drop, ilp_line_read_drop}; pub use quota::{ diff --git a/nodedb/src/engine/graph/edge_store/cascade.rs b/nodedb/src/engine/graph/edge_store/cascade.rs index 1b47f7a33..382809926 100644 --- a/nodedb/src/engine/graph/edge_store/cascade.rs +++ b/nodedb/src/engine/graph/edge_store/cascade.rs @@ -5,29 +5,41 @@ use redb::{ReadableDatabase, ReadableTable}; use std::collections::HashMap; -use super::store::{BaseKey, EDGES, EdgeStore, redb_err}; -use super::temporal::{EdgeRef, is_sentinel, parse_versioned_edge_key}; +use super::store::{BaseKey, EDGES, EdgeStore, NODE_SURROGATES, redb_err}; +use super::temporal::write::write_sentinel_in; +use super::temporal::{ + EdgeRef, EdgeValuePayload, EdgeVersionWrite, TOMBSTONE_SENTINEL, is_sentinel, + parse_versioned_edge_key, +}; use nodedb_types::{DatabaseId, TenantId}; -/// A single cascaded edge removal captured for transactional rollback: -/// `(collection, src, label, dst, old_properties)`. `old_properties` is the -/// edge's current-state value read BEFORE the soft-delete, so an -/// `UndoEntry::DeleteEdge` can re-insert the exact edge into both the CSR -/// partition and the persistent edge store on rollback. -pub type EdgeRestore = (String, String, String, String, Vec); +/// A single cascaded edge removal captured for transactional rollback. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct EdgeRestore { + pub collection: String, + pub src: String, + pub label: String, + pub dst: String, + /// The edge's properties before the tombstone. The CSR restore reads its + /// weight from them. + pub old_properties: Vec, + /// The tombstone version the cascade added. A rollback removes it. + pub tombstone: EdgeVersionWrite, +} impl EdgeStore { /// Soft-delete every edge incident on `node` (as either src or dst) in - /// the caller's tenant, across all collections. Emits a tombstone - /// version at `system_from` for each distinct base edge that has a - /// live (non-sentinel) latest version. + /// the caller's tenant, across all collections, and drop the node's + /// identity binding. Emits a tombstone version at `system_from` for each + /// distinct base edge that has a live (non-sentinel) latest version. + /// + /// One transaction holds every tombstone and the binding removal, so the + /// cascade lands whole or not at all. /// - /// Returns the set of edges actually soft-deleted, each paired with its - /// pre-delete `old_properties`, so a transactional caller can push one - /// `UndoEntry::DeleteEdge` per edge and fully reverse the cascade on - /// rollback. The returned edges are exactly the live bases that existed - /// before this call (already-tombstoned bases are skipped and not - /// returned — they were not removed by this op). + /// Returns the edges actually soft-deleted, each with its pre-delete + /// properties and its tombstone, so a transactional caller can push one + /// `UndoEntry::EdgeWrite` per edge and fully reverse the cascade on + /// rollback. Already-tombstoned bases are skipped and not returned. pub fn delete_edges_for_node( &self, db: u64, @@ -35,45 +47,58 @@ impl EdgeStore { node: &str, system_from: i64, ) -> crate::Result> { - // Snapshot all live bases touching `node`. Done in a read txn first - // so the write txn can call soft_delete_edge without nested locks. + // Snapshot all live bases touching `node` in a read txn first. let bases = self.live_bases_touching_node(db, tid, node)?; + let database = DatabaseId::new(db); + let write_txn = self + .db + .begin_write() + .map_err(|e| redb_err("begin_write", e))?; let mut removed = Vec::with_capacity(bases.len()); - for (collection, src, label, dst) in &bases { - // Capture the current-state properties BEFORE the soft-delete so a - // rolled-back transactional delete can restore the exact edge value. - let old_properties = self - .get_edge(db, tid, collection, src, label, dst)? - .unwrap_or_default(); - self.soft_delete_edge( - EdgeRef::new(DatabaseId::new(db), tid, collection, src, label, dst), + for ((collection, src, label, dst), old_properties) in bases { + let tombstone = write_sentinel_in( + &write_txn, + EdgeRef::new(database, tid, &collection, &src, &label, &dst), system_from, + TOMBSTONE_SENTINEL, + true, )?; - removed.push(( - collection.clone(), - src.clone(), - label.clone(), - dst.clone(), + removed.push(EdgeRestore { + collection, + src, + label, + dst, old_properties, - )); + tombstone, + }); } // The node itself is going away, so its identity binding goes with it. // Only this node's: the neighbours survive and keep theirs. A rolled-back // delete restores the binding along with the edges (see the transaction // undo path), so this is not a one-way loss. - self.delete_node_surrogate(DatabaseId::new(db), tid, node)?; + { + let mut surrogates = write_txn + .open_table(NODE_SURROGATES) + .map_err(|e| redb_err("open node_surrogates", e))?; + surrogates + .remove((db, tid.as_u64(), node)) + .map_err(|e| redb_err("remove node surrogate", e))?; + } + write_txn + .commit() + .map_err(|e| redb_err("commit node edge cascade", e))?; Ok(removed) } - /// Enumerate `(collection, src, label, dst)` tuples for every base edge - /// in this `(database, tenant)` whose latest version touches `node` as src - /// or dst and is not a sentinel. + /// Every base edge in this `(database, tenant)` whose latest version + /// touches `node` as src or dst and is live, with that version's + /// properties. fn live_bases_touching_node( &self, db: u64, tid: TenantId, node: &str, - ) -> crate::Result> { + ) -> crate::Result)>> { let t = tid.as_u64(); let read_txn = self .db @@ -83,7 +108,8 @@ impl EdgeStore { .open_table(EDGES) .map_err(|e| redb_err("open edges", e))?; - let mut latest: HashMap = HashMap::new(); + // Latest version per base: its system time and its raw value. + let mut latest: HashMap)> = HashMap::new(); // DB-scoped range: a node-delete in database A must NOT cascade into // the same tenant's edges in database B. let range = table @@ -104,21 +130,27 @@ impl EdgeStore { label.to_string(), dst.to_string(), ); - let is_sent = is_sentinel(v.value()); - latest - .entry(base) - .and_modify(|(cur, cur_sent)| { - if sys > *cur { - *cur = sys; - *cur_sent = is_sent; - } - }) - .or_insert((sys, is_sent)); + let value = v.value(); + match latest.get_mut(&base) { + Some((cur, bytes)) if sys > *cur => { + *cur = sys; + *bytes = value.to_vec(); + } + Some(_) => {} + None => { + latest.insert(base, (sys, value.to_vec())); + } + } + } + let mut live = Vec::with_capacity(latest.len()); + for (base, (_sys, bytes)) in latest { + if is_sentinel(&bytes) { + continue; + } + let properties = EdgeValuePayload::decode(&bytes)?.properties; + live.push((base, properties)); } - Ok(latest - .into_iter() - .filter_map(|(base, (_sys, is_sent))| if is_sent { None } else { Some(base) }) - .collect()) + Ok(live) } } @@ -170,12 +202,12 @@ mod tests { assert!( removed .iter() - .any(|(_, s, _, d, p)| s == "alice" && d == "bob" && p == b"1") + .any(|r| r.src == "alice" && r.dst == "bob" && r.old_properties == b"1") ); assert!( removed .iter() - .any(|(_, s, _, d, p)| s == "dave" && d == "alice" && p == b"3") + .any(|r| r.src == "dave" && r.dst == "alice" && r.old_properties == b"3") ); assert!( diff --git a/nodedb/src/engine/graph/edge_store/mod.rs b/nodedb/src/engine/graph/edge_store/mod.rs index 716f45802..b0efb7eaf 100644 --- a/nodedb/src/engine/graph/edge_store/mod.rs +++ b/nodedb/src/engine/graph/edge_store/mod.rs @@ -15,7 +15,7 @@ pub use node_identity::NodeSurrogateRecord; pub use stats::CollectionStats; pub use store::{Direction, Edge, EdgeRecord, EdgeStore}; pub use temporal::{ - EdgeRef, EdgeValuePayload, GDPR_ERASURE_SENTINEL, NeighborsAsOfParams, SYSTEM_TIME_WIDTH, - TOMBSTONE_SENTINEL, edge_version_prefix, is_gdpr_erasure, is_sentinel, is_tombstone, - parse_versioned_edge_key, versioned_edge_key, + EdgeCountChange, EdgeRef, EdgeValuePayload, EdgeVersionWrite, GDPR_ERASURE_SENTINEL, + NeighborsAsOfParams, SYSTEM_TIME_WIDTH, TOMBSTONE_SENTINEL, edge_version_prefix, + is_gdpr_erasure, is_sentinel, is_tombstone, parse_versioned_edge_key, versioned_edge_key, }; diff --git a/nodedb/src/engine/graph/edge_store/node_identity.rs b/nodedb/src/engine/graph/edge_store/node_identity.rs index fe20f737d..439bd63c9 100644 --- a/nodedb/src/engine/graph/edge_store/node_identity.rs +++ b/nodedb/src/engine/graph/edge_store/node_identity.rs @@ -86,6 +86,33 @@ impl EdgeStore { write_txn.commit().map_err(|e| redb_err("commit", e))?; Ok(()) } + + /// Bind `node` to the surrogate `raw`, replacing any binding it had. + /// + /// A rolled-back node delete uses it to put back the binding the delete + /// dropped. + pub fn bind_node_surrogate( + &self, + db: DatabaseId, + tid: TenantId, + node: &str, + raw: u32, + ) -> crate::Result<()> { + let write_txn = self + .db + .begin_write() + .map_err(|e| redb_err("begin_write", e))?; + { + let mut table = write_txn + .open_table(NODE_SURROGATES) + .map_err(|e| redb_err("open node_surrogates", e))?; + table + .insert((db.as_u64(), tid.as_u64(), node), raw) + .map_err(|e| redb_err("insert node surrogate", e))?; + } + write_txn.commit().map_err(|e| redb_err("commit", e))?; + Ok(()) + } } #[cfg(test)] diff --git a/nodedb/src/engine/graph/edge_store/stats/update.rs b/nodedb/src/engine/graph/edge_store/stats/update.rs index d2710deee..7fbe7f783 100644 --- a/nodedb/src/engine/graph/edge_store/stats/update.rs +++ b/nodedb/src/engine/graph/edge_store/stats/update.rs @@ -47,11 +47,23 @@ pub struct EdgeStatsKey<'a> { /// `node[dst].refcount` (bumping `summary.distinct_node_count` per new node). /// /// If a prior live version exists this is an update — counters are unchanged. +/// +/// Returns whether the counters changed. pub fn increment_for_insert( write_txn: &WriteTransaction, key: EdgeStatsKey<'_>, current_system_from: i64, -) -> crate::Result<()> { +) -> crate::Result { + if prior_live_exists(write_txn, key, current_system_from)? { + return Ok(false); + } + increment_counts(write_txn, key)?; + Ok(true) +} + +/// Count one more live edge for `key`: the summary, its label and both +/// endpoints. +pub fn increment_counts(write_txn: &WriteTransaction, key: EdgeStatsKey<'_>) -> crate::Result<()> { let EdgeStatsKey { db, tid, @@ -60,10 +72,6 @@ pub fn increment_for_insert( src, dst, } = key; - if prior_live_exists(write_txn, key, current_system_from)? { - return Ok(()); - } - let mut stats = write_txn .open_table(GRAPH_STATS) .map_err(|e| redb_err("open graph_stats (increment)", e))?; @@ -120,11 +128,23 @@ pub fn increment_for_insert( /// decrementing `summary.distinct_node_count` when refcount reaches zero). /// /// If there was no prior live version, this is a no-op for the counters. +/// +/// Returns whether the counters changed. pub fn decrement_for_delete( write_txn: &WriteTransaction, key: EdgeStatsKey<'_>, sentinel_system_from: i64, -) -> crate::Result<()> { +) -> crate::Result { + if !prior_live_exists(write_txn, key, sentinel_system_from)? { + return Ok(false); + } + decrement_counts(write_txn, key)?; + Ok(true) +} + +/// Count one live edge fewer for `key`. The exact inverse of +/// [`increment_counts`]. +pub fn decrement_counts(write_txn: &WriteTransaction, key: EdgeStatsKey<'_>) -> crate::Result<()> { let EdgeStatsKey { db, tid, @@ -133,10 +153,6 @@ pub fn decrement_for_delete( src, dst, } = key; - if !prior_live_exists(write_txn, key, sentinel_system_from)? { - return Ok(()); - } - let mut stats = write_txn .open_table(GRAPH_STATS) .map_err(|e| redb_err("open graph_stats (decrement)", e))?; diff --git a/nodedb/src/engine/graph/edge_store/temporal/mod.rs b/nodedb/src/engine/graph/edge_store/temporal/mod.rs index 2f1595a6b..2580f7204 100644 --- a/nodedb/src/engine/graph/edge_store/temporal/mod.rs +++ b/nodedb/src/engine/graph/edge_store/temporal/mod.rs @@ -25,6 +25,7 @@ pub mod payload; pub mod purge; pub mod query; pub mod read; +pub mod revert; pub mod write; pub use keys::{ @@ -33,3 +34,4 @@ pub use keys::{ }; pub use payload::EdgeValuePayload; pub use query::NeighborsAsOfParams; +pub use revert::{EdgeCountChange, EdgeVersionWrite}; diff --git a/nodedb/src/engine/graph/edge_store/temporal/revert.rs b/nodedb/src/engine/graph/edge_store/temporal/revert.rs new file mode 100644 index 000000000..6ed951e84 --- /dev/null +++ b/nodedb/src/engine/graph/edge_store/temporal/revert.rs @@ -0,0 +1,343 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Removal of one versioned edge write, for a rollback. +//! +//! A write adds a version at its `system_from`. A rollback removes that +//! version, so no read at any system time sees it. It also restores what the +//! write changed beside the version: the edge counters, the zero summary row +//! and the endpoint identity bindings. + +use crate::engine::graph::edge_store::stats::table::{GRAPH_STATS, summary_key}; +use crate::engine::graph::edge_store::stats::update::{ + EdgeStatsKey, decrement_counts, increment_counts, +}; +use crate::engine::graph::edge_store::store::{ + EDGES, EdgeStore, NODE_SURROGATES, REVERSE_EDGES, redb_err, +}; + +use super::keys::{EdgeRef, versioned_edge_key}; + +/// How one write changed the edge counters. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum EdgeCountChange { + /// The write made the edge live: it counted one more edge. + Added, + /// The write ended a live edge: it counted one edge fewer. + Removed, +} + +/// What one versioned edge write changed. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct EdgeVersionWrite { + /// The system-time key of the version the write added. + pub system_from: i64, + /// The forward value the key held before the write, `None` when absent. + pub prior_forward: Option>, + /// The reverse value the key held before the write, `None` when absent. + pub prior_reverse: Option>, + /// The counter change the write made, `None` when it made none. + pub counted: Option, + /// Whether the write created the collection's zero summary row. + pub summary_created: bool, + /// Every endpoint binding the write set, with the binding it held before. + pub prior_bindings: Vec<(String, Option)>, +} + +impl EdgeVersionWrite { + /// A write at `system_from` that changed nothing yet. + pub(super) fn at(system_from: i64) -> Self { + Self { + system_from, + prior_forward: None, + prior_reverse: None, + counted: None, + summary_created: false, + prior_bindings: Vec::new(), + } + } +} + +impl EdgeStore { + /// Remove the version `written` describes and restore what the write + /// changed beside it, in one transaction. + /// + /// The caller removes later writes to the same edge first. Rollback runs + /// in reverse write order, so it does. + pub fn remove_edge_version( + &self, + edge: EdgeRef<'_>, + written: &EdgeVersionWrite, + ) -> crate::Result<()> { + let sys = written.system_from; + let fwd = versioned_edge_key(edge.collection, edge.src, edge.label, edge.dst, sys)?; + let rev = versioned_edge_key(edge.collection, edge.dst, edge.label, edge.src, sys)?; + let d = edge.db.as_u64(); + let t = edge.tid.as_u64(); + + let write_txn = self + .db + .begin_write() + .map_err(|e| redb_err("begin_write", e))?; + { + let mut edges = write_txn + .open_table(EDGES) + .map_err(|e| redb_err("open edges", e))?; + let restored = match &written.prior_forward { + Some(prior) => edges + .insert((d, t, fwd.as_str()), prior.as_slice()) + .map(drop), + None => edges.remove((d, t, fwd.as_str())).map(drop), + }; + restored.map_err(|e| redb_err("remove edge version", e))?; + drop(edges); + + let mut reverse = write_txn + .open_table(REVERSE_EDGES) + .map_err(|e| redb_err("open reverse", e))?; + let restored = match &written.prior_reverse { + Some(prior) => reverse + .insert((d, t, rev.as_str()), prior.as_slice()) + .map(drop), + None => reverse.remove((d, t, rev.as_str())).map(drop), + }; + restored.map_err(|e| redb_err("remove reverse edge version", e))?; + drop(reverse); + + let key = EdgeStatsKey { + db: d, + tid: t, + collection: edge.collection, + label: edge.label, + src: edge.src, + dst: edge.dst, + }; + match written.counted { + Some(EdgeCountChange::Added) => decrement_counts(&write_txn, key)?, + Some(EdgeCountChange::Removed) => increment_counts(&write_txn, key)?, + None => {} + } + if written.summary_created { + let summary = summary_key(edge.collection); + let mut stats = write_txn + .open_table(GRAPH_STATS) + .map_err(|e| redb_err("open graph_stats", e))?; + stats + .remove((d, t, summary.as_str())) + .map_err(|e| redb_err("remove zero graph summary", e))?; + } + + let mut surrogates = write_txn + .open_table(NODE_SURROGATES) + .map_err(|e| redb_err("open node_surrogates", e))?; + for (node, prior) in &written.prior_bindings { + let restored = match prior { + Some(raw) => surrogates.insert((d, t, node.as_str()), *raw).map(drop), + None => surrogates.remove((d, t, node.as_str())).map(drop), + }; + restored.map_err(|e| redb_err("restore node surrogate", e))?; + } + } + write_txn + .commit() + .map_err(|e| redb_err("commit edge version removal", e))?; + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::{DatabaseId, Surrogate, TenantId}; + use redb::{ReadableDatabase, ReadableTable}; + + use super::*; + use crate::engine::graph::edge_store::stats::table::CollectionStats; + + const T: TenantId = TenantId::new(1); + const DB: DatabaseId = DatabaseId::DEFAULT; + const COLL: &str = "people"; + + fn make_store() -> (EdgeStore, tempfile::TempDir) { + let dir = tempfile::tempdir().expect("tempdir"); + let store = EdgeStore::open(&dir.path().join("graph.redb")).expect("open edge store"); + (store, dir) + } + + fn e<'a>(src: &'a str, dst: &'a str) -> EdgeRef<'a> { + EdgeRef::new(DB, T, COLL, src, "L", dst) + } + + fn version_count(store: &EdgeStore) -> usize { + let txn = store.db.begin_read().expect("begin read"); + let edges = txn.open_table(EDGES).expect("open edges"); + let reverse = txn.open_table(REVERSE_EDGES).expect("open reverse"); + let forward = edges.iter().expect("iter edges").count(); + let backward = reverse.iter().expect("iter reverse").count(); + assert_eq!(forward, backward, "every version has its reverse entry"); + forward + } + + fn stats(store: &EdgeStore) -> CollectionStats { + store + .collection_stats(DB.as_u64(), T, COLL, None) + .expect("collection stats") + } + + #[test] + fn removing_an_inserted_version_leaves_no_trace_at_any_system_time() { + let (store, _dir) = make_store(); + store + .put_edge_versioned(e("a", "b"), b"v1", 100, 100, i64::MAX) + .expect("seed"); + let before = stats(&store); + + let written = store + .put_edge_version_recorded( + e("a", "b").with_surrogates(Surrogate::new(7), Surrogate::new(8)), + b"v2", + 200, + 200, + i64::MAX, + true, + ) + .expect("write"); + store + .remove_edge_version(e("a", "b"), &written) + .expect("remove"); + + assert_eq!(version_count(&store), 1, "only the seeded version remains"); + for as_of in [150, 200, 250, i64::MAX] { + assert_eq!( + store + .ceiling_resolve_edge(e("a", "b"), as_of, None) + .expect("read"), + Some(b"v1".to_vec()), + "the edge reads as seeded at system time {as_of}" + ); + } + assert_eq!(stats(&store), before); + assert!( + store + .scan_all_node_surrogates() + .expect("scan bindings") + .is_empty(), + "the bindings the write set are gone" + ); + } + + #[test] + fn removing_a_tombstone_restores_the_live_edge_and_its_counters() { + let (store, _dir) = make_store(); + store + .put_edge_versioned(e("a", "b"), b"v1", 100, 100, i64::MAX) + .expect("seed"); + let before = stats(&store); + + let written = store + .soft_delete_edge_recorded(e("a", "b"), 200, true) + .expect("tombstone"); + assert_eq!(written.counted, Some(EdgeCountChange::Removed)); + store + .remove_edge_version(e("a", "b"), &written) + .expect("remove"); + + assert_eq!(version_count(&store), 1); + assert_eq!( + store + .ceiling_resolve_edge(e("a", "b"), 300, None) + .expect("read"), + Some(b"v1".to_vec()) + ); + assert_eq!(stats(&store), before); + } + + #[test] + fn removing_a_first_insert_uncounts_the_edge() { + let (store, _dir) = make_store(); + let written = store + .put_edge_version_recorded(e("a", "b"), b"v1", 100, 100, i64::MAX, true) + .expect("write"); + assert_eq!(written.counted, Some(EdgeCountChange::Added)); + store + .remove_edge_version(e("a", "b"), &written) + .expect("remove"); + + assert_eq!(version_count(&store), 0); + let after = stats(&store); + assert_eq!(after.edge_count, 0); + assert_eq!(after.distinct_node_count, 0); + assert_eq!(after.distinct_label_count, 0); + } + + #[test] + fn removing_a_version_that_replaced_one_at_the_same_key_restores_it() { + let (store, _dir) = make_store(); + store + .put_edge_versioned(e("a", "b"), b"v1", 100, 100, i64::MAX) + .expect("seed"); + let written = store + .put_edge_version_recorded(e("a", "b"), b"v2", 100, 100, i64::MAX, true) + .expect("overwrite"); + store + .remove_edge_version(e("a", "b"), &written) + .expect("remove"); + + assert_eq!( + store + .ceiling_resolve_edge(e("a", "b"), 100, None) + .expect("read"), + Some(b"v1".to_vec()) + ); + } + + #[test] + fn removing_a_write_restores_the_binding_it_replaced() { + let (store, _dir) = make_store(); + store + .put_edge_versioned( + e("a", "b").with_surrogates(Surrogate::new(1), Surrogate::new(2)), + b"v1", + 100, + 100, + i64::MAX, + ) + .expect("seed"); + let written = store + .put_edge_version_recorded( + e("a", "c").with_surrogates(Surrogate::new(9), Surrogate::new(3)), + b"v1", + 200, + 200, + i64::MAX, + true, + ) + .expect("write"); + store + .remove_edge_version(e("a", "c"), &written) + .expect("remove"); + + let mut bindings: Vec<(String, u32)> = store + .scan_all_node_surrogates() + .expect("scan bindings") + .into_iter() + .map(|record| (record.2, record.3)) + .collect(); + bindings.sort(); + assert_eq!(bindings, vec![("a".to_string(), 1), ("b".to_string(), 2)]); + } + + #[test] + fn removing_a_destination_replica_write_drops_the_zero_summary_it_created() { + let (store, _dir) = make_store(); + let written = store + .put_edge_version_recorded(e("a", "b"), b"v1", 100, 100, i64::MAX, false) + .expect("write"); + assert!(written.summary_created); + store + .remove_edge_version(e("a", "b"), &written) + .expect("remove"); + + let txn = store.db.begin_read().expect("begin read"); + let table = txn.open_table(GRAPH_STATS).expect("open stats"); + assert_eq!(table.iter().expect("iter stats").count(), 0); + } +} diff --git a/nodedb/src/engine/graph/edge_store/temporal/write.rs b/nodedb/src/engine/graph/edge_store/temporal/write.rs index 2faa38e02..aa358a8c6 100644 --- a/nodedb/src/engine/graph/edge_store/temporal/write.rs +++ b/nodedb/src/engine/graph/edge_store/temporal/write.rs @@ -3,13 +3,14 @@ //! Bitemporal write paths on `EdgeStore`: //! `put_edge_versioned`, `soft_delete_edge`, `gdpr_erase_edge`. -use redb::{ReadableDatabase, ReadableTable}; +use redb::{ReadableDatabase, ReadableTable, WriteTransaction}; use super::keys::{ EdgeRef, GDPR_ERASURE_SENTINEL, TOMBSTONE_SENTINEL, edge_version_prefix, is_sentinel, versioned_edge_key, }; use super::payload::EdgeValuePayload; +use super::revert::{EdgeCountChange, EdgeVersionWrite}; use crate::engine::graph::edge_store::stats::table::{GRAPH_STATS, SummaryRow, summary_key}; use crate::engine::graph::edge_store::stats::update::{ EdgeStatsKey, decrement_for_delete, increment_for_insert, @@ -57,6 +58,29 @@ impl EdgeStore { valid_until_ms: i64, account_stats: bool, ) -> crate::Result<()> { + self.put_edge_version_recorded( + edge, + properties, + system_from, + valid_from_ms, + valid_until_ms, + account_stats, + ) + .map(drop) + } + + /// Write a version like [`Self::put_edge_versioned_with_stats`] and + /// return what it changed, so [`Self::remove_edge_version`] can remove it. + pub fn put_edge_version_recorded( + &self, + edge: EdgeRef<'_>, + properties: &[u8], + system_from: i64, + valid_from_ms: i64, + valid_until_ms: i64, + account_stats: bool, + ) -> crate::Result { + let mut written = EdgeVersionWrite::at(system_from); let fwd = versioned_edge_key(edge.collection, edge.src, edge.label, edge.dst, system_from)?; let rev = versioned_edge_key(edge.collection, edge.dst, edge.label, edge.src, system_from)?; let payload = @@ -72,21 +96,23 @@ impl EdgeStore { let mut edges = write_txn .open_table(EDGES) .map_err(|e| redb_err("open edges", e))?; - edges + written.prior_forward = edges .insert((d, t, fwd.as_str()), payload.as_slice()) - .map_err(|e| redb_err("insert versioned edge", e))?; + .map_err(|e| redb_err("insert versioned edge", e))? + .map(|prior| prior.value().to_vec()); drop(edges); let mut rev_t = write_txn .open_table(REVERSE_EDGES) .map_err(|e| redb_err("open reverse", e))?; - rev_t + written.prior_reverse = rev_t .insert((d, t, rev.as_str()), &[] as &[u8]) - .map_err(|e| redb_err("insert reverse", e))?; + .map_err(|e| redb_err("insert reverse", e))? + .map(|prior| prior.value().to_vec()); drop(rev_t); if account_stats { - increment_for_insert( + written.counted = increment_for_insert( &write_txn, EdgeStatsKey { db: d, @@ -97,7 +123,8 @@ impl EdgeStore { dst: edge.dst, }, system_from, - )?; + )? + .then_some(EdgeCountChange::Added); } else { // Suppress the lazy edge-scan rebuild on a destination-only // replica: absence of a summary means "legacy stats missing", @@ -118,6 +145,7 @@ impl EdgeStore { stats .insert((d, t, key.as_str()), zero.as_slice()) .map_err(|e| redb_err("insert zero graph summary", e))?; + written.summary_created = true; } } @@ -136,13 +164,19 @@ impl EdgeStore { if raw == 0 { continue; } - surrogates + let prior = surrogates .insert((d, t, node), raw) - .map_err(|e| redb_err("insert node surrogate", e))?; + .map_err(|e| redb_err("insert node surrogate", e))? + .map(|prior| prior.value()); + // A self-loop binds one node twice. Its first prior is the + // one the node held before this write. + if !written.prior_bindings.iter().any(|(seen, _)| seen == node) { + written.prior_bindings.push((node.to_string(), prior)); + } } } write_txn.commit().map_err(|e| redb_err("commit", e))?; - Ok(()) + Ok(written) } /// BiTemporalFK enforcement: close a referrer edge by appending a new @@ -212,6 +246,7 @@ impl EdgeStore { /// Append a tombstone version at `system_from`. pub fn soft_delete_edge(&self, edge: EdgeRef<'_>, system_from: i64) -> crate::Result<()> { self.write_sentinel(edge, system_from, TOMBSTONE_SENTINEL, true) + .map(drop) } /// Append a tombstone without double-decrementing global statistics on the @@ -222,6 +257,18 @@ impl EdgeStore { system_from: i64, account_stats: bool, ) -> crate::Result<()> { + self.soft_delete_edge_recorded(edge, system_from, account_stats) + .map(drop) + } + + /// Append a tombstone like [`Self::soft_delete_edge_with_stats`] and + /// return what it changed, so [`Self::remove_edge_version`] can remove it. + pub fn soft_delete_edge_recorded( + &self, + edge: EdgeRef<'_>, + system_from: i64, + account_stats: bool, + ) -> crate::Result { self.write_sentinel(edge, system_from, TOMBSTONE_SENTINEL, account_stats) } @@ -229,6 +276,7 @@ impl EdgeStore { /// can distinguish user-visible removal from regulatory erasure. pub fn gdpr_erase_edge(&self, edge: EdgeRef<'_>, system_from: i64) -> crate::Result<()> { self.write_sentinel(edge, system_from, GDPR_ERASURE_SENTINEL, true) + .map(drop) } fn write_sentinel( @@ -237,57 +285,72 @@ impl EdgeStore { system_from: i64, sentinel: &[u8], account_stats: bool, - ) -> crate::Result<()> { - debug_assert!( - is_sentinel(sentinel), - "write_sentinel called with non-sentinel bytes" - ); - let fwd = versioned_edge_key(edge.collection, edge.src, edge.label, edge.dst, system_from)?; - let rev = versioned_edge_key(edge.collection, edge.dst, edge.label, edge.src, system_from)?; - let d = edge.db.as_u64(); - let t = edge.tid.as_u64(); - + ) -> crate::Result { let write_txn = self .db .begin_write() .map_err(|e| redb_err("begin_write", e))?; - { - let mut edges = write_txn - .open_table(EDGES) - .map_err(|e| redb_err("open edges", e))?; - edges - .insert((d, t, fwd.as_str()), sentinel) - .map_err(|e| redb_err("insert sentinel edge", e))?; - drop(edges); - - let mut rev_t = write_txn - .open_table(REVERSE_EDGES) - .map_err(|e| redb_err("open reverse", e))?; - rev_t - .insert((d, t, rev.as_str()), sentinel) - .map_err(|e| redb_err("insert sentinel reverse", e))?; - drop(rev_t); - - if account_stats { - decrement_for_delete( - &write_txn, - EdgeStatsKey { - db: d, - tid: t, - collection: edge.collection, - label: edge.label, - src: edge.src, - dst: edge.dst, - }, - system_from, - )?; - } - } + let written = write_sentinel_in(&write_txn, edge, system_from, sentinel, account_stats)?; write_txn .commit() .map_err(|e| redb_err("commit sentinel", e))?; - Ok(()) + Ok(written) + } +} + +/// Append a sentinel version at `system_from` inside `write_txn`, and return +/// what it changed. The caller commits. +pub(in crate::engine::graph::edge_store) fn write_sentinel_in( + write_txn: &WriteTransaction, + edge: EdgeRef<'_>, + system_from: i64, + sentinel: &[u8], + account_stats: bool, +) -> crate::Result { + let mut written = EdgeVersionWrite::at(system_from); + debug_assert!( + is_sentinel(sentinel), + "write_sentinel called with non-sentinel bytes" + ); + let fwd = versioned_edge_key(edge.collection, edge.src, edge.label, edge.dst, system_from)?; + let rev = versioned_edge_key(edge.collection, edge.dst, edge.label, edge.src, system_from)?; + let d = edge.db.as_u64(); + let t = edge.tid.as_u64(); + + let mut edges = write_txn + .open_table(EDGES) + .map_err(|e| redb_err("open edges", e))?; + written.prior_forward = edges + .insert((d, t, fwd.as_str()), sentinel) + .map_err(|e| redb_err("insert sentinel edge", e))? + .map(|prior| prior.value().to_vec()); + drop(edges); + + let mut rev_t = write_txn + .open_table(REVERSE_EDGES) + .map_err(|e| redb_err("open reverse", e))?; + written.prior_reverse = rev_t + .insert((d, t, rev.as_str()), sentinel) + .map_err(|e| redb_err("insert sentinel reverse", e))? + .map(|prior| prior.value().to_vec()); + drop(rev_t); + + if account_stats { + written.counted = decrement_for_delete( + write_txn, + EdgeStatsKey { + db: d, + tid: t, + collection: edge.collection, + label: edge.label, + src: edge.src, + dst: edge.dst, + }, + system_from, + )? + .then_some(EdgeCountChange::Removed); } + Ok(written) } #[cfg(test)] From 185f7e2b6e5bf148e959ea71f665200c9765ec3b Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 09:04:43 +0800 Subject: [PATCH 18/64] feat(bridge): track a node-wide outcome floor for restart replay Introduce OutcomeFloor: the highest WAL LSN at or below which every record dispatched to a Data Plane core has a final outcome (applied, or refused with a durable WriteAborted marker). A write opens a window before it mints its LSN and settles it once the outcome is final; the dispatcher opens a window for every accepted request that carries a WAL LSN and settles it on final response, dead core, or abandoned drain. Every request pushed onto a core's ring now carries the floor read at enqueue time, and each core keeps the highest floor it has read via a new AppliedPrefix tracker, surfaced in checkpoint snapshot logging. Wire the funnel to open a window per control-plane write and abort undispatched writes whose dispatch was refused before settling. BridgeRequest and its constructors change shape throughout the test suite to carry the new field via BridgeRequest::unfloored() where no floor tracking is under test. --- nodedb-test-support/src/tx_batch_helpers.rs | 12 +- nodedb/src/bridge/dispatch/core_channel.rs | 11 +- nodedb/src/bridge/dispatch/dispatched_lsns.rs | 42 +++ nodedb/src/bridge/dispatch/dispatcher.rs | 30 ++ nodedb/src/bridge/dispatch/drain.rs | 5 +- nodedb/src/bridge/dispatch/enqueue.rs | 29 +- nodedb/src/bridge/dispatch/mod.rs | 3 + nodedb/src/bridge/dispatch/outcome_floor.rs | 328 ++++++++++++++++++ nodedb/src/bridge/dispatch/response_poll.rs | 98 +++++- .../submit_write/funnel/dispatch.rs | 15 +- .../submit_write/funnel/driver.rs | 46 ++- .../submit_write/funnel/response.rs | 21 ++ .../server/dispatch_utils/write_abort.rs | 29 ++ nodedb/src/control/state/fields.rs | 5 +- nodedb/src/control/state/init.rs | 1 + nodedb/src/control/state/init_prod/open.rs | 1 + .../src/data/executor/applied_prefix/mod.rs | 5 + .../data/executor/applied_prefix/tracker.rs | 52 +++ nodedb/src/data/executor/array_checkpoint.rs | 42 ++- .../core_loop/checkpoint_floors/init.rs | 3 + .../core_loop/checkpoint_floors/state.rs | 4 + .../executor/core_loop/pressure/fixtures.rs | 6 +- nodedb/src/data/executor/core_loop/tick.rs | 115 +++--- .../data/executor/dispatch/array/aggregate.rs | 4 +- .../executor/dispatch/array/elementwise.rs | 4 +- .../data/executor/dispatch/array/mutate.rs | 156 ++++----- .../src/data/executor/dispatch/array/read.rs | 4 +- .../executor/dispatch/array/surrogate_scan.rs | 4 +- .../executor/handlers/control/snapshot.rs | 1 + nodedb/src/data/executor/mod.rs | 1 + .../executor/timeseries_checkpoint/flush.rs | 42 ++- .../cases/calvin_determinism_contract.rs | 6 +- .../inproc/cases/calvin_executor_apply.rs | 6 +- .../cases/calvin_executor_panic_recovery.rs | 39 +-- .../inproc/cases/calvin_two_phase_apply.rs | 8 +- .../cases/cross_engine_bitmap_currency.rs | 14 +- .../cross_engine_three_way_fts_vector_doc.rs | 14 +- .../inproc/cases/executor_tests/helpers.rs | 20 +- .../test_cross_engine_validation.rs | 64 ++-- .../cases/executor_tests/test_document.rs | 6 +- .../inproc/cases/executor_tests/test_graph.rs | 32 +- .../test_graph_savepoint_overlay.rs | 2 +- .../inproc/cases/executor_tests/test_kv.rs | 6 +- .../executor_tests/test_kv_ttl_overlay.rs | 2 +- .../cases/executor_tests/test_transaction.rs | 6 +- .../cases/executor_tests/test_vector.rs | 26 +- .../inproc/cases/intake_throttle/helpers.rs | 2 +- .../inproc/cases/surrogate_round_trip.rs | 12 +- .../transaction_batch_cross_engine_crash.rs | 8 +- nodedb/tests/inproc/cases/wal_catchup.rs | 46 ++- 50 files changed, 1026 insertions(+), 412 deletions(-) create mode 100644 nodedb/src/bridge/dispatch/dispatched_lsns.rs create mode 100644 nodedb/src/bridge/dispatch/outcome_floor.rs create mode 100644 nodedb/src/data/executor/applied_prefix/mod.rs create mode 100644 nodedb/src/data/executor/applied_prefix/tracker.rs diff --git a/nodedb-test-support/src/tx_batch_helpers.rs b/nodedb-test-support/src/tx_batch_helpers.rs index 0230b6f37..b9ab4bec3 100644 --- a/nodedb-test-support/src/tx_batch_helpers.rs +++ b/nodedb-test-support/src/tx_batch_helpers.rs @@ -69,10 +69,8 @@ pub fn send_ok( rx: &mut Consumer, plan: PhysicalPlan, ) -> Vec { - tx.try_push(BridgeRequest { - inner: make_request(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_request(plan))) + .unwrap(); core.tick(); let resp = rx.try_pop().unwrap(); assert_eq!( @@ -90,10 +88,8 @@ pub fn send_raw( rx: &mut Consumer, plan: PhysicalPlan, ) -> nodedb::bridge::envelope::Response { - tx.try_push(BridgeRequest { - inner: make_request(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_request(plan))) + .unwrap(); core.tick(); rx.try_pop().unwrap().inner } diff --git a/nodedb/src/bridge/dispatch/core_channel.rs b/nodedb/src/bridge/dispatch/core_channel.rs index c462d4b04..e093d978b 100644 --- a/nodedb/src/bridge/dispatch/core_channel.rs +++ b/nodedb/src/bridge/dispatch/core_channel.rs @@ -11,6 +11,7 @@ use nodedb_bridge::wfq::WeightedFairQueue; use crate::bridge::envelope; use crate::data::eventfd::EventFdNotifier; +use crate::types::Lsn; use super::dispatcher::{BridgeRequest, BridgeResponse}; @@ -88,7 +89,10 @@ impl CoreChannel { /// /// The disconnect is also checked before the first pop, so a core known /// to be dead never has another request moved into a doomed `try_push`. - pub(super) fn flush_wfq(&mut self) -> usize { + /// + /// Every pushed request carries `outcome_floor`, the floor the caller read + /// just before this flush. + pub(super) fn flush_wfq(&mut self, outcome_floor: Lsn) -> usize { let mut flushed = 0; if self.request_tx.is_disconnected() { return 0; @@ -99,7 +103,10 @@ impl CoreChannel { }; let db_id = req.database_id.as_u64(); let req_id = req.request_id.as_u64(); - match self.request_tx.try_push(BridgeRequest { inner: req }) { + match self.request_tx.try_push(BridgeRequest { + inner: req, + outcome_floor, + }) { Ok(()) => { flushed += 1; self.update_db_pressure(db_id); diff --git a/nodedb/src/bridge/dispatch/dispatched_lsns.rs b/nodedb/src/bridge/dispatch/dispatched_lsns.rs new file mode 100644 index 000000000..708931ab3 --- /dev/null +++ b/nodedb/src/bridge/dispatch/dispatched_lsns.rs @@ -0,0 +1,42 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The dispatcher's hold on the outcome floor for every accepted request that +//! carries a WAL LSN. +//! +//! A request holds its window from the moment the dispatcher accepts it until +//! the core's final response arrives. A core that died, or a drain that gave +//! up on a core, also settles the window: that core never publishes a +//! watermark again. + +use std::collections::HashMap; +use std::sync::Arc; + +use crate::types::Lsn; + +use super::outcome_floor::{OutcomeFloor, WriteWindow}; + +/// Open windows of dispatched requests, by request id. +#[derive(Debug, Default)] +pub(super) struct DispatchedLsns { + windows: HashMap, +} + +impl DispatchedLsns { + /// Hold the floor below `lsn` until request `request_id` is answered. + pub(super) fn track(&mut self, floor: &Arc, request_id: u64, lsn: Lsn) { + self.windows.insert(request_id, floor.open_dispatched(lsn)); + } + + /// Release the hold of request `request_id`, if it has one. + pub(super) fn settle(&mut self, request_id: u64) { + if let Some(window) = self.windows.remove(&request_id) { + window.settle(); + } + } + + /// Number of requests holding the floor. + #[cfg(test)] + pub(super) fn len(&self) -> usize { + self.windows.len() + } +} diff --git a/nodedb/src/bridge/dispatch/dispatcher.rs b/nodedb/src/bridge/dispatch/dispatcher.rs index 44b1766b7..35981fd38 100644 --- a/nodedb/src/bridge/dispatch/dispatcher.rs +++ b/nodedb/src/bridge/dispatch/dispatcher.rs @@ -18,8 +18,11 @@ use tokio::sync::Notify; use crate::bridge::envelope; use crate::control::router::vshard::VShardRouter; use crate::data::eventfd::EventFdNotifier; +use crate::types::Lsn; use super::core_channel::{CoreChannel, CoreChannelDataSide}; +use super::dispatched_lsns::DispatchedLsns; +use super::outcome_floor::OutcomeFloor; /// Per-core request queue capacity of the server's bridge dispatcher. /// @@ -36,6 +39,20 @@ pub const DATA_PLANE_QUEUE_CAPACITY: usize = 1024; pub struct BridgeRequest { /// The full typed request envelope. pub inner: envelope::Request, + /// The outcome floor when this request entered the ring: every record at + /// or below it that any core receives has a final outcome. + pub outcome_floor: Lsn, +} + +impl BridgeRequest { + /// A request that carries no outcome floor. A core that reads it learns + /// nothing about the floor. + pub fn unfloored(inner: envelope::Request) -> Self { + Self { + inner, + outcome_floor: Lsn::ZERO, + } + } } /// Serialized form of a response coming back from the Data Plane. @@ -108,6 +125,12 @@ pub struct Dispatcher { /// Capacity freed on the bridge dispatcher. Every path that releases an /// in-flight slot wakes all waiters once per call. pub(super) capacity_freed: Arc, + + /// The node's outcome floor. Every ring push carries its current value. + pub(super) outcome_floor: Arc, + + /// The floor windows of accepted requests that carry a WAL LSN. + pub(super) dispatched_lsns: DispatchedLsns, } impl Dispatcher { @@ -162,6 +185,8 @@ impl Dispatcher { priority_resolver, data_plane_draining: false, capacity_freed: Arc::new(Notify::new()), + outcome_floor: OutcomeFloor::new(), + dispatched_lsns: DispatchedLsns::default(), }, data_sides, ) @@ -177,6 +202,11 @@ impl Dispatcher { Arc::clone(&self.capacity_freed) } + /// The node's outcome floor, shared with every write that opens a window. + pub fn outcome_floor(&self) -> Arc { + Arc::clone(&self.outcome_floor) + } + /// Maximum SPSC request queue utilization across all cores (0-100). pub fn max_utilization(&self) -> u8 { self.cores diff --git a/nodedb/src/bridge/dispatch/drain.rs b/nodedb/src/bridge/dispatch/drain.rs index c05cf3ed5..064ba8155 100644 --- a/nodedb/src/bridge/dispatch/drain.rs +++ b/nodedb/src/bridge/dispatch/drain.rs @@ -63,8 +63,9 @@ impl Dispatcher { /// Idempotent: a second call re-flushes and changes nothing else. pub fn begin_data_plane_drain(&mut self) { self.data_plane_draining = true; + let outcome_floor = self.outcome_floor.floor(); for channel in self.cores.iter_mut() { - channel.flush_wfq(); + channel.flush_wfq(outcome_floor); } } @@ -152,6 +153,8 @@ impl Dispatcher { "data plane drain deadline expired — failing the requests this core still holds" ); for rid in ids { + // The node is shutting down, and this core publishes nothing more. + self.dispatched_lsns.settle(rid); freed |= release_inflight_slot(&mut self.request_tenant, &mut self.tenant_inflight, rid); abandoned.push(envelope::Response { diff --git a/nodedb/src/bridge/dispatch/enqueue.rs b/nodedb/src/bridge/dispatch/enqueue.rs index 15992c6bf..ada1ef612 100644 --- a/nodedb/src/bridge/dispatch/enqueue.rs +++ b/nodedb/src/bridge/dispatch/enqueue.rs @@ -13,6 +13,7 @@ use tracing::warn; use crate::DispatchCapacityScope; use crate::bridge::admission_chokepoint::{assert_write_admitted, reject_uninjected_write}; use crate::bridge::envelope; +use crate::types::Lsn; use super::dispatcher::Dispatcher; use super::refusal::DispatchRefusal; @@ -42,6 +43,7 @@ impl Dispatcher { let tenant_id = request.tenant_id.as_u64(); let req_id = request.request_id.as_u64(); let database_id = request.database_id.as_u64(); + let wal_lsn = request.wal_lsn; // Per-tenant fairness: refuse while the tenant holds its in-flight cap. if self.max_per_tenant_inflight > 0 { @@ -96,7 +98,7 @@ impl Dispatcher { )); } - self.commit_enqueued(core_id, database_id, tenant_id, req_id); + self.commit_enqueued(core_id, database_id, tenant_id, req_id, wal_lsn); Ok(()) } @@ -121,6 +123,7 @@ impl Dispatcher { let tenant_id = request.tenant_id.as_u64(); let req_id = request.request_id.as_u64(); let database_id = request.database_id.as_u64(); + let wal_lsn = request.wal_lsn; let channel = &mut self.cores[core_id]; let cls = self.priority_resolver.priority_for(database_id); @@ -135,7 +138,7 @@ impl Dispatcher { } })?; - self.commit_enqueued(core_id, database_id, tenant_id, req_id); + self.commit_enqueued(core_id, database_id, tenant_id, req_id, wal_lsn); Ok(()) } @@ -148,16 +151,30 @@ impl Dispatcher { } /// Bookkeeping once a request sits in `core_id`'s weighted-fair queue: - /// flush it toward the ring, record pressure, track it as outstanding and - /// in flight for its tenant, and wake the core. - fn commit_enqueued(&mut self, core_id: usize, database_id: u64, tenant_id: u64, req_id: u64) { + /// hold the outcome floor below its WAL LSN, flush it toward the ring, + /// record pressure, track it as outstanding and in flight for its tenant, + /// and wake the core. + fn commit_enqueued( + &mut self, + core_id: usize, + database_id: u64, + tenant_id: u64, + req_id: u64, + wal_lsn: Option, + ) { + // The hold starts before the flush below, so no push carries a floor + // at or above this request's LSN while it is unanswered. + if let Some(lsn) = wal_lsn { + self.dispatched_lsns.track(&self.outcome_floor, req_id, lsn); + } + let outcome_floor = self.outcome_floor.floor(); let channel = &mut self.cores[core_id]; // Update per-DB pressure. channel.update_db_pressure(database_id); // Flush WFQ → physical ring. - channel.flush_wfq(); + channel.flush_wfq(outcome_floor); // Update global backpressure based on ring utilization. let util = channel.request_tx.utilization(); diff --git a/nodedb/src/bridge/dispatch/mod.rs b/nodedb/src/bridge/dispatch/mod.rs index ee2bfb1cc..9ddf0fca1 100644 --- a/nodedb/src/bridge/dispatch/mod.rs +++ b/nodedb/src/bridge/dispatch/mod.rs @@ -1,9 +1,11 @@ // SPDX-License-Identifier: BUSL-1.1 mod core_channel; +mod dispatched_lsns; mod dispatcher; mod drain; mod enqueue; +mod outcome_floor; mod refusal; mod response_poll; #[cfg(test)] @@ -15,4 +17,5 @@ pub use dispatcher::{ DefaultPriorityResolver, Dispatcher, }; pub use drain::CorePending; +pub use outcome_floor::{OutcomeFloor, WriteWindow}; pub use refusal::DispatchRefusal; diff --git a/nodedb/src/bridge/dispatch/outcome_floor.rs b/nodedb/src/bridge/dispatch/outcome_floor.rs new file mode 100644 index 000000000..312a381a7 --- /dev/null +++ b/nodedb/src/bridge/dispatch/outcome_floor.rs @@ -0,0 +1,328 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The outcome floor: the highest WAL LSN at or below which every record sent +//! to a Data Plane core has a final outcome. +//! +//! A record's outcome is final when the core applied it, or refused it and its +//! `WriteAborted` marker is durable. A watermark that says "restart may skip +//! every record at or below me" is sound only at or below this floor: a record +//! above it can still be on its way to a core, or still be applying. +//! +//! ## Windows +//! +//! A write opens a [`WriteWindow`] before it mints its LSN, notes each LSN it +//! mints, and settles the window once its outcome is final. The dispatcher +//! also opens a window for every accepted request that carries a WAL LSN, and +//! settles it when the core's final response arrives. A record minted outside +//! any window must never reach a core. +//! +//! Each open window has a horizon, a lower bound on every LSN it holds back: +//! +//! - A window opened before its mint takes `max_noted + 1`. WAL LSNs strictly +//! increase, so an LSN minted after the open exceeds every LSN noted before +//! it. +//! - A dispatcher window takes the request's LSN. +//! +//! ## The floor +//! +//! F is the smallest open horizon minus one, or the highest noted LSN when no +//! window is open. F never decreases: each computed value is raised to the +//! last published one. +//! +//! F never passes an open mint window: +//! +//! 1. Every computed value is at most `max_noted`, so every published F is at +//! most `max_noted`. +//! 2. A mint window opens with horizon `max_noted + 1`, above every F published +//! before it. +//! 3. While it stays open, every computed value is at most its horizon minus +//! one, so the published F stays below its horizon too. +//! +//! A dispatcher window whose LSN is at or below the published F cannot hold F +//! back. Its record was minted outside a window, and the floor passed it +//! before it reached the dispatcher. The open logs that record. +//! +//! ## Settling +//! +//! [`WriteWindow::settle`] states that the outcome is final. A window dropped +//! without it stays open for the rest of the process. Its record can still +//! need restart replay, so F must never pass it. The drop logs an error. + +use std::collections::{BTreeMap, HashMap}; +use std::sync::{Arc, Mutex, MutexGuard}; + +use tracing::{error, warn}; + +use crate::types::Lsn; + +/// The node's registry of open write windows. +#[derive(Debug, Default)] +pub struct OutcomeFloor { + windows: Mutex, +} + +#[derive(Debug, Default)] +struct Windows { + next_ticket: u64, + /// Horizon of each open window, by ticket. + open: HashMap, + /// Number of open windows at each horizon. + horizons: BTreeMap, + /// Highest LSN any window noted. + max_noted: u64, + /// Highest floor handed out. + published: u64, +} + +impl Windows { + fn open(&mut self, horizon: u64) -> u64 { + let ticket = self.next_ticket; + self.next_ticket += 1; + self.open.insert(ticket, horizon); + *self.horizons.entry(horizon).or_insert(0) += 1; + ticket + } + + fn close(&mut self, ticket: u64) { + let Some(horizon) = self.open.remove(&ticket) else { + return; + }; + if let Some(count) = self.horizons.get_mut(&horizon) { + *count -= 1; + if *count == 0 { + self.horizons.remove(&horizon); + } + } + } + + fn note(&mut self, lsn: u64) { + self.max_noted = self.max_noted.max(lsn); + } + + fn floor(&mut self) -> u64 { + let computed = match self.horizons.first_key_value() { + Some((&horizon, _)) => horizon.saturating_sub(1).min(self.max_noted), + None => self.max_noted, + }; + self.published = self.published.max(computed); + self.published + } +} + +impl OutcomeFloor { + /// An empty registry. Its floor starts at zero. + pub fn new() -> Arc { + Arc::new(Self::default()) + } + + fn lock(&self) -> MutexGuard<'_, Windows> { + self.windows.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Open a window before the write mints its LSN. + pub fn open_write(self: &Arc) -> WriteWindow { + let ticket = { + let mut windows = self.lock(); + let horizon = windows.max_noted.saturating_add(1); + windows.open(horizon) + }; + WriteWindow::new(Arc::clone(self), ticket) + } + + /// Open a window for a request that carries `lsn` as it enters the + /// dispatcher. + pub fn open_dispatched(self: &Arc, lsn: Lsn) -> WriteWindow { + let ticket = { + let mut windows = self.lock(); + if lsn.as_u64() <= windows.published { + warn!( + lsn = lsn.as_u64(), + floor = windows.published, + "a record reached the dispatcher after the outcome floor passed it; \ + it was minted outside a write window" + ); + } + windows.note(lsn.as_u64()); + windows.open(lsn.as_u64()) + }; + WriteWindow::new(Arc::clone(self), ticket) + } + + /// The current floor. Never lower than a value returned before. + pub fn floor(&self) -> Lsn { + Lsn::new(self.lock().floor()) + } +} + +/// One write's hold on the outcome floor. Settle it once the write's outcome +/// is final. Dropped unsettled, it holds the floor for the rest of the +/// process. +#[derive(Debug)] +pub struct WriteWindow { + owner: Arc, + ticket: u64, + settled: bool, +} + +impl WriteWindow { + fn new(owner: Arc, ticket: u64) -> Self { + Self { + owner, + ticket, + settled: false, + } + } + + /// Record an LSN this window minted. Call it before the window settles. + pub fn note_minted(&self, lsn: Lsn) { + self.owner.lock().note(lsn.as_u64()); + } + + /// Close the window: the write's outcome is final. + pub fn settle(mut self) { + self.settled = true; + self.owner.lock().close(self.ticket); + } +} + +impl Drop for WriteWindow { + fn drop(&mut self) { + if !self.settled { + error!( + ticket = self.ticket, + "a write window was dropped before its outcome was final; the outcome \ + floor stays below it until restart" + ); + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicU64, Ordering}; + + use super::*; + + #[test] + fn an_empty_registry_has_floor_zero() { + assert_eq!(OutcomeFloor::new().floor(), Lsn::ZERO); + } + + #[test] + fn the_floor_never_passes_an_open_window() { + let floor = OutcomeFloor::new(); + let early = floor.open_write(); + early.note_minted(Lsn::new(10)); + let late = floor.open_write(); + late.note_minted(Lsn::new(11)); + late.settle(); + assert!( + floor.floor() < Lsn::new(10), + "the open window at 10 holds F" + ); + early.settle(); + assert_eq!(floor.floor(), Lsn::new(11)); + } + + #[test] + fn the_floor_advances_when_windows_settle() { + let floor = OutcomeFloor::new(); + let first = floor.open_write(); + first.note_minted(Lsn::new(5)); + first.settle(); + assert_eq!(floor.floor(), Lsn::new(5)); + let second = floor.open_write(); + second.note_minted(Lsn::new(9)); + assert_eq!(floor.floor(), Lsn::new(5)); + second.settle(); + assert_eq!(floor.floor(), Lsn::new(9)); + } + + #[test] + fn a_window_opened_before_its_mint_holds_the_floor_below_the_mint() { + let floor = OutcomeFloor::new(); + let done = floor.open_write(); + done.note_minted(Lsn::new(3)); + done.settle(); + let pending = floor.open_write(); + assert_eq!(floor.floor(), Lsn::new(3)); + pending.note_minted(Lsn::new(4)); + assert_eq!(floor.floor(), Lsn::new(3)); + pending.settle(); + } + + #[test] + fn a_dropped_window_holds_the_floor() { + let floor = OutcomeFloor::new(); + let before = floor.open_write(); + before.note_minted(Lsn::new(6)); + before.settle(); + let lost = floor.open_write(); + lost.note_minted(Lsn::new(7)); + drop(lost); + let later = floor.open_write(); + later.note_minted(Lsn::new(8)); + later.settle(); + assert_eq!(floor.floor(), Lsn::new(6)); + } + + #[test] + fn a_dispatched_window_holds_the_floor_below_its_lsn() { + let floor = OutcomeFloor::new(); + let dispatched = floor.open_dispatched(Lsn::new(20)); + assert_eq!(floor.floor(), Lsn::new(19)); + dispatched.settle(); + assert_eq!(floor.floor(), Lsn::new(20)); + } + + #[test] + fn a_dispatched_window_below_the_floor_does_not_lower_it() { + let floor = OutcomeFloor::new(); + let window = floor.open_write(); + window.note_minted(Lsn::new(30)); + window.settle(); + assert_eq!(floor.floor(), Lsn::new(30)); + let late = floor.open_dispatched(Lsn::new(12)); + assert_eq!(floor.floor(), Lsn::new(30), "the floor never decreases"); + late.settle(); + } + + /// Writers mint from a shared counter the way the WAL does, each inside a + /// window. A reader samples the floor while they run. No sampled floor may + /// reach an LSN whose window was still open when the sample was taken. + #[test] + fn concurrent_writers_never_see_the_floor_pass_their_open_window() { + let floor = OutcomeFloor::new(); + let wal = Arc::new(AtomicU64::new(1)); + let writers: Vec<_> = (0..4) + .map(|_| { + let floor = Arc::clone(&floor); + let wal = Arc::clone(&wal); + std::thread::spawn(move || { + for _ in 0..500 { + let window = floor.open_write(); + let lsn = Lsn::new(wal.fetch_add(1, Ordering::SeqCst)); + window.note_minted(lsn); + let seen = floor.floor(); + assert!(seen < lsn, "floor {seen:?} passed open lsn {lsn:?}"); + window.settle(); + } + }) + }) + .collect(); + let mut last = Lsn::ZERO; + for _ in 0..2000 { + let seen = floor.floor(); + assert!( + seen >= last, + "the floor went back from {last:?} to {seen:?}" + ); + last = seen; + } + for writer in writers { + writer.join().expect("writer thread"); + } + let minted = wal.load(Ordering::SeqCst) - 1; + assert_eq!(floor.floor(), Lsn::new(minted), "every window settled"); + } +} diff --git a/nodedb/src/bridge/dispatch/response_poll.rs b/nodedb/src/bridge/dispatch/response_poll.rs index 8539eaec1..a999c5007 100644 --- a/nodedb/src/bridge/dispatch/response_poll.rs +++ b/nodedb/src/bridge/dispatch/response_poll.rs @@ -42,6 +42,7 @@ impl Dispatcher { // tenant's in-flight slot mid-stream. if !br.inner.partial { channel.outstanding.remove(&rid); + self.dispatched_lsns.settle(rid); freed |= release_inflight_slot( &mut self.request_tenant, &mut self.tenant_inflight, @@ -53,7 +54,7 @@ impl Dispatcher { if !(producer_gone || channel.request_tx.is_disconnected()) { // Opportunistically flush WFQ after draining responses to fill headroom. - channel.flush_wfq(); + channel.flush_wfq(self.outcome_floor.floor()); continue; } @@ -82,6 +83,8 @@ impl Dispatcher { // and emits nothing, so a permanently dead core costs one pass // over two empty containers rather than a repeating failure storm. for rid in lost { + // A dead core never publishes a watermark again. + self.dispatched_lsns.settle(rid); freed |= release_inflight_slot(&mut self.request_tenant, &mut self.tenant_inflight, rid); responses.push(envelope::Response { @@ -386,4 +389,97 @@ mod tests { assert!(r.error_code.is_some()); } } + + // --- Outcome floor --- + + fn ok_response(request_id: u64) -> BridgeResponse { + BridgeResponse { + inner: envelope::Response { + request_id: RequestId::new(request_id), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + } + } + + #[test] + fn a_dispatched_lsn_holds_the_outcome_floor_until_its_response() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let floor = dispatcher.outcome_floor(); + let mut request = make_request_for_db(0, 0, 5); + request.wal_lsn = Some(Lsn::new(40)); + dispatcher.dispatch(request).unwrap(); + assert_eq!(dispatcher.dispatched_lsns.len(), 1); + assert_eq!(floor.floor(), Lsn::new(39)); + + let pushed = data_sides[0].request_rx.try_pop().unwrap(); + assert!( + pushed.outcome_floor < Lsn::new(40), + "the request's own push carries a floor below its lsn" + ); + + data_sides[0].response_tx.try_push(ok_response(5)).unwrap(); + assert_eq!(dispatcher.poll_responses().len(), 1); + assert_eq!(dispatcher.dispatched_lsns.len(), 0); + assert_eq!(floor.floor(), Lsn::new(40)); + + dispatcher.dispatch(make_request_for_db(0, 0, 6)).unwrap(); + let next = data_sides[0].request_rx.try_pop().unwrap(); + assert_eq!( + next.outcome_floor, + Lsn::new(40), + "a push after the response carries the advanced floor" + ); + } + + #[test] + fn a_partial_response_keeps_the_outcome_floor_held() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let floor = dispatcher.outcome_floor(); + let mut request = make_request_for_db(0, 0, 8); + request.wal_lsn = Some(Lsn::new(12)); + dispatcher.dispatch(request).unwrap(); + let _req = data_sides[0].request_rx.try_pop().unwrap(); + + let mut partial = ok_response(8); + partial.inner.partial = true; + data_sides[0].response_tx.try_push(partial).unwrap(); + dispatcher.poll_responses(); + assert_eq!(floor.floor(), Lsn::new(11)); + + data_sides[0].response_tx.try_push(ok_response(8)).unwrap(); + dispatcher.poll_responses(); + assert_eq!(floor.floor(), Lsn::new(12)); + } + + #[test] + fn a_dead_core_releases_the_outcome_floor_it_held() { + let (mut dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let floor = dispatcher.outcome_floor(); + let mut request = make_request_for_db(0, 0, 3); + request.wal_lsn = Some(Lsn::new(25)); + dispatcher.dispatch(request).unwrap(); + assert_eq!(floor.floor(), Lsn::new(24)); + + drop(data_sides.remove(0)); + let responses = dispatcher.poll_responses(); + assert_eq!(responses.len(), 1); + assert_eq!(dispatcher.dispatched_lsns.len(), 0); + assert_eq!(floor.floor(), Lsn::new(25)); + } + + #[test] + fn a_request_without_an_lsn_holds_nothing() { + let (mut dispatcher, _data_sides) = Dispatcher::new(1, 64); + dispatcher.dispatch(make_request_for_db(0, 0, 4)).unwrap(); + assert_eq!(dispatcher.dispatched_lsns.len(), 0); + assert_eq!(dispatcher.outcome_floor().floor(), Lsn::ZERO); + } } diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs index e17e7a632..5c19844c6 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs @@ -90,14 +90,15 @@ pub(super) fn dispatch_to_data_plane( let rx = shared.tracker.register(request_id); - match shared.dispatcher.lock() { - Ok(mut d) => rollback_on_err(shared, ddl_transition, d.dispatch(request))?, - Err(poisoned) => rollback_on_err( - shared, - ddl_transition, - poisoned.into_inner().dispatch(request), - )?, + let dispatched = match shared.dispatcher.lock() { + Ok(mut d) => d.dispatch(request), + Err(poisoned) => poisoned.into_inner().dispatch(request), }; + if dispatched.is_err() { + // No response will ever arrive for a refused request. + shared.tracker.cancel(&request_id); + } + rollback_on_err(shared, ddl_transition, dispatched)?; // Release the write-admission guards immediately after the enqueue, before // the Data-Plane round-trip. The per-database WFQ is strict FIFO, so once LSN diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index e9ae4c244..877855271 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -5,6 +5,7 @@ use crate::control::server::dispatch_utils::change_events::extract_write_change_set; use crate::control::server::dispatch_utils::durability_barrier::funnel_minted_redo_engine; +use crate::control::server::dispatch_utils::write_abort::{AbortTarget, abort_undispatched_write}; use crate::control::server::shared::session::statement_deadline; use crate::control::server::shared::write_admission::{bare_ok_response, route_write_to_calvin}; use crate::control::server::wal_dispatch; @@ -13,7 +14,7 @@ use crate::control::state::SharedState; use super::super::params::{ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability}; use super::admission::{AdmissionOutcome, admit_write}; use super::dispatch::{DispatchTarget, dispatch_to_data_plane}; -use super::response::{ResponsePhaseInput, collect_classify_and_finish}; +use super::response::{ResponsePhaseInput, collect_classify_and_finish, settle_window}; use super::wal_append::authorize_and_append; /// Admit, make durable, enqueue, collect, and publish one write. @@ -119,17 +120,31 @@ pub(crate) async fn submit_write( } }; + // A write that mints its own LSN opens its outcome-floor window before the + // mint. The window settles once the outcome is final. + let window = appends_here.then(|| shared.outcome_floor.open_write()); + // Array DDL authorization + durability, under the admission guard, // immediately before the enqueue below. let wal_append_outcome = - authorize_and_append(shared, tenant_id, database_id, vshard_id, plan, durability)?; + match authorize_and_append(shared, tenant_id, database_id, vshard_id, plan, durability) { + Ok(outcome) => outcome, + Err(error) => { + // Nothing minted here reaches a core. + settle_window(window); + return Err(error); + } + }; let ddl_transition = wal_append_outcome.ddl_transition; let plan = wal_append_outcome.plan; let wal_lsn = wal_append_outcome.wal_lsn; let resolved_now_ms = wal_append_outcome.resolved_now_ms; + if let (Some(window), Some(lsn)) = (&window, wal_lsn) { + window.note_minted(lsn); + } // Build the wire request and hand it to the Data-Plane dispatcher. - let dispatch_outcome = dispatch_to_data_plane( + let dispatched = dispatch_to_data_plane( shared, &ddl_transition, DispatchTarget { @@ -149,7 +164,29 @@ pub(crate) async fn submit_write( admission_guard, order_guard, post_apply.is_some(), - )?; + ); + let dispatch_outcome = match dispatched { + Ok(outcome) => outcome, + Err(error) => { + // The dispatcher refused the request, so no core applied it. The + // record is cancelled before the window settles. A failed cancel + // leaves the window open: restart must still replay from here. + abort_undispatched_write( + shared, + AbortTarget { + tenant_id, + database_id, + vshard_id, + wal_lsn, + appends_here, + final_refusal_key: 0, + }, + ) + .await?; + settle_window(window); + return Err(error); + } + }; // Collect response(s), classify the outcome, and run the post-apply steps // a successful write still owes. @@ -174,6 +211,7 @@ pub(crate) async fn submit_write( change_set, ddl_transition, deferred_guards: dispatch_outcome.deferred_guards, + window, }, ) .await diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs index 8c10d1aa3..684328dfa 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -8,6 +8,7 @@ use std::time::Instant; use tokio::sync::mpsc; +use crate::bridge::dispatch::WriteWindow; use crate::bridge::envelope::{Response, Status}; use crate::control::array_catalog::ddl::AuthorizedDdlTransition; use crate::control::server::dispatch_utils::change_events::{WriteChangeSet, publish_change_set}; @@ -47,6 +48,15 @@ pub(super) struct ResponsePhaseInput { pub change_set: Option, pub ddl_transition: AuthorizedDdlTransition, pub deferred_guards: super::dispatch::DeferredGuards, + /// The write's outcome-floor window, when the funnel minted its LSN. + pub window: Option, +} + +/// Settle a write's outcome-floor window: its outcome is final. +pub(super) fn settle_window(window: Option) { + if let Some(window) = window { + window.settle(); + } } /// Collect the response(s), classify the outcome, and run every step a @@ -81,6 +91,7 @@ pub(super) async fn collect_classify_and_finish( change_set, ddl_transition, deferred_guards, + window, } = input; let vshard_u32 = vshard_id.as_u32(); @@ -100,6 +111,9 @@ pub(super) async fn collect_classify_and_finish( { Ok(response) => response, Err(_) => { + // The dispatcher holds the floor below this record until the core + // answers, so this window can settle. + settle_window(window); observe(shared); // Dispatch completed, but the Data Plane may have applied CREATE // or ALTER before this deadline. Never roll that catalog state @@ -115,6 +129,8 @@ pub(super) async fn collect_classify_and_finish( let response = match response { Ok(r) => r, Err(DispatchCollectError::OverBudget { bytes }) => { + // The dispatcher holds the floor until the core's final response. + settle_window(window); shared.tracker.cancel(&request_id); observe(shared); // A partial response proves dispatch began but not whether an @@ -131,6 +147,8 @@ pub(super) async fn collect_classify_and_finish( }); } Err(DispatchCollectError::ChannelClosed) => { + // The dispatcher holds the floor until the core's final response. + settle_window(window); observe(shared); // The producer can close after applying but before sending its // response. CREATE/ALTER must remain catalog-finalized here. @@ -166,6 +184,9 @@ pub(super) async fn collect_classify_and_finish( ) .await?; } + // The core's outcome is final, and a refusal's abort marker is durable. A + // failed abort above returns first and leaves the window open. + settle_window(window); // Mint the post-apply redo record while the guards are still held, then // release them. A PointUpdate whose collection carries a secondary vector diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index ddbf57f81..89e3f612e 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -99,6 +99,35 @@ pub(crate) async fn abort_refused_write( } else { 0 }; + append_abort_marker(shared, &target, wal_lsn, marker_key).await +} + +/// Cancel a forward write record whose request the dispatcher refused. +/// +/// The request never reached a core, so the Data Plane applied nothing. The +/// marker carries no proposal key: a dispatch refusal depends on this node's +/// load at that moment, so a redelivery of the same entry can apply it. +pub(crate) async fn abort_undispatched_write( + shared: &SharedState, + target: AbortTarget, +) -> crate::Result<()> { + if !target.appends_here { + return Ok(()); + } + let Some(wal_lsn) = target.wal_lsn else { + return Ok(()); + }; + append_abort_marker(shared, &target, wal_lsn, 0).await +} + +/// Append a `WriteAborted` marker naming `wal_lsn` and wait until it is +/// durable. +async fn append_abort_marker( + shared: &SharedState, + target: &AbortTarget, + wal_lsn: Lsn, + marker_key: u64, +) -> crate::Result<()> { let abort_lsn = shared.wal.appender(marker_key).append_write_aborted( target.tenant_id, target.vshard_id, diff --git a/nodedb/src/control/state/fields.rs b/nodedb/src/control/state/fields.rs index d66729a9d..d40afb94b 100644 --- a/nodedb/src/control/state/fields.rs +++ b/nodedb/src/control/state/fields.rs @@ -6,7 +6,7 @@ use std::sync::{Arc, Mutex, OnceLock, RwLock}; use nodedb_types::config::TuningConfig; use nodedb_types::protocol::Limits; -use crate::bridge::dispatch::Dispatcher; +use crate::bridge::dispatch::{Dispatcher, OutcomeFloor}; use crate::control::request_tracker::RequestTracker; use crate::control::security::apikey::ApiKeyStore; use crate::control::security::audit::AuditLog; @@ -27,6 +27,9 @@ pub(super) struct AsyncRaftProposerPair { pub struct SharedState { pub dispatcher: Mutex, + /// The node's outcome floor. Every write that mints a WAL LSN for the Data + /// Plane holds a window on it until its outcome is final. + pub outcome_floor: Arc, pub tracker: RequestTracker, pub wal: Arc, /// Collection-scoped scan quiesce registry for safe `PurgeCollection` reclaim. diff --git a/nodedb/src/control/state/init.rs b/nodedb/src/control/state/init.rs index 0184c9413..5d7d49db6 100644 --- a/nodedb/src/control/state/init.rs +++ b/nodedb/src/control/state/init.rs @@ -218,6 +218,7 @@ impl SharedState { let rate_limit_config = RateLimitConfig::default(); let state = Arc::new(Self { + outcome_floor: dispatcher.outcome_floor(), dispatcher: Mutex::new(dispatcher), tracker: RequestTracker::new(), wal, diff --git a/nodedb/src/control/state/init_prod/open.rs b/nodedb/src/control/state/init_prod/open.rs index 938a43cb0..6aad3157c 100644 --- a/nodedb/src/control/state/init_prod/open.rs +++ b/nodedb/src/control/state/init_prod/open.rs @@ -168,6 +168,7 @@ impl SharedState { )?; let state = Arc::new(Self { + outcome_floor: dispatcher.outcome_floor(), dispatcher: Mutex::new(dispatcher), tracker: RequestTracker::new(), wal, diff --git a/nodedb/src/data/executor/applied_prefix/mod.rs b/nodedb/src/data/executor/applied_prefix/mod.rs new file mode 100644 index 000000000..52611e21d --- /dev/null +++ b/nodedb/src/data/executor/applied_prefix/mod.rs @@ -0,0 +1,5 @@ +// SPDX-License-Identifier: BUSL-1.1 + +mod tracker; + +pub(in crate::data::executor) use tracker::AppliedPrefix; diff --git a/nodedb/src/data/executor/applied_prefix/tracker.rs b/nodedb/src/data/executor/applied_prefix/tracker.rs new file mode 100644 index 000000000..d94d8a855 --- /dev/null +++ b/nodedb/src/data/executor/applied_prefix/tracker.rs @@ -0,0 +1,52 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What this core knows about the applied prefix of the WAL. +//! +//! Every request from the Control Plane carries the node's outcome floor: every +//! record at or below it that any core receives has a final outcome. The core +//! keeps the highest floor it has read. + +use crate::types::Lsn; + +/// The core's view of the node's outcome floor. +#[derive(Debug)] +pub(in crate::data::executor) struct AppliedPrefix { + outcome_floor: Lsn, +} + +impl AppliedPrefix { + /// A core that has read no floor yet. + pub(in crate::data::executor) fn new() -> Self { + Self { + outcome_floor: Lsn::ZERO, + } + } + + /// Read the floor a request carried. The kept floor never decreases. + pub(in crate::data::executor) fn observe_outcome_floor(&mut self, floor: Lsn) { + if floor > self.outcome_floor { + self.outcome_floor = floor; + } + } + + /// The highest outcome floor this core has read. + pub(in crate::data::executor) fn outcome_floor(&self) -> Lsn { + self.outcome_floor + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_kept_floor_never_decreases() { + let mut prefix = AppliedPrefix::new(); + assert_eq!(prefix.outcome_floor(), Lsn::ZERO); + prefix.observe_outcome_floor(Lsn::new(9)); + prefix.observe_outcome_floor(Lsn::new(4)); + assert_eq!(prefix.outcome_floor(), Lsn::new(9)); + prefix.observe_outcome_floor(Lsn::new(12)); + assert_eq!(prefix.outcome_floor(), Lsn::new(12)); + } +} diff --git a/nodedb/src/data/executor/array_checkpoint.rs b/nodedb/src/data/executor/array_checkpoint.rs index aea6f9060..f689029f0 100644 --- a/nodedb/src/data/executor/array_checkpoint.rs +++ b/nodedb/src/data/executor/array_checkpoint.rs @@ -211,28 +211,26 @@ mod tests { let id = self.next_id; self.next_id += 1; self.req_tx - .try_push(BridgeRequest { - inner: Request { - request_id: RequestId::new(id), - tenant_id: TenantId::new(TID), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - plan: PhysicalPlan::Array(op), - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Admitted, - }, - }) + .try_push(BridgeRequest::unfloored(Request { + request_id: RequestId::new(id), + tenant_id: TenantId::new(TID), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan: PhysicalPlan::Array(op), + deadline: Instant::now() + Duration::from_secs(5), + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: crate::bridge::envelope::Admission::Admitted, + })) .expect("push request"); self.core.tick(); self.resp_rx.try_pop().expect("response").inner diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs index 322d76875..97186db1e 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs @@ -2,6 +2,7 @@ //! What a freshly-opened core's floors start at, and why each starts there. +use crate::data::executor::applied_prefix::AppliedPrefix; use crate::data::executor::replay_floors::ReplayFloors; use crate::types::Lsn; @@ -49,6 +50,8 @@ impl CheckpointFloors { crdt_durable_lsn: Lsn::ZERO, spatial_durable_lsn: Lsn::ZERO, replay_floors: ReplayFloors::default(), + // No request has arrived yet, so no floor has been read. + applied_prefix: AppliedPrefix::new(), } } } diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs index 482568022..b9956aea7 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs @@ -7,6 +7,7 @@ //! nothing may be claimed that was not actually put on stable storage. Failing to //! advance one of these costs WAL growth; overstating one costs data. +use crate::data::executor::applied_prefix::AppliedPrefix; use crate::data::executor::replay_floors::ReplayFloors; use crate::types::Lsn; @@ -182,4 +183,7 @@ pub(in crate::data::executor) struct CheckpointFloors { /// so records already folded into a restored checkpoint are not applied a /// second time. Empty outside boot, and empty means "replay everything". pub(in crate::data::executor) replay_floors: ReplayFloors, + + /// The node's outcome floor as this core last read it from a request. + pub(in crate::data::executor) applied_prefix: AppliedPrefix, } diff --git a/nodedb/src/data/executor/core_loop/pressure/fixtures.rs b/nodedb/src/data/executor/core_loop/pressure/fixtures.rs index cf3eca4fb..1035b0efd 100644 --- a/nodedb/src/data/executor/core_loop/pressure/fixtures.rs +++ b/nodedb/src/data/executor/core_loop/pressure/fixtures.rs @@ -96,10 +96,8 @@ pub(super) fn make_stub_request(id: u64) -> Request { /// Push one request onto the inbound ring. pub(super) fn push_request(tx: &mut Producer, id: u64) { - tx.try_push(BridgeRequest { - inner: make_stub_request(id), - }) - .expect("ring has room"); + tx.try_push(BridgeRequest::unfloored(make_stub_request(id))) + .expect("ring has room"); } /// A task wrapping [`make_stub_request`], for the per-handler pressure gate. diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index b89d7bf00..116a35a09 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -31,6 +31,9 @@ impl CoreLoop { // action for the core to take, so the flag is not acted on. let (_drained, _control_plane_gone) = self.request_rx.drain_into(&mut batch, depth); for br in batch { + self.floors + .applied_prefix + .observe_outcome_floor(br.outcome_floor); self.task_queue.push(ExecutionTask::new(br.inner)); } } @@ -236,20 +239,18 @@ mod tests { fn expired_task_returns_deadline_exceeded() { let (mut core, mut req_tx, mut resp_rx, _dir) = make_core(); req_tx - .try_push(BridgeRequest { - inner: Request { - deadline: Instant::now() - Duration::from_secs(1), - ..make_request(PhysicalPlan::Document(DocumentOp::PointGet { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), - document_id: "y".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - })) - }, - }) + .try_push(BridgeRequest::unfloored(Request { + deadline: Instant::now() - Duration::from_secs(1), + ..make_request(PhysicalPlan::Document(DocumentOp::PointGet { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), + document_id: "y".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + })) + })) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -269,8 +270,8 @@ mod tests { ); for _ in 0..2 { req_tx - .try_push(BridgeRequest { - inner: make_request(PhysicalPlan::Document(DocumentOp::PointGet { + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), document_id: "y".into(), surrogate: nodedb_types::Surrogate::ZERO, @@ -278,8 +279,8 @@ mod tests { rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, - })), - }) + }), + ))) .expect("queue request"); } @@ -312,8 +313,8 @@ mod tests { ) .unwrap(); req_tx - .try_push(BridgeRequest { - inner: make_request(PhysicalPlan::Document(DocumentOp::PointGet { + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), document_id: "y".into(), surrogate: nodedb_types::Surrogate::ZERO, @@ -321,8 +322,8 @@ mod tests { rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, - })), - }) + }), + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -333,36 +334,32 @@ mod tests { fn cancel_removes_pending_task() { let (mut core, mut req_tx, _resp_rx, _dir) = make_core(); req_tx - .try_push(BridgeRequest { - inner: Request { - request_id: RequestId::new(10), - deadline: Instant::now() + Duration::from_secs(60), - ..make_request(PhysicalPlan::Document(DocumentOp::PointGet { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), - document_id: "y".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - })) - }, - }) + .try_push(BridgeRequest::unfloored(Request { + request_id: RequestId::new(10), + deadline: Instant::now() + Duration::from_secs(60), + ..make_request(PhysicalPlan::Document(DocumentOp::PointGet { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "x"), + document_id: "y".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + })) + })) .unwrap(); core.drain_requests(); assert_eq!(core.pending_count(), 1); req_tx - .try_push(BridgeRequest { - inner: Request { - request_id: RequestId::new(99), - priority: Priority::Critical, - consistency: ReadConsistency::Eventual, - ..make_request(PhysicalPlan::Meta(MetaOp::Cancel { - target_request_id: RequestId::new(10), - })) - }, - }) + .try_push(BridgeRequest::unfloored(Request { + request_id: RequestId::new(99), + priority: Priority::Critical, + consistency: ReadConsistency::Eventual, + ..make_request(PhysicalPlan::Meta(MetaOp::Cancel { + target_request_id: RequestId::new(10), + })) + })) .unwrap(); // Cancel runs at Critical priority and is drained before the Normal-priority // target. The cancel removes id=10 from the queue, so only the Cancel itself @@ -387,8 +384,8 @@ mod tests { let tagged = zerompk::to_msgpack_vec(&nodedb_types::Value::Object(obj)).unwrap(); req_tx - .try_push(BridgeRequest { - inner: make_request(PhysicalPlan::Document(DocumentOp::PointPut { + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Document(DocumentOp::PointPut { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), document_id: "o1".into(), value: tagged, @@ -397,8 +394,8 @@ mod tests { returning: None, rls_filters: Vec::new(), resolved_sum_targets: Vec::new(), - })), - }) + }), + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -436,8 +433,8 @@ mod tests { ); let bytes = zerompk::to_msgpack_vec(&nodedb_types::Value::Object(obj)).unwrap(); req_tx - .try_push(BridgeRequest { - inner: make_request(PhysicalPlan::Document(DocumentOp::PointPut { + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Document(DocumentOp::PointPut { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "things"), document_id: format!("doc_{sur_val}"), value: bytes, @@ -446,8 +443,8 @@ mod tests { returning: None, rls_filters: Vec::new(), resolved_sum_targets: Vec::new(), - })), - }) + }), + ))) .unwrap(); core.tick(); let _ = resp_rx.try_pop().unwrap(); @@ -458,8 +455,8 @@ mod tests { // Issue a scan with the prefilter. req_tx - .try_push(BridgeRequest { - inner: make_request(PhysicalPlan::Document(DocumentOp::Scan { + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Document(DocumentOp::Scan { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "things"), limit: 100, offset: 0, @@ -472,8 +469,8 @@ mod tests { system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, prefilter: Some(prefilter), - })), - }) + }), + ))) .unwrap(); core.tick(); diff --git a/nodedb/src/data/executor/dispatch/array/aggregate.rs b/nodedb/src/data/executor/dispatch/array/aggregate.rs index 47cb17486..8dca43332 100644 --- a/nodedb/src/data/executor/dispatch/array/aggregate.rs +++ b/nodedb/src/data/executor/dispatch/array/aggregate.rs @@ -461,9 +461,7 @@ mod tests { let id = self.next_id; self.next_id += 1; self.req_tx - .try_push(BridgeRequest { - inner: make_request(plan, id), - }) + .try_push(BridgeRequest::unfloored(make_request(plan, id))) .unwrap(); self.core.tick(); let resp = self.resp_rx.try_pop().unwrap(); diff --git a/nodedb/src/data/executor/dispatch/array/elementwise.rs b/nodedb/src/data/executor/dispatch/array/elementwise.rs index 49ea5bc05..85f3a703a 100644 --- a/nodedb/src/data/executor/dispatch/array/elementwise.rs +++ b/nodedb/src/data/executor/dispatch/array/elementwise.rs @@ -394,9 +394,7 @@ mod tests { let id = self.next_id; self.next_id += 1; self.req_tx - .try_push(BridgeRequest { - inner: make_request(plan, id), - }) + .try_push(BridgeRequest::unfloored(make_request(plan, id))) .unwrap(); self.core.tick(); let resp = self.resp_rx.try_pop().unwrap(); diff --git a/nodedb/src/data/executor/dispatch/array/mutate.rs b/nodedb/src/data/executor/dispatch/array/mutate.rs index 217154a66..22eb04304 100644 --- a/nodedb/src/data/executor/dispatch/array/mutate.rs +++ b/nodedb/src/data/executor/dispatch/array/mutate.rs @@ -342,19 +342,17 @@ mod tests { // 1) OpenArray req_tx - .try_push(BridgeRequest { - inner: make_request( - PhysicalPlan::Array(ArrayOp::OpenArray { - array_id: aid.clone(), - schema_msgpack: schema_bytes.clone(), - schema_hash, - prefix_bits: 8, - audit_retain_ms: None, - minimum_audit_retain_ms: None, - }), - 1, - ), - }) + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Array(ArrayOp::OpenArray { + array_id: aid.clone(), + schema_msgpack: schema_bytes.clone(), + schema_hash, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + }), + 1, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -376,17 +374,15 @@ mod tests { }]; let cells_bytes = zerompk::to_msgpack_vec(&cells).unwrap(); req_tx - .try_push(BridgeRequest { - inner: make_request( - PhysicalPlan::Array(ArrayOp::Put { - array_id: aid.clone(), - cells_msgpack: cells_bytes, - wal_lsn: 42, - provenance: None, - }), - 2, - ), - }) + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Array(ArrayOp::Put { + array_id: aid.clone(), + cells_msgpack: cells_bytes, + wal_lsn: 42, + provenance: None, + }), + 2, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -399,15 +395,13 @@ mod tests { // 3) Flush req_tx - .try_push(BridgeRequest { - inner: make_request( - PhysicalPlan::Array(ArrayOp::Flush { - array_id: aid.clone(), - wal_lsn: 99, - }), - 3, - ), - }) + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Array(ArrayOp::Flush { + array_id: aid.clone(), + wal_lsn: 99, + }), + 3, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -459,19 +453,17 @@ mod tests { // 1) Open v1. req_tx - .try_push(BridgeRequest { - inner: make_request( - PhysicalPlan::Array(ArrayOp::OpenArray { - array_id: aid.clone(), - schema_msgpack: v1_bytes.clone(), - schema_hash: 0xAAAA, - prefix_bits: 8, - audit_retain_ms: None, - minimum_audit_retain_ms: None, - }), - 1, - ), - }) + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Array(ArrayOp::OpenArray { + array_id: aid.clone(), + schema_msgpack: v1_bytes.clone(), + schema_hash: 0xAAAA, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + }), + 1, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -488,17 +480,15 @@ mod tests { }]; let cells_bytes = zerompk::to_msgpack_vec(&cells).unwrap(); req_tx - .try_push(BridgeRequest { - inner: make_request( - PhysicalPlan::Array(ArrayOp::Put { - array_id: aid.clone(), - cells_msgpack: cells_bytes, - wal_lsn: 7, - provenance: None, - }), - 2, - ), - }) + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Array(ArrayOp::Put { + array_id: aid.clone(), + cells_msgpack: cells_bytes, + wal_lsn: 7, + provenance: None, + }), + 2, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -506,14 +496,12 @@ mod tests { // 3) DropArray — releases per-core store + on-disk segment dir. req_tx - .try_push(BridgeRequest { - inner: make_request( - PhysicalPlan::Array(ArrayOp::DropArray { - array_id: aid.clone(), - }), - 3, - ), - }) + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Array(ArrayOp::DropArray { + array_id: aid.clone(), + }), + 3, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -530,14 +518,12 @@ mod tests { // Finalization purges the reversible tombstone before recreation. req_tx - .try_push(BridgeRequest { - inner: make_request( - PhysicalPlan::Array(ArrayOp::PurgeArrayDrop { - array_id: aid.clone(), - }), - 4, - ), - }) + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Array(ArrayOp::PurgeArrayDrop { + array_id: aid.clone(), + }), + 4, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -547,19 +533,17 @@ mod tests { // would fail with `SchemaMismatch`. The finalized post-drop state // must accept the new hash. req_tx - .try_push(BridgeRequest { - inner: make_request( - PhysicalPlan::Array(ArrayOp::OpenArray { - array_id: aid.clone(), - schema_msgpack: v1_bytes, - schema_hash: 0xBBBB, - prefix_bits: 8, - audit_retain_ms: None, - minimum_audit_retain_ms: None, - }), - 5, - ), - }) + .try_push(BridgeRequest::unfloored(make_request( + PhysicalPlan::Array(ArrayOp::OpenArray { + array_id: aid.clone(), + schema_msgpack: v1_bytes, + schema_hash: 0xBBBB, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + }), + 5, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); diff --git a/nodedb/src/data/executor/dispatch/array/read.rs b/nodedb/src/data/executor/dispatch/array/read.rs index b47990525..86553e64b 100644 --- a/nodedb/src/data/executor/dispatch/array/read.rs +++ b/nodedb/src/data/executor/dispatch/array/read.rs @@ -585,9 +585,7 @@ mod tests { let id = self.next_id; self.next_id += 1; self.req_tx - .try_push(BridgeRequest { - inner: make_request(plan, id), - }) + .try_push(BridgeRequest::unfloored(make_request(plan, id))) .unwrap(); self.core.tick(); let resp = self.resp_rx.try_pop().unwrap(); diff --git a/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs b/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs index 94b8dcdd0..af78ecd01 100644 --- a/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs +++ b/nodedb/src/data/executor/dispatch/array/surrogate_scan.rs @@ -223,9 +223,7 @@ mod tests { let id = self.next_id; self.next_id += 1; self.req_tx - .try_push(BridgeRequest { - inner: make_request(plan, id), - }) + .try_push(BridgeRequest::unfloored(make_request(plan, id))) .unwrap(); self.core.tick(); let resp = self.resp_rx.try_pop().unwrap(); diff --git a/nodedb/src/data/executor/handlers/control/snapshot.rs b/nodedb/src/data/executor/handlers/control/snapshot.rs index add64f95a..9445b9595 100644 --- a/nodedb/src/data/executor/handlers/control/snapshot.rs +++ b/nodedb/src/data/executor/handlers/control/snapshot.rs @@ -473,6 +473,7 @@ impl CoreLoop { core = self.core_id, checkpoint_lsn, watermark = self.watermark.as_u64(), + outcome_floor = self.floors.applied_prefix.outcome_floor().as_u64(), kv_durable_lsn = self.floors.kv_durable_lsn.as_u64(), sparse_vector_durable_lsn = self.floors.sparse_vector_durable_lsn.as_u64(), sync_hwm_durable_lsn = self.floors.sync_hwm_durable_lsn.as_u64(), diff --git a/nodedb/src/data/executor/mod.rs b/nodedb/src/data/executor/mod.rs index 75a3ed728..c9c68ce13 100644 --- a/nodedb/src/data/executor/mod.rs +++ b/nodedb/src/data/executor/mod.rs @@ -1,5 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 +mod applied_prefix; pub(crate) mod array_checkpoint; pub(crate) mod checkpoint_decode_error; pub(crate) mod checkpoint_encoding; diff --git a/nodedb/src/data/executor/timeseries_checkpoint/flush.rs b/nodedb/src/data/executor/timeseries_checkpoint/flush.rs index ea2d5304f..d510d6187 100644 --- a/nodedb/src/data/executor/timeseries_checkpoint/flush.rs +++ b/nodedb/src/data/executor/timeseries_checkpoint/flush.rs @@ -145,28 +145,26 @@ mod tests { let id = self.next_id; self.next_id += 1; self.req_tx - .try_push(BridgeRequest { - inner: Request { - request_id: RequestId::new(id), - tenant_id: TenantId::new(TID), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - plan, - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: wal_lsn.map(crate::types::Lsn::new), - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Admitted, - }, - }) + .try_push(BridgeRequest::unfloored(Request { + request_id: RequestId::new(id), + tenant_id: TenantId::new(TID), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan, + deadline: Instant::now() + Duration::from_secs(5), + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: wal_lsn.map(crate::types::Lsn::new), + resolved_now_ms: None, + admission: crate::bridge::envelope::Admission::Admitted, + })) .expect("push request"); self.core.tick(); self.resp_rx.try_pop().expect("response").inner diff --git a/nodedb/tests/inproc/cases/calvin_determinism_contract.rs b/nodedb/tests/inproc/cases/calvin_determinism_contract.rs index eb7107cb3..3476d65e4 100644 --- a/nodedb/tests/inproc/cases/calvin_determinism_contract.rs +++ b/nodedb/tests/inproc/cases/calvin_determinism_contract.rs @@ -94,10 +94,8 @@ fn send_one( rx: &mut Consumer, plan: PhysicalPlan, ) -> nodedb::bridge::envelope::Response { - tx.try_push(BridgeRequest { - inner: make_request(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_request(plan))) + .unwrap(); core.tick(); rx.try_pop().unwrap().inner } diff --git a/nodedb/tests/inproc/cases/calvin_executor_apply.rs b/nodedb/tests/inproc/cases/calvin_executor_apply.rs index 5826bbbd4..9ed445679 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_apply.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_apply.rs @@ -69,10 +69,8 @@ fn send_raw( rx: &mut Consumer, plan: PhysicalPlan, ) -> nodedb::bridge::envelope::Response { - tx.try_push(BridgeRequest { - inner: make_request(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_request(plan))) + .unwrap(); core.tick(); rx.try_pop().unwrap().inner } diff --git a/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs b/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs index 92dc8d945..8ec994ca0 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs @@ -383,13 +383,9 @@ fn calvin_static_replay_sees_only_committed_data() { }; // Commit a value before the panic batch. - tx.try_push(BridgeRequest { - inner: make_req(tx_batch(vec![kv_put_in( - "replay_coll", - b"pre_commit", - b"alive", - )])), - }) + tx.try_push(BridgeRequest::unfloored(make_req(tx_batch(vec![ + kv_put_in("replay_coll", b"pre_commit", b"alive"), + ])))) .unwrap(); core.tick(); let pre_resp = rx.try_pop().unwrap().inner; @@ -404,15 +400,13 @@ fn calvin_static_replay_sees_only_committed_data() { // Ok), then flush (where the panic fires during the replay). let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); - tx.try_push(BridgeRequest { - inner: make_req(calvin_static( - 1, - vec![ - kv_put_in("replay_coll", b"should_not_exist", b"gone"), - kv_put_in("replay_coll", b"should_not_exist2", b"gone"), - ], - )), - }) + tx.try_push(BridgeRequest::unfloored(make_req(calvin_static( + 1, + vec![ + kv_put_in("replay_coll", b"should_not_exist", b"gone"), + kv_put_in("replay_coll", b"should_not_exist2", b"gone"), + ], + )))) .unwrap(); core.tick(); let stage_resp = rx.try_pop().unwrap().inner; @@ -423,10 +417,8 @@ fn calvin_static_replay_sees_only_committed_data() { stage_resp.error_code ); - tx.try_push(BridgeRequest { - inner: make_req(calvin_flush(1)), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_req(calvin_flush(1)))) + .unwrap(); core.tick(); let panic_resp = rx.try_pop().unwrap().inner; assert_eq!( @@ -498,9 +490,10 @@ fn calvin_static_replay_sees_only_committed_data() { // never let the bad writes reach durable storage. // The rolled-back key must not exist on the fresh core. - tx2.try_push(BridgeRequest { - inner: make_req2(kv_get_in("replay_coll", b"should_not_exist")), - }) + tx2.try_push(BridgeRequest::unfloored(make_req2(kv_get_in( + "replay_coll", + b"should_not_exist", + )))) .unwrap(); core2.tick(); let gone_get = rx2.try_pop().unwrap().inner; diff --git a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs index 28f040b40..ad6f9daa9 100644 --- a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs +++ b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs @@ -78,9 +78,9 @@ fn send( vshard: u32, wal_lsn: Option, ) -> Response { - tx.try_push(BridgeRequest { - inner: make_request(plan, vshard, wal_lsn), - }) + tx.try_push(BridgeRequest::unfloored(make_request( + plan, vshard, wal_lsn, + ))) .unwrap(); core.tick(); rx.try_pop().unwrap().inner @@ -719,7 +719,7 @@ fn send_request( rx: &mut Consumer, request: Request, ) -> Response { - tx.try_push(BridgeRequest { inner: request }).unwrap(); + tx.try_push(BridgeRequest::unfloored(request)).unwrap(); core.tick(); rx.try_pop().unwrap().inner } diff --git a/nodedb/tests/inproc/cases/cross_engine_bitmap_currency.rs b/nodedb/tests/inproc/cases/cross_engine_bitmap_currency.rs index 725b4ba88..f5c16e3c7 100644 --- a/nodedb/tests/inproc/cases/cross_engine_bitmap_currency.rs +++ b/nodedb/tests/inproc/cases/cross_engine_bitmap_currency.rs @@ -86,10 +86,8 @@ fn send_ok( rx: &mut Consumer, plan: PhysicalPlan, ) -> Vec { - tx.try_push(BridgeRequest { - inner: make_req(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_req(plan))) + .unwrap(); core.tick(); let resp = rx.try_pop().unwrap(); assert_eq!( @@ -193,8 +191,8 @@ fn fts_derived_bitmap_filters_vector_search() { (s3, [-1.0, 0.0, 0.1]), ]; for (surrogate, vec) in vectors { - tx.try_push(BridgeRequest { - inner: make_req(PhysicalPlan::Vector(VectorOp::Insert { + tx.try_push(BridgeRequest::unfloored(make_req(PhysicalPlan::Vector( + VectorOp::Insert { collection: nodedb_types::QualifiedCollection::new( nodedb_types::DatabaseId::DEFAULT, "articles", @@ -205,8 +203,8 @@ fn fts_derived_bitmap_filters_vector_search() { surrogate: *surrogate, pk_bytes: None, provenance: None, - })), - }) + }, + )))) .unwrap(); } core.tick(); diff --git a/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs b/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs index 27e9cd469..26f28589c 100644 --- a/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs +++ b/nodedb/tests/inproc/cases/cross_engine_three_way_fts_vector_doc.rs @@ -88,10 +88,8 @@ fn send_ok( rx: &mut Consumer, plan: PhysicalPlan, ) -> Vec { - tx.try_push(BridgeRequest { - inner: make_req(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_req(plan))) + .unwrap(); core.tick(); let resp = rx.try_pop().unwrap(); assert_eq!( @@ -235,8 +233,8 @@ fn three_way_fts_vector_doc_bitmap() { // Insert vector embeddings for all 10 rows. for &s in LEARNING_SURS.iter().chain(NON_LEARNING_SURS) { - tx.try_push(BridgeRequest { - inner: make_req(PhysicalPlan::Vector(VectorOp::Insert { + tx.try_push(BridgeRequest::unfloored(make_req(PhysicalPlan::Vector( + VectorOp::Insert { collection: nodedb_types::QualifiedCollection::new( nodedb_types::DatabaseId::DEFAULT, COLLECTION, @@ -251,8 +249,8 @@ fn three_way_fts_vector_doc_bitmap() { surrogate: Surrogate::new(s), pk_bytes: None, provenance: None, - })), - }) + }, + )))) .unwrap(); } core.tick(); diff --git a/nodedb/tests/inproc/cases/executor_tests/helpers.rs b/nodedb/tests/inproc/cases/executor_tests/helpers.rs index c3e60bcc8..36d723739 100644 --- a/nodedb/tests/inproc/cases/executor_tests/helpers.rs +++ b/nodedb/tests/inproc/cases/executor_tests/helpers.rs @@ -112,9 +112,7 @@ pub fn send_ok( plan: PhysicalPlan, ) -> Vec { req_tx - .try_push(BridgeRequest { - inner: make_request(plan), - }) + .try_push(BridgeRequest::unfloored(make_request(plan))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -135,9 +133,7 @@ pub fn send_raw( plan: PhysicalPlan, ) -> nodedb::bridge::envelope::Response { req_tx - .try_push(BridgeRequest { - inner: make_request(plan), - }) + .try_push(BridgeRequest::unfloored(make_request(plan))) .unwrap(); core.tick(); resp_rx.try_pop().unwrap().inner @@ -176,9 +172,9 @@ pub fn send_ok_as_tenant( plan: PhysicalPlan, ) -> Vec { req_tx - .try_push(BridgeRequest { - inner: make_request_for_tenant(tenant_id, plan), - }) + .try_push(BridgeRequest::unfloored(make_request_for_tenant( + tenant_id, plan, + ))) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); @@ -200,9 +196,9 @@ pub fn send_raw_as_tenant( plan: PhysicalPlan, ) -> nodedb::bridge::envelope::Response { req_tx - .try_push(BridgeRequest { - inner: make_request_for_tenant(tenant_id, plan), - }) + .try_push(BridgeRequest::unfloored(make_request_for_tenant( + tenant_id, plan, + ))) .unwrap(); core.tick(); resp_rx.try_pop().unwrap().inner diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_engine_validation.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_engine_validation.rs index 40ba6d91a..a84d79d29 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_engine_validation.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_engine_validation.rs @@ -51,23 +51,21 @@ fn cross_model_query_vector_graph_relational() { // 2. Insert vectors for each document. for i in 0..10u32 { - tx.try_push(BridgeRequest { - inner: make_request_with_id( - 100 + i as u64, - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "papers", - ), - vector: vec![i as f32, (i as f32).sin(), (i as f32).cos()], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ), - }) + tx.try_push(BridgeRequest::unfloored(make_request_with_id( + 100 + i as u64, + PhysicalPlan::Vector(VectorOp::Insert { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "papers", + ), + vector: vec![i as f32, (i as f32).sin(), (i as f32).cos()], + dim: 3, + field_name: String::new(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: None, + provenance: None, + }), + ))) .unwrap(); } core.tick(); @@ -257,23 +255,21 @@ fn rrf_fusion_mathematically_correct() { // Insert vectors. for i in 0..20u32 { - tx.try_push(BridgeRequest { - inner: make_request_with_id( - 200 + i as u64, - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "docs", - ), - vector: vec![i as f32, 0.0, 0.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ), - }) + tx.try_push(BridgeRequest::unfloored(make_request_with_id( + 200 + i as u64, + PhysicalPlan::Vector(VectorOp::Insert { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "docs", + ), + vector: vec![i as f32, 0.0, 0.0], + dim: 3, + field_name: String::new(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: None, + provenance: None, + }), + ))) .unwrap(); } core.tick(); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_document.rs b/nodedb/tests/inproc/cases/executor_tests/test_document.rs index 12ced9fd6..b7cfcd46d 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_document.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_document.rs @@ -46,10 +46,8 @@ fn run_scan_and_count( rx: &mut Consumer, plan: PhysicalPlan, ) -> (usize, Status, Option) { - tx.try_push(BridgeRequest { - inner: make_request(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_request(plan))) + .unwrap(); core.tick(); let mut total = 0usize; diff --git a/nodedb/tests/inproc/cases/executor_tests/test_graph.rs b/nodedb/tests/inproc/cases/executor_tests/test_graph.rs index 5adf5fa9f..ae725a550 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_graph.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_graph.rs @@ -216,23 +216,21 @@ fn graph_rag_fusion_pipeline() { // Insert vectors. for i in 0..10u32 { - tx.try_push(BridgeRequest { - inner: make_request_with_id( - 100 + i as u64, - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "docs", - ), - vector: vec![i as f32, 0.0, 0.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ), - }) + tx.try_push(BridgeRequest::unfloored(make_request_with_id( + 100 + i as u64, + PhysicalPlan::Vector(VectorOp::Insert { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "docs", + ), + vector: vec![i as f32, 0.0, 0.0], + dim: 3, + field_name: String::new(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: None, + provenance: None, + }), + ))) .unwrap(); } core.tick(); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_graph_savepoint_overlay.rs b/nodedb/tests/inproc/cases/executor_tests/test_graph_savepoint_overlay.rs index 205c38b4b..0415b8656 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_graph_savepoint_overlay.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_graph_savepoint_overlay.rs @@ -40,7 +40,7 @@ fn send_txn( ..make_request(plan) }; req_tx - .try_push(nodedb::bridge::dispatch::BridgeRequest { inner: request }) + .try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored(request)) .unwrap(); core.tick(); resp_rx.try_pop().unwrap().inner diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv.rs index 0c2c68a7d..39577ca63 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv.rs @@ -529,7 +529,7 @@ fn kv_tenant_isolation() { rls_filters: Vec::new(), })) }; - tx.try_push(nodedb::bridge::dispatch::BridgeRequest { inner: req }) + tx.try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored(req)) .unwrap(); core.tick(); let resp = rx.try_pop().unwrap(); @@ -551,7 +551,7 @@ fn kv_tenant_isolation() { rls_filters: Vec::new(), })) }; - tx.try_push(nodedb::bridge::dispatch::BridgeRequest { inner: req }) + tx.try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored(req)) .unwrap(); core.tick(); let resp = rx.try_pop().unwrap(); @@ -570,7 +570,7 @@ fn kv_tenant_isolation() { surrogate_ceiling: None, })) }; - tx.try_push(nodedb::bridge::dispatch::BridgeRequest { inner: req }) + tx.try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored(req)) .unwrap(); core.tick(); let resp = rx.try_pop().unwrap(); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs index 0ca7b6cd1..f25942443 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs @@ -37,7 +37,7 @@ fn send_txn( ..make_request(plan) }; req_tx - .try_push(nodedb::bridge::dispatch::BridgeRequest { inner: request }) + .try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored(request)) .unwrap(); core.tick(); resp_rx.try_pop().unwrap().inner diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction.rs index 8afae4476..1e10785d2 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_transaction.rs @@ -97,8 +97,8 @@ fn transaction_batch_commits_atomically() { fn transaction_batch_response_uses_outer_request_id() { let (mut core, mut tx, mut rx, _dir) = make_core(); - tx.try_push(nodedb::bridge::dispatch::BridgeRequest { - inner: make_request_with_id( + tx.try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored( + make_request_with_id( 42, PhysicalPlan::Meta(MetaOp::TransactionBatch { txn_id: None, @@ -117,7 +117,7 @@ fn transaction_batch_response_uses_outer_request_id() { })], }), ), - }) + )) .unwrap(); core.tick(); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_vector.rs b/nodedb/tests/inproc/cases/executor_tests/test_vector.rs index d72ac4c62..1d4f83292 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_vector.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_vector.rs @@ -15,20 +15,18 @@ fn vector_insert_and_search() { let (mut core, mut tx, mut rx, _dir) = make_core(); for i in 0..10u32 { - tx.try_push(BridgeRequest { - inner: make_request_with_id( - 100 + i as u64, - PhysicalPlan::Vector(VectorOp::Insert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "embeddings"), - vector: vec![i as f32, 0.0, 0.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ), - }) + tx.try_push(BridgeRequest::unfloored(make_request_with_id( + 100 + i as u64, + PhysicalPlan::Vector(VectorOp::Insert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "embeddings"), + vector: vec![i as f32, 0.0, 0.0], + dim: 3, + field_name: String::new(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: None, + provenance: None, + }), + ))) .unwrap(); } diff --git a/nodedb/tests/inproc/cases/intake_throttle/helpers.rs b/nodedb/tests/inproc/cases/intake_throttle/helpers.rs index eb6997b85..26f992919 100644 --- a/nodedb/tests/inproc/cases/intake_throttle/helpers.rs +++ b/nodedb/tests/inproc/cases/intake_throttle/helpers.rs @@ -78,7 +78,7 @@ pub(super) fn fill_response_ring( deadline: Instant::now() - Duration::from_secs(1), ..crate::cases::core_loop::helpers::make_request_with_id(id, plan) }; - tx.try_push(BridgeRequest { inner }) + tx.try_push(BridgeRequest::unfloored(inner)) .expect("request ring has room"); } core.tick(); diff --git a/nodedb/tests/inproc/cases/surrogate_round_trip.rs b/nodedb/tests/inproc/cases/surrogate_round_trip.rs index 5733ad9a3..9a9e9977d 100644 --- a/nodedb/tests/inproc/cases/surrogate_round_trip.rs +++ b/nodedb/tests/inproc/cases/surrogate_round_trip.rs @@ -100,10 +100,8 @@ fn send_ok( rx: &mut Consumer, plan: PhysicalPlan, ) -> Vec { - tx.try_push(BridgeRequest { - inner: make_req(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_req(plan))) + .unwrap(); core.tick(); let resp = rx.try_pop().unwrap(); assert_eq!( @@ -126,10 +124,8 @@ fn send_batch_ok( ) { let n = plans.len(); for plan in plans { - tx.try_push(BridgeRequest { - inner: make_req(plan), - }) - .unwrap(); + tx.try_push(BridgeRequest::unfloored(make_req(plan))) + .unwrap(); } core.tick(); for _ in 0..n { diff --git a/nodedb/tests/inproc/cases/transaction_batch_cross_engine_crash.rs b/nodedb/tests/inproc/cases/transaction_batch_cross_engine_crash.rs index b74606eaa..e454efc93 100644 --- a/nodedb/tests/inproc/cases/transaction_batch_cross_engine_crash.rs +++ b/nodedb/tests/inproc/cases/transaction_batch_cross_engine_crash.rs @@ -32,12 +32,12 @@ fn send_batch_expecting_panic_rollback( rx: &mut nodedb_bridge::buffer::Consumer, plans: Vec, ) -> nodedb::bridge::envelope::Response { - tx.try_push(BridgeRequest { - inner: make_request(PhysicalPlan::Meta(MetaOp::TransactionBatch { + tx.try_push(BridgeRequest::unfloored(make_request(PhysicalPlan::Meta( + MetaOp::TransactionBatch { plans, txn_id: None, - })), - }) + }, + )))) .unwrap(); core.tick(); rx.try_pop().unwrap().inner diff --git a/nodedb/tests/inproc/cases/wal_catchup.rs b/nodedb/tests/inproc/cases/wal_catchup.rs index 68b9e2a57..067f5282e 100644 --- a/nodedb/tests/inproc/cases/wal_catchup.rs +++ b/nodedb/tests/inproc/cases/wal_catchup.rs @@ -574,30 +574,28 @@ fn startup_replay_recovers_all_wal_data() { use nodedb::bridge::envelope::{Priority, Request}; req_tx - .try_push(BridgeRequest { - inner: Request { - request_id: RequestId::new(1), - tenant_id: TenantId::new(1), - vshard_id: VShardId::new(0), - database_id: nodedb::types::DatabaseId::DEFAULT, - plan: scan_plan, - deadline: std::time::Instant::now() + Duration::from_secs(10), - priority: Priority::Normal, - trace_id: nodedb_types::TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: nodedb::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: nodedb::bridge::envelope::Admission::Exempt( - nodedb::bridge::envelope::ExemptReason::Read, - ), - }, - }) + .try_push(BridgeRequest::unfloored(Request { + request_id: RequestId::new(1), + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + database_id: nodedb::types::DatabaseId::DEFAULT, + plan: scan_plan, + deadline: std::time::Instant::now() + Duration::from_secs(10), + priority: Priority::Normal, + trace_id: nodedb_types::TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: nodedb::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: nodedb::bridge::envelope::Admission::Exempt( + nodedb::bridge::envelope::ExemptReason::Read, + ), + })) .unwrap(); core.tick(); let resp = resp_rx.try_pop().unwrap(); From c7b6223ddff885d5d18e191f59569e9c6a445320 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 09:04:54 +0800 Subject: [PATCH 19/64] fix(control): apply every permission-cache event on both consumer paths handle_permission_event used a non-blocking try_write and silently dropped the update on contention with no later repair, since each grant or edge event is the only update for its row. Make it async and wait for the write lock instead. Also route the WAL-catchup path's lone re-dispatched event through the same Normal-mode batch handling as the events that follow it, so a grant or hierarchy row consumed there reaches the permission cache too rather than only the trigger and CDC side effects. --- .../security/permission_tree/event_handler.rs | 91 +++++++++++++++++-- nodedb/src/event/consumer.rs | 17 +++- nodedb/src/event/consumer_helpers.rs | 14 ++- 3 files changed, 106 insertions(+), 16 deletions(-) diff --git a/nodedb/src/control/security/permission_tree/event_handler.rs b/nodedb/src/control/security/permission_tree/event_handler.rs index bc7f97769..9c227c3db 100644 --- a/nodedb/src/control/security/permission_tree/event_handler.rs +++ b/nodedb/src/control/security/permission_tree/event_handler.rs @@ -17,21 +17,23 @@ use super::types::PermissionGrant; /// Process a WriteEvent and update the permission cache if relevant. /// -/// Called from the Event Plane consumer after CDC routing. Checks if the -/// event's collection is a permission table or resource graph for any -/// registered permission tree, and updates the cache accordingly. -pub fn handle_permission_event( +/// Called from the Event Plane consumer for every data event, on both the +/// Normal-mode and WAL-catchup paths. Checks if the event's collection is a +/// permission table or resource graph for any registered permission tree, +/// and updates the cache accordingly. +/// +/// Waits for the write lock. Each grant or edge event is the only update for +/// that row, so a skipped event leaves the cache wrong with no later repair. +/// This function holds the lock across no await. The wait ends when the +/// planners that hold a read lock finish. +pub async fn handle_permission_event( event: &WriteEvent, cache: &Arc>, ) { let collection = event.collection.as_ref(); let tenant_id = event.tenant_id.as_u64(); - // Non-blocking lock — skip if contended; next event will catch up. - let mut guard = match cache.try_write() { - Ok(g) => g, - Err(_) => return, - }; + let mut guard = cache.write().await; let is_permission_table = guard.tree_defs_using_permission_table(tenant_id, collection); let is_resource_graph = guard.tree_defs_using_graph(tenant_id, collection); @@ -106,3 +108,74 @@ fn extract_grant(val: &serde_json::Value) -> Option { .unwrap_or(true), }) } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::event::types::{EventSource, RowId, WriteOp}; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + + const TENANT: u64 = 1; + + fn cache_with_tree() -> Arc> { + let def = sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"docs_tree","permission_table":"grants"}"#, + ) + .expect("tree def"); + let mut cache = PermissionCache::new(); + cache.register_tree_def(TENANT, "docs", def); + Arc::new(tokio::sync::RwLock::new(cache)) + } + + fn grant_insert() -> WriteEvent { + let row = serde_json::json!({ + "resource_id": "d1", + "grantee": "role_a", + "level": "viewer", + "inherited": false, + }); + WriteEvent { + sequence: 1, + collection: Arc::from("grants"), + op: WriteOp::Insert, + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("g1")), + lsn: Lsn::new(1), + database_id: DatabaseId::DEFAULT, + tenant_id: TenantId::new(TENANT), + vshard_id: VShardId::new(0), + source: EventSource::User, + new_value: Some(Arc::from( + nodedb_types::json_to_msgpack(&row).expect("encode grant row"), + )), + old_value: None, + system_time_ms: None, + valid_time_ms: None, + user_id: None, + statement_digest: None, + } + } + + #[tokio::test] + async fn grant_event_applies_after_a_held_read_lock_is_released() { + let cache = cache_with_tree(); + let reader = cache.read().await; + + let apply = { + let cache = Arc::clone(&cache); + tokio::spawn(async move { handle_permission_event(&grant_insert(), &cache).await }) + }; + tokio::task::yield_now().await; + assert!(!apply.is_finished(), "the update waits for the reader"); + drop(reader); + apply.await.expect("apply task"); + + let guard = cache.read().await; + assert_eq!( + guard.get_grant(TENANT, "d1", "role_a"), + Some(("viewer", false)) + ); + assert_eq!(guard.tenant_version(TENANT), 1); + } +} diff --git a/nodedb/src/event/consumer.rs b/nodedb/src/event/consumer.rs index fbdba1d84..ff6268c04 100644 --- a/nodedb/src/event/consumer.rs +++ b/nodedb/src/event/consumer.rs @@ -438,12 +438,19 @@ async fn consumer_loop(config: ConsumerConfig, metrics: Arc) { // Discard ring-buffer events already re-dispatched by replay // (`lsn <= last_lsn`); the first event past the replay point is - // returned and dispatched here rather than dropped (the ring is - // SPSC and cannot un-receive it). Everything after it is fresh - // and is served by the Normal-mode drain on the next iteration. + // returned and processed here rather than dropped (the ring is + // SPSC and cannot un-receive it). It is a live ring event, so it + // takes the same Normal-mode path as the events after it, which + // the Normal-mode drain serves on the next iteration. if let Some(event) = drain_and_skip_stale(&mut rx, last_lsn) { record_event(core_id, &event, &metrics); - dispatch_event(&event, &shared_state, &mut retry_queue, &cdc_router).await; + process_normal_batch( + std::slice::from_ref(&event), + &shared_state, + &mut retry_queue, + &cdc_router, + ) + .await; last_sequence = event.sequence; if event.lsn.is_ahead_of(last_lsn) { last_lsn = event.lsn; @@ -551,7 +558,7 @@ async fn process_normal_batch( dispatch_event_actions(event, shared_state, retry_queue).await; // Non-trigger side effects (watermark, CDC, permission cache, MVs, CRDT). - accumulate_data_event(event, shared_state, cdc_router); + accumulate_data_event(event, shared_state, cdc_router).await; } } diff --git a/nodedb/src/event/consumer_helpers.rs b/nodedb/src/event/consumer_helpers.rs index 872e60d31..0c7b46a1d 100644 --- a/nodedb/src/event/consumer_helpers.rs +++ b/nodedb/src/event/consumer_helpers.rs @@ -224,6 +224,15 @@ pub async fn dispatch_event( .watermark_tracker .advance_lsn_only(event.vshard_id.as_u32(), event.lsn.as_u64()); cdc_router.route_event(event, &shared_state.watermark_tracker); + // A grant or hierarchy row consumed here never reaches the Normal-mode + // batch, so catchup applies it to the permission cache too. + if event.op.is_data_event() { + crate::control::security::permission_tree::event_handler::handle_permission_event( + event, + &shared_state.permission_cache, + ) + .await; + } } /// Dispatch the awaited trigger actions shared by normal Event Plane delivery @@ -260,7 +269,7 @@ fn event_actions_required(event: &WriteEvent) -> bool { /// by [`dispatch_triggers`] (called once per event by both the Normal-mode /// and WAL-catchup paths) so a per-row event fires its AFTER-ROW trigger /// exactly once regardless of which path consumed it. -pub fn accumulate_data_event( +pub async fn accumulate_data_event( event: &WriteEvent, shared_state: &Arc, cdc_router: &Arc, @@ -279,7 +288,8 @@ pub fn accumulate_data_event( crate::control::security::permission_tree::event_handler::handle_permission_event( event, &shared_state.permission_cache, - ); + ) + .await; let matching_streams = shared_state.stream_registry.find_matching( event.database_id, event.tenant_id.as_u64(), From 1117e7d9b2b192187c5918115547f39e7358b35d Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 16:02:20 +0800 Subject: [PATCH 20/64] feat(bridge): bound write windows to the outcome floor and fix partial-refusal replay MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Introduces an outcome-floor mechanism that tracks in-flight write windows (minted WAL records not yet resolved) so restart replay and checkpoint truncation never advance past a write whose outcome is still undecided. Every mint site opens a window, and settles or holds it on every exit path including panics, deferrals, and Calvin scheduler halts; a dropped window that never settled or held is reported as a diagnostic leak. Handlers that commit rows incrementally (columnar ingest, bulk update/ delete, CRDT apply, graph edge writes, vector writes, timeseries ingest) now track whether any row landed before a later failure, and answer with a refusal code that keeps their WAL records instead of one that claims nothing applied — otherwise the Control Plane would cancel records for a write that partially landed, and recovery would silently drop it. The same correction applies to transaction sub-plans with no engine-specific undo handling and to Calvin's redo-record bookkeeping. Adds `ErrorCode::ExpiredBeforeExecution` for requests whose deadline passed before a core started them, distinct from `DeadlineExceeded` for a task that ran partway; both surface as the same query-cancelled error to clients. Replaces the WAL appender's apply-key thread-local with an explicit appender parameter, and extracts gateway response-shaping into a dedicated module so remote and local dispatch share one response shape. --- Cargo.lock | 1 + .../src/rpc_codec/data_plane_error.rs | 3 + nodedb/src/bootstrap/state_wiring.rs | 40 +- nodedb/src/bridge/dispatch/mod.rs | 2 +- nodedb/src/bridge/dispatch/outcome_floor.rs | 281 +++++++++- nodedb/src/bridge/envelope/error_code.rs | 5 + .../backup/restore/columnar_reissue.rs | 22 +- .../control/backup/restore/crdt_reissue.rs | 2 +- .../backup/restore/timeseries_reissue.rs | 22 +- .../control/backup/restore/vector_reissue.rs | 22 +- .../post_apply/async_dispatch/core_fanout.rs | 140 ++++- .../post_apply/async_dispatch/dispatcher.rs | 4 +- .../post_apply/async_dispatch/vector.rs | 238 +++++++-- .../control/catalog_entry/post_apply/mod.rs | 1 + .../scheduler/driver/core/commit_redo.rs | 25 +- .../scheduler/driver/core/completion_route.rs | 45 +- .../calvin/scheduler/driver/core/deferred.rs | 5 +- .../driver/core/dispatch/active_dispatch.rs | 2 + .../driver/core/dispatch/static_dispatch.rs | 2 + .../calvin/scheduler/driver/core/halt.rs | 3 + .../calvin/scheduler/driver/core/mod.rs | 1 + .../calvin/scheduler/driver/core/process.rs | 4 + .../scheduler/driver/core/redo_window.rs | 123 +++++ .../calvin/scheduler/driver/core/scheduler.rs | 6 +- .../scheduler/driver/core/test_support.rs | 1 + .../cluster/calvin/scheduler/driver/types.rs | 5 + .../control/cluster/data_plane_error_wire.rs | 9 + nodedb/src/control/gateway/core.rs | 36 +- nodedb/src/control/gateway/dispatch_remote.rs | 3 + nodedb/src/control/gateway/dispatcher.rs | 18 +- nodedb/src/control/gateway/mod.rs | 1 + nodedb/src/control/gateway/outcome.rs | 154 ++++++ nodedb/src/control/gateway/sql_execute.rs | 4 +- nodedb/src/control/mod.rs | 2 +- .../procedural/executor/core/dispatch.rs | 62 ++- nodedb/src/control/request_tracker.rs | 158 +++++- .../control/server/dispatch_utils/collect.rs | 28 +- .../control/server/dispatch_utils/dispatch.rs | 243 ++++++++- .../server/dispatch_utils/error_status.rs | 24 +- .../server/dispatch_utils/minted/mod.rs | 12 + .../server/dispatch_utils/minted/owned.rs | 260 +++++++++ .../server/dispatch_utils/minted/records.rs | 353 ++++++++++++ .../server/dispatch_utils/minted/resolve.rs | 241 +++++++++ .../src/control/server/dispatch_utils/mod.rs | 10 +- .../submit_write/funnel/dispatch.rs | 6 +- .../submit_write/funnel/driver.rs | 67 ++- .../dispatch_utils/submit_write/funnel/mod.rs | 10 +- .../submit_write/funnel/response.rs | 148 ++--- .../submit_write/funnel/wal_append.rs | 22 +- .../dispatch_utils/submit_write/params.rs | 18 + .../control/server/dispatch_utils/types.rs | 4 + .../server/dispatch_utils/write_abort.rs | 140 +---- nodedb/src/control/server/exchange/gather.rs | 2 +- .../src/control/server/http/routes/health.rs | 46 +- .../src/control/server/http/routes/metrics.rs | 36 ++ .../server/native/dispatch/conversion.rs | 42 +- .../server/native/dispatch/raw_dispatch.rs | 37 +- .../server/native/dispatch/sql_gateway.rs | 48 +- .../server/native/dispatch/transaction.rs | 5 +- .../server/pgwire/handler/dispatch/local.rs | 8 +- .../pgwire/handler/transaction_cmds/commit.rs | 3 +- .../control/server/resp/gateway_dispatch.rs | 40 +- nodedb/src/control/server/result_stream.rs | 44 +- .../ddl/neutral/collection/index/teardown.rs | 107 +++- .../shared/ddl/neutral/dsl/vector_index.rs | 6 +- .../shared/ddl/neutral/graph_ops/edge.rs | 30 +- .../neutral/maintenance/vector_index_set.rs | 6 +- .../ddl/neutral/materialized_view/refresh.rs | 41 +- .../ddl/neutral/tree_ops/create_index.rs | 87 +-- .../src/control/server/shared/ddl/sqlstate.rs | 2 +- .../shared/ddl/sync_dispatch/dispatch.rs | 134 +++-- .../shared/ddl/sync_dispatch/system_task.rs | 12 + .../shared/session/commit/single_shard.rs | 2 +- .../server/shared/session/lifecycle.rs | 1 - .../control/server/shared/session/outcome.rs | 11 +- .../server/shared/session/overlay_drop.rs | 4 +- .../server/shared/session/savepoint_ops.rs | 3 +- .../control/server/sync/columnar_handler.rs | 44 +- nodedb/src/control/server/sync/fts_handler.rs | 53 +- .../raft_dispatch/durability_test_support.rs | 34 +- .../server/sync/raft_dispatch/minted.rs | 60 +++ .../control/server/sync/raft_dispatch/mod.rs | 11 +- .../server/sync/raft_dispatch/response.rs | 119 +++-- .../server/sync/raft_dispatch/write.rs | 193 +++++-- nodedb/src/control/server/sync/refusal.rs | 1 + .../control/server/sync/spatial_handler.rs | 54 +- .../control/server/sync/timeseries_handler.rs | 50 +- .../src/control/server/sync/vector_handler.rs | 90 ++-- nodedb/src/control/shutdown/data_plane.rs | 2 +- nodedb/src/control/system_txn/data_plane.rs | 6 +- nodedb/src/control/system_txn/run.rs | 1 + nodedb/src/control/wal_catchup.rs | 12 + nodedb/src/data/executor/core_loop/tick.rs | 7 +- .../src/data/executor/dispatch/timeseries.rs | 51 ++ .../data/executor/handlers/bulk_dml/delete.rs | 235 +++++--- .../data/executor/handlers/bulk_dml/update.rs | 171 ++++-- .../handlers/bulk_dml/update_project.rs | 51 ++ .../handlers/columnar_write/row_ingest.rs | 504 ++++++++++++------ .../executor/handlers/control/calvin/flush.rs | 36 +- .../handlers/control/calvin/shared.rs | 10 - .../handlers/control/calvin/static_stage.rs | 5 +- .../handlers/control/crdt_apply/local.rs | 163 ++++-- .../handlers/graph_edge_write/put_batch.rs | 82 ++- nodedb/src/data/executor/handlers/mod.rs | 1 + .../data/executor/handlers/partial_refusal.rs | 73 +++ .../executor/handlers/timeseries/admission.rs | 24 + .../executor/handlers/timeseries/ingest.rs | 46 +- .../handlers/timeseries/ingest_schema.rs | 31 ++ .../executor/handlers/transaction/batch.rs | 41 +- .../handlers/transaction/batch_crdt.rs | 11 +- .../transaction/batch_irreversible.rs | 146 +++++ .../data/executor/handlers/transaction/mod.rs | 1 + .../handlers/update_from_join_write.rs | 196 +++++-- .../handlers/vector_direct_resolve/apply.rs | 95 +++- .../data/executor/handlers/vector_write.rs | 48 +- nodedb/src/data/mod.rs | 1 + nodedb/src/data/panic_payload.rs | 16 + nodedb/src/data/runtime/event_loop.rs | 14 +- nodedb/src/diag/context/mod.rs | 2 + nodedb/src/diag/context/outcome_floor.rs | 83 +++ nodedb/src/diag/mod.rs | 1 + nodedb/src/diag/recording/mod.rs | 2 + nodedb/src/diag/recording/outcome_floor.rs | 49 ++ nodedb/src/engine/timeseries/ilp_ingest.rs | 2 +- nodedb/src/engine/timeseries/ilp_schema.rs | 21 +- nodedb/src/error_from_data_plane.rs | 4 +- nodedb/src/wal/manager/appender.rs | 44 +- .../cases/request_tracker_backpressure.rs | 12 +- .../tests/inproc/cases/snapshot_round_trip.rs | 2 +- 129 files changed, 5439 insertions(+), 1367 deletions(-) create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs create mode 100644 nodedb/src/control/gateway/outcome.rs create mode 100644 nodedb/src/control/server/dispatch_utils/minted/mod.rs create mode 100644 nodedb/src/control/server/dispatch_utils/minted/owned.rs create mode 100644 nodedb/src/control/server/dispatch_utils/minted/records.rs create mode 100644 nodedb/src/control/server/dispatch_utils/minted/resolve.rs create mode 100644 nodedb/src/control/server/sync/raft_dispatch/minted.rs create mode 100644 nodedb/src/data/executor/handlers/partial_refusal.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/batch_irreversible.rs create mode 100644 nodedb/src/data/panic_payload.rs create mode 100644 nodedb/src/diag/context/outcome_floor.rs create mode 100644 nodedb/src/diag/recording/outcome_floor.rs diff --git a/Cargo.lock b/Cargo.lock index c47be165a..0c28306a1 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4648,6 +4648,7 @@ dependencies = [ name = "nodedb-test-support" version = "0.5.0" dependencies = [ + "async-trait", "base64 0.23.1", "bytes", "futures", diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index 13756ee80..d84e13d86 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -127,6 +127,9 @@ pub enum DataPlaneErrorCode { DispatchCapacity { reason: String, }, + /// The request's deadline passed before the core started it; nothing + /// ran. + ExpiredBeforeExecution, } /// Wire mirror of `nodedb::bridge::envelope::CounterFault`. diff --git a/nodedb/src/bootstrap/state_wiring.rs b/nodedb/src/bootstrap/state_wiring.rs index 0b5034c87..155949d57 100644 --- a/nodedb/src/bootstrap/state_wiring.rs +++ b/nodedb/src/bootstrap/state_wiring.rs @@ -244,24 +244,10 @@ pub async fn wire_state( state.scheduler_config = config.scheduler.clone(); } - // Construct and install the gateway + DDL plan-cache invalidator. - // - // `Gateway` holds a `Weak` back-reference to its own - // `SharedState`, so it cannot be installed via `Arc::get_mut` (which - // requires strong count 1 AND weak count 0 — the gateway's own `Weak` - // violates the latter). `gateway`/`gateway_invalidator` are therefore - // `OnceLock`s, set through `&self` exactly once here at boot. - // - // That weak reference outlives this call, so every `Arc::get_mut` install - // above depends on running BEFORE this block: one placed after it no-ops. - { - let gateway = Arc::new(crate::control::gateway::Gateway::new(Arc::clone(shared))); - let invalidator = Arc::new(crate::control::gateway::PlanCacheInvalidator::new( - &gateway.plan_cache, - )); - let _ = shared.gateway.set(gateway); - let _ = shared.gateway_invalidator.set(invalidator); - } + // The gateway's weak back-reference outlives this call, so every + // `Arc::get_mut` install above must run BEFORE it: one placed after + // it no-ops. + install_gateway(shared); // Hydrate bitemporal retention registry from array catalog. { @@ -301,3 +287,21 @@ pub async fn wire_state( Ok(()) } + +/// Construct and install the gateway and the DDL plan-cache invalidator. +/// +/// `Gateway` holds a `Weak` back-reference to its own +/// `SharedState`. `Arc::get_mut` requires strong count 1 and weak count 0, +/// so it cannot install the gateway. `gateway`/`gateway_invalidator` are +/// `OnceLock`s instead, set through `&self` exactly once. +/// +/// Every `Arc::get_mut` install on `shared` must run before this call. +/// A later `get_mut` sees the weak reference and no-ops. +pub fn install_gateway(shared: &Arc) { + let gateway = Arc::new(crate::control::gateway::Gateway::new(Arc::clone(shared))); + let invalidator = Arc::new(crate::control::gateway::PlanCacheInvalidator::new( + &gateway.plan_cache, + )); + let _ = shared.gateway.set(gateway); + let _ = shared.gateway_invalidator.set(invalidator); +} diff --git a/nodedb/src/bridge/dispatch/mod.rs b/nodedb/src/bridge/dispatch/mod.rs index 9ddf0fca1..8c5d6e55f 100644 --- a/nodedb/src/bridge/dispatch/mod.rs +++ b/nodedb/src/bridge/dispatch/mod.rs @@ -17,5 +17,5 @@ pub use dispatcher::{ DefaultPriorityResolver, Dispatcher, }; pub use drain::CorePending; -pub use outcome_floor::{OutcomeFloor, WriteWindow}; +pub use outcome_floor::{OutcomeFloor, StuckFloor, WriteWindow}; pub use refusal::DispatchRefusal; diff --git a/nodedb/src/bridge/dispatch/outcome_floor.rs b/nodedb/src/bridge/dispatch/outcome_floor.rs index 312a381a7..ff8049fdb 100644 --- a/nodedb/src/bridge/dispatch/outcome_floor.rs +++ b/nodedb/src/bridge/dispatch/outcome_floor.rs @@ -42,14 +42,18 @@ //! back. Its record was minted outside a window, and the floor passed it //! before it reached the dispatcher. The open logs that record. //! -//! ## Settling +//! ## Closing a window //! -//! [`WriteWindow::settle`] states that the outcome is final. A window dropped -//! without it stays open for the rest of the process. Its record can still -//! need restart replay, so F must never pass it. The drop logs an error. +//! [`WriteWindow::settle`] states that the outcome is final. +//! [`WriteWindow::hold`] states that the record has no final outcome in this +//! process: restart replay must reach it, so the window stays open until the +//! process exits. A window dropped without either is a leak. It stays open +//! too, and the drop counts it, logs an error, and files a report. -use std::collections::{BTreeMap, HashMap}; +use std::collections::BTreeMap; +use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::{Arc, Mutex, MutexGuard}; +use std::time::{Duration, Instant}; use tracing::{error, warn}; @@ -59,42 +63,77 @@ use crate::types::Lsn; #[derive(Debug, Default)] pub struct OutcomeFloor { windows: Mutex, + /// Windows dropped without a settle or a hold. + leaked: AtomicU64, +} + +#[derive(Debug)] +struct OpenWindow { + horizon: u64, + opened_at: Instant, + /// Held until restart: the floor stays below it by design. + held: bool, } #[derive(Debug, Default)] struct Windows { next_ticket: u64, - /// Horizon of each open window, by ticket. - open: HashMap, + /// Each open window, by ticket. Tickets increase, so the first entry is + /// the oldest open window. + open: BTreeMap, /// Number of open windows at each horizon. horizons: BTreeMap, /// Highest LSN any window noted. max_noted: u64, /// Highest floor handed out. published: u64, + /// Number of held windows. + held: usize, } impl Windows { fn open(&mut self, horizon: u64) -> u64 { let ticket = self.next_ticket; self.next_ticket += 1; - self.open.insert(ticket, horizon); + self.open.insert( + ticket, + OpenWindow { + horizon, + opened_at: Instant::now(), + held: false, + }, + ); *self.horizons.entry(horizon).or_insert(0) += 1; ticket } fn close(&mut self, ticket: u64) { - let Some(horizon) = self.open.remove(&ticket) else { + let Some(window) = self.open.remove(&ticket) else { return; }; - if let Some(count) = self.horizons.get_mut(&horizon) { + if let Some(count) = self.horizons.get_mut(&window.horizon) { *count -= 1; if *count == 0 { - self.horizons.remove(&horizon); + self.horizons.remove(&window.horizon); } } } + /// Mark a window held. Returns its horizon and age. + fn mark_held(&mut self, ticket: u64) -> Option<(u64, Duration)> { + let window = self.open.get_mut(&ticket)?; + if !window.held { + window.held = true; + self.held += 1; + } + Some((window.horizon, window.opened_at.elapsed())) + } + + /// The oldest window that is not held. + fn oldest_unheld(&self) -> Option<&OpenWindow> { + self.open.values().find(|window| !window.held) + } + fn note(&mut self, lsn: u64) { self.max_noted = self.max_noted.max(lsn); } @@ -109,6 +148,19 @@ impl Windows { } } +/// The oldest window that holds the floor, and how long it has held it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct StuckFloor { + /// The current floor. + pub floor: Lsn, + /// The oldest open window's horizon: the floor stays below it. + pub horizon: Lsn, + /// How long the oldest open window has been open. + pub open_for: Duration, + /// Number of open windows that are not held. + pub open_windows: usize, +} + impl OutcomeFloor { /// An empty registry. Its floor starts at zero. pub fn new() -> Arc { @@ -148,20 +200,95 @@ impl OutcomeFloor { WriteWindow::new(Arc::clone(self), ticket) } + /// Open a window for an existing record at `lsn` that is sent to a core + /// again. `None` when the floor already passed `lsn`: the record's + /// outcome is final, and a second apply would land below the floor. + pub fn open_existing(self: &Arc, lsn: Lsn) -> Option { + let ticket = { + let mut windows = self.lock(); + if lsn.as_u64() <= windows.floor() { + return None; + } + windows.note(lsn.as_u64()); + windows.open(lsn.as_u64()) + }; + Some(WriteWindow::new(Arc::clone(self), ticket)) + } + /// The current floor. Never lower than a value returned before. pub fn floor(&self) -> Lsn { Lsn::new(self.lock().floor()) } + + /// Windows dropped without a settle or a hold since the process started. + pub fn leaked_windows(&self) -> u64 { + self.leaked.load(Ordering::Relaxed) + } + + /// Windows held until restart. The floor stays below each by design. + pub fn held_windows(&self) -> usize { + self.lock().held + } + + /// The oldest open window that is not held, when it has held the floor + /// for longer than `bound`. A held window is expected to stay open, so + /// it never makes the floor stuck. + pub fn stuck(&self, bound: Duration) -> Option { + let mut windows = self.lock(); + let floor = windows.floor(); + let open_windows = windows.open.len() - windows.held; + let oldest = windows.oldest_unheld()?; + let open_for = oldest.opened_at.elapsed(); + (open_for > bound).then_some(StuckFloor { + floor: Lsn::new(floor), + horizon: Lsn::new(oldest.horizon), + open_for, + open_windows, + }) + } + + /// How long the oldest open window that is not held has been open, or + /// zero when none is. + pub fn oldest_open_for(&self) -> Duration { + self.lock() + .oldest_unheld() + .map_or(Duration::ZERO, |oldest| oldest.opened_at.elapsed()) + } + + fn close(&self, ticket: u64) { + self.lock().close(ticket); + } + + /// Count a leaked window and report it. The window stays open. + fn leak(&self, ticket: u64) { + self.leaked.fetch_add(1, Ordering::Relaxed); + let (horizon, open_for) = self + .lock() + .open + .get(&ticket) + .map_or((0, Duration::ZERO), |window| { + (window.horizon, window.opened_at.elapsed()) + }); + error!( + ticket, + horizon, + open_for_ms = u64::try_from(open_for.as_millis()).unwrap_or(u64::MAX), + "a write window was dropped before its outcome was final; the outcome \ + floor stays below it until restart" + ); + crate::diag::write_window_leaked(ticket, horizon, open_for); + } } /// One write's hold on the outcome floor. Settle it once the write's outcome -/// is final. Dropped unsettled, it holds the floor for the rest of the -/// process. +/// is final, or hold it when the record has no final outcome in this process. +/// Dropped without either, it leaks: it holds the floor until the process +/// exits, and the drop reports it. #[derive(Debug)] pub struct WriteWindow { owner: Arc, ticket: u64, - settled: bool, + closed: bool, } impl WriteWindow { @@ -169,7 +296,7 @@ impl WriteWindow { Self { owner, ticket, - settled: false, + closed: false, } } @@ -180,19 +307,36 @@ impl WriteWindow { /// Close the window: the write's outcome is final. pub fn settle(mut self) { - self.settled = true; - self.owner.lock().close(self.ticket); + self.closed = true; + self.owner.close(self.ticket); + } + + /// Keep the window open until the process exits: the record has no final + /// outcome here, and restart replay must reach it. Files a report naming + /// the caller. + #[track_caller] + pub fn hold(mut self) { + self.closed = true; + let site = std::panic::Location::caller(); + let (horizon, open_for) = self + .owner + .lock() + .mark_held(self.ticket) + .unwrap_or((0, Duration::ZERO)); + warn!( + ticket = self.ticket, + horizon, + site = %site, + "a write window is held until restart; the outcome floor stays below it" + ); + crate::diag::write_window_held(site, self.ticket, horizon, open_for); } } impl Drop for WriteWindow { fn drop(&mut self) { - if !self.settled { - error!( - ticket = self.ticket, - "a write window was dropped before its outcome was final; the outcome \ - floor stays below it until restart" - ); + if !self.closed { + self.owner.leak(self.ticket); } } } @@ -266,6 +410,97 @@ mod tests { assert_eq!(floor.floor(), Lsn::new(6)); } + #[test] + fn a_dropped_window_counts_as_a_leak() { + let floor = OutcomeFloor::new(); + assert_eq!(floor.leaked_windows(), 0); + drop(floor.open_write()); + assert_eq!(floor.leaked_windows(), 1); + floor.open_write().settle(); + floor.open_write().hold(); + assert_eq!( + floor.leaked_windows(), + 1, + "a settled or held window is not a leak" + ); + } + + #[test] + fn a_held_window_holds_the_floor() { + let floor = OutcomeFloor::new(); + let before = floor.open_write(); + before.note_minted(Lsn::new(4)); + before.settle(); + let held = floor.open_write(); + held.note_minted(Lsn::new(5)); + held.hold(); + let later = floor.open_write(); + later.note_minted(Lsn::new(6)); + later.settle(); + assert_eq!(floor.floor(), Lsn::new(4)); + } + + #[test] + fn a_window_open_past_the_bound_reports_a_stuck_floor() { + let floor = OutcomeFloor::new(); + assert!(floor.stuck(Duration::ZERO).is_none(), "no open window"); + let window = floor.open_write(); + window.note_minted(Lsn::new(3)); + std::thread::sleep(Duration::from_millis(2)); + let stuck = floor + .stuck(Duration::ZERO) + .expect("the window is older than a zero bound"); + assert_eq!(stuck.horizon, Lsn::new(1)); + assert_eq!(stuck.floor, Lsn::ZERO); + assert_eq!(stuck.open_windows, 1); + assert!(floor.oldest_open_for() > Duration::ZERO); + assert!( + floor.stuck(Duration::from_secs(3600)).is_none(), + "a young window is inside the bound" + ); + window.settle(); + assert!(floor.stuck(Duration::ZERO).is_none()); + assert_eq!(floor.oldest_open_for(), Duration::ZERO); + } + + /// A held window stays open by design. It never makes the floor stuck, + /// and the held count reports it. + #[test] + fn a_held_window_is_counted_and_never_stuck() { + let floor = OutcomeFloor::new(); + let held = floor.open_write(); + held.note_minted(Lsn::new(3)); + held.hold(); + std::thread::sleep(Duration::from_millis(2)); + assert_eq!(floor.held_windows(), 1); + assert!(floor.stuck(Duration::ZERO).is_none()); + assert_eq!(floor.oldest_open_for(), Duration::ZERO); + + let open = floor.open_write(); + std::thread::sleep(Duration::from_millis(2)); + let stuck = floor + .stuck(Duration::ZERO) + .expect("the window that is not held is older than a zero bound"); + assert_eq!(stuck.open_windows, 1); + open.settle(); + assert!(floor.stuck(Duration::ZERO).is_none()); + } + + #[test] + fn an_existing_record_the_floor_passed_opens_no_window() { + let floor = OutcomeFloor::new(); + let window = floor.open_write(); + window.note_minted(Lsn::new(8)); + window.settle(); + assert!(floor.open_existing(Lsn::new(8)).is_none()); + let resent = floor + .open_existing(Lsn::new(9)) + .expect("the floor has not passed 9"); + assert_eq!(floor.floor(), Lsn::new(8)); + resent.settle(); + assert_eq!(floor.floor(), Lsn::new(9)); + } + #[test] fn a_dispatched_window_holds_the_floor_below_its_lsn() { let floor = OutcomeFloor::new(); diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index 92b75f6c5..812cc01da 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -140,6 +140,11 @@ pub enum ErrorCode { /// nothing was enqueued or applied. Transient: the same request succeeds /// once capacity frees. `reason` names the limit and its counts. DispatchCapacity { reason: String }, + /// The request's deadline passed before the core started it, so nothing + /// ran. Distinct from [`Self::DeadlineExceeded`], which a core also + /// answers for a task it stopped part way. Surfaces as the same + /// query-cancelled error. + ExpiredBeforeExecution, } impl From for ErrorCode { diff --git a/nodedb/src/control/backup/restore/columnar_reissue.rs b/nodedb/src/control/backup/restore/columnar_reissue.rs index 549a4d3c4..d10ed4225 100644 --- a/nodedb/src/control/backup/restore/columnar_reissue.rs +++ b/nodedb/src/control/backup/restore/columnar_reissue.rs @@ -18,8 +18,8 @@ use nodedb_types::value::Value; use crate::Error; use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::server::wal_dispatch::wal_append_if_write; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, VShardId}; use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp}; @@ -165,7 +165,8 @@ pub fn build_columnar_insert_plan( /// /// Branches identically to a normal write: /// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: `wal_append_if_write` then `sync_dispatch::dispatch_system`. +/// - Single-node: append the redo under an outcome-floor window, then +/// `sync_dispatch::dispatch_system`, which closes the window. pub async fn reissue_columnar_durably( state: &SharedState, tenant_id: TenantId, @@ -193,7 +194,19 @@ pub async fn reissue_columnar_durably( } // Single-node: WAL first (durable for restart replay), then install live. - wal_append_if_write(&state.wal, tenant_id, vshard, database_id, &plan)?; + // The record's outcome-floor window opens before the append and closes + // from the install's outcome. + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id: vshard, + }; + let minted = MintedRecords::open(&state.outcome_floor); + if let Err(error) = minted.append_plan(&state.wal, owner, &plan) { + // Any record appended before the error never reaches a core. + minted.cancel(&state.wal, owner, 0).await?; + return Err(error); + } sync_dispatch::dispatch_system( state, sync_dispatch::SystemTask::new( @@ -202,7 +215,8 @@ pub async fn reissue_columnar_durably( database_id, collection, plan, - ), + ) + .with_minted(minted), REISSUE_TIMEOUT, ) .await?; diff --git a/nodedb/src/control/backup/restore/crdt_reissue.rs b/nodedb/src/control/backup/restore/crdt_reissue.rs index 9f99877e1..4603205c0 100644 --- a/nodedb/src/control/backup/restore/crdt_reissue.rs +++ b/nodedb/src/control/backup/restore/crdt_reissue.rs @@ -29,7 +29,7 @@ const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); /// /// Branches identically to a normal write (and to `reissue_timeseries_durably`): /// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: `wal_append_if_write` then `sync_dispatch::dispatch_system`. +/// - Single-node: the autocommit funnel appends the redo and installs it. async fn reissue_crdt_collection( state: &SharedState, tenant_id: TenantId, diff --git a/nodedb/src/control/backup/restore/timeseries_reissue.rs b/nodedb/src/control/backup/restore/timeseries_reissue.rs index 3329ad0a9..bd4457e3b 100644 --- a/nodedb/src/control/backup/restore/timeseries_reissue.rs +++ b/nodedb/src/control/backup/restore/timeseries_reissue.rs @@ -19,8 +19,8 @@ use nodedb_types::value::Value; use crate::Error; use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::server::wal_dispatch::wal_append_if_write; use crate::control::state::SharedState; use crate::engine::timeseries::columnar_memtable::{ ColumnData, ColumnType, ColumnarMemtable, ColumnarMemtableConfig, MemtableSnapshot, @@ -316,7 +316,8 @@ pub fn build_timeseries_ingest_plan( /// /// Branches identically to a normal write (and to `reissue_columnar_durably`): /// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: `wal_append_if_write` then `sync_dispatch::dispatch_system`. +/// - Single-node: append the redo under an outcome-floor window, then +/// `sync_dispatch::dispatch_system`, which closes the window. pub async fn reissue_timeseries_durably( state: &SharedState, tenant_id: TenantId, @@ -344,7 +345,19 @@ pub async fn reissue_timeseries_durably( } // Single-node: WAL first (durable for restart replay), then install live. - wal_append_if_write(&state.wal, tenant_id, vshard, database_id, &plan)?; + // The record's outcome-floor window opens before the append and closes + // from the install's outcome. + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id: vshard, + }; + let minted = MintedRecords::open(&state.outcome_floor); + if let Err(error) = minted.append_plan(&state.wal, owner, &plan) { + // Any record appended before the error never reaches a core. + minted.cancel(&state.wal, owner, 0).await?; + return Err(error); + } sync_dispatch::dispatch_system( state, sync_dispatch::SystemTask::new( @@ -353,7 +366,8 @@ pub async fn reissue_timeseries_durably( database_id, collection, plan, - ), + ) + .with_minted(minted), REISSUE_TIMEOUT, ) .await?; diff --git a/nodedb/src/control/backup/restore/vector_reissue.rs b/nodedb/src/control/backup/restore/vector_reissue.rs index a8c54b3f6..0a856f0a7 100644 --- a/nodedb/src/control/backup/restore/vector_reissue.rs +++ b/nodedb/src/control/backup/restore/vector_reissue.rs @@ -13,8 +13,8 @@ use nodedb_types::surrogate::Surrogate; use crate::Error; use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::server::wal_dispatch::wal_append_if_write; use crate::control::state::SharedState; use crate::engine::vector::index_config::{IndexConfig, IndexType}; use crate::types::{DatabaseId, TenantId, VShardId}; @@ -112,7 +112,8 @@ pub fn build_vector_set_params_plan( /// /// Branches identically to a normal write: /// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: `wal_append_if_write` then `sync_dispatch::dispatch_system`. +/// - Single-node: append the redo under an outcome-floor window, then +/// `sync_dispatch::dispatch_system`, which closes the window. pub async fn reissue_vector_durably( state: &SharedState, tenant_id: TenantId, @@ -140,7 +141,19 @@ pub async fn reissue_vector_durably( } // Single-node: WAL first (durable for restart replay), then install live. - wal_append_if_write(&state.wal, tenant_id, vshard, database_id, &plan)?; + // The record's outcome-floor window opens before the append and closes + // from the install's outcome. + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id: vshard, + }; + let minted = MintedRecords::open(&state.outcome_floor); + if let Err(error) = minted.append_plan(&state.wal, owner, &plan) { + // Any record appended before the error never reaches a core. + minted.cancel(&state.wal, owner, 0).await?; + return Err(error); + } sync_dispatch::dispatch_system( state, sync_dispatch::SystemTask::new( @@ -149,7 +162,8 @@ pub async fn reissue_vector_durably( database_id, collection, plan, - ), + ) + .with_minted(minted), REISSUE_TIMEOUT, ) .await?; diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs index ac7a3eae0..f5a8a0804 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs @@ -11,12 +11,13 @@ use std::time::Duration; use tracing::debug; -use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Status}; +use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Response, Status}; +use crate::control::ResponseReceiver; use crate::control::state::SharedState; use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; /// Deadline for one core's acknowledgement of a post-apply meta op. -const DISPATCH_TIMEOUT: Duration = Duration::from_secs(30); +pub(super) const DISPATCH_TIMEOUT: Duration = Duration::from_secs(30); /// Scope and naming for one fan-out, as every log line and error reports it. pub(super) struct CoreFanout<'a> { @@ -30,6 +31,31 @@ pub(super) struct CoreFanout<'a> { pub detail: &'a str, } +/// Every core's answer to one fan-out, as far as the dispatch deadline saw. +pub(super) struct FanoutAnswers { + /// Cores that were not reached, or that answered anything but `Ok`. + pub(super) refused: Vec, + /// Cores still working at the deadline, each with the receiver its + /// answer arrives on. + pub(super) pending: Vec<(usize, ResponseReceiver)>, +} + +impl FanoutAnswers { + /// Every core that did not apply the plan, once each pending core gave + /// its final answer. Waits with no deadline. A core whose receiver + /// closes without a final answer counts as not applied. + pub(super) async fn into_final_refusals(self) -> Vec { + let mut refused = self.refused; + for (core_id, mut rx) in self.pending { + match final_response(&mut rx).await { + Some(resp) if resp.status == Status::Ok => {} + _ => refused.push(core_id), + } + } + refused + } +} + /// Dispatch `plan` to every core on this node, raising when any core failed to /// apply it. pub(super) async fn dispatch_to_every_core( @@ -47,21 +73,35 @@ pub(super) async fn dispatch_to_every_core( } /// Dispatch `plan` to every core on this node, returning the cores that -/// answered with anything but `Ok`. -/// -/// A caller that dispatches two plans covering each other reads the two lists -/// per core instead of raising on the first refusal. -pub(super) async fn unacked_cores( +/// did not answer `Ok` by the dispatch deadline. +async fn unacked_cores( shared: &SharedState, target: &CoreFanout<'_>, plan: &PhysicalPlan, ) -> Vec { + let answers = fan_out(shared, target, plan).await; + let mut unacked = answers.refused; + unacked.extend(answers.pending.into_iter().map(|(core_id, _)| core_id)); + unacked +} + +/// Dispatch `plan` to every core on this node and collect the answers that +/// arrive by the dispatch deadline. +/// +/// A caller that dispatches two plans covering each other reads the two +/// answer sets per core instead of raising on the first refusal. +pub(super) async fn fan_out( + shared: &SharedState, + target: &CoreFanout<'_>, + plan: &PhysicalPlan, +) -> FanoutAnswers { let num_cores = { let d = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); d.num_cores() }; + let deadline = std::time::Instant::now() + DISPATCH_TIMEOUT; let mut receivers = Vec::with_capacity(num_cores); - let mut unreached: Vec = Vec::new(); + let mut refused: Vec = Vec::new(); { let mut d = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); @@ -73,7 +113,7 @@ pub(super) async fn unacked_cores( database_id: DatabaseId::new(target.database_id), vshard_id: VShardId::new(core_id as u32), plan: plan.clone(), - deadline: std::time::Instant::now() + DISPATCH_TIMEOUT, + deadline, priority: Priority::Background, trace_id: TraceId::generate(), consistency: ReadConsistency::Eventual, @@ -92,16 +132,18 @@ pub(super) async fn unacked_cores( let rx = shared.tracker.register(request_id); if d.dispatch_to_core(core_id, request).is_err() { shared.tracker.cancel(&request_id); - unreached.push(core_id); + refused.push(core_id); continue; } receivers.push((core_id, rx)); } } + let mut pending: Vec<(usize, ResponseReceiver)> = Vec::new(); + let wait_until = tokio::time::Instant::from_std(deadline); for (core_id, mut rx) in receivers { - match tokio::time::timeout(DISPATCH_TIMEOUT, async { rx.recv().await.ok_or(()) }).await { - Ok(Ok(resp)) if resp.status == Status::Ok => { + match tokio::time::timeout_at(wait_until, final_response(&mut rx)).await { + Ok(Some(resp)) if resp.status == Status::Ok => { debug!( tenant = target.tenant_id, collection = %target.collection, @@ -111,9 +153,79 @@ pub(super) async fn unacked_cores( "post-apply core ack" ); } - _ => unreached.push(core_id), + Ok(_) => refused.push(core_id), + Err(_) => pending.push((core_id, rx)), + } + } + + FanoutAnswers { refused, pending } +} + +/// The request's final response, skipping partial frames. `None` once the +/// receiver closes without one. Cancel-safe. +async fn final_response(rx: &mut ResponseReceiver) -> Option { + while let Some(resp) = rx.recv().await { + if !resp.partial { + return Some(resp); } } + None +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::Payload; + use crate::control::RequestTracker; + use crate::types::{Lsn, RequestId}; - unreached + fn answer(id: u64, status: Status, partial: bool) -> Response { + Response { + request_id: RequestId::new(id), + status, + attempt: 1, + partial, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A core that outlived the dispatch deadline counts by its final + /// answer. Partial frames before it do not decide anything. + #[tokio::test] + async fn a_pending_core_counts_by_its_final_answer() { + let tracker = RequestTracker::new(); + let applied = tracker.register(RequestId::new(1)); + let refused = tracker.register(RequestId::new(2)); + let answers = FanoutAnswers { + refused: vec![3], + pending: vec![(0, applied), (1, refused)], + }; + let waiter = tokio::spawn(answers.into_final_refusals()); + + assert!(tracker.complete(answer(1, Status::Partial, true))); + assert!(tracker.complete(answer(1, Status::Ok, false))); + assert!(tracker.complete(answer(2, Status::Error, false))); + + assert_eq!(waiter.await.expect("waiter"), vec![3, 1]); + } + + /// A receiver that closes without a final answer leaves the core's + /// outcome unknown, so it counts as not applied. + #[tokio::test] + async fn a_core_whose_receiver_closes_counts_as_not_applied() { + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(4)); + tracker.cancel(&RequestId::new(4)); + let answers = FanoutAnswers { + refused: Vec::new(), + pending: vec![(2, rx)], + }; + + assert_eq!(answers.into_final_refusals().await, vec![2]); + } } diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs index 8eaeadbcd..e9d232cb5 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs @@ -185,7 +185,7 @@ pub fn spawn_post_apply_async_side_effects( CatalogEntry::PutVectorIndexParams(stored) => { tokio::task::block_in_place(|| { tokio::runtime::Handle::current().block_on(async move { - super::vector::put_async(*stored, &shared).await; + super::vector::put_async(*stored, Arc::clone(&shared)).await; }); }); } @@ -204,7 +204,7 @@ pub fn spawn_post_apply_async_side_effects( tenant_id, collection, field_name, - &shared, + Arc::clone(&shared), ) .await; }); diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs index 52fb4f832..b4fa63e60 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs @@ -26,13 +26,25 @@ //! committed. Every failed stage files a `Capture` instead, because a node //! silently missing an index is the defect this module exists to stop. +use std::sync::Arc; + +use tokio::sync::oneshot; + use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use nodedb_physical::physical_plan::VectorOp; use nodedb_types::StoredVectorIndexParams; -use super::core_fanout::{CoreFanout, dispatch_to_every_core, unacked_cores}; +use super::core_fanout::{CoreFanout, DISPATCH_TIMEOUT, FanoutAnswers, fan_out}; + +/// The longest a parameter install waits for its cores to answer: the +/// `SetParams` dispatch deadline, then the reshape's. Its record's window +/// closes within this of its open. +pub(crate) fn longest_core_wait() -> std::time::Duration { + DISPATCH_TIMEOUT.saturating_mul(2) +} /// One vector index, named the way every stage below reports it. struct IndexTarget<'a> { @@ -42,6 +54,30 @@ struct IndexTarget<'a> { field_name: &'a str, } +impl<'a> IndexTarget<'a> { + fn of(entry: &'a StoredVectorIndexParams) -> Self { + Self { + database_id: entry.database_id, + tenant_id: entry.tenant_id, + collection: &entry.collection, + field_name: &entry.field_name, + } + } +} + +/// Where a parameter install stands when a core outlives the dispatch +/// deadline. +enum PutStage { + /// Some cores have not answered `SetParams` yet. + SetParams(FanoutAnswers), + /// Every core answered `SetParams`. `refused` took the reshape instead, + /// and some cores have not answered it yet. + Rebuild { + refused: Vec, + rebuild: FanoutAnswers, + }, +} + /// Build the `SetParams` plan the boot seed and the CREATE handler both /// reproduce, so runtime and restart install identical parameters. fn set_params_plan(entry: &StoredVectorIndexParams) -> PhysicalPlan { @@ -94,66 +130,166 @@ fn drop_index_plan(database_id: u64, collection: &str, field_name: &str) -> Phys /// Install one vector index's build parameters on this node: append the redo /// record, then bring every core to the committed parameters. /// +/// The install owns its record's outcome-floor window, so it runs in a task +/// the caller does not own: a caller dropped mid-install leaves the task to +/// close the window. The call returns once every core answered or the +/// dispatch deadline passed. Cores still working then are waited for by the +/// task. +/// /// The single-node DDL handlers call this directly, where no applier runs and /// the post-apply lane never fires. -pub async fn put_async(entry: StoredVectorIndexParams, shared: &SharedState) { - let target = IndexTarget { - database_id: entry.database_id, - tenant_id: entry.tenant_id, - collection: &entry.collection, - field_name: &entry.field_name, - }; +pub async fn put_async(entry: StoredVectorIndexParams, shared: Arc) { + let (ready_tx, ready_rx) = oneshot::channel(); + tokio::spawn(install_params(entry, shared, ready_tx)); + if ready_rx.await.is_err() { + tracing::error!("the vector index install task ended before it reported"); + } +} + +async fn install_params( + entry: StoredVectorIndexParams, + shared: Arc, + ready: oneshot::Sender<()>, +) { + let target = IndexTarget::of(&entry); let plan = set_params_plan(&entry); // The record makes this node's log self-sufficient: replay rebuilds the - // index from it in LSN order alongside the vector writes around it. - if let Err(error) = append_redo(shared, &target, &plan) { + // index from it in LSN order alongside the vector writes around it. Its + // outcome-floor window closes once every core gave its final answer. + let minted = MintedRecords::open(&shared.outcome_floor); + if let Err(error) = append_redo(&shared, &target, &plan, &minted) { report(&error, "set_params_wal_append", &target); } - let target_fanout = fanout(&target); - let refused = unacked_cores(shared, &target_fanout, &plan).await; - if refused.is_empty() { - return; - } + let set_params = fan_out(&shared, &fanout(&target), &plan).await; + let stage = if set_params.pending.is_empty() { + let refused = set_params.refused; + if refused.is_empty() { + minted.settle(); + // The caller can be gone. The install completes either way. + let _ = ready.send(()); + return; + } + // Every refused core already holds a materialized index, which only + // the in-place reshape reaches. A core that took `SetParams` answers + // this with `NotFound` and stays on the parameters it accepted. + let rebuild = fan_out(&shared, &fanout(&target), &rebuild_plan(&entry)).await; + if rebuild.pending.is_empty() { + close_put(&target, refused, rebuild.refused, minted); + let _ = ready.send(()); + return; + } + PutStage::Rebuild { refused, rebuild } + } else { + PutStage::SetParams(set_params) + }; + // Cores outlived the dispatch deadline. The caller moves on while this + // task waits for their final answers. + let _ = ready.send(()); + finish_put(&shared, &entry, stage, minted).await; +} + +/// Resolve a handed-off parameter install once every core answered. +async fn finish_put( + shared: &SharedState, + entry: &StoredVectorIndexParams, + stage: PutStage, + minted: MintedRecords, +) { + let target = IndexTarget::of(entry); + let (refused, rebuild) = match stage { + PutStage::SetParams(set_params) => { + let refused = set_params.into_final_refusals().await; + if refused.is_empty() { + minted.settle(); + return; + } + let rebuild = fan_out(shared, &fanout(&target), &rebuild_plan(entry)).await; + (refused, rebuild) + } + PutStage::Rebuild { refused, rebuild } => (refused, rebuild), + }; + let reshaped = rebuild.into_final_refusals().await; + close_put(&target, refused, reshaped, minted); +} - // Every refused core already holds a materialized index, which only the - // in-place reshape reaches. A core that took `SetParams` answers this with - // `NotFound` and stays on the parameters it just accepted. - let reshaped = unacked_cores(shared, &target_fanout, &rebuild_plan(&entry)).await; +/// Close a parameter install's window from both answer sets. A core that +/// refused `SetParams` and the reshape missed the change. +fn close_put( + target: &IndexTarget<'_>, + refused: Vec, + reshaped: Vec, + minted: MintedRecords, +) { let missed: Vec = refused .into_iter() .filter(|core_id| reshaped.contains(core_id)) .collect(); - if !missed.is_empty() { - let error = crate::Error::Internal { - detail: format!("cores did not apply the vector index change: {missed:?}"), - }; - report(&error, "set_params_dispatch", &target); + if missed.is_empty() { + minted.settle(); + return; } + let error = crate::Error::Internal { + detail: format!("cores did not apply the vector index change: {missed:?}"), + }; + report(&error, "set_params_dispatch", target); + // A core that missed the change still needs restart replay to reach the + // record. + minted.hold(); } /// Remove one vector index from this node: append and fsync the drop record, /// then dispatch `DropIndex` to every core. +/// +/// Runs in a task the caller does not own, and returns once every core +/// answered or the dispatch deadline passed, as [`put_async`] does. pub async fn delete_async( database_id: u64, tenant_id: u64, collection: String, field_name: String, - shared: &SharedState, + shared: Arc, ) { + let (ready_tx, ready_rx) = oneshot::channel(); + tokio::spawn(drop_index( + IndexName { + database_id, + tenant_id, + collection, + field_name, + }, + shared, + ready_tx, + )); + if ready_rx.await.is_err() { + tracing::error!("the vector index drop task ended before it reported"); + } +} + +/// An owned index name, for the task that drops the index. +struct IndexName { + database_id: u64, + tenant_id: u64, + collection: String, + field_name: String, +} + +async fn drop_index(name: IndexName, shared: Arc, ready: oneshot::Sender<()>) { let target = IndexTarget { - database_id, - tenant_id, - collection: &collection, - field_name: &field_name, + database_id: name.database_id, + tenant_id: name.tenant_id, + collection: &name.collection, + field_name: &name.field_name, }; - let plan = drop_index_plan(database_id, &collection, &field_name); + let plan = drop_index_plan(name.database_id, &name.collection, &name.field_name); // The vector writes this drop cancels are already fsynced in this node's // log, so replay rebuilds the dropped index unless the drop record is - // durable too. Append and fsync before touching the cores. - match append_redo(shared, &target, &plan) { + // durable too. Append and fsync before touching the cores. The record's + // outcome-floor window closes once every core gave its final answer. + let minted = MintedRecords::open(&shared.outcome_floor); + match append_redo(&shared, &target, &plan, &minted) { Ok(Some(lsn)) => { if let Err(error) = shared.wal.wait_durable(lsn).await { report(&error, "drop_index_fsync", &target); @@ -168,26 +304,44 @@ pub async fn delete_async( Err(error) => report(&error, "drop_index_wal_append", &target), } - if let Err(error) = dispatch_to_every_core(shared, &fanout(&target), &plan).await { - report(&error, "drop_index_dispatch", &target); + let answers = fan_out(&shared, &fanout(&target), &plan).await; + // Cores still working past the deadline answer later. The caller moves + // on while this task waits for their final answers. + let _ = ready.send(()); + let refused = answers.into_final_refusals().await; + close_drop(&target, refused, minted); +} + +/// Close a drop's window from the cores that did not drop the index. +fn close_drop(target: &IndexTarget<'_>, refused: Vec, minted: MintedRecords) { + if refused.is_empty() { + minted.settle(); + return; } + let error = crate::Error::Internal { + detail: format!("cores did not apply the vector index change: {refused:?}"), + }; + report(&error, "drop_index_dispatch", target); + // A core that kept the index still needs restart replay to reach the + // drop record. + minted.hold(); } -/// Append `plan`'s redo record to this node's WAL, returning its LSN. +/// Append `plan`'s redo record to this node's WAL under `minted`'s window, +/// returning its LSN. fn append_redo( shared: &SharedState, target: &IndexTarget<'_>, plan: &PhysicalPlan, + minted: &MintedRecords, ) -> crate::Result> { let database_id = DatabaseId::new(target.database_id); - let vshard = VShardId::from_collection_in_database(database_id, target.collection); - let outcome = crate::control::server::wal_dispatch::wal_append_if_write( - &shared.wal, - TenantId::new(target.tenant_id), - vshard, + let owner = RecordOwner { + tenant_id: TenantId::new(target.tenant_id), database_id, - plan, - )?; + vshard_id: VShardId::from_collection_in_database(database_id, target.collection), + }; + let outcome = minted.append_plan(&shared.wal, owner, plan)?; Ok(outcome.lsn) } diff --git a/nodedb/src/control/catalog_entry/post_apply/mod.rs b/nodedb/src/control/catalog_entry/post_apply/mod.rs index f0e63bd7c..564e1af61 100644 --- a/nodedb/src/control/catalog_entry/post_apply/mod.rs +++ b/nodedb/src/control/catalog_entry/post_apply/mod.rs @@ -56,5 +56,6 @@ pub(crate) use async_dispatch::crdt_compact::compact_async; pub use async_dispatch::spawn_post_apply_async_side_effects; pub(crate) use async_dispatch::synonym_group::delete_async as remove_synonym_group; pub(crate) use async_dispatch::synonym_group::put_async as install_synonym_group; +pub(crate) use async_dispatch::vector::longest_core_wait as vector_install_longest_core_wait; pub(crate) use async_dispatch::vector::put_async as install_vector_index_params; pub use sync::apply_post_apply_side_effects_sync; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index 6124816a7..34301b461 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -17,6 +17,7 @@ use super::halt::{HaltReason, HaltStep, error_response_text}; use super::scheduler::Scheduler; use crate::bridge::envelope::{Response, Status}; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::server::dispatch_utils::MintedRecords; use crate::types::VShardId; use crate::wal::{CalvinStamp, RedoRecord}; use nodedb_physical::physical_plan::PhysicalPlan; @@ -74,21 +75,26 @@ impl Scheduler { let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; - let redo_lsn = if redo.ops.is_empty() { - None + // The record's outcome-floor window opens before the append. It stays + // with the pending txn until the flush completes. + let (redo_lsn, redo_records) = if redo.ops.is_empty() { + (None, None) } else { - match self - .shared - .wal - .appender(crate::wal::manager::NO_APPLY_KEY) + let records = MintedRecords::open(&self.shared.outcome_floor); + let appended = records + .appender(&self.shared.wal, crate::wal::manager::NO_APPLY_KEY) .append_transaction_redo( tenant_id, VShardId::new(self.vshard_id), database_id, &redo, - ) { - Ok(lsn) => Some(lsn), + ); + match appended { + Ok(lsn) => (Some(lsn), Some(records)), Err(e) => { + // The txn stays pending and unapplied, and a failed append + // leaves no record for restart replay to reach. + records.settle(); self.halt_apply( txn_id, HaltReason::WalAppendFailed, @@ -99,6 +105,9 @@ impl Scheduler { } } }; + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.redo_records = redo_records; + } // A flush refused at capacity is parked for re-send. The txn awaits its // flush response either way, so the state below is the same. diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs index 90ba3f6dc..3e376bac0 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs @@ -56,13 +56,20 @@ impl Scheduler { .unwrap_or(0); self.metrics.record_executor_txn_duration_ms(elapsed_ms); + // A staged transaction resolves through its commit-barrier state, OLLP + // answer included. A flush under a commit verdict that answers + // `OllpRetryRequired` wrote nothing on this replica while the others + // applied, so it halts and holds. It never settles as a retry. + let commit_state = self.pending.get(&txn_id).and_then(|p| p.commit_state); + // OLLP mismatch: the active executor detected predicate drift and returned // OllpRetryRequired without writing. The retry loop is now COORDINATOR-owned // (`run_dependent_with_retry`): the scheduler must NOT re-submit a stale // prediction. Instead it (1) releases the aborted attempt's locks and // (2) signals the coordinator's completion waiter via the registry so it // can run a FRESH reconnaissance and resubmit. - if response.status == crate::bridge::envelope::Status::Error + if commit_state.is_none() + && response.status == crate::bridge::envelope::Status::Error && response.error_code.as_deref() == Some(&crate::bridge::envelope::ErrorCode::OllpRetryRequired) { @@ -109,7 +116,6 @@ impl Scheduler { // vote; drive the vote-and-park and let the verdict resume the // flush-or-drop. Dependent / active txns carry no `commit_state` and apply // directly below. - let commit_state = self.pending.get(&txn_id).and_then(|p| p.commit_state); match commit_state { Some(CommitState::Staged) => { // Stage failures remain participants in the barrier: @@ -198,7 +204,7 @@ mod tests { use super::*; use crate::control::cluster::calvin::scheduler::driver::core::halt::HaltReason; use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ - make_sequenced_txn, scheduler_with_pending, staged_pending, + error_response, make_sequenced_txn, scheduler_with_pending, staged_pending, }; use crate::control::cluster::calvin::scheduler::metrics::apply_halt_reason; @@ -260,4 +266,37 @@ mod tests { Some((5, 1, "stage")) ); } + + /// A committed flush that answers `OllpRetryRequired` wrote nothing on + /// this replica. The txn stays pending and unapplied, its redo records + /// stay open, and the scheduler halts on the flush. + #[tokio::test] + async fn an_ollp_answer_to_a_committed_flush_halts_and_holds() { + let txn_id = TxnId::new(5, 1); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }, + ); + + scheduler.handle_completion( + txn_id, + RequestId::new(9), + Some(error_response( + crate::bridge::envelope::ErrorCode::OllpRetryRequired, + )), + ); + + assert!(!scheduler.applied.is_applied(5, 1)); + assert!( + scheduler.pending.contains_key(&txn_id), + "the txn must stay pending" + ); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::FlushFailed) + ); + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs index 37d5faefa..5d69915c5 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs @@ -15,12 +15,11 @@ use std::time::Instant; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::MetaOp; -use tokio::sync::mpsc; use super::halt::{HaltReason, HaltStep}; use super::scheduler::Scheduler; use crate::bridge::dispatch::DispatchRefusal; -use crate::bridge::envelope::{Request, Response}; +use crate::bridge::envelope::Request; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::types::RequestId; @@ -245,7 +244,7 @@ impl Scheduler { txn_id: TxnId, step: DispatchStep, request_id: RequestId, - resp_rx: mpsc::Receiver, + resp_rx: crate::control::ResponseReceiver, ) { match step { DispatchStep::StageStatic | DispatchStep::StageActive => { diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs index d5adbc5d0..bae382334 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs @@ -135,6 +135,8 @@ impl Scheduler { // Set only once the txn parks in `AwaitingVerdict`. verdict_deadline: None, stage_error: None, + // Set once a committed txn appends its redo record. + redo_records: None, }, ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs index 37a761a0d..6a574abc6 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs @@ -278,6 +278,8 @@ impl Scheduler { // Set only once the txn parks in `AwaitingVerdict`. verdict_deadline: None, stage_error: None, + // Set once a committed txn appends its redo record. + redo_records: None, }, ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs index d37b8e6bb..aaeb8e559 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs @@ -192,6 +192,9 @@ impl Scheduler { step: HaltStep, error: String, ) { + // The txn stays unapplied here, so restart replay must reach its redo + // record. + self.hold_redo_records(txn_id); let reason = if self.node_shutting_down() { HaltReason::Draining } else { diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index 2c5dce1dd..a3cf69916 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -27,6 +27,7 @@ pub mod owed; pub mod process; pub mod propose; pub mod read_result; +mod redo_window; pub mod request; pub mod routing; pub mod scheduler; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs index 23bf6c66f..cde8a5cad 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs @@ -262,6 +262,10 @@ impl Scheduler { ); return; }; + // The flush applied the txn's redo record. + if let Some(records) = pending.redo_records { + records.settle(); + } self.release_and_mark_applied(txn_id, pending.lock_owner); } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs new file mode 100644 index 000000000..e91945f96 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs @@ -0,0 +1,123 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Close the outcome-floor window of a committed txn's `TransactionRedo` +//! record. +//! +//! The record joins the pending txn when it is appended. The window settles +//! when the txn completes: the flush applied the record. It holds when the +//! txn stays unapplied here, because restart replay must reach the record. + +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Hold the redo record of `txn_id`: the scheduler halted with the txn + /// unapplied. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn hold_redo_records( + &mut self, + txn_id: TxnId, + ) { + if let Some(records) = self + .pending + .get_mut(&txn_id) + .and_then(|pending| pending.redo_records.take()) + { + records.hold(); + } + } + + /// Hold the redo record of every pending txn: the scheduler stops with + /// them unapplied. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn hold_all_redo_records( + &mut self, + ) { + for pending in self.pending.values_mut() { + if let Some(records) = pending.redo_records.take() { + records.hold(); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::halt::{HaltReason, HaltStep}; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::scheduler_with_pending; + use crate::control::cluster::calvin::scheduler::driver::types::CommitState; + use crate::control::server::dispatch_utils::MintedRecords; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + + /// Attach an appended redo record to the pending txn, as a committed + /// resolve does. + fn attach_redo_record(scheduler: &mut Scheduler, txn_id: TxnId) -> Lsn { + let records = MintedRecords::open(&scheduler.shared.outcome_floor); + let lsn = records + .appender(&scheduler.shared.wal, crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"redo", + ) + .expect("append"); + if let Some(pending) = scheduler.pending.get_mut(&txn_id) { + pending.redo_records = Some(records); + } + lsn + } + + fn awaiting_flush() -> CommitState { + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + } + } + + #[tokio::test] + async fn a_halted_txn_holds_its_redo_record() { + let txn_id = TxnId::new(3, 1); + let (mut scheduler, _dir) = scheduler_with_pending(txn_id, awaiting_flush()); + let lsn = attach_redo_record(&mut scheduler, txn_id); + + scheduler.halt_apply( + txn_id, + HaltReason::FlushFailed, + HaltStep::Flush, + "flush refused".to_string(), + ); + + let floor = &scheduler.shared.outcome_floor; + assert!( + floor.floor() < lsn, + "the held record keeps the floor below it" + ); + assert_eq!(floor.leaked_windows(), 0); + } + + #[tokio::test] + async fn a_completed_txn_settles_its_redo_record() { + let txn_id = TxnId::new(3, 2); + let (mut scheduler, _dir) = scheduler_with_pending(txn_id, awaiting_flush()); + let lsn = attach_redo_record(&mut scheduler, txn_id); + + scheduler.on_txn_complete(txn_id); + + let floor = &scheduler.shared.outcome_floor; + assert_eq!(floor.floor(), lsn); + assert_eq!(floor.leaked_windows(), 0); + } + + #[tokio::test] + async fn a_stopping_scheduler_holds_every_pending_redo_record() { + let txn_id = TxnId::new(3, 3); + let (mut scheduler, _dir) = scheduler_with_pending(txn_id, awaiting_flush()); + let lsn = attach_redo_record(&mut scheduler, txn_id); + + scheduler.hold_all_redo_records(); + + let floor = &scheduler.shared.outcome_floor; + assert!(floor.floor() < lsn); + assert_eq!(floor.leaked_windows(), 0); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index a42f4ad24..eb0e66a74 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -119,7 +119,7 @@ pub struct Scheduler { /// Fan-in receiver for executor responses. /// /// Each dispatched transaction spawns a lightweight bridge task that - /// awaits the per-request `mpsc::Receiver` and forwards the + /// awaits the per-request `ResponseReceiver` and forwards the /// result here as a [`CompletionItem`]. The scheduler's `select!` loop /// includes this channel as a first-class arm so it wakes the moment /// any executor response is ready — no polling, no sleep. @@ -326,7 +326,7 @@ impl Scheduler { &self, txn_id: TxnId, request_id: RequestId, - mut response_rx: mpsc::Receiver, + mut response_rx: crate::control::ResponseReceiver, ) { let tx = self.completion_tx.clone(); tokio::spawn(async move { @@ -453,6 +453,8 @@ impl Scheduler { } } } + // Every txn still pending stays unapplied on this replica. + self.hold_all_redo_records(); } /// Allocate a fresh request ID for a dispatch. diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs index d749ad559..d3ff23762 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -428,6 +428,7 @@ pub(super) fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { commit_state: Some(CommitState::Staged), verdict_deadline: None, stage_error: None, + redo_records: None, } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs index af7e39652..914e97a9c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs @@ -70,6 +70,11 @@ pub(super) struct PendingTxn { /// as usual. A COMMIT verdict halts the scheduler, because the txn cannot /// apply here while its peers apply it. pub stage_error: Option, + /// The `TransactionRedo` record appended for a committed txn's flush, + /// under its outcome-floor window. The window settles when the txn + /// completes, and holds when the scheduler halts or stops with the txn + /// still pending. + pub redo_records: Option, } /// Commit-resolution state of a staged static Calvin transaction. diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index e29189546..fb2b04d73 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -159,6 +159,7 @@ impl From for DataPlaneErrorCode { }, ErrorCode::DivisionByZero => Self::DivisionByZero, ErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, + ErrorCode::ExpiredBeforeExecution => Self::ExpiredBeforeExecution, } } } @@ -264,6 +265,7 @@ impl From for ErrorCode { } DataPlaneErrorCode::DivisionByZero => Self::DivisionByZero, DataPlaneErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, + DataPlaneErrorCode::ExpiredBeforeExecution => Self::ExpiredBeforeExecution, } } } @@ -352,6 +354,13 @@ mod tests { assert_eq!(ErrorCode::from(wire), original); } + #[test] + fn expired_before_execution_roundtrips_verbatim() { + let wire = DataPlaneErrorCode::from(ErrorCode::ExpiredBeforeExecution); + assert_eq!(wire, DataPlaneErrorCode::ExpiredBeforeExecution); + assert_eq!(ErrorCode::from(wire), ErrorCode::ExpiredBeforeExecution); + } + #[test] fn counter_fault_roundtrips_verbatim() { for fault in [ diff --git a/nodedb/src/control/gateway/core.rs b/nodedb/src/control/gateway/core.rs index 02e5d201e..8dd63dac9 100644 --- a/nodedb/src/control/gateway/core.rs +++ b/nodedb/src/control/gateway/core.rs @@ -34,6 +34,7 @@ use nodedb_physical::physical_plan::PhysicalPlan; use super::dispatcher::{DispatchRouteParams, dispatch_route, statement_deadline_ms}; use super::fuser::fuse_payloads; use super::key_extractor::UnwiredKeyExtractor; +use super::outcome::GatewayOutcome; use super::plan_cache::PlanCache; use super::retry::retry_not_leader; use super::route::TaskRoute; @@ -163,7 +164,9 @@ impl Gateway { ctx: &QueryContext, plan: PhysicalPlan, ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { - self.execute_plan_with_watermarks(ctx, plan).await + self.execute_plan_outcome(ctx, plan) + .await + .map(GatewayOutcome::into_parts) } /// Execute a pre-planned `PhysicalPlan`, returning both the raw payloads and @@ -180,14 +183,17 @@ impl Gateway { checked: CloneCheckedTask, ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { let plan = authorized_plan_for_context(ctx, checked)?; - self.execute_plan_with_watermarks(ctx, plan).await + self.execute_plan_outcome(ctx, plan) + .await + .map(GatewayOutcome::into_parts) } - async fn execute_plan_with_watermarks( + /// Execute an authorized plan and keep every route's result detail. + pub(super) async fn execute_plan_outcome( &self, ctx: &QueryContext, plan: PhysicalPlan, - ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { + ) -> Result { let shared = self.shared()?; let span = info_span!( "gateway.execute", @@ -240,9 +246,13 @@ impl Gateway { ctx: &QueryContext, plan: PhysicalPlan, version_set: GatewayVersionSet, - ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { + ) -> Result { let shared = self.shared()?; let routes = self.compute_routes(plan, ctx)?; + // A fan-out reads a `NotFound` route as a shard with no slice. A + // single-route plan keeps it as the Data Plane's verdict. + let single_route = routes.len() == 1; + let mut not_found = false; let deadline_ms = statement_deadline_ms(&shared); // Gateway-level byte ceiling: per-route `dispatch_to_data_plane` @@ -341,6 +351,7 @@ impl Gateway { // participating shard, never collapsed to a scalar, so a multi-route // read produces one read-set entry per shard. all_shard_watermarks.extend(outcome.shard_watermarks); + not_found = single_route && outcome.not_found; if outcome.read_version_lsn > max_read_version { max_read_version = outcome.read_version_lsn; } @@ -362,12 +373,17 @@ impl Gateway { // For broadcast scans, fuse all shard payloads into one. The per-shard // watermarks are NOT fused — each participating shard keeps its own // read-set entry. - if all_payloads.len() > 1 { - let fused = fuse_payloads(all_payloads)?; - Ok((vec![fused.payload], all_shard_watermarks, max_read_version)) + let payloads = if all_payloads.len() > 1 { + vec![fuse_payloads(all_payloads)?.payload] } else { - Ok((all_payloads, all_shard_watermarks, max_read_version)) - } + all_payloads + }; + Ok(GatewayOutcome { + payloads, + shard_watermarks: all_shard_watermarks, + read_version_lsn: max_read_version, + not_found, + }) } /// Compute routing decisions for a plan. diff --git a/nodedb/src/control/gateway/dispatch_remote.rs b/nodedb/src/control/gateway/dispatch_remote.rs index d572ed4aa..d86038f50 100644 --- a/nodedb/src/control/gateway/dispatch_remote.rs +++ b/nodedb/src/control/gateway/dispatch_remote.rs @@ -97,6 +97,7 @@ pub(super) async fn dispatch_remote( // and stamped it on this response, so carry it through. This // route did not observe a version of its own to report. read_version_lsn: resp.read_version_lsn, + not_found: false, }); } crate::control::server::exchange::Resolved::Plan(p) => *p, @@ -117,6 +118,7 @@ pub(super) async fn dispatch_remote( // serves an in-transaction read and no read-set entry consumes // this value. read_version_lsn: Lsn::ZERO, + not_found: false, }); } }; @@ -184,6 +186,7 @@ pub(super) async fn dispatch_remote( )], payloads: resp.payloads, read_version_lsn: Lsn::new(resp.read_version_lsn), + not_found: false, }) } } diff --git a/nodedb/src/control/gateway/dispatcher.rs b/nodedb/src/control/gateway/dispatcher.rs index 3c818c405..e0f3b7507 100644 --- a/nodedb/src/control/gateway/dispatcher.rs +++ b/nodedb/src/control/gateway/dispatcher.rs @@ -12,7 +12,7 @@ use std::sync::Arc; use nodedb_cluster::rpc_codec::TypedClusterError; use crate::Error; -use crate::bridge::envelope::PhysicalPlan; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; use crate::control::server::dispatch_utils::{ dispatch_to_data_plane_with_txn, reject_data_plane_error, }; @@ -40,6 +40,11 @@ pub struct DispatchOutcome { /// read targets one collection, so one non-zero value survives — for /// cross-shard OCC read validation. pub read_version_lsn: Lsn, + /// The owning core refused the task with `ErrorCode::NotFound`. + /// + /// A fan-out reads it as a shard that holds no slice. A single-route + /// task reports it as the Data Plane's verdict on that task. + pub not_found: bool, } /// Parameters for [`dispatch_route`]. `txn_id` is session-transaction @@ -287,6 +292,7 @@ async fn dispatch_local( payloads: vec![resp.payload.to_vec()], shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], read_version_lsn: resp.read_version_lsn, + not_found: is_not_found(&resp), }); } @@ -308,6 +314,7 @@ async fn dispatch_local( // `coll_write_lsn` is surfaced via `read_version_lsn` instead. shard_watermarks: vec![(vshard_id, Lsn::ZERO)], read_version_lsn: write_version, + not_found: false, }); } @@ -330,9 +337,18 @@ async fn dispatch_local( payloads: vec![resp.payload.to_vec()], shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], read_version_lsn: resp.read_version_lsn, + not_found: is_not_found(&resp), }) } +/// Whether the core refused the task with `ErrorCode::NotFound`. +/// +/// `reject_data_plane_error` passes this refusal as an empty success. +/// The flag keeps the verdict for a caller that needs it. +fn is_not_found(resp: &Response) -> bool { + resp.status == Status::Error && resp.error_code.as_deref() == Some(&ErrorCode::NotFound) +} + /// Map a [`TypedClusterError`] to an internal [`Error`]. /// /// `NotLeader` is mapped such that the gateway retry loop can extract the diff --git a/nodedb/src/control/gateway/mod.rs b/nodedb/src/control/gateway/mod.rs index 50058b703..da868d892 100644 --- a/nodedb/src/control/gateway/mod.rs +++ b/nodedb/src/control/gateway/mod.rs @@ -10,6 +10,7 @@ pub mod fuser; pub mod invalidation; pub mod key_extractor; pub mod lowered_plan; +pub mod outcome; pub mod plan_cache; pub mod retry; pub mod route; diff --git a/nodedb/src/control/gateway/outcome.rs b/nodedb/src/control/gateway/outcome.rs new file mode 100644 index 000000000..767b88ef2 --- /dev/null +++ b/nodedb/src/control/gateway/outcome.rs @@ -0,0 +1,154 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Gateway execution results, and their Data-Plane response shape. +//! +//! A transport that renders a Data-Plane `Response` gets the same shape from +//! the gateway that local SPSC dispatch gives it. The shape keeps the error +//! status and code of a `NotFound` verdict, the read watermark, and the +//! read-version LSN. + +use crate::Error; +use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; +use crate::control::server::shared::clone_write::CloneCheckedTask; +use crate::types::{Lsn, RequestId, VShardId}; + +use super::core::{Gateway, QueryContext, authorized_plan_for_context}; + +/// Everything one gateway execution observed across its routes. +pub struct GatewayOutcome { + /// One payload, fused when several routes answered. + pub payloads: Vec>, + /// One `(vshard, watermark_lsn)` per participating shard. + pub shard_watermarks: Vec<(VShardId, Lsn)>, + /// Max-folded per-collection read-version LSN. + pub read_version_lsn: Lsn, + /// A single-route plan's owning core refused it with `ErrorCode::NotFound`. + /// + /// Always `false` for a fan-out: there it means a shard holds no slice. + pub not_found: bool, +} + +impl GatewayOutcome { + /// Payloads, per-shard watermarks, and read-version LSN. + pub fn into_parts(self) -> (Vec>, Vec<(VShardId, Lsn)>, Lsn) { + (self.payloads, self.shard_watermarks, self.read_version_lsn) + } + + /// The response local SPSC dispatch returns for the same task. + pub fn into_response(self) -> Response { + let watermark_lsn = self + .shard_watermarks + .iter() + .map(|(_, lsn)| *lsn) + .max() + .unwrap_or(Lsn::ZERO); + if self.not_found { + return not_found_response(watermark_lsn, self.read_version_lsn); + } + let payload = self + .payloads + .into_iter() + .next() + .map(Payload::from_vec) + .unwrap_or_else(Payload::empty); + response( + Status::Ok, + None, + payload, + watermark_lsn, + self.read_version_lsn, + ) + } +} + +impl Gateway { + /// Execute one authorized task and return its Data-Plane response shape. + /// + /// A `NotFound` verdict returns as an error-status response, the same as + /// local SPSC dispatch returns it. The caller records a phantom read from + /// it and renders the verdict. Every other error returns as its typed + /// `Err`. + pub async fn execute_response( + &self, + ctx: &QueryContext, + checked: CloneCheckedTask, + ) -> Result { + let plan = authorized_plan_for_context(ctx, checked)?; + match self.execute_plan_outcome(ctx, plan).await { + Ok(outcome) => Ok(outcome.into_response()), + // A remote leaseholder returns its `NotFound` verdict as a typed + // error. It gets the same shape as a local one. + Err(Error::DataPlane(ErrorCode::NotFound)) => { + Ok(not_found_response(Lsn::ZERO, Lsn::ZERO)) + } + Err(error) => Err(error), + } + } +} + +fn not_found_response(watermark_lsn: Lsn, read_version_lsn: Lsn) -> Response { + response( + Status::Error, + Some(ErrorCode::NotFound), + Payload::empty(), + watermark_lsn, + read_version_lsn, + ) +} + +fn response( + status: Status, + error_code: Option, + payload: Payload, + watermark_lsn: Lsn, + read_version_lsn: Lsn, +) -> Response { + Response { + request_id: RequestId::new(0), + status, + attempt: 0, + partial: false, + payload, + watermark_lsn, + error_code: error_code.map(Box::new), + read_set_valid: None, + read_version_lsn, + write_set: Vec::new(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn outcome(not_found: bool) -> GatewayOutcome { + GatewayOutcome { + payloads: vec![vec![0x90]], + shard_watermarks: vec![ + (VShardId::new(3), Lsn::new(7)), + (VShardId::new(4), Lsn::new(9)), + ], + read_version_lsn: Lsn::new(5), + not_found, + } + } + + #[test] + fn a_not_found_verdict_keeps_its_error_status_and_code() { + let resp = outcome(true).into_response(); + assert_eq!(resp.status, Status::Error); + assert_eq!(resp.error_code.as_deref(), Some(&ErrorCode::NotFound)); + assert_eq!(resp.watermark_lsn, Lsn::new(9)); + assert_eq!(resp.read_version_lsn, Lsn::new(5)); + } + + #[test] + fn a_success_keeps_its_payload_and_lsns() { + let resp = outcome(false).into_response(); + assert_eq!(resp.status, Status::Ok); + assert!(resp.error_code.is_none()); + assert_eq!(resp.payload.to_vec(), vec![0x90u8]); + assert_eq!(resp.watermark_lsn, Lsn::new(9)); + assert_eq!(resp.read_version_lsn, Lsn::new(5)); + } +} diff --git a/nodedb/src/control/gateway/sql_execute.rs b/nodedb/src/control/gateway/sql_execute.rs index fc063cf2b..34fa083a2 100644 --- a/nodedb/src/control/gateway/sql_execute.rs +++ b/nodedb/src/control/gateway/sql_execute.rs @@ -89,7 +89,7 @@ impl Gateway { return self .execute_with_version_set(ctx, plan, stored_vs) .await - .map(|(payloads, _watermarks, _read_version)| payloads); + .map(|outcome| outcome.payloads); } } } @@ -122,6 +122,6 @@ impl Gateway { let plan = authorized_plan_for_context(ctx, checked)?; self.execute_with_version_set(ctx, plan, actual_vs) .await - .map(|(payloads, _watermarks, _read_version)| payloads) + .map(|outcome| outcome.payloads) } } diff --git a/nodedb/src/control/mod.rs b/nodedb/src/control/mod.rs index fb48758ae..a39ce8645 100644 --- a/nodedb/src/control/mod.rs +++ b/nodedb/src/control/mod.rs @@ -64,7 +64,7 @@ pub mod wal_replication; pub mod write_resolve; pub use exec_receiver::LocalPlanExecutor; -pub use request_tracker::RequestTracker; +pub use request_tracker::{RequestTracker, ResponseReceiver}; pub use rolling_upgrade::ClusterVersionView; pub use state::SharedState; pub use wal_replication::{DistributedApplier, ProposeTracker, create_distributed_applier}; diff --git a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs index 0606e485e..9ab86cf58 100644 --- a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs +++ b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs @@ -8,6 +8,7 @@ use super::sql_literal_concat::fold_literal_string_concat; use crate::control::planner::procedural::ast::SqlExpr; use crate::control::planner::procedural::executor::bindings::RowBindings; use crate::control::planner::procedural::executor::eval; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::types::TraceId; impl<'a> StatementExecutor<'a> { @@ -172,13 +173,23 @@ impl<'a> StatementExecutor<'a> { } } - let outcome = crate::control::server::wal_dispatch::wal_append_if_write( - &self.state.wal, - task.tenant_id, - task.vshard_id, - task.database_id, - &task.plan, - )?; + // The window opens before the append and closes from the + // write's outcome inside the funnel. + let owner = RecordOwner { + tenant_id: task.tenant_id, + database_id: task.database_id, + vshard_id: task.vshard_id, + }; + let minted = MintedRecords::open(&self.state.outcome_floor); + let outcome = match minted.append_plan(&self.state.wal, owner, &task.plan) { + Ok(outcome) => outcome, + Err(error) => { + // Any record appended before the error never reaches + // a core. + minted.cancel(&self.state.wal, owner, 0).await?; + return Err(error); + } + }; crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( self.state, @@ -192,6 +203,7 @@ impl<'a> StatementExecutor<'a> { txn_id: None, wal_lsn: outcome.lsn, resolved_now_ms: outcome.resolved_now_ms, + minted: Some(minted), }, ) .await?; @@ -319,21 +331,33 @@ impl<'a> StatementExecutor<'a> { // resolving that properly for N>1 would need `MetaOp::TransactionBatch` // to carry a per-plan `Vec>`, a separate, wider change to // the procedural batch-flush path, not this KV-write fix. - let mut max_wal_lsn: Option = None; + // + // Every record the loop appends is held under one outcome-floor + // window, which the funnel closes from the batch's outcome. + let owner = RecordOwner { + tenant_id: tasks[0].tenant_id, + database_id: tasks[0].database_id, + vshard_id: tasks[0].vshard_id, + }; + let minted = MintedRecords::open(&self.state.outcome_floor); let mut single_task_resolved_now_ms: Option = None; for task in &tasks { - let outcome = crate::control::server::wal_dispatch::wal_append_if_write( - &self.state.wal, - task.tenant_id, - task.vshard_id, - task.database_id, - &task.plan, - )?; - if let Some(lsn) = outcome.lsn { - max_wal_lsn = Some(max_wal_lsn.map_or(lsn, |cur| cur.max(lsn))); - } + let task_owner = RecordOwner { + tenant_id: task.tenant_id, + database_id: task.database_id, + vshard_id: task.vshard_id, + }; + let outcome = match minted.append_plan(&self.state.wal, task_owner, &task.plan) { + Ok(outcome) => outcome, + Err(error) => { + // The records appended so far never reach a core. + minted.cancel(&self.state.wal, owner, 0).await?; + return Err(error); + } + }; single_task_resolved_now_ms = outcome.resolved_now_ms; } + let max_wal_lsn = minted.highest(); if tasks.len() == 1 { if let Some(task) = tasks.into_iter().next() { @@ -349,6 +373,7 @@ impl<'a> StatementExecutor<'a> { txn_id: None, wal_lsn: max_wal_lsn, resolved_now_ms: single_task_resolved_now_ms, + minted: Some(minted), }, ) .await?; @@ -378,6 +403,7 @@ impl<'a> StatementExecutor<'a> { // N>1 batch: no single instant represents every task's // resolved TTL — see the comment above the WAL-append loop. resolved_now_ms: None, + minted: Some(minted), }, ) .await?; diff --git a/nodedb/src/control/request_tracker.rs b/nodedb/src/control/request_tracker.rs index 27d641b21..6fac98f4c 100644 --- a/nodedb/src/control/request_tracker.rs +++ b/nodedb/src/control/request_tracker.rs @@ -3,12 +3,12 @@ use std::collections::HashMap; use std::sync::{Mutex, MutexGuard}; -use tokio::sync::mpsc; +use tokio::sync::{mpsc, oneshot}; use crate::bridge::envelope::Response; use crate::types::RequestId; -/// Per-request channel capacity. A streaming scan produces at most +/// Per-request partial-response capacity. A streaming scan produces at most /// `ceil(rows / STREAM_CHUNK_SIZE)` partials — a few hundred for the /// largest realistic queries. Capacity here bounds how many chunks can /// sit in RAM while the Control-Plane session's TCP write buffer is @@ -16,19 +16,83 @@ use crate::types::RequestId; /// observe backpressure instead of silently growing RSS. pub const REQUEST_CHANNEL_CAPACITY: usize = 256; +/// The sending ends of one tracked request. +struct PendingRequest { + partials: mpsc::Sender, + final_tx: oneshot::Sender, +} + +/// The receiving end of one tracked request. +/// +/// Partial responses arrive through a bounded channel. The final response +/// has a slot of its own, so a full partial channel never drops it. +pub struct ResponseReceiver { + partials: mpsc::Receiver, + final_rx: Option>, +} + +impl ResponseReceiver { + /// The next response: every buffered partial in order, then the final + /// one. `None` once the final response was taken, or once the request + /// ended without one. + /// + /// Cancel-safe: a dropped `recv` future loses no response. + pub async fn recv(&mut self) -> Option { + if let Some(response) = self.partials.recv().await { + return Some(response); + } + let final_rx = self.final_rx.as_mut()?; + let answer = final_rx.await; + self.final_rx = None; + answer.ok() + } + + /// The next response when one is ready, without waiting. `None` when + /// nothing is ready, or once the request ended. + pub fn try_recv(&mut self) -> Option { + match self.partials.try_recv() { + Ok(response) => return Some(response), + Err(mpsc::error::TryRecvError::Empty) => return None, + Err(mpsc::error::TryRecvError::Disconnected) => {} + } + let final_rx = self.final_rx.as_mut()?; + match final_rx.try_recv() { + Ok(response) => { + self.final_rx = None; + Some(response) + } + Err(oneshot::error::TryRecvError::Empty) => None, + Err(oneshot::error::TryRecvError::Closed) => { + self.final_rx = None; + None + } + } + } + + /// A receiver fed by `partials` alone: every response, final included, + /// arrives on it in order. + #[cfg(test)] + pub(crate) fn from_channel(partials: mpsc::Receiver) -> Self { + Self { + partials, + final_rx: None, + } + } +} + /// Routes Data Plane responses back to the waiting Control Plane session. /// -/// Each dispatched request registers an mpsc sender here. The background +/// Each dispatched request registers its senders here. The background /// response poller forwards responses as they arrive. For streaming queries, /// multiple partial responses arrive before the final one. /// /// - Partial responses (`response.partial == true`): forwarded but request /// stays in the map for more chunks. -/// - Final response (`response.partial == false`): forwarded and request -/// removed from the map. +/// - Final response (`response.partial == false`): forwarded into the +/// request's final slot and request removed from the map. #[derive(Default)] pub struct RequestTracker { - pending: Mutex>>, + pending: Mutex>, } impl RequestTracker { @@ -38,48 +102,58 @@ impl RequestTracker { } } - fn lock_pending(&self) -> MutexGuard<'_, HashMap>> { + fn lock_pending(&self) -> MutexGuard<'_, HashMap> { match self.pending.lock() { Ok(guard) => guard, Err(poisoned) => poisoned.into_inner(), } } - /// Register a pending request. Returns a bounded receiver the session awaits. + /// Register a pending request. Returns the receiver the session awaits. /// /// For non-streaming requests, exactly one response arrives. /// For streaming requests, multiple partial responses arrive before the final one. - /// Channel capacity applies backpressure when the session is slow. - pub fn register(&self, id: RequestId) -> mpsc::Receiver { - let (tx, rx) = mpsc::channel(REQUEST_CHANNEL_CAPACITY); - self.lock_pending().insert(id, tx); - rx + /// Partial capacity applies backpressure when the session is slow. + pub fn register(&self, id: RequestId) -> ResponseReceiver { + let (partials_tx, partials_rx) = mpsc::channel(REQUEST_CHANNEL_CAPACITY); + let (final_tx, final_rx) = oneshot::channel(); + self.lock_pending().insert( + id, + PendingRequest { + partials: partials_tx, + final_tx, + }, + ); + ResponseReceiver { + partials: partials_rx, + final_rx: Some(final_rx), + } } /// Forward a response from the Data Plane to the waiting session. /// /// - If `response.partial` is true: sends the chunk but keeps the /// request in the map for subsequent chunks. - /// - If `response.partial` is false: sends the final chunk and - /// removes the request from the map. + /// - If `response.partial` is false: puts the response in the final + /// slot and removes the request from the map. A full partial channel + /// never refuses it. /// - /// Returns `false` if the request was cancelled, timed out, or the - /// session buffer is full (backpressure signal — the Data Plane should - /// stop producing further chunks for this request). + /// Returns `false` if the request was cancelled or its receiver dropped, + /// or if a partial found the session buffer full (backpressure signal — + /// the Data Plane must stop producing further chunks for this request). pub fn complete(&self, response: Response) -> bool { let is_final = !response.partial; let mut pending = self.lock_pending(); if is_final { - if let Some(tx) = pending.remove(&response.request_id) { - tx.try_send(response).is_ok() - } else { - false + match pending.remove(&response.request_id) { + Some(request) => request.final_tx.send(response).is_ok(), + None => false, } } else { let request_id = response.request_id; - if let Some(tx) = pending.get(&request_id) { - match tx.try_send(response) { + if let Some(request) = pending.get(&request_id) { + match request.partials.try_send(response) { Ok(()) => true, Err(_) => { // Full channel (session stalled) or closed (cancelled): @@ -205,4 +279,40 @@ mod tests { // Entry was evicted on first full-channel hit. assert_eq!(tracker.in_flight(), 0); } + + /// A session that stalls with its partial buffer full still receives + /// the final response, after every buffered partial. + #[tokio::test] + async fn a_full_partial_buffer_never_drops_the_final_response() { + let tracker = RequestTracker::new(); + let mut rx = tracker.register(RequestId::new(11)); + for i in 0..REQUEST_CHANNEL_CAPACITY { + assert!(tracker.complete(make_partial(11, &format!("chunk-{i}")))); + } + + assert!(tracker.complete(make_response(11))); + + for _ in 0..REQUEST_CHANNEL_CAPACITY { + let partial = rx.recv().await.expect("buffered partial"); + assert!(partial.partial); + } + let last = rx.recv().await.expect("final response"); + assert!(!last.partial); + assert!(rx.recv().await.is_none()); + } + + /// A `recv` dropped while it waits keeps the final response for the + /// next `recv`. + #[tokio::test] + async fn a_cancelled_recv_keeps_the_final_response() { + let tracker = RequestTracker::new(); + let mut rx = tracker.register(RequestId::new(12)); + let waited = tokio::time::timeout(std::time::Duration::from_millis(10), rx.recv()).await; + assert!(waited.is_err(), "nothing has arrived yet"); + + assert!(tracker.complete(make_response(12))); + + let last = rx.recv().await.expect("final response"); + assert_eq!(last.request_id, RequestId::new(12)); + } } diff --git a/nodedb/src/control/server/dispatch_utils/collect.rs b/nodedb/src/control/server/dispatch_utils/collect.rs index 561ec422b..67ff384eb 100644 --- a/nodedb/src/control/server/dispatch_utils/collect.rs +++ b/nodedb/src/control/server/dispatch_utils/collect.rs @@ -21,7 +21,7 @@ pub(crate) enum DispatchCollectError { /// concatenated payload) or an error if the channel closed without a /// final chunk or if the accumulated payload would exceed the ceiling. pub(crate) async fn collect_bounded_response( - rx: &mut tokio::sync::mpsc::Receiver, + rx: &mut crate::control::ResponseReceiver, max_result_bytes: usize, ) -> Result { // Each streamed chunk is its OWN msgpack array (`encode_raw_document_rows` @@ -108,7 +108,7 @@ pub(crate) struct DeadlineCollect<'a> { /// the symptom and hand the client a generic internal error for its own /// timeout. pub(crate) async fn collect_under_deadline( - rx: &mut tokio::sync::mpsc::Receiver, + rx: &mut crate::control::ResponseReceiver, params: DeadlineCollect<'_>, ) -> crate::Result { let DeadlineCollect { @@ -223,7 +223,8 @@ mod collect_budget_tests { #[tokio::test] async fn non_streaming_single_response_passes_through() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(final_bytes(100)).await.unwrap(); drop(tx); // Single terminal frame returns unmodified — no merge, exact bytes. @@ -235,7 +236,8 @@ mod collect_budget_tests { async fn streaming_merges_all_chunk_arrays() { // Three standalone array chunks must merge into ONE array with every // element — the regression: raw concatenation kept only the first array. - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_rows(1000)).await.unwrap(); tx.send(partial_rows(1000)).await.unwrap(); tx.send(final_rows(500)).await.unwrap(); @@ -251,7 +253,8 @@ mod collect_budget_tests { #[tokio::test] async fn streaming_over_budget_on_partial_aborts() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(600)).await.unwrap(); tx.send(partial_bytes(600)).await.unwrap(); drop(tx); @@ -264,7 +267,8 @@ mod collect_budget_tests { #[tokio::test] async fn streaming_over_budget_on_final_chunk_aborts() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(500)).await.unwrap(); tx.send(final_bytes(600)).await.unwrap(); drop(tx); @@ -274,7 +278,8 @@ mod collect_budget_tests { #[tokio::test] async fn a_collect_past_the_deadline_reports_the_deadline() { - let (_tx, mut rx) = mpsc::channel(4); + let (_tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); let result = collect_under_deadline( &mut rx, DeadlineCollect { @@ -298,7 +303,8 @@ mod collect_budget_tests { // The channel closes rather than answering, and the deadline has // already passed: the closure follows from the statement running out // of time, so reporting it would report the symptom. - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(10)).await.unwrap(); drop(tx); let result = collect_under_deadline( @@ -319,7 +325,8 @@ mod collect_budget_tests { #[tokio::test] async fn a_producer_that_stopped_inside_the_budget_reports_the_closure() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(10)).await.unwrap(); drop(tx); let result = collect_under_deadline( @@ -340,7 +347,8 @@ mod collect_budget_tests { #[tokio::test] async fn channel_closed_without_final_is_explicit_error() { - let (tx, mut rx) = mpsc::channel(4); + let (tx, rx) = mpsc::channel(4); + let mut rx = crate::control::ResponseReceiver::from_channel(rx); tx.send(partial_bytes(10)).await.unwrap(); drop(tx); let err = collect_bounded_response(&mut rx, 1024).await.unwrap_err(); diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index 7636c4d46..fa2828feb 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -9,6 +9,7 @@ use crate::control::server::shared::clone_write::CloneCheckedTask; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use super::minted::{MintedRecords, RecordOwner, resolve_on_response}; use super::submit_write::{ ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; @@ -34,6 +35,38 @@ pub async fn dispatch_authorized_to_data_plane( durability: WalDurability::CallerSupplied { wal_lsn: None, resolved_now_ms: None, + minted: None, + }, + }, + ) + .await +} + +/// Dispatch a clone-checked task whose records the caller already appended +/// under `minted`. The request carries the highest of their LSNs, so the +/// write is durable before it is acknowledged. The funnel closes their +/// outcome-floor window from the task's outcome. +pub(crate) async fn dispatch_authorized_minted_to_data_plane( + shared: &SharedState, + checked: CloneCheckedTask, + trace_id: TraceId, + minted: MintedRecords, +) -> crate::Result { + let task = checked.into_authorized().into_physical_task(); + dispatch_to_data_plane_inner( + shared, + DataPlaneDispatch { + tenant_id: task.tenant_id, + database_id: task.database_id, + vshard_id: task.vshard_id, + plan: task.plan, + trace_id, + event_source: crate::event::EventSource::User, + txn_id: task.txn_id, + durability: WalDurability::CallerSupplied { + wal_lsn: minted.highest(), + resolved_now_ms: None, + minted: Some(minted), }, }, ) @@ -152,6 +185,7 @@ pub(crate) async fn dispatch_to_data_plane_with_source( durability: WalDurability::CallerSupplied { wal_lsn: None, resolved_now_ms: None, + minted: None, }, }, ) @@ -182,6 +216,7 @@ pub(crate) async fn dispatch_trusted_internal_write_to_data_plane( txn_id, wal_lsn, resolved_now_ms, + minted, } = write; dispatch_to_data_plane_inner( shared, @@ -200,6 +235,7 @@ pub(crate) async fn dispatch_trusted_internal_write_to_data_plane( durability: WalDurability::CallerSupplied { wal_lsn, resolved_now_ms, + minted, }, }, ) @@ -280,6 +316,7 @@ pub(crate) async fn dispatch_to_data_plane_with_txn( durability: WalDurability::CallerSupplied { wal_lsn: None, resolved_now_ms: None, + minted: None, }, }, ) @@ -298,8 +335,13 @@ async fn dispatch_to_data_plane_inner( trace_id, event_source, txn_id, - durability, + mut durability, } = params; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id, + }; // Resolve any Exchange data-movement nodes before dispatch: a root-level // Gather fans the child to all cores and returns the merged response here; // a Broadcast join child is gathered and embedded so the plan reaching a @@ -308,7 +350,7 @@ async fn dispatch_to_data_plane_inner( // and already done upstream on the pgwire/native paths. // Internal funnel (COPY, cursors, materialized-view refresh, constraint // subqueries): not session-transaction-scoped, so `None`. - let plan = match crate::control::server::exchange::resolve_exchange_in_plan( + let resolved = crate::control::server::exchange::resolve_exchange_in_plan( shared, database_id, tenant_id, @@ -316,21 +358,42 @@ async fn dispatch_to_data_plane_inner( trace_id, None, ) - .await? - { - crate::control::server::exchange::Resolved::Gathered( + .await; + // A plan that never reaches the funnel closes the caller's records here. + let plan = match resolved { + Ok(crate::control::server::exchange::Resolved::Plan(p)) => *p, + Ok(crate::control::server::exchange::Resolved::Gathered( resp, _shard_watermarks, _shuffle_reads, - ) => { + )) => { + if let Some(minted) = durability.take_minted() { + resolve_on_response(&shared.wal, owner, 0, &resp, minted).await?; + } return Ok(resp); } - crate::control::server::exchange::Resolved::Plan(p) => *p, // Internal funnel callers want a fully-collected Response, not a lazy // stream: materialize the stream into one merged-array Response, // preserving the prior gather-then-return behaviour on this path. - crate::control::server::exchange::Resolved::Stream(s) => { - return crate::control::server::exchange::gather::stream_to_response(s).await; + Ok(crate::control::server::exchange::Resolved::Stream(s)) => { + let collected = crate::control::server::exchange::gather::stream_to_response(s).await; + if let Some(minted) = durability.take_minted() { + match &collected { + Ok(resp) => resolve_on_response(&shared.wal, owner, 0, resp, minted).await?, + // The gather failed part way: what reached the cores is + // unknown here. + Err(_) => minted.hold(), + } + } + return collected; + } + Err(error) => { + // A write plan carries no exchange node, so a failed resolution + // dispatched none of it. + if let Some(minted) = durability.take_minted() { + minted.cancel(&shared.wal, owner, 0).await?; + } + return Err(error); } }; @@ -519,4 +582,166 @@ mod tests { "the minted redo must be fsync-durable before the write is acknowledged" ); } + + // --- Caller records under the outcome floor --- + + /// A read plan: the funnel admits it without a gate, so these tests reach + /// the dispatch and response paths with the caller's records attached. + fn point_get_plan() -> crate::bridge::envelope::PhysicalPlan { + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::PointGet { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + document_id: "u1".into(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }, + ) + } + + fn minted_record(state: &SharedState) -> (super::MintedRecords, Lsn) { + let minted = super::MintedRecords::open(&state.outcome_floor); + let lsn = minted + .appender(&state.wal, crate::wal::manager::NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"row", + ) + .expect("append"); + (minted, lsn) + } + + fn write_with(minted: super::MintedRecords, lsn: Lsn) -> super::WriteDispatch { + super::WriteDispatch { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan: point_get_plan(), + trace_id: crate::types::TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + wal_lsn: Some(lsn), + resolved_now_ms: None, + minted: Some(minted), + } + } + + fn replayed(state: &SharedState) -> Vec { + state.wal.sync().expect("sync"); + state + .wal + .replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect() + } + + /// Answer one request with `status` and `code`. + async fn respond_once_with( + state: Arc, + mut side: CoreChannelDataSide, + status: Status, + code: Option, + ) { + let deadline = Instant::now() + Duration::from_secs(5); + let mut handled = false; + while !handled && Instant::now() < deadline { + if let Ok(request) = side.request_rx.try_pop() { + side.response_tx + .try_push(BridgeResponse { + inner: crate::bridge::envelope::Response { + request_id: request.inner.request_id, + status, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.clone().map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .expect("fake data-plane response queue has capacity"); + handled = true; + } + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + assert!(handled, "fake data plane received the dispatched request"); + state.poll_and_route_responses(); + } + + #[tokio::test] + async fn a_refused_dispatch_cancels_the_callers_records() { + let (state, _side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + state + .dispatcher + .lock() + .expect("dispatcher") + .begin_data_plane_drain(); + + let result = + super::dispatch_trusted_internal_write_to_data_plane(&state, write_with(minted, lsn)) + .await; + + assert!(result.is_err(), "a draining dispatcher refuses the request"); + assert!(!replayed(&state).contains(&lsn.as_u64())); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + } + + #[tokio::test] + async fn a_refusal_that_applied_nothing_cancels_the_callers_records() { + let (state, side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let responder = tokio::spawn(respond_once_with( + Arc::clone(&state), + side, + Status::Error, + Some(crate::bridge::envelope::ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }), + )); + + let response = + super::dispatch_trusted_internal_write_to_data_plane(&state, write_with(minted, lsn)) + .await + .expect("the refusal is a response"); + responder.await.expect("responder completes"); + + assert_eq!(response.status, Status::Error); + assert!(!replayed(&state).contains(&lsn.as_u64())); + assert!(state.outcome_floor.floor() >= lsn); + } + + #[tokio::test] + async fn an_applied_write_settles_the_callers_records() { + let (state, side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let responder = tokio::spawn(respond_once_with( + Arc::clone(&state), + side, + Status::Ok, + None, + )); + + let response = + super::dispatch_trusted_internal_write_to_data_plane(&state, write_with(minted, lsn)) + .await + .expect("the write applies"); + responder.await.expect("responder completes"); + + assert_eq!(response.status, Status::Ok); + assert!(replayed(&state).contains(&lsn.as_u64())); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + } } diff --git a/nodedb/src/control/server/dispatch_utils/error_status.rs b/nodedb/src/control/server/dispatch_utils/error_status.rs index 5f3727e18..b2fa8c0c6 100644 --- a/nodedb/src/control/server/dispatch_utils/error_status.rs +++ b/nodedb/src/control/server/dispatch_utils/error_status.rs @@ -15,8 +15,9 @@ use crate::bridge::envelope::{ErrorCode, Response, Status}; /// other code crosses as `Error::DataPlane` so its SQLSTATE survives. An /// error status carrying no code fails closed rather than reading as success. /// -/// `DeadlineExceeded` is the exception, and it crosses as -/// [`crate::Error::DeadlineExceeded`]. A shard refusing an expired task is the +/// `DeadlineExceeded` and `ExpiredBeforeExecution` are the exception, and +/// they cross as [`crate::Error::DeadlineExceeded`]. A shard refusing an +/// expired task is the /// statement running out of time — the same condition the Control-Plane timer /// reports — so both produce one variant and one SQLSTATE. Leaving it wrapped /// would make the SQLSTATE a client sees depend on which half of that race @@ -27,9 +28,11 @@ pub(crate) fn reject_data_plane_error(resp: &Response) -> crate::Result<()> { } match resp.error_code.as_deref() { Some(ErrorCode::NotFound) => Ok(()), - Some(ErrorCode::DeadlineExceeded) => Err(crate::Error::DeadlineExceeded { - request_id: resp.request_id, - }), + Some(ErrorCode::DeadlineExceeded | ErrorCode::ExpiredBeforeExecution) => { + Err(crate::Error::DeadlineExceeded { + request_id: resp.request_id, + }) + } Some(code) => Err(crate::Error::DataPlane(code.clone())), None => Err(crate::Error::DataPlane(ErrorCode::Internal { detail: "data plane returned an error status with no error code".into(), @@ -70,6 +73,17 @@ mod tests { } } + /// A task that expired before it started reports the same deadline. + #[test] + fn a_task_that_never_started_reports_the_deadline() { + match reject_data_plane_error(&refusal(ErrorCode::ExpiredBeforeExecution)) { + Err(crate::Error::DeadlineExceeded { request_id }) => { + assert_eq!(request_id, RequestId::new(9)); + } + other => panic!("expected the deadline variant, got {other:?}"), + } + } + #[test] fn every_other_verdict_keeps_its_data_plane_code() { match reject_data_plane_error(&refusal(ErrorCode::DivisionByZero)) { diff --git a/nodedb/src/control/server/dispatch_utils/minted/mod.rs b/nodedb/src/control/server/dispatch_utils/minted/mod.rs new file mode 100644 index 000000000..3b8fb9b5e --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/mod.rs @@ -0,0 +1,12 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The records a write appends for one Data-Plane dispatch, and how their +//! outcome-floor window closes. + +mod owned; +mod records; +mod resolve; + +pub(crate) use owned::{Collect, OwnedResponse, OwnedWait, await_response_owned}; +pub(crate) use records::{MintedRecords, RecordOwner}; +pub(crate) use resolve::resolve_on_response; diff --git a/nodedb/src/control/server/dispatch_utils/minted/owned.rs b/nodedb/src/control/server/dispatch_utils/minted/owned.rs new file mode 100644 index 000000000..023ef6ab2 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/owned.rs @@ -0,0 +1,260 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Waiting for a dispatched write's response in a task the caller does not +//! own. +//! +//! Once a write is enqueued, its records close from the core's final +//! response. A caller future dropped mid-wait would drop the records with it +//! and leak their window. The wait and the close run in a spawned task +//! instead. The caller awaits what the task reports, and a dropped caller +//! leaves the task running until the records close. + +use std::sync::Arc; +use std::time::Instant; + +use tokio::sync::oneshot; + +use crate::bridge::envelope::Response; +use crate::control::ResponseReceiver; +use crate::wal::WalManager; + +use super::super::collect::{DispatchCollectError, collect_bounded_response}; +use super::records::{MintedRecords, RecordOwner}; +use super::resolve::{resolve_at_final, resolve_on_response}; + +/// How the owned task reads the response it reports. +#[derive(Debug, Clone, Copy)] +pub(crate) enum Collect { + /// Every frame up to the final one, merged, under a byte budget. + Merged { max_result_bytes: usize }, + /// The first frame. A partial first frame leaves the task closing the + /// records from the final one. + First, +} + +/// Where the owned task waits, and how the records close. +pub(crate) struct OwnedWait { + pub wal: Arc, + pub owner: RecordOwner, + /// The key a final refusal's abort marker carries, `0` when this write's + /// refusals are never final. + pub final_refusal_key: u64, + /// The instant the caller stops waiting. + pub deadline: Instant, + pub collect: Collect, +} + +/// What the owned task reports to the caller. +pub(crate) enum OwnedResponse { + /// A response arrived by the deadline. `closed` is the result of closing + /// the records from it: a failed cancel returns its error and holds the + /// window. A partial response reports `Ok` here, and the task closes the + /// records from the final one. + Answered { + response: Response, + closed: crate::Result<()>, + }, + /// The deadline passed first. The task closes the records once the + /// final response arrives. + DeadlineExceeded, + /// The merged response outgrew its byte budget. The task closes the + /// records once the final response arrives. + OverBudget { bytes: usize }, + /// The channel closed without a final response. The records are held. + ChannelClosed, +} + +/// Wait for `rx`'s response in a spawned task that owns `minted`, and +/// return what it reports. +pub(crate) async fn await_response_owned( + wait: OwnedWait, + rx: ResponseReceiver, + minted: MintedRecords, +) -> crate::Result { + let (report_tx, report_rx) = oneshot::channel(); + tokio::spawn(wait_and_close(wait, rx, minted, report_tx)); + report_rx.await.map_err(|_| crate::Error::Internal { + detail: "the task waiting for a dispatched write's response ended without \ + reporting" + .into(), + }) +} + +async fn wait_and_close( + wait: OwnedWait, + mut rx: ResponseReceiver, + minted: MintedRecords, + report: oneshot::Sender, +) { + let OwnedWait { + wal, + owner, + final_refusal_key, + deadline, + collect, + } = wait; + let until = tokio::time::Instant::from_std(deadline); + let collected = match collect { + Collect::Merged { max_result_bytes } => { + tokio::time::timeout_at(until, collect_bounded_response(&mut rx, max_result_bytes)) + .await + } + Collect::First => { + tokio::time::timeout_at(until, async { + rx.recv().await.ok_or(DispatchCollectError::ChannelClosed) + }) + .await + } + }; + match collected { + Ok(Ok(response)) if response.partial => { + let _ = report.send(OwnedResponse::Answered { + response, + closed: Ok(()), + }); + resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + } + Ok(Ok(response)) => { + let closed = + resolve_on_response(&wal, owner, final_refusal_key, &response, minted).await; + let _ = report.send(OwnedResponse::Answered { response, closed }); + } + Ok(Err(DispatchCollectError::OverBudget { bytes })) => { + let _ = report.send(OwnedResponse::OverBudget { bytes }); + resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + } + Ok(Err(DispatchCollectError::ChannelClosed)) => { + minted.hold(); + let _ = report.send(OwnedResponse::ChannelClosed); + } + Err(_) => { + let _ = report.send(OwnedResponse::DeadlineExceeded); + resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + } + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + use crate::bridge::dispatch::OutcomeFloor; + use crate::bridge::envelope::{ErrorCode, Payload, Status}; + use crate::control::RequestTracker; + use crate::types::{DatabaseId, Lsn, RequestId, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; + + fn owner() -> RecordOwner { + RecordOwner { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + } + } + + fn minted_record(wal: &WalManager, floor: &Arc) -> (MintedRecords, Lsn) { + let minted = MintedRecords::open(floor); + let lsn = minted + .appender(wal, NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"x", + ) + .expect("append"); + (minted, lsn) + } + + fn refusal(id: u64) -> Response { + Response { + request_id: RequestId::new(id), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: Some(Box::new(ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + })), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + fn wait(wal: &Arc, deadline: Instant) -> OwnedWait { + OwnedWait { + wal: Arc::clone(wal), + owner: owner(), + final_refusal_key: 0, + deadline, + collect: Collect::Merged { + max_result_bytes: 1 << 20, + }, + } + } + + /// The caller future is dropped while it waits. The task still closes + /// the records from the refusal that arrives afterwards. + #[tokio::test] + async fn a_dropped_caller_still_closes_its_records() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(1)); + let deadline = Instant::now() + Duration::from_secs(30); + + let caller = tokio::time::timeout( + Duration::from_millis(10), + await_response_owned(wait(&wal, deadline), rx, minted), + ) + .await; + assert!( + caller.is_err(), + "the caller gave up before the core answered" + ); + assert!(floor.floor() < lsn, "the window holds while the core works"); + + assert!(tracker.complete(refusal(1))); + for _ in 0..200 { + if floor.floor() >= lsn { + break; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + assert!(floor.floor() >= lsn, "the refusal closed the records"); + assert_eq!(floor.leaked_windows(), 0); + } + + /// The deadline passes first. The caller hears it at once, and the task + /// closes the records from the final response that follows. + #[tokio::test] + async fn a_deadline_reports_at_once_and_the_records_close_later() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(2)); + + let outcome = await_response_owned(wait(&wal, Instant::now()), rx, minted) + .await + .expect("report"); + assert!(matches!(outcome, OwnedResponse::DeadlineExceeded)); + assert!(floor.floor() < lsn); + + assert!(tracker.complete(refusal(2))); + for _ in 0..200 { + if floor.floor() >= lsn { + break; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + assert!(floor.floor() >= lsn); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/minted/records.rs b/nodedb/src/control/server/dispatch_utils/minted/records.rs new file mode 100644 index 000000000..f36ea4a6d --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/records.rs @@ -0,0 +1,353 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The records a write appended to the WAL for one Data-Plane dispatch, held +//! under the outcome-floor window opened before the first append. +//! +//! The window holds the outcome floor below every record until the write's +//! outcome is final. Each path that ends the write picks one close: +//! +//! - [`MintedRecords::settle`]: the records applied, or their outcome at the +//! core is final. +//! - [`MintedRecords::cancel`]: nothing applied. A `WriteAborted` marker names +//! each record, and the window settles once the markers are durable. +//! - [`MintedRecords::hold`]: the records have no final outcome in this +//! process. Restart replay must reach them, so the floor stays below them +//! until the process exits. +//! +//! Appends go through [`MintedRecords::appender`], which records the LSN of +//! every record it writes. A plan that appends several records is cancelled +//! whole. + +use std::sync::{Arc, Mutex, MutexGuard}; + +use crate::bridge::dispatch::{OutcomeFloor, WriteWindow}; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::wal_dispatch::{WalAppendOutcome, WalAppendRequest, wal_append}; +use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::wal::WalManager; +use crate::wal::manager::{NO_APPLY_KEY, WalAppender}; + +/// Where a write's records live. The abort markers that cancel them carry it. +#[derive(Debug, Clone, Copy)] +pub(crate) struct RecordOwner { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard_id: VShardId, +} + +/// The records one write appended, and the window that holds the outcome +/// floor below them. +#[derive(Debug)] +#[must_use = "minted records hold the outcome floor until they settle, cancel, or hold"] +pub(crate) struct MintedRecords { + window: WriteWindow, + lsns: Mutex>, + /// Whether this write appended the records. A resent record belongs to + /// the write that appended it, and only that write can cancel it. + appended_here: bool, +} + +impl MintedRecords { + /// Open the window. Call it before the first record is appended. + pub(crate) fn open(floor: &Arc) -> Self { + Self { + window: floor.open_write(), + lsns: Mutex::new(Vec::new()), + appended_here: true, + } + } + + /// Hold an existing record at `lsn` that is sent to a core again. `None` + /// when the floor already passed it: its outcome is final, and a second + /// apply would land below the floor. + pub(crate) fn resend(floor: &Arc, lsn: Lsn) -> Option { + let window = floor.open_existing(lsn)?; + Some(Self { + window, + lsns: Mutex::new(vec![lsn]), + appended_here: false, + }) + } + + fn recorded(&self) -> MutexGuard<'_, Vec> { + self.lsns.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// An appender whose records carry `apply_key` and join this set. + pub(crate) fn appender<'a>(&'a self, wal: &'a WalManager, apply_key: u64) -> WalAppender<'a> { + wal.recording_appender(apply_key, &self.lsns) + } + + /// Append `plan`'s redo records under this window. + pub(crate) fn append_plan( + &self, + wal: &WalManager, + owner: RecordOwner, + plan: &PhysicalPlan, + ) -> crate::Result { + wal_append(WalAppendRequest { + wal: self.appender(wal, NO_APPLY_KEY), + tenant_id: owner.tenant_id, + vshard_id: owner.vshard_id, + database_id: owner.database_id, + plan, + credentials: None, + now_override: None, + }) + } + + /// The highest appended LSN, or `None` when nothing was appended. + pub(crate) fn highest(&self) -> Option { + self.recorded().iter().copied().max() + } + + /// Every appended LSN, in append order. + #[cfg(test)] + pub(crate) fn lsns(&self) -> Vec { + self.recorded().clone() + } + + /// Note every recorded LSN on the window and take the list out. + fn into_parts(self) -> (WriteWindow, Vec, bool) { + let Self { + window, + lsns, + appended_here, + } = self; + let lsns = lsns.into_inner().unwrap_or_else(|p| p.into_inner()); + if let Some(highest) = lsns.iter().copied().max() { + window.note_minted(highest); + } + (window, lsns, appended_here) + } + + /// The outcome of every record is final. + pub(crate) fn settle(self) { + let (window, _, _) = self.into_parts(); + window.settle(); + } + + /// The records have no final outcome in this process. + #[track_caller] + pub(crate) fn hold(self) { + let (window, _, _) = self.into_parts(); + window.hold(); + } + + /// Cancel every record with a `WriteAborted` marker that carries + /// `marker_key`, wait until the markers are durable, then settle. + /// + /// A failed append or fsync holds the window and returns the error: the + /// records stay replayable, so the floor must not pass them. + /// + /// A crash before the markers are durable still leaves the records + /// replayable. The markers make the refusal durable once it is reported. + /// + /// A resent record is not cancelled here: the write that appended it + /// cancels it from its own outcome. The window settles. + /// + /// The cancel runs in a task the caller does not own, so a caller + /// dropped mid-cancel leaves the task to close the window. + pub(crate) async fn cancel( + self, + wal: &Arc, + owner: RecordOwner, + marker_key: u64, + ) -> crate::Result<()> { + let wal = Arc::clone(wal); + tokio::spawn(async move { self.cancel_in_place(&wal, owner, marker_key).await }) + .await + .map_err(|error| crate::Error::Internal { + detail: format!("the task cancelling a write's records failed: {error}"), + })? + } + + async fn cancel_in_place( + self, + wal: &WalManager, + owner: RecordOwner, + marker_key: u64, + ) -> crate::Result<()> { + let (window, lsns, appended_here) = self.into_parts(); + if !appended_here { + window.settle(); + return Ok(()); + } + let mut last_marker = None; + for lsn in &lsns { + match wal.appender(marker_key).append_write_aborted( + owner.tenant_id, + owner.vshard_id, + owner.database_id, + *lsn, + ) { + Ok(marker) => last_marker = Some(marker), + Err(error) => { + window.hold(); + return Err(error); + } + } + } + if let Some(marker) = last_marker + && let Err(error) = wal.wait_durable(marker).await + { + window.hold(); + return Err(error); + } + tracing::debug!( + cancelled = lsns.len(), + "refused write records cancelled in the WAL" + ); + window.settle(); + Ok(()) + } + + /// Cancel records whose write another path carries to its outcome, such + /// as a Calvin route or a Raft proposal, in a task the caller does not + /// own. That path's result stands. A cancel error holds the window, + /// which files its report, and is logged with `path` naming the route. + /// + /// The caller awaits [`Superseded::finish`] once the other path returns. + /// A caller dropped before then leaves the task to finish the cancel. + pub(crate) fn supersede( + self, + wal: Arc, + owner: RecordOwner, + path: &'static str, + ) -> Superseded { + Superseded { + task: tokio::spawn(async move { + if let Err(error) = self.cancel_in_place(&wal, owner, 0).await { + tracing::error!( + path, + %error, + "records of a write carried by another path could not be \ + cancelled; their window is held until restart" + ); + } + }), + } + } +} + +/// The task cancelling records another path superseded. +pub(crate) struct Superseded { + task: tokio::task::JoinHandle<()>, +} + +impl Superseded { + /// Wait until the cancel ended. + pub(crate) async fn finish(self) { + if let Err(error) = self.task.await { + tracing::error!(%error, "the task cancelling superseded records failed"); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::wal::manager::NO_APPLY_KEY; + + fn owner() -> RecordOwner { + RecordOwner { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + } + } + + fn append(wal: &WalManager, minted: &MintedRecords, body: &[u8]) -> Lsn { + minted + .appender(wal, NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + body, + ) + .expect("append") + } + + #[test] + fn settled_records_release_the_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("wal"); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + assert!(floor.floor() < lsn); + minted.settle(); + assert_eq!(floor.floor(), lsn); + } + + #[test] + fn held_records_keep_the_floor_below_them() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("wal"); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + minted.hold(); + assert!(floor.floor() < lsn); + assert_eq!(floor.leaked_windows(), 0, "a hold is not a leak"); + } + + #[tokio::test] + async fn a_resent_record_is_never_cancelled() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let lsn = wal + .appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"a", + ) + .expect("append"); + let resent = MintedRecords::resend(&floor, lsn).expect("the floor is below the record"); + assert!(floor.floor() < lsn); + resent.cancel(&wal, owner(), 0).await.expect("cancel"); + wal.sync().expect("sync"); + let replayed: Vec = wal + .replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect(); + assert!(replayed.contains(&lsn.as_u64()), "no marker names it"); + assert_eq!(floor.floor(), lsn); + assert!( + MintedRecords::resend(&floor, lsn).is_none(), + "the floor passed the record" + ); + } + + #[tokio::test] + async fn cancelled_records_are_dropped_from_replay_and_release_the_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let first = append(&wal, &minted, b"a"); + let second = append(&wal, &minted, b"b"); + assert_eq!(minted.lsns(), vec![first, second]); + assert_eq!(minted.highest(), Some(second)); + minted.cancel(&wal, owner(), 0).await.expect("cancel"); + assert!( + wal.durable_through() > second.as_u64(), + "markers are durable" + ); + let replayed: Vec = wal + .replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect(); + assert!(!replayed.contains(&first.as_u64())); + assert!(!replayed.contains(&second.as_u64())); + assert!(floor.floor() >= second, "the window settled"); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs new file mode 100644 index 000000000..26abbd17e --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs @@ -0,0 +1,241 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Close a write's outcome-floor window from the core's final response. +//! +//! A refusal whose code proves nothing applied cancels the records. Any other +//! final response settles them: the core's outcome is final, and restart +//! replay from the floor reproduces it. + +use std::sync::Arc; + +use crate::bridge::envelope::{Response, Status}; +use crate::wal::WalManager; + +use super::super::write_abort::{refusal_is_final, write_definitely_not_applied}; +use super::records::{MintedRecords, RecordOwner}; + +/// Close `minted` from the core's final `response`. +/// +/// `final_refusal_key` is the proposal key a final refusal's marker carries, +/// `0` when this write's refusals are never final. A failed cancel returns +/// the error and holds the window. +pub(crate) async fn resolve_on_response( + wal: &Arc, + owner: RecordOwner, + final_refusal_key: u64, + response: &Response, + minted: MintedRecords, +) -> crate::Result<()> { + let refusal = response + .error_code + .as_deref() + .filter(|code| response.status != Status::Ok && write_definitely_not_applied(code)); + match refusal { + Some(code) => { + let marker_key = if refusal_is_final(code) { + final_refusal_key + } else { + 0 + }; + minted.cancel(wal, owner, marker_key).await + } + None => { + minted.settle(); + Ok(()) + } + } +} + +/// Close `minted` once the core's final response arrives on `rx`, after the +/// caller stopped waiting for it. +/// +/// A refusal that arrives after the caller timed out still cancels its +/// records, so a restart cannot apply a write the caller never saw applied. A +/// channel that closes before a final response leaves the outcome unknown: +/// the window holds. +pub(crate) async fn resolve_at_final( + wal: &Arc, + owner: RecordOwner, + final_refusal_key: u64, + mut rx: crate::control::ResponseReceiver, + minted: MintedRecords, +) { + loop { + match rx.recv().await { + Some(response) if response.partial => continue, + Some(response) => { + if let Err(error) = + resolve_on_response(wal, owner, final_refusal_key, &response, minted).await + { + tracing::error!( + %error, + "a late refusal's abort marker failed; its records stay replayable" + ); + } + return; + } + None => { + minted.hold(); + return; + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::dispatch::OutcomeFloor; + use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::types::{DatabaseId, Lsn, RequestId, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; + + fn owner() -> RecordOwner { + RecordOwner { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + } + } + + fn response(status: Status, code: Option, partial: bool) -> Response { + Response { + request_id: RequestId::new(1), + status, + attempt: 1, + partial, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + fn refusal() -> Response { + response( + Status::Error, + Some(ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }), + false, + ) + } + + fn minted_record(wal: &WalManager, floor: &Arc) -> (MintedRecords, Lsn) { + let minted = MintedRecords::open(floor); + let lsn = minted + .appender(wal, NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"row", + ) + .expect("append"); + (minted, lsn) + } + + fn replayed(wal: &WalManager) -> Vec { + wal.replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect() + } + + #[tokio::test] + async fn an_applied_response_settles_and_keeps_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let ok = response(Status::Ok, None, false); + resolve_on_response(&wal, owner(), 0, &ok, minted) + .await + .expect("resolve"); + wal.sync().expect("sync"); + assert!(replayed(&wal).contains(&lsn.as_u64())); + assert_eq!(floor.floor(), lsn); + } + + #[tokio::test] + async fn an_ambiguous_failure_settles_and_keeps_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let internal = response( + Status::Error, + Some(ErrorCode::Internal { + detail: "io".into(), + }), + false, + ); + resolve_on_response(&wal, owner(), 0, &internal, minted) + .await + .expect("resolve"); + wal.sync().expect("sync"); + assert!(replayed(&wal).contains(&lsn.as_u64())); + assert_eq!(floor.floor(), lsn); + } + + #[tokio::test] + async fn a_definite_refusal_cancels_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + resolve_on_response(&wal, owner(), 0, &refusal(), minted) + .await + .expect("resolve"); + assert!(!replayed(&wal).contains(&lsn.as_u64())); + assert!(floor.floor() >= lsn); + } + + /// The caller stopped waiting before the core answered. The refusal that + /// arrives later still writes its abort marker before the window settles. + #[tokio::test] + async fn a_refusal_after_the_caller_timed_out_still_writes_its_abort_marker() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let (tx, rx) = tokio::sync::mpsc::channel(4); + let rx = crate::control::ResponseReceiver::from_channel(rx); + let waiter = { + let wal = Arc::clone(&wal); + tokio::spawn(async move { resolve_at_final(&wal, owner(), 0, rx, minted).await }) + }; + assert!(floor.floor() < lsn, "the window holds while the core works"); + + tx.send(response(Status::Ok, None, true)) + .await + .expect("send partial"); + tx.send(refusal()).await.expect("send refusal"); + waiter.await.expect("waiter"); + + assert!(!replayed(&wal).contains(&lsn.as_u64())); + assert!(floor.floor() >= lsn); + } + + #[tokio::test] + async fn a_channel_closed_before_a_final_response_holds_the_window() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let (tx, rx) = tokio::sync::mpsc::channel::(4); + let rx = crate::control::ResponseReceiver::from_channel(rx); + let waiter = { + let wal = Arc::clone(&wal); + tokio::spawn(async move { resolve_at_final(&wal, owner(), 0, rx, minted).await }) + }; + drop(tx); + waiter.await.expect("waiter"); + assert!(floor.floor() < lsn); + assert_eq!(floor.leaked_windows(), 0); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index 05a1975a6..5821fb7c8 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -7,6 +7,7 @@ mod collect; mod dispatch; mod durability_barrier; mod error_status; +mod minted; mod submit_write; mod types; mod write_abort; @@ -20,14 +21,17 @@ pub(crate) use collect::{ }; pub use dispatch::{dispatch_authorized_autocommit_write, dispatch_authorized_to_data_plane}; pub(crate) use dispatch::{ - dispatch_authorized_autocommit_write_with_source, dispatch_autocommit_write, - dispatch_to_data_plane, dispatch_to_data_plane_with_txn, + dispatch_authorized_autocommit_write_with_source, dispatch_authorized_minted_to_data_plane, + dispatch_autocommit_write, dispatch_to_data_plane, dispatch_to_data_plane_with_txn, dispatch_trusted_internal_write_to_data_plane, }; pub use durability_barrier::writes_acked_without_durability; pub(crate) use error_status::reject_data_plane_error; +pub(crate) use minted::{ + Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, +}; pub(crate) use submit_write::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; pub(crate) use types::{AutocommitWrite, WriteDispatch}; -pub(crate) use write_abort::refusal_is_final; +pub(crate) use write_abort::{refusal_is_final, write_definitely_not_applied}; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs index 5c19844c6..d14e851e7 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs @@ -4,9 +4,9 @@ use std::time::Instant; -use tokio::sync::{OwnedMutexGuard, mpsc}; +use tokio::sync::OwnedMutexGuard; -use crate::bridge::envelope::{Admission, PhysicalPlan, Priority, Request, Response}; +use crate::bridge::envelope::{Admission, PhysicalPlan, Priority, Request}; use crate::control::array_catalog::ddl::AuthorizedDdlTransition; use crate::control::server::shared::write_admission::WriteAdmissionGuard; use crate::control::state::SharedState; @@ -43,7 +43,7 @@ pub(super) struct DispatchTarget { /// any). pub(super) struct DispatchOutcome { pub request_id: RequestId, - pub rx: mpsc::Receiver, + pub rx: crate::control::ResponseReceiver, pub dispatch_started: Instant, pub deferred_guards: DeferredGuards, } diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index 877855271..0ebf8f2e9 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -5,7 +5,7 @@ use crate::control::server::dispatch_utils::change_events::extract_write_change_set; use crate::control::server::dispatch_utils::durability_barrier::funnel_minted_redo_engine; -use crate::control::server::dispatch_utils::write_abort::{AbortTarget, abort_undispatched_write}; +use crate::control::server::dispatch_utils::minted::{MintedRecords, RecordOwner}; use crate::control::server::shared::session::statement_deadline; use crate::control::server::shared::write_admission::{bare_ok_response, route_write_to_calvin}; use crate::control::server::wal_dispatch; @@ -14,7 +14,7 @@ use crate::control::state::SharedState; use super::super::params::{ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability}; use super::admission::{AdmissionOutcome, admit_write}; use super::dispatch::{DispatchTarget, dispatch_to_data_plane}; -use super::response::{ResponsePhaseInput, collect_classify_and_finish, settle_window}; +use super::response::{ResponsePhaseInput, collect_classify_and_finish}; use super::wal_append::authorize_and_append; /// Admit, make durable, enqueue, collect, and publish one write. @@ -33,10 +33,18 @@ pub(crate) async fn submit_write( event_source, txn_id, user_id, - durability, + mut durability, ordering, change_feed, } = params; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id, + }; + // Records the caller appended for this write, under their outcome-floor + // window. Every path below closes the window. + let caller_minted = durability.take_minted(); // The running statement's deadline, pinned once at the session boundary and // shared by every request the statement fans out into. Used for both the @@ -110,8 +118,17 @@ pub(crate) async fn submit_write( order_guard, } => (admission, admission_guard, order_guard), AdmissionOutcome::RouteToCalvin => { + // The scheduler applies the write from its own records, so + // the caller's records never apply. + let superseded = caller_minted.map(|minted| { + minted.supersede(std::sync::Arc::clone(&shared.wal), owner, "calvin_route") + }); let routed = - route_write_to_calvin(shared, tenant_id, database_id, vshard_id, plan).await?; + route_write_to_calvin(shared, tenant_id, database_id, vshard_id, plan).await; + if let Some(superseded) = superseded { + superseded.finish().await; + } + let routed = routed?; return Ok(SubmitOutcome { response: routed .unwrap_or_else(|| bare_ok_response(crate::types::RequestId::new(0))), @@ -121,17 +138,22 @@ pub(crate) async fn submit_write( }; // A write that mints its own LSN opens its outcome-floor window before the - // mint. The window settles once the outcome is final. - let window = appends_here.then(|| shared.outcome_floor.open_write()); + // mint, and appends through it. + let minted = match caller_minted { + Some(minted) => Some(minted), + None => appends_here.then(|| MintedRecords::open(&shared.outcome_floor)), + }; // Array DDL authorization + durability, under the admission guard, // immediately before the enqueue below. let wal_append_outcome = - match authorize_and_append(shared, tenant_id, database_id, vshard_id, plan, durability) { + match authorize_and_append(shared, owner, plan, durability, minted.as_ref()) { Ok(outcome) => outcome, Err(error) => { - // Nothing minted here reaches a core. - settle_window(window); + // No record of this write reaches a core. + if let Some(minted) = minted { + minted.cancel(&shared.wal, owner, 0).await?; + } return Err(error); } }; @@ -139,9 +161,6 @@ pub(crate) async fn submit_write( let plan = wal_append_outcome.plan; let wal_lsn = wal_append_outcome.wal_lsn; let resolved_now_ms = wal_append_outcome.resolved_now_ms; - if let (Some(window), Some(lsn)) = (&window, wal_lsn) { - window.note_minted(lsn); - } // Build the wire request and hand it to the Data-Plane dispatcher. let dispatched = dispatch_to_data_plane( @@ -168,22 +187,12 @@ pub(crate) async fn submit_write( let dispatch_outcome = match dispatched { Ok(outcome) => outcome, Err(error) => { - // The dispatcher refused the request, so no core applied it. The - // record is cancelled before the window settles. A failed cancel - // leaves the window open: restart must still replay from here. - abort_undispatched_write( - shared, - AbortTarget { - tenant_id, - database_id, - vshard_id, - wal_lsn, - appends_here, - final_refusal_key: 0, - }, - ) - .await?; - settle_window(window); + // The dispatcher refused the request, so no core applied it. A + // dispatch refusal depends on this node's load, so the markers + // carry no proposal key. + if let Some(minted) = minted { + minted.cancel(&shared.wal, owner, 0).await?; + } return Err(error); } }; @@ -211,7 +220,7 @@ pub(crate) async fn submit_write( change_set, ddl_transition, deferred_guards: dispatch_outcome.deferred_guards, - window, + minted, }, ) .await diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs index 843372a53..dcaa1ca98 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs @@ -17,12 +17,12 @@ //! only surfaces as lost data after a crash, or as a change stream that never //! fires. Add the step here, once, and every caller gets it. //! -//! It also owns the mirror of the redo append: when it appended the record -//! itself and the Data Plane then REFUSED the write, it cancels that record -//! before returning the error. See +//! It also owns the mirror of the redo append: every record of the write, the +//! ones it appended and the ones a caller appended under their outcome-floor +//! window, is cancelled when the Data Plane refuses the write with a verdict +//! that proves nothing applied. See //! [`crate::control::server::dispatch_utils::write_abort`] for which verdicts -//! qualify, the residual crash window it does not close, and the latency it -//! costs a rejection. +//! qualify, and the `minted` module for how each path closes the window. //! //! Split by concern, run in this fixed order by [`driver::submit_write`]: //! - [`admission`]: the write-admission gate. diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs index 684328dfa..77bb6dc2d 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -4,20 +4,20 @@ //! post-apply steps a successful write still owes: the post-apply redo, the //! durable-at-ack barrier, DDL finalization, and the change-event publish. +use std::sync::Arc; use std::time::Instant; -use tokio::sync::mpsc; - -use crate::bridge::dispatch::WriteWindow; -use crate::bridge::envelope::{Response, Status}; +use crate::bridge::envelope::Status; use crate::control::array_catalog::ddl::AuthorizedDdlTransition; use crate::control::server::dispatch_utils::change_events::{WriteChangeSet, publish_change_set}; use crate::control::server::dispatch_utils::collect::{ DispatchCollectError, collect_bounded_response, }; use crate::control::server::dispatch_utils::durability_barrier::assert_durable_before_ack; +use crate::control::server::dispatch_utils::minted::{ + Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, +}; use crate::control::server::dispatch_utils::submit_write::ambiguous_ddl::preserve_ambiguous_array_ddl; -use crate::control::server::dispatch_utils::write_abort::{AbortTarget, abort_refused_write}; use crate::control::server::wal_dispatch; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, RequestId, TenantId, VShardId}; @@ -29,7 +29,7 @@ use super::wal_append::rollback_on_err; /// append, and dispatch phases that ran before it. pub(super) struct ResponsePhaseInput { pub request_id: RequestId, - pub rx: mpsc::Receiver, + pub rx: crate::control::ResponseReceiver, pub deadline: Instant, pub dispatch_started: Instant, pub tenant_id: TenantId, @@ -41,30 +41,24 @@ pub(super) struct ResponsePhaseInput { /// `WalDurability::AppendHere`). pub apply_key: u64, /// The key a final refusal's abort marker carries, `0` when this write's - /// refusals are not final (see `AbortTarget::final_refusal_key`). + /// refusals are not final. A final refusal is the proposal's outcome: the + /// proposal ledger rebuilt at boot counts the key as applied. pub final_refusal_key: u64, pub post_apply: Option, pub funnel_redo_engine: Option<&'static str>, pub change_set: Option, pub ddl_transition: AuthorizedDdlTransition, pub deferred_guards: super::dispatch::DeferredGuards, - /// The write's outcome-floor window, when the funnel minted its LSN. - pub window: Option, -} - -/// Settle a write's outcome-floor window: its outcome is final. -pub(super) fn settle_window(window: Option) { - if let Some(window) = window { - window.settle(); - } + /// The records minted for this write, under their outcome-floor window. + pub minted: Option, } /// Collect the response(s), classify the outcome, and run every step a /// completed write still owes before the funnel returns. /// /// For non-streaming queries, exactly one response arrives. For streaming -/// queries, multiple partial chunks arrive before the final. The mpsc channel -/// is bounded (see `RequestTracker::register`); here the *total* accumulated +/// queries, multiple partial chunks arrive before the final. The partial +/// channel is bounded (see `RequestTracker::register`); here the *total* accumulated /// payload is additionally capped so a runaway scan can't pin Control-Plane /// RAM — any query whose combined result exceeds /// `tuning.network.max_query_result_bytes` is cancelled with a typed @@ -76,7 +70,7 @@ pub(super) async fn collect_classify_and_finish( ) -> crate::Result { let ResponsePhaseInput { request_id, - mut rx, + rx, deadline, dispatch_started, tenant_id, @@ -91,8 +85,13 @@ pub(super) async fn collect_classify_and_finish( change_set, ddl_transition, deferred_guards, - window, + minted, } = input; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id, + }; let vshard_u32 = vshard_id.as_u32(); let observe = |shared: &SharedState| { @@ -100,20 +99,40 @@ pub(super) async fn collect_classify_and_finish( shared.per_vshard_metrics.observe(vshard_u32, latency_us); }; - // The same instant the envelope carries. The Data Plane normally answers - // with `DeadlineExceeded` first; this bounds the wait when it is inside a - // stage that carries no safe point yet. - let response = match tokio::time::timeout_at( - tokio::time::Instant::from_std(deadline), - collect_bounded_response(&mut rx, max_result_bytes), - ) - .await - { - Ok(response) => response, - Err(_) => { - // The dispatcher holds the floor below this record until the core - // answers, so this window can settle. - settle_window(window); + // Wait to the same instant the envelope carries. The Data Plane normally + // answers with `DeadlineExceeded` first; this bounds the wait when it is + // inside a stage that carries no safe point yet. A write's records close + // in a task this future does not own, so a caller dropped mid-wait still + // closes them. A refusal that arrives after the deadline still cancels + // them. + let outcome = match minted { + Some(minted) => { + await_response_owned( + OwnedWait { + wal: Arc::clone(&shared.wal), + owner, + final_refusal_key, + deadline, + collect: Collect::Merged { max_result_bytes }, + }, + rx, + minted, + ) + .await? + } + None => collect_unminted(shared, request_id, rx, deadline, max_result_bytes).await, + }; + + let response = match outcome { + OwnedResponse::Answered { response, closed } => { + if response.status != Status::Ok { + let _ = ddl_transition.rollback(shared); + } + // A failed cancel holds the window and fails the write here. + closed?; + response + } + OwnedResponse::DeadlineExceeded => { observe(shared); // Dispatch completed, but the Data Plane may have applied CREATE // or ALTER before this deadline. Never roll that catalog state @@ -124,14 +143,7 @@ pub(super) async fn collect_classify_and_finish( } return Err(crate::Error::DeadlineExceeded { request_id }); } - }; - - let response = match response { - Ok(r) => r, - Err(DispatchCollectError::OverBudget { bytes }) => { - // The dispatcher holds the floor until the core's final response. - settle_window(window); - shared.tracker.cancel(&request_id); + OwnedResponse::OverBudget { bytes } => { observe(shared); // A partial response proves dispatch began but not whether an // Array DDL completed; preserve CREATE/ALTER and fail-stop. @@ -146,9 +158,7 @@ pub(super) async fn collect_classify_and_finish( ), }); } - Err(DispatchCollectError::ChannelClosed) => { - // The dispatcher holds the floor until the core's final response. - settle_window(window); + OwnedResponse::ChannelClosed => { observe(shared); // The producer can close after applying but before sending its // response. CREATE/ALTER must remain catalog-finalized here. @@ -168,26 +178,6 @@ pub(super) async fn collect_classify_and_finish( } }; - if response.status != Status::Ok { - let _ = ddl_transition.rollback(shared); - abort_refused_write( - shared, - AbortTarget { - tenant_id, - database_id, - vshard_id, - wal_lsn, - appends_here, - final_refusal_key, - }, - &response, - ) - .await?; - } - // The core's outcome is final, and a refusal's abort marker is durable. A - // failed abort above returns first and leaves the window open. - settle_window(window); - // Mint the post-apply redo record while the guards are still held, then // release them. A PointUpdate whose collection carries a secondary vector // index returns its surrogate + post-image in `write_set`; without this @@ -273,3 +263,31 @@ pub(super) async fn collect_classify_and_finish( observe(shared); Ok(SubmitOutcome { response, wal_lsn }) } + +/// Collect a response that carries no records, under the same deadline and +/// byte budget a write's owned wait applies. +async fn collect_unminted( + shared: &SharedState, + request_id: RequestId, + mut rx: crate::control::ResponseReceiver, + deadline: Instant, + max_result_bytes: usize, +) -> OwnedResponse { + let collected = tokio::time::timeout_at( + tokio::time::Instant::from_std(deadline), + collect_bounded_response(&mut rx, max_result_bytes), + ) + .await; + match collected { + Ok(Ok(response)) => OwnedResponse::Answered { + response, + closed: Ok(()), + }, + Ok(Err(DispatchCollectError::OverBudget { bytes })) => { + shared.tracker.cancel(&request_id); + OwnedResponse::OverBudget { bytes } + } + Ok(Err(DispatchCollectError::ChannelClosed)) => OwnedResponse::ChannelClosed, + Err(_) => OwnedResponse::DeadlineExceeded, + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs index efc382bc6..b974711d3 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs @@ -11,9 +11,10 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::array_catalog::ddl::AuthorizedDdlTransition; +use crate::control::server::dispatch_utils::minted::{MintedRecords, RecordOwner}; use crate::control::server::wal_dispatch::{self, WalAppendRequest}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::types::Lsn; use super::super::params::WalDurability; @@ -59,14 +60,21 @@ pub(super) fn rollback_on_err( /// Plane is about to execute must name the record that reproduces it. This is /// the only place that knows both, and it knows them for every caller: no /// upstream path may allocate an LSN of its own and hope it matches. +/// +/// An `AppendHere` write appends through `minted`, so every record it writes +/// joins the write's outcome-floor window. pub(super) fn authorize_and_append( shared: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - vshard_id: VShardId, + owner: RecordOwner, mut plan: PhysicalPlan, durability: WalDurability, + minted: Option<&MintedRecords>, ) -> crate::Result { + let RecordOwner { + tenant_id, + database_id, + vshard_id, + } = owner; let ddl_transition = crate::control::array_catalog::ddl::apply_authorized_ddl( shared, tenant_id, @@ -83,7 +91,10 @@ pub(super) fn authorize_and_append( shared, &ddl_transition, wal_dispatch::wal_append(WalAppendRequest { - wal: shared.wal.appender(apply_key), + wal: match minted { + Some(minted) => minted.appender(&shared.wal, apply_key), + None => shared.wal.appender(apply_key), + }, tenant_id, vshard_id, database_id, @@ -97,6 +108,7 @@ pub(super) fn authorize_and_append( WalDurability::CallerSupplied { wal_lsn, resolved_now_ms, + .. } => (wal_lsn, resolved_now_ms), }; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs index 05a57d7e0..0468a3630 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs @@ -10,6 +10,7 @@ use std::sync::Arc; use crate::bridge::envelope::{PhysicalPlan, Response}; +use crate::control::server::dispatch_utils::minted::MintedRecords; use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; /// Who owns this write's durable redo record. @@ -34,12 +35,29 @@ pub(crate) enum WalDurability { /// sync path that owns its own funnel — and supplies the LSN it minted. /// The funnel appends nothing and stamps these values through unchanged; /// the supplied LSN names the record that replays this write. + /// + /// `minted` holds the records the caller appended for this write under + /// their outcome-floor window. The funnel closes the window from the + /// write's outcome: it cancels the records on a refusal that applied + /// nothing, and on a Calvin route that applies the write from its own + /// records. CallerSupplied { wal_lsn: Option, resolved_now_ms: Option, + minted: Option, }, } +impl WalDurability { + /// Take the caller's minted records out, leaving `None` in their place. + pub(crate) fn take_minted(&mut self) -> Option { + match self { + Self::AppendHere { .. } => None, + Self::CallerSupplied { minted, .. } => minted.take(), + } + } +} + /// Where this write's ordering was decided. pub(crate) enum WriteOrdering { /// Run the write-admission gate: fast path, per-key order lock, or a route diff --git a/nodedb/src/control/server/dispatch_utils/types.rs b/nodedb/src/control/server/dispatch_utils/types.rs index ae600589a..317b11d77 100644 --- a/nodedb/src/control/server/dispatch_utils/types.rs +++ b/nodedb/src/control/server/dispatch_utils/types.rs @@ -39,6 +39,10 @@ pub(crate) struct WriteDispatch { /// record carries instead of re-reading the clock at apply time. `None` /// for reads and other writes. pub resolved_now_ms: Option, + /// The records the caller appended for this write, under their + /// outcome-floor window. The funnel closes the window from the write's + /// outcome. `None` when the caller appended nothing for this dispatch. + pub minted: Option, } /// Inputs for `dispatch_to_data_plane_inner`: the Data Plane request identity diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index 89e3f612e..746251ebf 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -22,126 +22,11 @@ //! The match is exhaustive on purpose. A new [`ErrorCode`] must be classified by //! whoever adds it, not silently inherit either answer. //! -//! [`abort_refused_write`] is the one place that acts on that verdict, and the -//! write funnel is its only caller — every `AppendHere` write in every engine, -//! including the Raft apply loop, passes through there. +//! [`resolve_on_response`](super::minted::resolve_on_response) is the one +//! place that acts on that verdict. Every write that mints a record for a +//! Data-Plane dispatch resolves its records there. -use crate::bridge::envelope::{ErrorCode, Response}; -use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; - -/// Identity of the forward record an abort marker would name. -pub(crate) struct AbortTarget { - pub tenant_id: TenantId, - pub database_id: DatabaseId, - pub vshard_id: VShardId, - /// The forward write's redo LSN, if one was minted at all. - pub wal_lsn: Option, - /// Whether the write funnel appended the forward record. A caller that - /// recorded durability elsewhere owns the undo semantics of its own record. - pub appends_here: bool, - /// The idempotency key of the replicated proposal whose refusal is final, - /// `0` otherwise. A final refusal is the proposal's outcome: its abort - /// marker carries the key, and the proposal ledger rebuilt at boot counts - /// the proposal as applied. The cancelled forward record never counts. - pub final_refusal_key: u64, -} - -/// Cancel a forward write record the Data Plane refused. -/// -/// The write funnel appends the redo record before the Data Plane has decided, -/// so a refusal arrives with the record already in the log and restart replay -/// would re-apply the very write the client was told was rejected. This writes a -/// `WriteAborted` marker naming that record, then waits for it to be fsynced -/// BEFORE the error is returned: the refusal path performs no fsync of its own, -/// so an abort left buffered is volatile while the forward record may already be -/// durable via a concurrent writer's group commit. -/// -/// Cost: a rejected write now pays a WAL append plus an fsync wait it did not -/// pay before. That is a deliberate trade of refusal latency for the guarantee -/// that a refusal, once acknowledged, stays refused. -/// -/// **Known residual, not closed:** a crash BEFORE the abort record is durable -/// can still resurrect the write, because the forward record's durability is not -/// gated on the verdict. Closing that window means holding the per-key order -/// guard across the whole Data-Plane round trip, which would serialize same-key -/// writes on every engine's common path. What this guarantees is the ACKED case: -/// once the client has been told the write was refused, a restart cannot make it -/// appear. -/// -/// An append or fsync failure here is propagated, not logged: continuing would -/// return the refusal while leaving the forward record replayable, which is -/// exactly the bug this exists to prevent. -pub(crate) async fn abort_refused_write( - shared: &SharedState, - target: AbortTarget, - response: &Response, -) -> crate::Result<()> { - if !target.appends_here { - return Ok(()); - } - let Some(wal_lsn) = target.wal_lsn else { - return Ok(()); - }; - // A rejection is only cancellable when the verdict itself proves nothing - // was installed. An ambiguous failure keeps its forward record, because - // erasing a write that actually landed is worse than replaying one that - // did not. - let Some(code) = response.error_code.as_deref() else { - return Ok(()); - }; - if !write_definitely_not_applied(code) { - return Ok(()); - } - - let marker_key = if refusal_is_final(code) { - target.final_refusal_key - } else { - 0 - }; - append_abort_marker(shared, &target, wal_lsn, marker_key).await -} - -/// Cancel a forward write record whose request the dispatcher refused. -/// -/// The request never reached a core, so the Data Plane applied nothing. The -/// marker carries no proposal key: a dispatch refusal depends on this node's -/// load at that moment, so a redelivery of the same entry can apply it. -pub(crate) async fn abort_undispatched_write( - shared: &SharedState, - target: AbortTarget, -) -> crate::Result<()> { - if !target.appends_here { - return Ok(()); - } - let Some(wal_lsn) = target.wal_lsn else { - return Ok(()); - }; - append_abort_marker(shared, &target, wal_lsn, 0).await -} - -/// Append a `WriteAborted` marker naming `wal_lsn` and wait until it is -/// durable. -async fn append_abort_marker( - shared: &SharedState, - target: &AbortTarget, - wal_lsn: Lsn, - marker_key: u64, -) -> crate::Result<()> { - let abort_lsn = shared.wal.appender(marker_key).append_write_aborted( - target.tenant_id, - target.vshard_id, - target.database_id, - wal_lsn, - )?; - shared.wal.wait_durable(abort_lsn).await?; - tracing::debug!( - aborted_lsn = wal_lsn.as_u64(), - abort_lsn = abort_lsn.as_u64(), - "refused write cancelled in the WAL" - ); - Ok(()) -} +use crate::bridge::envelope::ErrorCode; /// Whether a replicated proposal refused with `code` is refused for good: a /// redelivery of the same entry against the same state refuses it again. @@ -149,7 +34,8 @@ async fn append_abort_marker( /// A verdict that depends on this node's momentary load or on a transient /// precondition is not final: another replica can apply the same entry, and /// a redelivery here can too. That covers admission and capacity verdicts, -/// concurrency retries, the staging byte budget, and `RetryableRefusal`, +/// a task that expired before it started, concurrency retries, the staging +/// byte budget, and `RetryableRefusal`, /// which a committed-redo apply answers with after it rolled a failed /// install back. pub(crate) fn refusal_is_final(code: &ErrorCode) -> bool { @@ -160,6 +46,7 @@ pub(crate) fn refusal_is_final(code: &ErrorCode) -> bool { | ErrorCode::RateExceeded { .. } | ErrorCode::CollectionDraining { .. } | ErrorCode::DispatchCapacity { .. } + | ErrorCode::ExpiredBeforeExecution | ErrorCode::ConflictRetry | ErrorCode::OllpRetryRequired | ErrorCode::TxnOverlayMemoryExceeded { .. } @@ -195,6 +82,8 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { | ErrorCode::CollectionDraining { .. } | ErrorCode::DispatchCapacity { .. } | ErrorCode::Unsupported { .. } + // The deadline passed before the core started the task. + | ErrorCode::ExpiredBeforeExecution // The target row or collection did not exist, so the write had nothing // to mutate. | ErrorCode::NotFound @@ -276,6 +165,17 @@ mod tests { })); } + /// A task that expired before its core started it ran nothing, so the + /// record aborts. A redelivery can still run it, so the refusal is not + /// final. + #[test] + fn a_task_that_never_started_aborts_the_record_but_is_not_final() { + assert!(write_definitely_not_applied( + &ErrorCode::ExpiredBeforeExecution + )); + assert!(!refusal_is_final(&ErrorCode::ExpiredBeforeExecution)); + } + /// The asymmetry that keeps this safe: an ambiguous outcome must never /// produce an abort, because the write it would erase may have landed. #[test] diff --git a/nodedb/src/control/server/exchange/gather.rs b/nodedb/src/control/server/exchange/gather.rs index 540afd36f..6c171a12e 100644 --- a/nodedb/src/control/server/exchange/gather.rs +++ b/nodedb/src/control/server/exchange/gather.rs @@ -60,7 +60,7 @@ pub(crate) fn eager_dispatch_to_all_cores( Vec<( usize, crate::types::RequestId, - tokio::sync::mpsc::Receiver, + crate::control::ResponseReceiver, )>, > { // Every core in this fan-out belongs to ONE statement, so all of them diff --git a/nodedb/src/control/server/http/routes/health.rs b/nodedb/src/control/server/http/routes/health.rs index 31e406c3a..d8dbc95e4 100644 --- a/nodedb/src/control/server/http/routes/health.rs +++ b/nodedb/src/control/server/http/routes/health.rs @@ -156,8 +156,33 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { return (status, axum::Json(body)); } + // A write window open past the longest statement deadline holds the + // outcome floor, so no checkpoint on this node advances past it. The node + // serves, so it reports degraded. + if let Some(stuck) = state + .shared + .outcome_floor + .stuck(outcome_floor_bound(&state)) + { + let body = json!({ + "status": "degraded", + "reason": "outcome_floor_stuck", + "node_id": state.shared.node_id, + "outcome_floor": stuck.floor.as_u64(), + "oldest_window_horizon": stuck.horizon.as_u64(), + "oldest_window_open_secs": stuck.open_for.as_secs(), + "open_windows": stuck.open_windows, + "leaked_windows": state.shared.outcome_floor.leaked_windows(), + "held_windows": state.shared.outcome_floor.held_windows(), + }); + return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); + } + let health = crate::control::startup::health::observe(&state.shared.startup); - let (status, body) = crate::control::startup::health::to_http_response(&health); + let (status, mut body) = crate::control::startup::health::to_http_response(&health); + // A held window keeps the outcome floor below it by design, so it never + // degrades readiness. The count shows how many restart replay will reach. + body["held_windows"] = json!(state.shared.outcome_floor.held_windows()); // Checked only once the startup gate is otherwise green, so a node still // advancing through phases keeps reporting the phase it is stuck in. if status == StatusCode::OK @@ -173,6 +198,25 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { (status, axum::Json(body)) } +/// How long a write window can hold the outcome floor before readiness reports +/// it: the longest path from a window's open to its final outcome. +/// +/// - A statement write reaches its outcome by the longest statement deadline, +/// plus the wait the node gives a committed entry to apply. +/// - A vector index install waits two core dispatch deadlines. +/// +/// A window older than both is stuck. +fn outcome_floor_bound(state: &AppState) -> std::time::Duration { + let network = &state.shared.tuning.network; + let statement = std::time::Duration::from_secs( + network + .default_deadline_secs + .max(network.copy_deadline_secs), + ) + .saturating_add(crate::control::metadata_proposer::DEFAULT_PROPOSE_TIMEOUT); + statement.max(crate::control::catalog_entry::post_apply::vector_install_longest_core_wait()) +} + /// Why a cross-shard Calvin write would be refused on this node right now, /// or `None` when one would be accepted. /// diff --git a/nodedb/src/control/server/http/routes/metrics.rs b/nodedb/src/control/server/http/routes/metrics.rs index 99ef5cb0f..8ee8d5c61 100644 --- a/nodedb/src/control/server/http/routes/metrics.rs +++ b/nodedb/src/control/server/http/routes/metrics.rs @@ -50,6 +50,42 @@ pub async fn metrics( output.push_str("# TYPE nodedb_wal_next_lsn gauge\n"); output.push_str(&format!("nodedb_wal_next_lsn {wal_lsn}\n\n")); + // Outcome floor: every engine watermark and WAL truncation stays at or + // below it. + let outcome_floor = &state.shared.outcome_floor; + output.push_str( + "# HELP nodedb_outcome_floor_lsn Highest WAL LSN at or below which every dispatched record has a final outcome.\n", + ); + output.push_str("# TYPE nodedb_outcome_floor_lsn gauge\n"); + output.push_str(&format!( + "nodedb_outcome_floor_lsn {}\n\n", + outcome_floor.floor().as_u64() + )); + output.push_str( + "# HELP nodedb_outcome_floor_oldest_window_seconds Age of the oldest write window holding the outcome floor.\n", + ); + output.push_str("# TYPE nodedb_outcome_floor_oldest_window_seconds gauge\n"); + output.push_str(&format!( + "nodedb_outcome_floor_oldest_window_seconds {}\n\n", + outcome_floor.oldest_open_for().as_secs_f64() + )); + output.push_str( + "# HELP nodedb_outcome_floor_windows_leaked_total Write windows dropped before their write's outcome was final.\n", + ); + output.push_str("# TYPE nodedb_outcome_floor_windows_leaked_total counter\n"); + output.push_str(&format!( + "nodedb_outcome_floor_windows_leaked_total {}\n\n", + outcome_floor.leaked_windows() + )); + output.push_str( + "# HELP nodedb_outcome_floor_windows_held Write windows held until restart; the floor stays below each.\n", + ); + output.push_str("# TYPE nodedb_outcome_floor_windows_held gauge\n"); + output.push_str(&format!( + "nodedb_outcome_floor_windows_held {}\n\n", + outcome_floor.held_windows() + )); + // Node ID. output.push_str("# HELP nodedb_node_id This node's cluster ID.\n"); output.push_str("# TYPE nodedb_node_id gauge\n"); diff --git a/nodedb/src/control/server/native/dispatch/conversion.rs b/nodedb/src/control/server/native/dispatch/conversion.rs index baa151254..a69792b3d 100644 --- a/nodedb/src/control/server/native/dispatch/conversion.rs +++ b/nodedb/src/control/server/native/dispatch/conversion.rs @@ -67,7 +67,14 @@ pub(crate) fn error_to_native(seq: u64, e: &crate::Error) -> NativeResponse { crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate(e.code()), e.message().to_string(), ), - other => ("XX000", format!("{other}")), + // Every other variant takes the protocol-neutral SQLSTATE pgwire + // renders for it. `XX000` is only for a variant that table leaves + // unclassified. + other => { + let (_severity, sqlstate, message) = + crate::control::server::pgwire::types::error_map::error_to_sqlstate(other); + (sqlstate, message) + } }; let ndb_code = crate::error_classify::classify(e).code().0; NativeResponse::error_with_code(seq, code, message, ndb_code) @@ -540,11 +547,10 @@ mod tests { ); } - /// The numeric code is populated for every variant, including the ones - /// whose SQLSTATE falls through to `XX000` — otherwise the fix would be a - /// per-variant special case rather than one classification. + /// A variant with no native arm takes the SQLSTATE pgwire renders for it, + /// and keeps its numeric code. #[test] - fn errors_without_a_dedicated_sqlstate_still_carry_a_code() { + fn errors_without_a_native_arm_take_the_pgwire_sqlstate() { let response = error_to_native( 1, &crate::Error::PlanError { @@ -555,11 +561,33 @@ mod tests { let error = response .error .expect("error responses must carry a payload"); - assert_eq!(error.code, "XX000"); + assert_eq!(error.code, nodedb_types::error::sqlstate::SYNTAX_ERROR); assert_eq!( error.ndb_code, nodedb_types::error::ErrorCode::PLAN_ERROR.0, - "an unmapped SQLSTATE must not also erase the numeric classification" + "the numeric classification must survive the SQLSTATE rendering" + ); + } + + /// A constraint refusal that crossed a node boundary keeps its SQLSTATE. + #[test] + fn a_rejected_constraint_keeps_its_sqlstate() { + let response = error_to_native( + 1, + &crate::Error::RejectedConstraint { + collection: "c".to_owned(), + constraint: "unique".to_owned(), + detail: "duplicate key".to_owned(), + }, + ); + + let error = response + .error + .expect("error responses must carry a payload"); + assert_eq!(error.code, nodedb_types::error::sqlstate::UNIQUE_VIOLATION); + assert_eq!( + error.ndb_code, + nodedb_types::error::ErrorCode::CONSTRAINT_VIOLATION.0 ); } } diff --git a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs index 540bb7a5e..dd4900a1d 100644 --- a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs @@ -5,7 +5,6 @@ use crate::bridge::envelope::{Payload, PhysicalPlan, Response, Status}; use std::sync::Arc; -use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext as GatewayQueryContext; use crate::control::gateway::router::is_task_vshard_scoped; use crate::control::server::shared::clone_write::CloneCheckedOutcome; @@ -73,19 +72,9 @@ pub(super) async fn dispatch_authorized_single_task( database_id: ctx.database_id(), txn_id, }; - gateway - .execute(&query, checked) - .await - .map(gateway_payloads_to_response) - .map_err(|error| match error { - // A capacity refusal keeps its type so the client sees - // the retryable overload class. - capacity @ crate::Error::DispatchCapacity { .. } => capacity, - other => { - let (_, detail) = GatewayErrorMap::to_native(&other); - crate::Error::Dispatch { detail } - } - }) + // The typed error passes through unchanged. The native frame + // renders its SQLSTATE and numeric code from it. + gateway.execute_response(&query, checked).await } None => dispatch_without_gateway(ctx, checked).await, } @@ -205,23 +194,3 @@ pub(super) async fn dispatch_without_gateway( write().await } } - -fn gateway_payloads_to_response(payloads: Vec>) -> Response { - let payload = payloads - .into_iter() - .next() - .map(Payload::from_vec) - .unwrap_or_else(Payload::empty); - Response { - request_id: RequestId::new(0), - status: Status::Ok, - attempt: 0, - partial: false, - payload, - watermark_lsn: Lsn::ZERO, - error_code: None, - read_set_valid: None, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - } -} diff --git a/nodedb/src/control/server/native/dispatch/sql_gateway.rs b/nodedb/src/control/server/native/dispatch/sql_gateway.rs index 94eade89b..2e350dffe 100644 --- a/nodedb/src/control/server/native/dispatch/sql_gateway.rs +++ b/nodedb/src/control/server/native/dispatch/sql_gateway.rs @@ -3,20 +3,19 @@ //! Gateway-based SQL task dispatch for the native protocol. //! //! When `SharedState.gateway` is `Some`, tasks are routed through -//! `Gateway::execute` which handles cluster-aware routing, typed `NotLeader` +//! `Gateway::execute_response` which handles cluster-aware routing, typed `NotLeader` //! retry, and plan caching. The `None` fallback retains the original //! `dispatch_to_data_plane` path for single-node boot before the gateway is //! wired. This is native's SQL-TEXT opcode path — distinct from //! `raw_dispatch.rs`, which serves only native's direct-op opcodes. -use crate::bridge::envelope::{Payload, Response, Status}; +use crate::bridge::envelope::Response; use std::sync::Arc; -use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext as GatewayQueryContext; use crate::control::gateway::router::is_task_vshard_scoped; use crate::control::server::shared::clone_write::CloneCheckedOutcome; -use crate::types::{Lsn, RequestId, TraceId}; +use crate::types::TraceId; use nodedb_physical::physical_task::PhysicalTask; use super::DispatchCtx; @@ -49,8 +48,8 @@ pub(super) fn authorize_native_task( /// Dispatch a single `PhysicalTask` through the gateway when available, /// falling back to the local SPSC path. /// -/// Returns a synthetic `Response` shaped identically to the SPSC path so that -/// the calling code in `sql.rs` is unchanged. +/// Both paths return the Data-Plane `Response` shape, with a `NotFound` +/// verdict as an error status. pub(super) async fn dispatch_task_via_gateway( ctx: &DispatchCtx<'_>, task: PhysicalTask, @@ -95,15 +94,9 @@ pub(super) async fn dispatch_task_via_gateway( // dispatch resolves the per-txn staging overlay. txn_id, }; - gw.execute(&gw_ctx, checked) - .await - .map_err(|e| { - let (code, msg) = GatewayErrorMap::to_native(&e); - crate::Error::Internal { - detail: format!("gateway error {code}: {msg}"), - } - }) - .map(payloads_to_response) + // The typed error passes through unchanged. The native frame + // renders its SQLSTATE and numeric code from it. + gw.execute_response(&gw_ctx, checked).await } None => { crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( @@ -115,28 +108,3 @@ pub(super) async fn dispatch_task_via_gateway( } } } - -/// Convert gateway `Vec>` payloads into a synthetic `Response`. -/// -/// Mirrors the same conversion used in the RESP gateway_dispatch module: -/// the first payload is used as the response body; an empty `Vec` yields an -/// empty payload with `Status::Ok`. -fn payloads_to_response(payloads: Vec>) -> Response { - let payload = payloads - .into_iter() - .next() - .map(Payload::from_vec) - .unwrap_or_else(Payload::empty); - Response { - request_id: RequestId::new(0), - status: Status::Ok, - attempt: 0, - partial: false, - payload, - watermark_lsn: Lsn::new(0), - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - } -} diff --git a/nodedb/src/control/server/native/dispatch/transaction.rs b/nodedb/src/control/server/native/dispatch/transaction.rs index 6a1a0dc6e..597cf3766 100644 --- a/nodedb/src/control/server/native/dispatch/transaction.rs +++ b/nodedb/src/control/server/native/dispatch/transaction.rs @@ -21,7 +21,6 @@ use crate::control::server::shared::session::{ AbortReason, CommitOutcome, TxnDataPlane, commit, lifecycle, }; use crate::control::state::SharedState; -use crate::types::Lsn; use nodedb_physical::physical_task::PhysicalTask; use super::super::super::dispatch_utils; @@ -43,7 +42,6 @@ impl TxnDataPlane for NativeTxnDp<'_> { fn dispatch_no_wal<'a>( &'a self, task: PhysicalTask, - wal_lsn: Option, ) -> Pin> + Send + 'a>> { let state = self.state; Box::pin(async move { @@ -57,10 +55,11 @@ impl TxnDataPlane for NativeTxnDp<'_> { trace_id: TraceId::ZERO, event_source: crate::event::EventSource::User, txn_id: None, - wal_lsn, + wal_lsn: None, // Batch COMMIT record, not per-task WAL append — see // `dispatch_task_no_wal`'s equivalent limitation. resolved_now_ms: None, + minted: None, }, ) .await diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs index cbbff6317..acdcc6a27 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs @@ -42,7 +42,6 @@ impl NodeDbPgHandler { &self, task: PhysicalTask, user_id: Option>, - wal_lsn: Option, ) -> crate::Result { // Without this, a transaction begun before the freeze could COMMIT mid-scan and // break the as-of contract. @@ -57,8 +56,8 @@ impl NodeDbPgHandler { } reject_unadmitted_crdt_apply(&task.plan)?; let txn_id = task.txn_id; - // Writes were durably recorded under one `RecordType::Transaction` record at - // COMMIT; per-task WAL append is skipped. `wal_lsn` stamps that record's LSN. + // The caller owns the transaction's durability, so the task carries no + // WAL record of its own. self.submit_to_data_plane(SubmitArgs { tenant_id: task.tenant_id, vshard_id: task.vshard_id, @@ -69,8 +68,9 @@ impl NodeDbPgHandler { // No per-task TTL instant (see `flush_transaction_buffer`), so a TTL-bearing // KV write falls back to `epoch_system_ms` at apply time. durability: WalDurability::CallerSupplied { - wal_lsn, + wal_lsn: None, resolved_now_ms: None, + minted: None, }, }) .await diff --git a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs index 4f8a6a6e2..d82e2f6f5 100644 --- a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs +++ b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs @@ -38,9 +38,8 @@ impl TxnDataPlane for PgwireTxnDp<'_> { fn dispatch_no_wal<'a>( &'a self, task: PhysicalTask, - wal_lsn: Option, ) -> Pin> + Send + 'a>> { - Box::pin(self.handler.dispatch_task_no_wal(task, None, wal_lsn)) + Box::pin(self.handler.dispatch_task_no_wal(task, None)) } fn event_source(&self) -> crate::event::EventSource { diff --git a/nodedb/src/control/server/resp/gateway_dispatch.rs b/nodedb/src/control/server/resp/gateway_dispatch.rs index e930ba858..b27844de9 100644 --- a/nodedb/src/control/server/resp/gateway_dispatch.rs +++ b/nodedb/src/control/server/resp/gateway_dispatch.rs @@ -2,7 +2,7 @@ //! RESP gateway dispatch helpers. //! -//! Routes KV operations through `Gateway::execute` when the gateway is +//! Routes KV operations through `Gateway::execute_response` when the gateway is //! available (cluster-aware routing), falling back to direct local SPSC //! dispatch on single-node boot. //! @@ -11,7 +11,7 @@ use std::sync::Arc; -use crate::bridge::envelope::{Payload, PhysicalPlan, Response, Status}; +use crate::bridge::envelope::{PhysicalPlan, Response}; use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext; use crate::control::security::identity::AuthenticatedIdentity; @@ -21,7 +21,7 @@ use crate::control::server::shared::clone_write::CloneCheckedOutcome; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, RequestId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId, VShardId}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::session::RespSession; @@ -74,12 +74,11 @@ pub(super) async fn dispatch_kv( database_id: checked.database_id(), txn_id: None, }; - gw.execute(&gw_ctx, checked) + gw.execute_response(&gw_ctx, checked) .await .map_err(|e| crate::Error::Bridge { detail: GatewayErrorMap::to_resp(&e), }) - .map(gateway_payloads_to_response) } None => dispatch_utils::dispatch_authorized_to_data_plane(state, checked, TraceId::ZERO) .await @@ -136,12 +135,11 @@ pub(super) async fn dispatch_kv_write( database_id: checked.database_id(), txn_id: None, }; - gw.execute(&gw_ctx, checked) + gw.execute_response(&gw_ctx, checked) .await .map_err(|e| crate::Error::Bridge { detail: GatewayErrorMap::to_resp(&e), }) - .map(gateway_payloads_to_response) } None => dispatch_utils::dispatch_authorized_autocommit_write(state, checked, TraceId::ZERO) .await @@ -341,31 +339,6 @@ fn resp_auth_scope<'a, 'p>( ClientRequestScope::for_database(identity, stores, database_id, peer_addr) } -/// Convert gateway `Vec>` payloads into a synthetic `Response`. -/// -/// The RESP sub-handlers inspect `resp.status` and `resp.payload`; we -/// synthesise a `Status::Ok` response carrying the first payload so that all -/// existing sub-handler logic continues to work without modification. -fn gateway_payloads_to_response(payloads: Vec>) -> Response { - let payload = payloads - .into_iter() - .next() - .map(Payload::from_vec) - .unwrap_or_else(Payload::empty); - Response { - request_id: RequestId::new(0), - status: Status::Ok, - attempt: 0, - partial: false, - payload, - watermark_lsn: Lsn::new(0), - error_code: None, - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - } -} - /// Map bridge/dispatch errors to a BUSY error for Redis client compatibility. /// /// When the SPSC ring buffer is full or the Data Plane core is overloaded, @@ -384,11 +357,12 @@ fn map_busy_error(e: crate::Error) -> crate::Error { #[cfg(test)] mod tests { + use crate::bridge::envelope::{Payload, Status}; use crate::control::security::identity::{AuthMethod, DatabaseSet, Role}; use crate::control::security::metering::quota::QuotaManager; use crate::control::security::request_scope::AuthStores; use crate::control::security::scope::grant::ScopeGrantStore; - use crate::types::TenantId; + use crate::types::{Lsn, TenantId}; use super::*; diff --git a/nodedb/src/control/server/result_stream.rs b/nodedb/src/control/server/result_stream.rs index fc58612fc..6682bca5d 100644 --- a/nodedb/src/control/server/result_stream.rs +++ b/nodedb/src/control/server/result_stream.rs @@ -3,7 +3,7 @@ //! Durable streaming result abstraction. //! //! A dispatched scan returns its rows as a sequence of `Response` frames over a -//! `tokio::sync::mpsc::Receiver` (see `RequestTracker::register`): +//! `ResponseReceiver` (see `RequestTracker::register`): //! several `partial: true` frames followed by one terminal (`partial: false`) //! frame, each carrying a standalone msgpack-array payload of rows //! (`encode_raw_document_rows`). @@ -15,7 +15,7 @@ //! merged msgpack array for byte-demanding consumers that still need the //! fully-collected result. -use crate::bridge::envelope::{Response, Status}; +use crate::bridge::envelope::Status; use crate::control::server::dispatch_utils::reject_data_plane_error; use crate::control::server::payload_merge::merge_msgpack_arrays; use crate::types::Lsn; @@ -51,7 +51,7 @@ pub type ResultStream = /// ends after the terminal (`!partial`) frame is yielded, or when the channel /// closes. pub(crate) fn stream_response_channel( - mut rx: tokio::sync::mpsc::Receiver, + mut rx: crate::control::ResponseReceiver, max_result_bytes: usize, tolerate_not_found: bool, ) -> ResultStream { @@ -134,7 +134,7 @@ pub(crate) async fn materialize(mut stream: ResultStream) -> crate::Result<(Vec< #[cfg(test)] mod tests { use super::*; - use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::bridge::envelope::{ErrorCode, Payload, Response}; use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; use crate::types::RequestId; use tokio::sync::mpsc; @@ -213,7 +213,11 @@ mod tests { tx.send(partial(1000)).await.unwrap(); tx.send(final_frame(500)).await.unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + false, + ); let (merged, _lsn) = materialize(stream).await.unwrap(); assert_eq!( extract_msgpack_elements(&merged).len(), @@ -228,7 +232,11 @@ mod tests { tx.send(raw_partial(600)).await.unwrap(); tx.send(raw_partial(600)).await.unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1000, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1000, + false, + ); let err = materialize(stream).await.unwrap_err(); assert!(matches!(err, crate::Error::ExecutionLimitExceeded { .. })); } @@ -244,7 +252,11 @@ mod tests { .await .unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + false, + ); match materialize(stream).await { Err(crate::Error::DataPlane(ErrorCode::ResourcesExhausted)) => {} other => panic!("expected the shard's own code, got {other:?}"), @@ -262,7 +274,11 @@ mod tests { .await .unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + false, + ); match materialize(stream).await { Err(crate::Error::DeadlineExceeded { request_id }) => { assert_eq!(request_id, RequestId::new(1)); @@ -278,7 +294,11 @@ mod tests { let (tx, rx) = mpsc::channel(8); tx.send(error_frame(ErrorCode::NotFound)).await.unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, false); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + false, + ); assert!(materialize(stream).await.is_err()); } @@ -287,7 +307,11 @@ mod tests { let (tx, rx) = mpsc::channel(8); tx.send(error_frame(ErrorCode::NotFound)).await.unwrap(); drop(tx); - let stream = stream_response_channel(rx, 1 << 20, true); + let stream = stream_response_channel( + crate::control::ResponseReceiver::from_channel(rx), + 1 << 20, + true, + ); let (merged, _lsn) = materialize(stream).await.unwrap(); assert_eq!( extract_msgpack_elements(&merged).len(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs index 1058ba637..f1d36d72d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs @@ -23,6 +23,7 @@ //! cannot propagate and files a `Capture` instead. use crate::control::security::catalog::{IndexKind, StoredIndexRecord}; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId}; @@ -105,7 +106,15 @@ async fn secondary( field, }, ); - dispatch(state, tenant_id, database_id, &record.collection, plan).await + dispatch( + state, + tenant_id, + database_id, + &record.collection, + plan, + None, + ) + .await } /// Remove the vector index's durable build parameters and its Data Plane @@ -144,30 +153,54 @@ async fn vector( // WAL first: the `VectorParams` record that created this index is still // in the log, so without a durable drop record a restart rebuilds the // index the user just dropped. + // + // The record's outcome-floor window opens before the append and closes + // from the drop's outcome. let vshard = crate::types::VShardId::from_collection_in_database(database_id, &record.collection); - let appended = crate::control::server::wal_dispatch::wal_append_if_write( - &state.wal, + let owner = RecordOwner { tenant_id, - vshard, database_id, - &plan, - ) - .map_err(|e| err("XX000", format!("persist vector index drop to WAL: {e}")))?; + vshard_id: vshard, + }; + let minted = MintedRecords::open(&state.outcome_floor); + let appended = match minted.append_plan(&state.wal, owner, &plan) { + Ok(appended) => appended, + Err(e) => { + // Any record appended before the error never reaches a core. + minted + .cancel(&state.wal, owner, 0) + .await + .map_err(|c| err("XX000", format!("cancel vector index drop record: {c}")))?; + return Err(err( + "XX000", + format!("persist vector index drop to WAL: {e}"), + )); + } + }; // An append only buffers. The records this drop cancels were already // fsynced by the writes that acked them, so a buffered-only drop is lost on // restart while replay still rebuilds the index from those records. - let lsn = appended - .lsn - .ok_or_else(|| err("XX000", "vector index drop minted no WAL record"))?; - state - .wal - .wait_durable(lsn) - .await - .map_err(|e| err("XX000", format!("fsync vector index drop: {e}")))?; + let Some(lsn) = appended.lsn else { + minted.settle(); + return Err(err("XX000", "vector index drop minted no WAL record")); + }; + if let Err(e) = state.wal.wait_durable(lsn).await { + // The record can still be on disk, so restart replay can reach it. + minted.hold(); + return Err(err("XX000", format!("fsync vector index drop: {e}"))); + } - dispatch(state, tenant_id, database_id, &record.collection, plan).await + dispatch( + state, + tenant_id, + database_id, + &record.collection, + plan, + Some(minted), + ) + .await } /// Reset the collection's FTS binding once its last full-text index is gone. @@ -206,29 +239,47 @@ async fn fulltext( fuzzy_default: Some(false), }, ); - dispatch(state, tenant_id, database_id, &record.collection, plan).await + dispatch( + state, + tenant_id, + database_id, + &record.collection, + plan, + None, + ) + .await } /// Dispatch one teardown plan to the Data Plane, surfacing both transport and -/// handler-side failures. +/// handler-side failures. `minted` holds the record appended for the plan; +/// the funnel closes its outcome-floor window from the plan's outcome. async fn dispatch( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, collection: &str, plan: crate::bridge::envelope::PhysicalPlan, + minted: Option, ) -> Result<(), DdlError> { let vshard = crate::types::VShardId::from_collection_in_database(database_id, collection); - let response = crate::control::server::dispatch_utils::dispatch_to_data_plane( - state, - tenant_id, - database_id, - vshard, - plan, - TraceId::ZERO, - ) - .await - .map_err(|e| err("XX000", format!("index teardown dispatch failed: {e}")))?; + let response = + crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( + state, + crate::control::server::dispatch_utils::WriteDispatch { + tenant_id, + database_id, + vshard_id: vshard, + plan, + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + minted, + }, + ) + .await + .map_err(|e| err("XX000", format!("index teardown dispatch failed: {e}")))?; if response.status == crate::bridge::envelope::Status::Error { let detail = match response.error_code.as_deref() { diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs index 47a498739..74899fa31 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs @@ -219,7 +219,11 @@ pub async fn create_vector_index( // record plus the fan-out that reaches every core, not just the one the // pre-flight dispatched to. if outcome.needs_local_apply() { - crate::control::catalog_entry::post_apply::install_vector_index_params(stored, state).await; + let shared = state + .self_arc() + .map_err(|e| ddl_err("XX000", format!("install vector index params: {e}")))?; + crate::control::catalog_entry::post_apply::install_vector_index_params(stored, shared) + .await; } propose_index_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs index 23e63430d..f6c4b78fe 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs @@ -433,25 +433,31 @@ pub async fn set_node_labels( }; // Single-keyed on `node_id`, so single-home: route to `from_key(node_id)`. - // No redb durability — a WAL record is the bitset's only backing. - crate::control::server::wal_dispatch::wal_append_if_write( - &state.wal, + // No redb durability — a WAL record is the bitset's only backing. The + // record's outcome-floor window opens before the append and closes from + // the dispatch's outcome. + let owner = crate::control::server::dispatch_utils::RecordOwner { tenant_id, + database_id: DatabaseId::DEFAULT, vshard_id, - DatabaseId::DEFAULT, - &plan, - ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + }; + let minted = crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); + if let Err(e) = minted.append_plan(&state.wal, owner, &plan) { + // Any record appended before the error never reaches a core. + minted + .cancel(&state.wal, owner, 0) + .await + .map_err(|c| ddl_err("XX000", c.to_string()))?; + return Err(ddl_err("XX000", e.to_string())); + } let response = - crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_sync_response( + crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_minted_sync_response( state, - tenant_id, - DatabaseId::DEFAULT, - vshard_id, + owner, plan, - TraceId::ZERO, crate::event::EventSource::User, + minted, ) .await .map_err(|e| ddl_err("XX000", e.to_string()))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs index 5250632d3..364575f06 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs @@ -79,7 +79,11 @@ pub async fn handle_alter_vector_index_set( // Single node: no applier runs, so post-apply never fires. Run the // per-node install the post-apply lane runs everywhere else. if outcome.needs_local_apply() { - crate::control::catalog_entry::post_apply::install_vector_index_params(merged, state).await; + let shared = state + .self_arc() + .map_err(|e| ddl_err("XX000", format!("install vector index params: {e}")))?; + crate::control::catalog_entry::post_apply::install_vector_index_params(merged, shared) + .await; } state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs index 89d9ee55f..c94404772 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs @@ -306,21 +306,32 @@ async fn dispatch_sql( checked } }; - crate::control::server::wal_dispatch::wal_append_if_write( - &state.wal, - identity.tenant_id, - checked.vshard_id(), - checked.database_id(), - checked.plan(), - ) - .map_err(|e| err(sqlstate::IO_ERROR, format!("wal append: {e}")))?; - let response = crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( - state, - checked, - TraceId::ZERO, - ) - .await - .map_err(|e| err(sqlstate::CONNECTION_FAILURE, format!("dispatch: {e}")))?; + // The record's outcome-floor window opens before the append and + // closes from the task's outcome inside the funnel. + let owner = crate::control::server::dispatch_utils::RecordOwner { + tenant_id: identity.tenant_id, + database_id: checked.database_id(), + vshard_id: checked.vshard_id(), + }; + let minted = + crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); + if let Err(e) = minted.append_plan(&state.wal, owner, checked.plan()) { + // Any record appended before the error never reaches a core. + minted + .cancel(&state.wal, owner, 0) + .await + .map_err(|c| err(sqlstate::IO_ERROR, format!("cancel refresh record: {c}")))?; + return Err(err(sqlstate::IO_ERROR, format!("wal append: {e}"))); + } + let response = + crate::control::server::dispatch_utils::dispatch_authorized_minted_to_data_plane( + state, + checked, + TraceId::ZERO, + minted, + ) + .await + .map_err(|e| err(sqlstate::CONNECTION_FAILURE, format!("dispatch: {e}")))?; require_ok_response(&response)?; } Ok(()) diff --git a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs index 9c4862551..43323d4f6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs @@ -236,28 +236,21 @@ pub async fn create_graph_index( let plan = PhysicalPlan::Graph(GraphOp::EdgePutBatch { edges: edges.clone(), }); - // Append locally first so the batch is durable even on a single-node + // Append locally first so the batch is durable on a single-node // deployment with no Raft proposer configured (see the module doc - // comment, point 4). `dispatch_sync_response` below additionally - // replicates via Raft under RF>1 — a second, independent durability - // mechanism, not a duplicate WAL record. - crate::control::server::wal_dispatch::wal_append_if_write( - &state.wal, - tenant_id, - shard, - DatabaseId::DEFAULT, - &plan, - ) - .map_err(|e| ddl_err("XX000", format!("edge-insert WAL append failed: {e}")))?; + // comment, point 4). Under RF>1 the dispatch below proposes through + // Raft: the entry's apply appends its own record, and the dispatch + // cancels this one. + let minted = append_edge_batch(state, tenant_id, shard, &plan) + .await + .map_err(|e| ddl_err("XX000", format!("edge-insert WAL append failed: {e}")))?; - match crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_sync_response( + match crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_minted_sync_response( state, - tenant_id, - DatabaseId::DEFAULT, - shard, + edge_batch_owner(tenant_id, shard), plan, - TraceId::ZERO, crate::event::EventSource::User, + minted, ) .await { @@ -289,6 +282,38 @@ pub async fn create_graph_index( ))]) } +/// Append an edge batch's records under an outcome-floor window opened before +/// the first append. The dispatch that follows closes the window. +async fn append_edge_batch( + state: &SharedState, + tenant_id: TenantId, + shard: VShardId, + plan: &PhysicalPlan, +) -> crate::Result { + let owner = edge_batch_owner(tenant_id, shard); + let minted = crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); + match minted.append_plan(&state.wal, owner, plan) { + Ok(_) => Ok(minted), + Err(e) => { + // Any record appended before the error never reaches a core. + minted.cancel(&state.wal, owner, 0).await?; + Err(e) + } + } +} + +/// Where an edge batch's record lives. +fn edge_batch_owner( + tenant_id: TenantId, + shard: VShardId, +) -> crate::control::server::dispatch_utils::RecordOwner { + crate::control::server::dispatch_utils::RecordOwner { + tenant_id, + database_id: DatabaseId::DEFAULT, + vshard_id: shard, + } +} + /// Surface a build-time failure. /// /// Runs rollback in parallel across all committed shards. If **every** @@ -315,29 +340,21 @@ async fn surface_failure( }); let shard = *shard; async move { - // Same local-WAL-then-Raft-dispatch discipline as the forward - // path above: append locally first so the rollback tombstones - // are durable on single-node, then dispatch (which additionally - // replicates via Raft under RF>1). - if let Err(e) = crate::control::server::wal_dispatch::wal_append_if_write( - &state.wal, - tenant_id, - shard, - DatabaseId::DEFAULT, - &plan, - ) { - return (shard, Err(e)); - } + // Same discipline as the forward path above: append locally + // first so the rollback tombstones are durable on single-node, + // then dispatch, which proposes through Raft under RF>1. + let minted = match append_edge_batch(state, tenant_id, shard, &plan).await { + Ok(minted) => minted, + Err(e) => return (shard, Err(e)), + }; ( shard, - crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_sync_response( + crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_minted_sync_response( state, - tenant_id, - DatabaseId::DEFAULT, - shard, + edge_batch_owner(tenant_id, shard), plan, - TraceId::ZERO, crate::event::EventSource::User, + minted, ) .await, ) diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index 30e3fddc2..fc4906dcd 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -9,7 +9,7 @@ use crate::bridge::envelope::ErrorCode; /// Map a Data Plane `ErrorCode` to SQLSTATE. pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, String) { match code { - ErrorCode::DeadlineExceeded => ( + ErrorCode::DeadlineExceeded | ErrorCode::ExpiredBeforeExecution => ( "ERROR", sqlstate::QUERY_CANCELED, "query cancelled due to deadline".into(), diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs index 2f4f0b0a7..5dcce8feb 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs @@ -2,9 +2,13 @@ //! Async Data-Plane dispatch for system-initiated and authorized work. +use std::sync::Arc; use std::time::{Duration, Instant}; use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Response, Status}; +use crate::control::server::dispatch_utils::{ + Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, +}; use crate::control::server::shared::clone_write::CloneCheckedTask; use crate::control::server::shared::session::statement_deadline; use crate::control::state::SharedState; @@ -75,12 +79,15 @@ pub(crate) async fn dispatch_system_response_with_source( ); dispatch_plan( state, - task.tenant_id, - task.database_id, - vshard_id, - task.plan, - timeout, - event_source, + PlanDispatch { + tenant_id: task.tenant_id, + database_id: task.database_id, + vshard_id, + plan: task.plan, + timeout, + event_source, + minted: task.minted, + }, ) .await } @@ -109,12 +116,15 @@ pub(crate) async fn dispatch_authorized( let tenant_id = task.tenant_id; let resp = dispatch_plan( state, - tenant_id, - task.database_id, - vshard_id, - task.plan, - timeout, - crate::event::EventSource::User, + PlanDispatch { + tenant_id, + database_id: task.database_id, + vshard_id, + plan: task.plan, + timeout, + event_source: crate::event::EventSource::User, + minted: None, + }, ) .await?; @@ -130,16 +140,35 @@ pub(crate) async fn dispatch_authorized( Ok(resp.payload.to_vec()) } -/// Shared transport: build the request envelope, dispatch, await the response. -async fn dispatch_plan( - state: &SharedState, +/// What [`dispatch_plan`] sends, and where. +struct PlanDispatch { tenant_id: TenantId, database_id: DatabaseId, vshard_id: VShardId, plan: PhysicalPlan, timeout: Duration, event_source: crate::event::EventSource, -) -> crate::Result { + /// Records the caller appended for this plan, under their outcome-floor + /// window. The transport closes the window from the plan's outcome. + minted: Option, +} + +/// Shared transport: build the request envelope, dispatch, await the response. +async fn dispatch_plan(state: &SharedState, dispatch: PlanDispatch) -> crate::Result { + let PlanDispatch { + tenant_id, + database_id, + vshard_id, + plan, + timeout, + event_source, + minted, + } = dispatch; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id, + }; let request_id = state.next_request_id(); // Whichever comes first: the running statement's deadline, or the caller's @@ -174,30 +203,69 @@ async fn dispatch_plan( let mut rx = state.tracker.register(request_id); - match state.dispatcher.lock() { - Ok(mut d) => d.dispatch(request).map_err(|e| crate::Error::Internal { - detail: e.to_string(), - })?, - Err(p) => p - .into_inner() - .dispatch(request) - .map_err(|e| crate::Error::Internal { - detail: e.to_string(), - })?, + let dispatched = match state.dispatcher.lock() { + Ok(mut d) => d.dispatch(request), + Err(p) => p.into_inner().dispatch(request), }; + if let Err(error) = dispatched { + // No response will arrive, and no core applied the plan. + state.tracker.cancel(&request_id); + if let Some(minted) = minted { + minted.cancel(&state.wal, owner, 0).await?; + } + return Err(crate::Error::Internal { + detail: error.to_string(), + }); + } // Await to the same instant the envelope carries — yields the thread so the // response poller can run. Reaching that instant is the statement running // out of time, so it reports the deadline, and so does a producer that // stopped after it: the closure there is the symptom, not the cause. - match tokio::time::timeout_at(tokio::time::Instant::from_std(deadline), rx.recv()).await { - Ok(Some(response)) => Ok(response), - Ok(None) if Instant::now() >= deadline => { - Err(crate::Error::DeadlineExceeded { request_id }) + let Some(minted) = minted else { + let received = + tokio::time::timeout_at(tokio::time::Instant::from_std(deadline), rx.recv()).await; + return match received { + Ok(Some(response)) => Ok(response), + Ok(None) => Err(closed_error(request_id, deadline)), + Err(_) => Err(crate::Error::DeadlineExceeded { request_id }), + }; + }; + // The records close in a task this future does not own, so a caller + // dropped mid-wait still closes them. A late refusal still cancels them. + let outcome = await_response_owned( + OwnedWait { + wal: Arc::clone(&state.wal), + owner, + final_refusal_key: 0, + deadline, + collect: Collect::First, + }, + rx, + minted, + ) + .await?; + match outcome { + OwnedResponse::Answered { response, closed } => { + closed?; + Ok(response) } - Ok(None) => Err(crate::Error::Internal { - detail: "response channel closed".into(), + OwnedResponse::ChannelClosed => Err(closed_error(request_id, deadline)), + OwnedResponse::DeadlineExceeded => Err(crate::Error::DeadlineExceeded { request_id }), + OwnedResponse::OverBudget { bytes } => Err(crate::Error::ExecutionLimitExceeded { + detail: format!("system task response exceeded its byte budget ({bytes} bytes)"), }), - Err(_) => Err(crate::Error::DeadlineExceeded { request_id }), + } +} + +/// The error for a response channel that closed before a response: the +/// deadline when it already passed, since the closure is then its symptom. +fn closed_error(request_id: crate::types::RequestId, deadline: Instant) -> crate::Error { + if Instant::now() >= deadline { + crate::Error::DeadlineExceeded { request_id } + } else { + crate::Error::Internal { + detail: "response channel closed".into(), + } } } diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs index eaa171810..5745cc66f 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs @@ -17,6 +17,7 @@ //! authorization instead. use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::MintedRecords; use crate::types::{DatabaseId, TenantId}; /// Why a Data-Plane dispatch carries no user identity. @@ -71,6 +72,9 @@ pub(crate) struct SystemTask<'a> { pub(super) database_id: DatabaseId, pub(super) collection: &'a str, pub(super) plan: PhysicalPlan, + /// Records the caller appended for this task, under their outcome-floor + /// window. `None` when the task appends nothing. + pub(super) minted: Option, } impl<'a> SystemTask<'a> { @@ -92,6 +96,14 @@ impl<'a> SystemTask<'a> { database_id, collection, plan, + minted: None, } } + + /// Attach the records the caller appended for this task. The dispatch + /// closes their outcome-floor window from the task's outcome. + pub(crate) fn with_minted(mut self, minted: MintedRecords) -> Self { + self.minted = Some(minted); + self + } } diff --git a/nodedb/src/control/server/shared/session/commit/single_shard.rs b/nodedb/src/control/server/shared/session/commit/single_shard.rs index fefb94750..2ffddc7af 100644 --- a/nodedb/src/control/server/shared/session/commit/single_shard.rs +++ b/nodedb/src/control/server/shared/session/commit/single_shard.rs @@ -64,7 +64,7 @@ pub(super) async fn dispatch_single_shard( post_set_op: PostSetOp::None, txn_id: None, }; - let resolve_resp = match dp.dispatch_no_wal(resolve_task, None).await { + let resolve_resp = match dp.dispatch_no_wal(resolve_task).await { Ok(r) if r.status == Status::Ok => r, Ok(r) => { return Some(AbortReason::BatchRejected { diff --git a/nodedb/src/control/server/shared/session/lifecycle.rs b/nodedb/src/control/server/shared/session/lifecycle.rs index 207b530a5..8f7889897 100644 --- a/nodedb/src/control/server/shared/session/lifecycle.rs +++ b/nodedb/src/control/server/shared/session/lifecycle.rs @@ -186,7 +186,6 @@ mod tests { fn dispatch_no_wal<'a>( &'a self, task: PhysicalTask, - _wal_lsn: Option, ) -> Pin> + Send + 'a>> { let vshard = task.vshard_id; let payload = if let PhysicalPlan::Meta(op) = &task.plan { diff --git a/nodedb/src/control/server/shared/session/outcome.rs b/nodedb/src/control/server/shared/session/outcome.rs index 18c971646..251ff177e 100644 --- a/nodedb/src/control/server/shared/session/outcome.rs +++ b/nodedb/src/control/server/shared/session/outcome.rs @@ -69,17 +69,12 @@ pub enum AbortReason { /// nested listener request pipeline, and a boxed (type-erased) future keeps that /// async type-layout depth bounded rather than compounding per transport impl. pub trait TxnDataPlane { - /// Dispatch one task to the Data Plane without a per-task WAL append (the - /// whole transaction is written as a single WAL record by the caller). - /// - /// `wal_lsn` is the LSN of that single transaction WAL record: it is - /// stamped onto the dispatched `Request` so the Data Plane records the - /// committed write version for every key in the batch. `None` when no WAL - /// record was written (empty / read-only commit). + /// Dispatch one task to the Data Plane without a per-task WAL append. + /// The task carries no WAL record: the caller owns the transaction's + /// durability. fn dispatch_no_wal<'a>( &'a self, task: PhysicalTask, - wal_lsn: Option, ) -> Pin> + Send + 'a>>; /// The source the transaction's committed writes carry into the Event diff --git a/nodedb/src/control/server/shared/session/overlay_drop.rs b/nodedb/src/control/server/shared/session/overlay_drop.rs index c72e1de04..b65914c27 100644 --- a/nodedb/src/control/server/shared/session/overlay_drop.rs +++ b/nodedb/src/control/server/shared/session/overlay_drop.rs @@ -33,7 +33,7 @@ pub(super) async fn drop_txn_overlay( }; match resolve_leader(&task, state) { RouteDecision::Local => { - dp.dispatch_no_wal(task, None).await?; + dp.dispatch_no_wal(task).await?; Ok(()) } _ => { @@ -44,7 +44,7 @@ pub(super) async fn drop_txn_overlay( let drop_plan = drop_plan.clone(); async move { match resolve_leader(&task, state) { - RouteDecision::Local => dp.dispatch_no_wal(task, None).await, + RouteDecision::Local => dp.dispatch_no_wal(task).await, remote => forward_to_leader(state, remote, task, &drop_plan).await, } } diff --git a/nodedb/src/control/server/shared/session/savepoint_ops.rs b/nodedb/src/control/server/shared/session/savepoint_ops.rs index 037fa81be..2c556c56a 100644 --- a/nodedb/src/control/server/shared/session/savepoint_ops.rs +++ b/nodedb/src/control/server/shared/session/savepoint_ops.rs @@ -59,7 +59,7 @@ async fn dispatch_overlay_savepoint( txn_id: None, }; // Savepoint overlay meta-ops are not writes — no WAL record, no version. - match dp.dispatch_no_wal(task, None).await { + match dp.dispatch_no_wal(task).await { Ok(resp) => Some(resp.payload.to_vec()), Err(e) => { tracing::warn!(error = %e, "savepoint overlay meta-op dispatch failed"); @@ -275,7 +275,6 @@ mod tests { fn dispatch_no_wal<'a>( &'a self, task: PhysicalTask, - _wal_lsn: Option, ) -> Pin> + Send + 'a>> { let vshard = task.vshard_id; let payload = if let PhysicalPlan::Meta(op) = &task.plan { diff --git a/nodedb/src/control/server/sync/columnar_handler.rs b/nodedb/src/control/server/sync/columnar_handler.rs index 3a0c687f2..57fdc7a19 100644 --- a/nodedb/src/control/server/sync/columnar_handler.rs +++ b/nodedb/src/control/server/sync/columnar_handler.rs @@ -18,8 +18,8 @@ use nodedb_types::value::Value; use super::session::SyncSession; use super::wire::*; +use crate::control::server::dispatch_utils::RecordOwner; use crate::types::{DatabaseId, TenantId, VShardId}; -use crate::wal::manager::NO_APPLY_KEY; // ── PK extraction helper ───────────────────────────────────────────────────── @@ -193,18 +193,29 @@ impl<'a> ColumnarDispatcher for SharedStateColumnarDispatcher<'a> { // WAL append — surrogates are persisted so followers never mint their // own divergent ids. - let appended_lsn = wal_append_columnar( - self.shared.wal.appender(NO_APPLY_KEY), + let owner = RecordOwner { tenant_id, - vshard, database_id, - ColumnarWalAppendArgs { - collection: &collection, - payload: &payload, - provenance: Some(&prov), - surrogates: &surrogates, - }, - )?; + vshard_id: vshard, + }; + // The record's outcome-floor window opens before the append and + // closes from the dispatch's outcome. + let (minted, appended_lsn) = + super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append_columnar( + wal, + tenant_id, + vshard, + database_id, + ColumnarWalAppendArgs { + collection: &collection, + payload: &payload, + provenance: Some(&prov), + surrogates: &surrogates, + }, + ) + }) + .await?; let wal_lsn = appended_lsn.map(|lsn| lsn.as_u64()); let plan = PhysicalPlan::Columnar(ColumnarOp::Insert { @@ -227,15 +238,14 @@ impl<'a> ColumnarDispatcher for SharedStateColumnarDispatcher<'a> { rls_filters: Vec::new(), }); - let authorized = super::raft_dispatch::authorize_sync_task( + super::raft_dispatch::authorize_and_dispatch_minted( self.shared, self.identity, - tenant_id, - database_id, - vshard, + owner, plan, - )?; - super::raft_dispatch::dispatch_sync_payload(self.shared, authorized, appended_lsn).await + minted, + ) + .await } } diff --git a/nodedb/src/control/server/sync/fts_handler.rs b/nodedb/src/control/server/sync/fts_handler.rs index 96d5c7d81..587460a9e 100644 --- a/nodedb/src/control/server/sync/fts_handler.rs +++ b/nodedb/src/control/server/sync/fts_handler.rs @@ -17,8 +17,8 @@ use async_trait::async_trait; use nodedb_types::Surrogate; +use crate::control::server::dispatch_utils::RecordOwner; use crate::types::{DatabaseId, TenantId, VShardId}; -use crate::wal::manager::NO_APPLY_KEY; // ── Dispatcher trait ───────────────────────────────────────────────────────── @@ -106,13 +106,17 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { &surrogate_hex, &text, ); - let wal_lsn = wal_append_fts_index( - self.shared.wal.appender(NO_APPLY_KEY), + let owner = RecordOwner { tenant_id, - vshard, database_id, - &fts_index_payload, - )?; + vshard_id: vshard, + }; + // The record's outcome-floor window opens before the append and + // closes from the dispatch's outcome. + let (minted, _) = super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append_fts_index(wal, tenant_id, vshard, database_id, &fts_index_payload).map(Some) + }) + .await?; let plan = PhysicalPlan::Text(TextOp::FtsIndexDoc { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -121,15 +125,14 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { provenance: Some(prov), }); - let authorized = super::raft_dispatch::authorize_sync_task( + super::raft_dispatch::authorize_and_dispatch_minted( self.shared, self.identity, - tenant_id, - database_id, - vshard, + owner, plan, - )?; - super::raft_dispatch::dispatch_sync_payload(self.shared, authorized, Some(wal_lsn)).await + minted, + ) + .await } async fn dispatch_delete( @@ -158,13 +161,18 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { crate::engine::document::store::StorageKey::for_surrogate(surrogate).to_string(); let fts_delete_payload = nodedb_wal::record::FtsDeletePayload::new(prov.clone(), &collection, &surrogate_hex); - let wal_lsn = wal_append_fts_delete( - self.shared.wal.appender(NO_APPLY_KEY), + let owner = RecordOwner { tenant_id, - vshard, database_id, - &fts_delete_payload, - )?; + vshard_id: vshard, + }; + // The record's outcome-floor window opens before the append and + // closes from the dispatch's outcome. + let (minted, _) = super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append_fts_delete(wal, tenant_id, vshard, database_id, &fts_delete_payload) + .map(Some) + }) + .await?; let plan = PhysicalPlan::Text(TextOp::FtsDeleteDoc { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -172,15 +180,14 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { provenance: Some(prov), }); - let authorized = super::raft_dispatch::authorize_sync_task( + super::raft_dispatch::authorize_and_dispatch_minted( self.shared, self.identity, - tenant_id, - database_id, - vshard, + owner, plan, - )?; - super::raft_dispatch::dispatch_sync_payload(self.shared, authorized, Some(wal_lsn)).await + minted, + ) + .await } fn assign_surrogate( diff --git a/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs b/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs index 2f817b2aa..764e21a02 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs @@ -49,6 +49,20 @@ pub(super) fn fixture() -> (Arc, CoreChannelDataSide, tempfile::Tem /// Append a real FTS-delete redo and return its LSN, buffered but not durable. pub(super) fn append_buffered_record(state: &SharedState) -> Lsn { + append_fts_delete(state.wal.appender(crate::wal::manager::NO_APPLY_KEY)) +} + +/// Append a real FTS-delete redo under an outcome-floor window, as a caller +/// that dispatches it does. The record is buffered but not durable. +pub(super) fn minted_buffered_record( + state: &SharedState, +) -> (crate::control::server::dispatch_utils::MintedRecords, Lsn) { + let minted = crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); + let lsn = append_fts_delete(minted.appender(&state.wal, crate::wal::manager::NO_APPLY_KEY)); + (minted, lsn) +} + +fn append_fts_delete(wal: crate::wal::manager::WalAppender<'_>) -> Lsn { let payload = nodedb_wal::record::FtsDeletePayload::new( nodedb_types::sync::wire::SyncProvenance { producer_id: 1, @@ -60,7 +74,7 @@ pub(super) fn append_buffered_record(state: &SharedState) -> Lsn { "00000001", ); crate::control::server::wal_dispatch::wal_append_fts_delete( - state.wal.appender(crate::wal::manager::NO_APPLY_KEY), + wal, tenant(), vshard(), DatabaseId::DEFAULT, @@ -71,15 +85,23 @@ pub(super) fn append_buffered_record(state: &SharedState) -> Lsn { /// A write-class plan matching the appended record, authorized for dispatch. pub(super) fn authorized_write(state: &SharedState) -> AuthorizedTask { - let task = PhysicalTask { - tenant_id: tenant(), - database_id: DatabaseId::DEFAULT, - vshard_id: vshard(), - plan: PhysicalPlan::Text(TextOp::FtsDeleteDoc { + authorized_plan( + state, + PhysicalPlan::Text(TextOp::FtsDeleteDoc { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLLECTION), surrogate: nodedb_types::Surrogate::ZERO, provenance: None, }), + ) +} + +/// `plan` on the test collection, authorized for dispatch. +pub(super) fn authorized_plan(state: &SharedState, plan: PhysicalPlan) -> AuthorizedTask { + let task = PhysicalTask { + tenant_id: tenant(), + database_id: DatabaseId::DEFAULT, + vshard_id: vshard(), + plan, post_set_op: PostSetOp::None, txn_id: None, }; diff --git a/nodedb/src/control/server/sync/raft_dispatch/minted.rs b/nodedb/src/control/server/sync/raft_dispatch/minted.rs new file mode 100644 index 000000000..6d3e73809 --- /dev/null +++ b/nodedb/src/control/server/sync/raft_dispatch/minted.rs @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Append a sync write's redo record under an outcome-floor window, and hand +//! the record to the dispatch that closes the window. + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; +use crate::control::state::SharedState; +use crate::types::Lsn; +use crate::wal::manager::{NO_APPLY_KEY, WalAppender}; + +use super::authorize::authorize_sync_task; +use super::response::dispatch_sync_payload; + +/// Open an outcome-floor window, then run `append` with an appender that +/// records every LSN it writes into the window. Returns the window's records +/// and the LSN `append` returned. +pub(crate) async fn append_under_window( + shared: &SharedState, + owner: RecordOwner, + append: impl FnOnce(WalAppender<'_>) -> crate::Result>, +) -> crate::Result<(MintedRecords, Option)> { + let minted = MintedRecords::open(&shared.outcome_floor); + let appended = append(minted.appender(&shared.wal, NO_APPLY_KEY)); + match appended { + Ok(lsn) => Ok((minted, lsn)), + Err(error) => { + // Any record appended before the error never reaches a core. + minted.cancel(&shared.wal, owner, 0).await?; + Err(error) + } + } +} + +/// Authorize `plan` and dispatch it with the records the caller appended. A +/// refused authorization reaches no core, so it cancels the records. +pub(crate) async fn authorize_and_dispatch_minted( + shared: &SharedState, + identity: Option<&AuthenticatedIdentity>, + owner: RecordOwner, + plan: PhysicalPlan, + minted: MintedRecords, +) -> crate::Result> { + let authorized = match authorize_sync_task( + shared, + identity, + owner.tenant_id, + owner.database_id, + owner.vshard_id, + plan, + ) { + Ok(authorized) => authorized, + Err(error) => { + minted.cancel(&shared.wal, owner, 0).await?; + return Err(error); + } + }; + dispatch_sync_payload(shared, authorized, Some(minted)).await +} diff --git a/nodedb/src/control/server/sync/raft_dispatch/mod.rs b/nodedb/src/control/server/sync/raft_dispatch/mod.rs index b8ce4ac4f..2dd638b97 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/mod.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/mod.rs @@ -7,13 +7,18 @@ pub mod admission_guard; pub mod authorize; #[cfg(test)] mod durability_test_support; +mod minted; pub mod outcome; pub mod propose; pub mod response; pub mod write; pub use authorize::{authorize_sync_collection, authorize_sync_task}; +pub(crate) use minted::{append_under_window, authorize_and_dispatch_minted}; pub use outcome::SyncDispatchOutcome; -pub(crate) use response::dispatch_trusted_internal_sync_response; -pub use response::{dispatch_authorized_sync_response, dispatch_sync_payload, noop_dispatch_error}; -pub use write::{dispatch_sync_bytes, dispatch_write_replicated}; +pub use response::noop_dispatch_error; +pub(crate) use response::{ + dispatch_trusted_internal_minted_sync_response, dispatch_trusted_internal_sync_response, +}; +pub use write::dispatch_sync_bytes; +pub(crate) use write::dispatch_write_replicated; diff --git a/nodedb/src/control/server/sync/raft_dispatch/response.rs b/nodedb/src/control/server/sync/raft_dispatch/response.rs index 47d8b9730..cf6d10a00 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/response.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/response.rs @@ -6,6 +6,7 @@ //! which need the raw `Response` to extract the payload themselves. use crate::bridge::envelope::{PhysicalPlan, Response, Status}; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::server::shared::authorization::AuthorizedTask; use crate::control::state::SharedState; use crate::control::wal_replication::{ReplicableWrite, to_replicated_entry}; @@ -24,21 +25,22 @@ struct SyncResponseDispatch { plan: PhysicalPlan, trace_id: TraceId, event_source: EventSource, - /// The LSN of the redo record the caller already appended for this - /// write, or `None` when the caller minted none. See + /// The redo records the caller already appended for this write, under + /// their outcome-floor window, or `None` when the caller minted none. See /// [`dispatch_authorized_sync_response`] for the durability contract. - wal_lsn: Option, + minted: Option, } -/// `wal_lsn` is the caller's already-appended redo record, or `None`. Threaded -/// into the write funnel so the durable-at-ack barrier fsyncs it before this -/// returns — sync handlers ack their peer off this return value. -pub async fn dispatch_authorized_sync_response( +/// `minted` holds the caller's already-appended redo records, or `None`. +/// Threaded into the write funnel so the durable-at-ack barrier fsyncs them +/// before this returns — sync handlers ack their peer off this return value. +/// The funnel closes their outcome-floor window from the write's outcome. +pub(crate) async fn dispatch_authorized_sync_response( state: &SharedState, authorized: AuthorizedTask, trace_id: TraceId, event_source: EventSource, - wal_lsn: Option, + minted: Option, ) -> crate::Result { let task = authorized.into_physical_task(); dispatch_sync_response_inner( @@ -50,7 +52,32 @@ pub async fn dispatch_authorized_sync_response( plan: task.plan, trace_id, event_source, - wal_lsn, + minted, + }, + ) + .await +} + +/// Trusted-internal sync-shaped dispatch of a write whose redo records the +/// caller already appended under `minted`. Used by DDL paths that append +/// their own record before dispatch. +pub(crate) async fn dispatch_trusted_internal_minted_sync_response( + state: &SharedState, + owner: RecordOwner, + plan: PhysicalPlan, + event_source: EventSource, + minted: MintedRecords, +) -> crate::Result { + dispatch_sync_response_inner( + state, + SyncResponseDispatch { + tenant_id: owner.tenant_id, + database_id: owner.database_id, + vshard_id: owner.vshard_id, + plan, + trace_id: TraceId::ZERO, + event_source, + minted: Some(minted), }, ) .await @@ -77,7 +104,7 @@ pub(crate) async fn dispatch_trusted_internal_sync_response( plan, trace_id, event_source, - wal_lsn: None, + minted: None, }, ) .await @@ -85,7 +112,9 @@ pub(crate) async fn dispatch_trusted_internal_sync_response( /// Cluster path: proposes through Raft, wraps the payload in `Status::Ok`. Gate /// verdict travels in the payload; non-`Ok` means a protocol error, not a gate -/// rejection. Single-node path carries `wal_lsn` through the write funnel. +/// rejection. The Raft entry's apply appends its own records, so the caller's +/// records are cancelled. Single-node path carries the caller's records +/// through the write funnel, which closes them from the write's outcome. async fn dispatch_sync_response_inner( state: &SharedState, params: SyncResponseDispatch, @@ -97,18 +126,47 @@ async fn dispatch_sync_response_inner( plan, trace_id, event_source, - wal_lsn, + minted, } = params; - reject_unadmitted_crdt_apply(&plan)?; - if let Some(proposer) = state.async_raft_proposer() - && let Some(entry) = to_replicated_entry( - tenant_id, - database_id, - vshard_id, - &ReplicableWrite::decide_for_replication(&plan)?, - )? - { - let payload = propose_sync_write(state, entry, proposer).await?; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id, + }; + if let Err(error) = reject_unadmitted_crdt_apply(&plan) { + // Nothing was dispatched. + if let Some(minted) = minted { + minted.cancel(&state.wal, owner, 0).await?; + } + return Err(error); + } + let replicated = match state.async_raft_proposer() { + Some(proposer) => { + let entry = ReplicableWrite::decide_for_replication(&plan).and_then(|replicable| { + to_replicated_entry(tenant_id, database_id, vshard_id, &replicable) + }); + match entry { + Ok(entry) => entry.map(|entry| (proposer, entry)), + Err(error) => { + if let Some(minted) = minted { + minted.cancel(&state.wal, owner, 0).await?; + } + return Err(error); + } + } + } + None => None, + }; + if let Some((proposer, entry)) = replicated { + // The Raft entry's apply appends its own records. + let superseded = minted.map(|minted| { + minted.supersede(std::sync::Arc::clone(&state.wal), owner, "raft_proposal") + }); + let proposed = propose_sync_write(state, entry, proposer).await; + if let Some(superseded) = superseded { + superseded.finish().await; + } + let payload = proposed?; let request_id = state.next_request_id(); return Ok(Response { request_id, @@ -136,28 +194,29 @@ async fn dispatch_sync_response_inner( txn_id: None, // Caller already appended this write's redo; funnel must not append a // second one — stamps this LSN and waits at the durable-at-ack barrier. - wal_lsn, + wal_lsn: minted.as_ref().and_then(MintedRecords::highest), // Only a TTL-bearing KV write resolves a wall-clock instant, and KV has no sync handler. resolved_now_ms: None, + minted, }, ) .await } /// Sync-path convenience: dispatches `plan` tagged [`EventSource::CrdtSync`], -/// returns just the payload bytes. `wal_lsn` isn't optional in spirit — these +/// returns the payload bytes. `minted` isn't optional in spirit — these /// engines rebuild only by WAL replay, so acking without it loses a write on `kill -9`. -pub async fn dispatch_sync_payload( +pub(crate) async fn dispatch_sync_payload( state: &SharedState, authorized: AuthorizedTask, - wal_lsn: Option, + minted: Option, ) -> crate::Result> { let response = dispatch_authorized_sync_response( state, authorized, TraceId::ZERO, EventSource::CrdtSync, - wal_lsn, + minted, ) .await?; Ok(response.payload.to_vec()) @@ -180,7 +239,7 @@ mod tests { use std::sync::Arc; use super::super::durability_test_support::{ - append_buffered_record, authorized_write, fixture, respond_once, + append_buffered_record, authorized_write, fixture, minted_buffered_record, respond_once, }; use super::dispatch_sync_payload; @@ -190,7 +249,7 @@ mod tests { #[tokio::test] async fn a_supplied_lsn_is_fsync_durable_before_the_payload_returns() { let (state, side, _directory) = fixture(); - let lsn = append_buffered_record(&state); + let (minted, lsn) = minted_buffered_record(&state); assert!( state.wal.durable_through() < lsn.as_u64(), "the append must only buffer, or this test proves nothing" @@ -198,7 +257,7 @@ mod tests { let authorized = authorized_write(&state); let responder = tokio::spawn(respond_once(Arc::clone(&state), side)); - dispatch_sync_payload(&state, authorized, Some(lsn)) + dispatch_sync_payload(&state, authorized, Some(minted)) .await .expect("sync dispatch succeeds"); responder.await.expect("responder completes"); diff --git a/nodedb/src/control/server/sync/raft_dispatch/write.rs b/nodedb/src/control/server/sync/raft_dispatch/write.rs index f8ad90135..038fd82b4 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/write.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/write.rs @@ -5,12 +5,13 @@ use std::time::Duration; use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::server::shared::authorization::AuthorizedTask; use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; use crate::control::wal_replication::{ReplicableWrite, to_replicated_entry}; use crate::event::EventSource; -use crate::types::{Lsn, VShardId}; +use crate::types::VShardId; use super::admission_guard::reject_unadmitted_crdt_apply; use super::outcome::SyncDispatchOutcome; @@ -60,72 +61,124 @@ pub async fn dispatch_sync_bytes( /// Dispatch a write so it is quorum-durable when the node is clustered. /// -/// Cluster path proposes through Raft and blocks until applied locally. Single-node -/// path waits on `wal_lsn` (the caller's already-appended redo) before returning. -pub async fn dispatch_write_replicated( +/// `minted` holds the redo records the caller already appended, under their +/// outcome-floor window. Cluster path proposes through Raft and blocks until +/// applied locally; the Raft entry's apply appends its own records, so the +/// caller's records are cancelled. Single-node path installs the write, +/// closes the records from its outcome, and waits until they are durable. +pub(crate) async fn dispatch_write_replicated( state: &SharedState, collection: &str, authorized: AuthorizedTask, timeout: Duration, event_source: EventSource, - wal_lsn: Option, + minted: Option, ) -> crate::Result> { let task = authorized.into_physical_task(); let tenant_id = task.tenant_id; let database_id = task.database_id; let vshard_id = task.vshard_id; let plan = task.plan; - reject_unadmitted_crdt_apply(&plan)?; - if vshard_id != VShardId::from_collection_in_database(database_id, collection) { - return Err(crate::Error::Internal { - detail: "authorized sync task vShard does not match collection".into(), - }); + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id, + }; + let refused = reject_unadmitted_crdt_apply(&plan).and_then(|()| { + if vshard_id == VShardId::from_collection_in_database(database_id, collection) { + Ok(()) + } else { + Err(crate::Error::Internal { + detail: "authorized sync task vShard does not match collection".into(), + }) + } + }); + if let Err(error) = refused { + // Nothing was dispatched. + if let Some(minted) = minted { + minted.cancel(&state.wal, owner, 0).await?; + } + return Err(error); } let local_frontier_mutation = matches!( &plan, PhysicalPlan::Crdt(op) if crate::control::crdt_admission::changes_crdt_frontier(op) ); - if let Some(proposer) = state.async_raft_proposer() - && let Some(entry) = to_replicated_entry( - tenant_id, - database_id, - vshard_id, - &ReplicableWrite::decide_for_replication(&plan)?, - )? - { - return propose_sync_write(state, entry, proposer).await; + if let Some(proposer) = state.async_raft_proposer() { + let entry = ReplicableWrite::decide_for_replication(&plan).and_then(|replicable| { + to_replicated_entry(tenant_id, database_id, vshard_id, &replicable) + }); + let entry = match entry { + Ok(entry) => entry, + Err(error) => { + if let Some(minted) = minted { + minted.cancel(&state.wal, owner, 0).await?; + } + return Err(error); + } + }; + if let Some(entry) = entry { + // The Raft entry's apply appends its own records. + let superseded = minted.map(|minted| { + minted.supersede(std::sync::Arc::clone(&state.wal), owner, "raft_proposal") + }); + let proposed = propose_sync_write(state, entry, proposer).await; + if let Some(superseded) = superseded { + superseded.finish().await; + } + return proposed; + } } + let wal_lsn = minted.as_ref().and_then(MintedRecords::highest); + let task = crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( + crate::control::server::shared::ddl::sync_dispatch::SystemReason::AdmittedContinuation, + tenant_id, + database_id, + collection, + plan, + ); let resp = if local_frontier_mutation { - state + // The sequencer can refuse before it runs the dispatch. The records + // stay here until the dispatch takes them, so such a refusal cancels + // them: nothing reached a core. + let unsent = std::sync::Mutex::new(minted); + let run = state .vshard_admission_sequencer .run(vshard_id, || async { + let minted = unsent.lock().unwrap_or_else(|p| p.into_inner()).take(); + let task = match minted { + Some(minted) => task.with_minted(minted), + None => task, + }; crate::control::server::shared::ddl::sync_dispatch::dispatch_system_response_with_source( state, - crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( - crate::control::server::shared::ddl::sync_dispatch::SystemReason::AdmittedContinuation, - tenant_id, - database_id, - collection, - plan, - ), + task, timeout, event_source, ) .await }) - .await? + .await; + match run { + Ok(resp) => resp, + Err(error) => { + let never_sent = unsent.into_inner().unwrap_or_else(|p| p.into_inner()); + if let Some(minted) = never_sent { + minted.cancel(&state.wal, owner, 0).await?; + } + return Err(error); + } + } } else { + let task = match minted { + Some(minted) => task.with_minted(minted), + None => task, + }; crate::control::server::shared::ddl::sync_dispatch::dispatch_system_response_with_source( state, - crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( - crate::control::server::shared::ddl::sync_dispatch::SystemReason::AdmittedContinuation, - tenant_id, - database_id, - collection, - plan, - ), + task, timeout, event_source, ) @@ -154,7 +207,8 @@ mod tests { use std::time::Duration; use super::super::durability_test_support::{ - COLLECTION, append_buffered_record, authorized_write, fixture, respond_once, + COLLECTION, append_buffered_record, authorized_write, fixture, minted_buffered_record, + respond_once, }; use super::dispatch_write_replicated; use crate::event::EventSource; @@ -164,7 +218,7 @@ mod tests { #[tokio::test] async fn a_supplied_lsn_is_fsync_durable_before_the_payload_returns() { let (state, side, _directory) = fixture(); - let lsn = append_buffered_record(&state); + let (minted, lsn) = minted_buffered_record(&state); assert!( state.wal.durable_through() < lsn.as_u64(), "the append must only buffer, or this test proves nothing" @@ -178,7 +232,7 @@ mod tests { authorized, Duration::from_secs(5), EventSource::CrdtSync, - Some(lsn), + Some(minted), ) .await .expect("replicated sync dispatch succeeds"); @@ -216,4 +270,67 @@ mod tests { "nothing appended by this dispatch means nothing to fsync" ); } + + /// The admission sequencer refuses a frontier write before it runs the + /// dispatch. Nothing reached a core, so the caller's records are + /// cancelled and their window settles. + #[tokio::test] + async fn a_sequencer_refusal_cancels_the_unsent_records() { + use super::super::durability_test_support::{authorized_plan, vshard}; + + let (state, _side, _directory) = fixture(); + let (minted, lsn) = minted_buffered_record(&state); + let sequencer = Arc::clone(&state.vshard_admission_sequencer); + let holders: Vec<_> = (0..crate::control::vshard_admission::VSHARD_ADMISSION_CAPACITY) + .map(|_| { + let sequencer = Arc::clone(&sequencer); + tokio::spawn(async move { + sequencer + .run(vshard(), std::future::pending::>) + .await + }) + }) + .collect(); + for _ in 0..8 { + tokio::task::yield_now().await; + } + let authorized = authorized_plan( + &state, + crate::bridge::envelope::PhysicalPlan::Crdt( + nodedb_physical::physical_plan::CrdtOp::DocDelete { + collection: nodedb_types::QualifiedCollection::new( + crate::types::DatabaseId::DEFAULT, + COLLECTION, + ), + document_id: "d1".into(), + surrogate: nodedb_types::Surrogate::ZERO, + returning: None, + rls_filters: Vec::new(), + }, + ), + ); + + let result = dispatch_write_replicated( + &state, + COLLECTION, + authorized, + Duration::from_secs(5), + EventSource::CrdtSync, + Some(minted), + ) + .await; + + assert!( + matches!( + result, + Err(crate::Error::VShardAdmissionCapacityExceeded { .. }) + ), + "got {result:?}" + ); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + for holder in holders { + holder.abort(); + } + } } diff --git a/nodedb/src/control/server/sync/refusal.rs b/nodedb/src/control/server/sync/refusal.rs index 72be91880..ed75eb617 100644 --- a/nodedb/src/control/server/sync/refusal.rs +++ b/nodedb/src/control/server/sync/refusal.rs @@ -51,6 +51,7 @@ fn is_indeterminate(error: &crate::Error) -> bool { | crate::Error::ConflictRetry { .. } | crate::Error::DataPlane( ErrorCode::DeadlineExceeded + | ErrorCode::ExpiredBeforeExecution | ErrorCode::ResourcesExhausted | ErrorCode::DispatchCapacity { .. } | ErrorCode::ConflictRetry diff --git a/nodedb/src/control/server/sync/spatial_handler.rs b/nodedb/src/control/server/sync/spatial_handler.rs index 144254b47..ece8ca732 100644 --- a/nodedb/src/control/server/sync/spatial_handler.rs +++ b/nodedb/src/control/server/sync/spatial_handler.rs @@ -18,8 +18,8 @@ use async_trait::async_trait; use nodedb_types::Surrogate; use nodedb_types::geometry::Geometry; +use crate::control::server::dispatch_utils::RecordOwner; use crate::types::{DatabaseId, TenantId, VShardId}; -use crate::wal::manager::NO_APPLY_KEY; // ── Dispatcher trait ───────────────────────────────────────────────────────── @@ -113,13 +113,18 @@ impl<'a> SpatialDispatcher for SharedStateSpatialDispatcher<'a> { )?; let spatial_put_payload = encode_spatial_put_payload(&collection, &field, surrogate, &geometry, &prov)?; - let wal_lsn = wal_append_spatial_put( - self.shared.wal.appender(NO_APPLY_KEY), + let owner = RecordOwner { tenant_id, - vshard, database_id, - &spatial_put_payload, - )?; + vshard_id: vshard, + }; + // The record's outcome-floor window opens before the append and + // closes from the dispatch's outcome. + let (minted, _) = super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append_spatial_put(wal, tenant_id, vshard, database_id, &spatial_put_payload) + .map(Some) + }) + .await?; let plan = PhysicalPlan::Spatial(SpatialOp::Insert { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -129,15 +134,14 @@ impl<'a> SpatialDispatcher for SharedStateSpatialDispatcher<'a> { provenance: Some(prov), }); - let authorized = super::raft_dispatch::authorize_sync_task( + super::raft_dispatch::authorize_and_dispatch_minted( self.shared, self.identity, - tenant_id, - database_id, - vshard, + owner, plan, - )?; - super::raft_dispatch::dispatch_sync_payload(self.shared, authorized, Some(wal_lsn)).await + minted, + ) + .await } async fn dispatch_delete( @@ -166,13 +170,18 @@ impl<'a> SpatialDispatcher for SharedStateSpatialDispatcher<'a> { let spatial_delete_payload = encode_spatial_delete_payload(&collection, &field, surrogate, &prov); - let wal_lsn = wal_append_spatial_delete( - self.shared.wal.appender(NO_APPLY_KEY), + let owner = RecordOwner { tenant_id, - vshard, database_id, - &spatial_delete_payload, - )?; + vshard_id: vshard, + }; + // The record's outcome-floor window opens before the append and + // closes from the dispatch's outcome. + let (minted, _) = super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append_spatial_delete(wal, tenant_id, vshard, database_id, &spatial_delete_payload) + .map(Some) + }) + .await?; let plan = PhysicalPlan::Spatial(SpatialOp::Delete { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -181,15 +190,14 @@ impl<'a> SpatialDispatcher for SharedStateSpatialDispatcher<'a> { provenance: Some(prov), }); - let authorized = super::raft_dispatch::authorize_sync_task( + super::raft_dispatch::authorize_and_dispatch_minted( self.shared, self.identity, - tenant_id, - database_id, - vshard, + owner, plan, - )?; - super::raft_dispatch::dispatch_sync_payload(self.shared, authorized, Some(wal_lsn)).await + minted, + ) + .await } fn assign_surrogate( diff --git a/nodedb/src/control/server/sync/timeseries_handler.rs b/nodedb/src/control/server/sync/timeseries_handler.rs index 7830824bf..eaf5b7dc8 100644 --- a/nodedb/src/control/server/sync/timeseries_handler.rs +++ b/nodedb/src/control/server/sync/timeseries_handler.rs @@ -15,8 +15,8 @@ use tracing::{debug, error}; use super::session::SyncSession; use super::wire::*; +use crate::control::server::dispatch_utils::RecordOwner; use crate::types::{DatabaseId, TenantId, VShardId}; -use crate::wal::manager::NO_APPLY_KEY; // ── Dispatcher trait ───────────────────────────────────────────────────────── @@ -93,18 +93,29 @@ impl<'a> TimeseriesDispatcher for SharedStateTimeseriesDispatcher<'a> { // Allocate a WAL LSN on the Control Plane before dispatching to the // Data Plane. This is the canonical LSN for dedup tracking. - let appended_lsn = wal_append_timeseries( - self.shared.wal.appender(NO_APPLY_KEY), - TimeseriesWalAppendContext { - tenant_id, - vshard_id: vshard, - database_id, - collection: &collection, - }, - &payload_bytes, - Some(&prov), - Some(&self.shared.credentials), - )?; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id: vshard, + }; + // The record's outcome-floor window opens before the append and + // closes from the dispatch's outcome. + let (minted, appended_lsn) = + super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append_timeseries( + wal, + TimeseriesWalAppendContext { + tenant_id, + vshard_id: vshard, + database_id, + collection: &collection, + }, + &payload_bytes, + Some(&prov), + Some(&self.shared.credentials), + ) + }) + .await?; let wal_lsn = appended_lsn.map(|lsn| lsn.as_u64()); let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { @@ -122,21 +133,28 @@ impl<'a> TimeseriesDispatcher for SharedStateTimeseriesDispatcher<'a> { rls_filters: Vec::new(), }); - let authorized = super::raft_dispatch::authorize_sync_task( + let authorized = match super::raft_dispatch::authorize_sync_task( self.shared, self.identity, tenant_id, database_id, vshard, plan, - )?; + ) { + Ok(authorized) => authorized, + Err(error) => { + // A refused authorization reaches no core. + minted.cancel(&self.shared.wal, owner, 0).await?; + return Err(error); + } + }; super::raft_dispatch::dispatch_write_replicated( self.shared, &collection, authorized, std::time::Duration::from_secs(self.shared.tuning.network.default_deadline_secs), crate::event::EventSource::CrdtSync, - appended_lsn, + Some(minted), ) .await } diff --git a/nodedb/src/control/server/sync/vector_handler.rs b/nodedb/src/control/server/sync/vector_handler.rs index e54a39e15..27a6949ca 100644 --- a/nodedb/src/control/server/sync/vector_handler.rs +++ b/nodedb/src/control/server/sync/vector_handler.rs @@ -15,8 +15,8 @@ use async_trait::async_trait; use nodedb_types::Surrogate; +use crate::control::server::dispatch_utils::RecordOwner; use crate::types::{DatabaseId, TenantId, VShardId}; -use crate::wal::manager::NO_APPLY_KEY; // ── Dispatcher trait ───────────────────────────────────────────────────────── @@ -106,20 +106,31 @@ impl<'a> VectorDispatcher for SharedStateVectorDispatcher<'a> { // Allocate WAL LSN on the Control Plane before dispatching to the // Data Plane. Sync path MUST write to WAL; non-sync path already does // this via `wal_append_if_write_with_creds` in the main dispatch. - let wal_lsn = wal_append_vector_put( - self.shared.wal.appender(NO_APPLY_KEY), + let owner = RecordOwner { tenant_id, - vshard, database_id, - VectorPutWalArgs { - collection: ¶ms.collection, - vector: ¶ms.vector, - dim: params.dim, - field_name: ¶ms.field_name, - surrogate: params.surrogate, - provenance: Some(&prov), - }, - )?; + vshard_id: vshard, + }; + // The record's outcome-floor window opens before the append and + // closes from the dispatch's outcome. + let (minted, _) = super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append_vector_put( + wal, + tenant_id, + vshard, + database_id, + VectorPutWalArgs { + collection: ¶ms.collection, + vector: ¶ms.vector, + dim: params.dim, + field_name: ¶ms.field_name, + surrogate: params.surrogate, + provenance: Some(&prov), + }, + ) + .map(Some) + }) + .await?; let plan = PhysicalPlan::Vector(VectorOp::Insert { collection: nodedb_types::QualifiedCollection::new(database_id, ¶ms.collection), @@ -131,15 +142,14 @@ impl<'a> VectorDispatcher for SharedStateVectorDispatcher<'a> { provenance: Some(prov), }); - let authorized = super::raft_dispatch::authorize_sync_task( + super::raft_dispatch::authorize_and_dispatch_minted( self.shared, self.identity, - tenant_id, - database_id, - vshard, + owner, plan, - )?; - super::raft_dispatch::dispatch_sync_payload(self.shared, authorized, Some(wal_lsn)).await + minted, + ) + .await } async fn dispatch_delete( @@ -169,18 +179,29 @@ impl<'a> VectorDispatcher for SharedStateVectorDispatcher<'a> { // Allocate WAL LSN on the Control Plane before dispatching to the // Data Plane. - let wal_lsn = wal_append_vector_delete_by_surrogate( - self.shared.wal.appender(NO_APPLY_KEY), + let owner = RecordOwner { tenant_id, - vshard, database_id, - VectorDeleteWalArgs { - collection: &collection, - surrogate, - field_name: &field_name, - provenance: Some(&prov), - }, - )?; + vshard_id: vshard, + }; + // The record's outcome-floor window opens before the append and + // closes from the dispatch's outcome. + let (minted, _) = super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append_vector_delete_by_surrogate( + wal, + tenant_id, + vshard, + database_id, + VectorDeleteWalArgs { + collection: &collection, + surrogate, + field_name: &field_name, + provenance: Some(&prov), + }, + ) + .map(Some) + }) + .await?; let plan = PhysicalPlan::Vector(VectorOp::DeleteBySurrogate { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -189,15 +210,14 @@ impl<'a> VectorDispatcher for SharedStateVectorDispatcher<'a> { provenance: Some(prov), }); - let authorized = super::raft_dispatch::authorize_sync_task( + super::raft_dispatch::authorize_and_dispatch_minted( self.shared, self.identity, - tenant_id, - database_id, - vshard, + owner, plan, - )?; - super::raft_dispatch::dispatch_sync_payload(self.shared, authorized, Some(wal_lsn)).await + minted, + ) + .await } fn assign_surrogate( diff --git a/nodedb/src/control/shutdown/data_plane.rs b/nodedb/src/control/shutdown/data_plane.rs index 1fcfd1cf7..519daf437 100644 --- a/nodedb/src/control/shutdown/data_plane.rs +++ b/nodedb/src/control/shutdown/data_plane.rs @@ -294,7 +294,7 @@ mod tests { /// Hand the request to the single core and register its waiter, exactly as /// a session would. - fn dispatch_one(shared: &SharedState, id: u64) -> tokio::sync::mpsc::Receiver { + fn dispatch_one(shared: &SharedState, id: u64) -> crate::control::ResponseReceiver { let rx = shared.tracker.register(RequestId::new(id)); shared .dispatcher diff --git a/nodedb/src/control/system_txn/data_plane.rs b/nodedb/src/control/system_txn/data_plane.rs index cb150569e..7608ae0b0 100644 --- a/nodedb/src/control/system_txn/data_plane.rs +++ b/nodedb/src/control/system_txn/data_plane.rs @@ -9,7 +9,7 @@ use crate::bridge::envelope::Response; use crate::control::server::dispatch_utils; use crate::control::server::shared::session::TxnDataPlane; use crate::control::state::SharedState; -use crate::types::{Lsn, TraceId}; +use crate::types::TraceId; use nodedb_physical::physical_task::PhysicalTask; /// Dispatches a system transaction's commit-time tasks straight to the core @@ -33,7 +33,6 @@ impl TxnDataPlane for SystemTxnDataPlane<'_> { fn dispatch_no_wal<'a>( &'a self, task: PhysicalTask, - wal_lsn: Option, ) -> Pin> + Send + 'a>> { let state = self.state; let event_source = self.event_source; @@ -48,8 +47,9 @@ impl TxnDataPlane for SystemTxnDataPlane<'_> { trace_id: TraceId::ZERO, event_source, txn_id: None, - wal_lsn, + wal_lsn: None, resolved_now_ms: None, + minted: None, }, ) .await diff --git a/nodedb/src/control/system_txn/run.rs b/nodedb/src/control/system_txn/run.rs index 07dc69c7d..72d48a45e 100644 --- a/nodedb/src/control/system_txn/run.rs +++ b/nodedb/src/control/system_txn/run.rs @@ -137,6 +137,7 @@ async fn dispatch_staged( txn_id: task.txn_id, wal_lsn: None, resolved_now_ms: None, + minted: None, }, ) .await diff --git a/nodedb/src/control/wal_catchup.rs b/nodedb/src/control/wal_catchup.rs index 264e0f538..74897fdcc 100644 --- a/nodedb/src/control/wal_catchup.rs +++ b/nodedb/src/control/wal_catchup.rs @@ -176,6 +176,17 @@ async fn run_catchup_cycle(shared: &SharedState) -> CatchupResult { continue; } + // A record at or below the outcome floor has a final outcome. Sending + // it again would apply it below the floor, where a published + // watermark already claims its outcome. + let Some(minted) = crate::control::server::dispatch_utils::MintedRecords::resend( + &shared.outcome_floor, + Lsn::new(record.header.lsn), + ) else { + max_lsn = max_lsn.max(record.header.lsn); + continue; + }; + let tenant_id = TenantId::new(record.header.tenant_id); let database_id = DatabaseId::new(record.header.database_id); let vshard_id = VShardId::new(record.header.vshard_id); @@ -213,6 +224,7 @@ async fn run_catchup_cycle(shared: &SharedState) -> CatchupResult { resolved_now_ms: decoded .default_timestamp_ms .and_then(|ms| u64::try_from(ms).ok()), + minted: Some(minted), }, ) .await diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index 116a35a09..af690811c 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -85,7 +85,8 @@ impl CoreLoop { partial: false, payload: Payload::empty(), watermark_lsn: self.watermark, - error_code: Some(Box::new(ErrorCode::DeadlineExceeded)), + // The task never started, so nothing it would write ran. + error_code: Some(Box::new(ErrorCode::ExpiredBeforeExecution)), read_set_valid: None, read_version_lsn: crate::types::Lsn::ZERO, write_set: Vec::new(), @@ -236,7 +237,7 @@ mod tests { } #[test] - fn expired_task_returns_deadline_exceeded() { + fn an_expired_task_answers_that_it_never_started() { let (mut core, mut req_tx, mut resp_rx, _dir) = make_core(); req_tx .try_push(BridgeRequest::unfloored(Request { @@ -257,7 +258,7 @@ mod tests { assert_eq!(resp.inner.status, Status::Error); assert_eq!( resp.inner.error_code.as_deref(), - Some(&ErrorCode::DeadlineExceeded) + Some(&ErrorCode::ExpiredBeforeExecution) ); } diff --git a/nodedb/src/data/executor/dispatch/timeseries.rs b/nodedb/src/data/executor/dispatch/timeseries.rs index 438e2844f..62ce17c16 100644 --- a/nodedb/src/data/executor/dispatch/timeseries.rs +++ b/nodedb/src/data/executor/dispatch/timeseries.rs @@ -409,4 +409,55 @@ mod tests { ); assert_eq!(h.core.ts_max_ingested_lsn.get(&key), None); } + + /// A `RETURNING` ingest whose tags overflow the cardinality limit is + /// refused before the first row lands. The refusal code claims nothing + /// applied, so no row and no memtable can exist afterwards. + #[test] + fn a_returning_ingest_over_the_tag_limit_writes_no_row() { + use nodedb_physical::physical_plan::document::{ReturningColumns, ReturningSpec}; + + let mut h = make_core(); + h.core.ts_tuning.max_tag_cardinality = 2; + let mut task = ingest_task( + format!( + "{COLLECTION},host=h0 value=1i\n\ + {COLLECTION},host=h1 value=2i\n\ + {COLLECTION},host=h2 value=3i\n" + ) + .into_bytes(), + Some(7), + ); + if let PhysicalPlan::Timeseries(TimeseriesOp::Ingest { returning, .. }) = + &mut task.request.plan + { + *returning = Some(ReturningSpec { + columns: ReturningColumns::Star, + }); + } + let PhysicalPlan::Timeseries(op) = task.request.plan.clone() else { + panic!("timeseries plan"); + }; + + let response = h.core.dispatch_timeseries(&task, &op); + assert_eq!(response.status, Status::Error); + assert!( + matches!( + response.error_code.as_deref(), + Some(crate::bridge::envelope::ErrorCode::RejectedPrevalidation { .. }) + ), + "got {:?}", + response.error_code + ); + + let key = ( + DatabaseId::DEFAULT, + TenantId::new(TENANT), + COLLECTION.to_string(), + ); + assert!( + !h.core.columnar_memtables.contains_key(&key), + "a refused ingest must not create the memtable" + ); + } } diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs index d4810a54e..38baacc1b 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete.rs @@ -6,6 +6,7 @@ use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::write_hook; +use crate::data::executor::handlers::partial_refusal::refusal_after_rows; use crate::data::executor::handlers::returning_doc; use crate::data::executor::handlers::returning_rows; use crate::data::executor::handlers::rls_write_gate; @@ -175,25 +176,48 @@ impl CoreLoop { // deleted. The pre-deletion image is the only image a delete has. A row // that is already absent is admitted: it removes nothing, so there is // no image for the policy to restrict. - if !matches!( + // The period lock is judged here too, on the same image: each row + // below commits in its own transaction, so a lock judged there + // refuses after the rows ahead of it were removed. + let gate_policy = !matches!( rls_write_check.decision(), nodedb_types::WriteGateDecision::AdmitAll - ) { + ); + let period_lock = self + .doc_configs + .get(&config_key) + .and_then(|config| config.enforcement.period_lock.as_ref()); + if gate_policy || period_lock.is_some() { for key in &apply_ids { let stored = match self.sparse.get(database_id, tid, collection, key) { Ok(Some(bytes)) => bytes, Ok(None) => continue, Err(e) => return self.response_error(task, e), }; - let identity = key.to_identity(); - if let Err(e) = rls_write_gate::admit_stored_row( - rls_write_check, - &stored, - &identity, - strict_schema.as_ref(), - tid, - collection, - ) { + if gate_policy + && let Err(e) = rls_write_gate::admit_stored_row( + rls_write_check, + &stored, + &key.to_identity(), + strict_schema.as_ref(), + tid, + collection, + ) + { + return self.response_error(task, e); + } + if let Some(lock) = period_lock + && let Err(e) = + crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + collection, + &stored, + lock, + resolved_sum_targets, + ) + { return self.response_error(task, e); } } @@ -241,35 +265,38 @@ impl CoreLoop { // will not decode is a different answer: it would silently drop out // of RETURNING and, worse, contribute no removed index tuples, so // its old secondary-index entries would survive the delete. - let pre_delete_doc: Option = - if returning.is_some() || !index_paths.is_empty() { - match self - .sparse - .get( - task.request.database_id.as_u64(), - tid, - collection, - storage_key, - ) - .ok() - .flatten() - { - Some(bytes) => { - let identity = storage_key.to_identity(); - match returning_doc::from_stored_json( - &bytes, - &identity, - strict_schema.as_ref(), - ) { - Ok(doc) => Some(doc), - Err(e) => return self.response_error(task, e), + let pre_delete_doc: Option = if returning.is_some() + || !index_paths.is_empty() + { + match self + .sparse + .get( + task.request.database_id.as_u64(), + tid, + collection, + storage_key, + ) + .ok() + .flatten() + { + Some(bytes) => { + let identity = storage_key.to_identity(); + match returning_doc::from_stored_json( + &bytes, + &identity, + strict_schema.as_ref(), + ) { + Ok(doc) => Some(doc), + Err(e) => { + return self.response_error(task, refusal_after_rows(affected, e)); } } - None => None, } - } else { - None - }; + None => None, + } + } else { + None + }; // The removal and the materialized-sum deltas it owes share ONE // transaction, so a debited target row can never outlive a removal @@ -279,7 +306,7 @@ impl CoreLoop { // index diff, and is not widened for this. let row_txn = match self.sparse.begin_write() { Ok(txn) => txn, - Err(e) => return self.response_error(task, e), + Err(e) => return self.response_error(task, refusal_after_rows(affected, e)), }; let deleted_bytes = self .sparse @@ -292,24 +319,6 @@ impl CoreLoop { ) .ok() .flatten(); - // Period lock, the pre-deletion image — a delete has no other. - // Checked before `write_hook::run` and before commit: dropping - // `row_txn` un-committed on a refusal reverses the removal. - if let Some(bytes) = deleted_bytes.as_deref() - && let Some(config) = self.doc_configs.get(&config_key) - && let Some(ref pl) = config.enforcement.period_lock - && let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( - &self.sparse, - database_id, - tid, - collection, - bytes, - pl, - resolved_sum_targets, - ) - { - return self.response_error(task, e); - } let mut target_writes = Vec::new(); if let Some(bytes) = deleted_bytes.as_deref() { match write_hook::run( @@ -333,15 +342,18 @@ impl CoreLoop { Ok(outcome) => target_writes = outcome.target_writes, // Dropping `row_txn` un-committed reverses both the removal // and every target it had already debited. - Err(e) => return self.response_error(task, e), + Err(e) => return self.response_error(task, refusal_after_rows(affected, e)), } } if let Err(e) = row_txn.commit() { return self.response_error( task, - ErrorCode::Internal { - detail: format!("bulk delete commit: {e}"), - }, + refusal_after_rows( + affected, + ErrorCode::Internal { + detail: format!("bulk delete commit: {e}"), + }, + ), ); } // One durable redo entry per debited target row, naming the TARGET @@ -417,3 +429,106 @@ impl CoreLoop { response } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::doc_format; + use crate::engine::document::store::CollectionConfig; + use crate::types::{DatabaseId, TenantId}; + use nodedb_physical::physical_plan::PeriodLockConfig; + use nodedb_types::{StorageKey, Surrogate}; + + const TID: u64 = 1; + const COLLECTION: &str = "journal"; + + fn seed(core: &mut CoreLoop, database_id: u64, surrogate: u32, row: serde_json::Value) { + core.sparse + .put( + database_id, + TID, + COLLECTION, + &StorageKey::for_surrogate(Surrogate(surrogate)), + &doc_format::encode_to_msgpack(&row), + ) + .expect("seed row"); + } + + /// A closed period holds one matched row. The refusal code claims + /// nothing applied, so no matched row can be removed, including the rows + /// the lock does not hold. + #[test] + fn a_period_lock_on_any_matched_row_removes_no_row() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let task = make_default_task(); + let database_id = task.request.database_id.as_u64(); + let mut config = CollectionConfig::new(COLLECTION); + config.enforcement.period_lock = Some(PeriodLockConfig { + period_column: "fiscal_period".into(), + ref_table: "fiscal_periods".into(), + ref_pk: "period_key".into(), + status_column: "status".into(), + allowed_statuses: vec!["OPEN".into()], + }); + core.doc_configs.insert( + ( + DatabaseId::new(database_id), + TenantId::new(TID), + COLLECTION.to_string(), + ), + config, + ); + seed(&mut core, database_id, 1, serde_json::json!({"amount": 1})); + // No reference row resolves this period, so the lock refuses it. + seed( + &mut core, + database_id, + 2, + serde_json::json!({"amount": 2, "fiscal_period": "2026-01"}), + ); + seed(&mut core, database_id, 3, serde_json::json!({"amount": 3})); + + let response = core.execute_bulk_delete( + &task, + TID, + BulkDeleteParams { + collection: COLLECTION, + filter_bytes: &[], + returning: None, + rls_filters: &[], + rls_write_check: &nodedb_types::RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: &[], + ollp: OllpPrediction { + surrogates: None, + edges: None, + }, + declared_primary_key: None, + }, + ); + + assert_eq!(response.status, Status::Error); + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::PeriodLocked { .. }) + ), + "got {:?}", + response.error_code + ); + for surrogate in [1, 2, 3] { + let stored = core + .sparse + .get( + database_id, + TID, + COLLECTION, + &StorageKey::for_surrogate(Surrogate(surrogate)), + ) + .expect("read row"); + assert!(stored.is_some(), "row {surrogate} must remain"); + } + } +} diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update.rs b/nodedb/src/data/executor/handlers/bulk_dml/update.rs index cbe7a232f..a86f0182c 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update.rs @@ -6,11 +6,11 @@ use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; use crate::bridge::scan_filter::ScanFilter; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::write_hook; +use crate::data::executor::handlers::partial_refusal::refusal_after_partial_apply; use crate::data::executor::handlers::point::update_reindex::NonbitemporalUpdateReindex; use crate::data::executor::handlers::point::update_reindex_vector::UpdateVectorReindex; use crate::data::executor::handlers::returning_doc; use crate::data::executor::handlers::returning_rows; -use crate::data::executor::handlers::rls_write_gate; use crate::data::executor::handlers::transaction::stage_write::stored_row_identity; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; @@ -225,6 +225,16 @@ impl CoreLoop { { return self.response_error(task, e); } + if let Err(code) = self.gate_bulk_update_rows( + task, + tid, + collection, + &projected, + rls_write_check, + resolved_sum_targets, + ) { + return self.response_error(task, code); + } for row in projected { let ProjectedUpdateRow { key: storage_key, @@ -233,45 +243,6 @@ impl CoreLoop { doc, updated_bytes, } = row; - // Period lock, both images — matching `execute_point_update`: a - // closed period must reject an edit to a row it already holds, - // and must reject an edit that assigns the period column into it. - if let Some(config) = self.doc_configs.get(&config_key) - && let Some(ref pl) = config.enforcement.period_lock - { - if let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( - &self.sparse, - database_id, - tid, - collection, - ¤t_bytes, - pl, - resolved_sum_targets, - ) { - return self.response_error(task, e); - } - if let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( - &self.sparse, - database_id, - tid, - collection, - &updated_bytes, - pl, - resolved_sum_targets, - ) { - return self.response_error(task, e); - } - } - // Gate the persist on the collection's write policy, decided - // against this row's post-update image — `doc` already has - // the assignments and any regenerated columns applied, so it - // is the row that would exist afterwards. A rejected row - // fails the statement rather than being skipped: a skipped - // row would be reported as unaffected while the rest of the - // predicate's matches were rewritten. - if let Err(e) = rls_write_gate::admit_row(rls_write_check, &doc, tid, collection) { - return self.response_error(task, e); - } // Both images are already materialized here — `old_doc_json` // for the secondary-index diff and `doc` as the post-image — // so the row's materialized-sum delta costs no extra read @@ -307,7 +278,12 @@ impl CoreLoop { // skipping it would report a smaller affected count as // the truth while the rest of the predicate's matches // were rewritten, and leave the stored total short of the - // `SUM(...)` over the rows that did land. + // `SUM(...)` over the rows that did land. The row's own + // transaction did not commit, but earlier rows did. + Err(e) if affected > 0 => { + return self + .response_error(task, refusal_after_partial_apply(ErrorCode::from(e))); + } Err(e) => return self.response_error(task, e), }; // One durable redo entry per derived target row, naming the @@ -353,7 +329,8 @@ impl CoreLoop { has_vectors, }) { - return self.response_error(task, e); + // The row's body already committed. + return self.response_error(task, refusal_after_partial_apply(ErrorCode::from(e))); } // Emit an update event per affected row to the Event Plane, // so AFTER-UPDATE triggers and CDC/change-stream consumers @@ -444,3 +421,113 @@ impl CoreLoop { response } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::doc_format; + use nodedb_physical::physical_plan::UpdateValue; + use nodedb_types::{StorageKey, Surrogate}; + + const TID: u64 = 1; + const COLLECTION: &str = "orders"; + + fn seed(core: &mut CoreLoop, database_id: u64, surrogate: u32, owner: &str) { + let row = serde_json::json!({"owner": owner, "note": "old"}); + core.sparse + .put( + database_id, + TID, + COLLECTION, + &StorageKey::for_surrogate(Surrogate(surrogate)), + &doc_format::encode_to_msgpack(&row), + ) + .expect("seed row"); + } + + fn note_of(core: &CoreLoop, database_id: u64, surrogate: u32) -> serde_json::Value { + let stored = core + .sparse + .get( + database_id, + TID, + COLLECTION, + &StorageKey::for_surrogate(Surrogate(surrogate)), + ) + .expect("read row") + .expect("row exists"); + doc_format::decode_document(&stored) + .expect("row decodes") + .get("note") + .cloned() + .unwrap_or(serde_json::Value::Null) + } + + fn owner_policy(owner: &str) -> Vec { + let filter = ScanFilter { + field: "owner".into(), + op: crate::bridge::scan_filter::FilterOp::Eq, + value: nodedb_types::Value::String(owner.into()), + clauses: Vec::new(), + expr: None, + }; + zerompk::to_msgpack_vec(&vec![filter]).expect("encode policy") + } + + /// The write policy refuses one matched row. The refusal code claims + /// nothing applied, so no matched row can be rewritten, including the + /// rows the policy admits. + #[test] + fn a_policy_refusal_on_any_matched_row_rewrites_no_row() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let task = make_default_task(); + let database_id = task.request.database_id.as_u64(); + seed(&mut core, database_id, 1, "alice"); + seed(&mut core, database_id, 2, "bob"); + seed(&mut core, database_id, 3, "alice"); + + let updates = vec![( + "note".to_string(), + UpdateValue::Literal( + nodedb_types::json_to_msgpack(&serde_json::json!("new")).expect("encode"), + ), + )]; + let policy = nodedb_types::RlsWriteCheck::Predicate(owner_policy("alice")); + let response = core.execute_bulk_update( + &task, + TID, + BulkUpdateParams { + collection: COLLECTION, + filter_bytes: &[], + updates: &updates, + returning: None, + ollp_predicted_surrogates: None, + ollp_predicted_edges: None, + rls_filters: &[], + rls_write_check: &policy, + resolved_sum_targets: &[], + declared_primary_key: None, + }, + ); + + assert_eq!(response.status, Status::Error); + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RejectedAuthz { .. }) + ), + "got {:?}", + response.error_code + ); + for surrogate in [1, 2, 3] { + assert_eq!( + note_of(&core, database_id, surrogate), + serde_json::json!("old"), + "row {surrogate} must be unchanged" + ); + } + } +} diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs index e2aed6754..b0b8aaa4b 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update_project.rs @@ -185,3 +185,54 @@ impl CoreLoop { Ok(projected) } } + +impl CoreLoop { + /// Run every per-row gate over the whole projected set: the period lock + /// on both images and the write policy on the post-image. The apply loop + /// commits one row at a time, so a gate judged there refuses after the + /// rows ahead of it landed. + pub(in crate::data::executor) fn gate_bulk_update_rows( + &self, + task: &crate::data::executor::task::ExecutionTask, + tid: u64, + collection: &str, + rows: &[ProjectedUpdateRow], + rls_write_check: &nodedb_types::RlsWriteCheck, + resolved_sum_targets: &[nodedb_physical::physical_plan::ResolvedSumTarget], + ) -> Result<(), crate::bridge::envelope::ErrorCode> { + let database_id = task.request.database_id.as_u64(); + let config_key = ( + task.request.database_id, + TenantId::new(tid), + collection.to_string(), + ); + let period_lock = self + .doc_configs + .get(&config_key) + .and_then(|config| config.enforcement.period_lock.as_ref()); + for row in rows { + // A closed period refuses an edit to a row it holds, and an edit + // that assigns the period column into it. + if let Some(lock) = period_lock { + for image in [&row.current_bytes, &row.updated_bytes] { + crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + collection, + image, + lock, + resolved_sum_targets, + )?; + } + } + crate::data::executor::handlers::rls_write_gate::admit_row( + rls_write_check, + &row.doc, + tid, + collection, + )?; + } + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs b/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs index 3df97e8ac..060654b3c 100644 --- a/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs +++ b/nodedb/src/data/executor/handlers/columnar_write/row_ingest.rs @@ -56,16 +56,91 @@ impl CoreLoop { /// (upsert-overwrite for `Insert` and `Put`, silent skip for /// `InsertIfAbsent`, merge-via-`apply_on_conflict_updates` for `Put` /// with non-empty `on_conflict_updates`, `RejectedConstraint` error for - /// `InsertUnique` on a PK the index already carries). + /// `InsertUnique` on a PK the index or an earlier row of the batch + /// already carries). /// - /// Returns the accepted row count (and, on request, the stored post-images), - /// or `Err(Response)` on the first unrecoverable error (short-circuits the - /// remaining rows). + /// Every row is resolved and checked before any row is written, so a + /// refusal applies nothing. Returns the accepted row count (and, on + /// request, the stored post-images), or `Err(Response)` on the first + /// error. pub(in crate::data::executor) fn insert_columnar_rows( &mut self, task: &ExecutionTask, params: RowIngestParams<'_>, ) -> Result { + let resolved = self.resolve_columnar_rows(task, ¶ms)?; + let RowIngestParams { + engine_key, + intent, + collect_stored_rows, + .. + } = params; + let mut accepted = 0u64; + let mut stored_rows: Vec> = Vec::new(); + + for row in resolved { + let engine = match self.columnar_engines.get_mut(engine_key) { + Some(e) => e, + None => { + return Err(self.response_error( + task, + ErrorCode::Internal { + detail: "columnar engine vanished during insert".into(), + }, + )); + } + }; + let result = match intent { + ColumnarInsertIntent::InsertIfAbsent => engine.insert_if_absent(&row.values), + ColumnarInsertIntent::InsertUnique + | ColumnarInsertIntent::Insert + | ColumnarInsertIntent::Put => match row.surrogate { + Some(s) => engine.insert_with_surrogate(&row.values, s), + None => engine.insert(&row.values), + }, + }; + + match result { + // An `insert_if_absent` that hit an existing key returns an + // EMPTY `wal_records` — that is the engine's documented no-op + // signal, and the only way to tell a skip from a write. Counting + // it reported an `INSERT 1` for a row that was never stored, and + // returning it would hand back a row that does not exist. + Ok(mutation) if mutation.wal_records.is_empty() => {} + Ok(_) => { + accepted += 1; + if collect_stored_rows { + stored_rows.push(row.values); + } + } + Err(e) => { + return Err(self.response_error( + task, + ErrorCode::Internal { + detail: format!("columnar insert failed: {e}"), + }, + )); + } + } + } + + Ok(RowIngestOutcome { + accepted, + stored_rows, + }) + } + + /// Resolve every row of the batch to the values it writes, and run every + /// check a row can fail, without writing anything. + /// + /// An ON CONFLICT DO UPDATE merge reads the prior row from the earlier rows + /// of this batch first, then from the engine: the same prior a + /// row-by-row write reads. + fn resolve_columnar_rows( + &self, + task: &ExecutionTask, + params: &RowIngestParams<'_>, + ) -> Result, Response> { let RowIngestParams { engine_key, schema, @@ -75,10 +150,16 @@ impl CoreLoop { surrogates, ndb_rows, rls_write_check, - collect_stored_rows, - } = params; - let mut accepted = 0u64; - let mut stored_rows: Vec> = Vec::new(); + .. + } = *params; + let mut resolved: Vec = Vec::with_capacity(ndb_rows.len()); + let merging = intent == ColumnarInsertIntent::Put && !on_conflict_updates.is_empty(); + let unique = intent == ColumnarInsertIntent::InsertUnique; + // Primary key → values of the latest earlier row of this batch. Kept + // only for the intents whose checks read it: a merge reads the prior + // row, and a unique insert reads the key alone. + let mut batch_rows: std::collections::HashMap, Vec> = + std::collections::HashMap::new(); for (row_idx, row) in ndb_rows.iter().enumerate() { let obj = match row { @@ -126,88 +207,52 @@ impl CoreLoop { } }; - // Resolve the actual row to write (merged for ON CONFLICT DO - // UPDATE, plain otherwise). This runs before the mutable - // engine borrow needed by the insert call. - let final_values: Vec = match intent { - ColumnarInsertIntent::Put if !on_conflict_updates.is_empty() => { - let pk_bytes = { - let engine = match self.columnar_engines.get(engine_key) { - Some(e) => e, - None => { - return Err(self.response_error( - task, - ErrorCode::Internal { - detail: "columnar engine vanished during insert".into(), - }, - )); - } - }; - match engine.encode_pk_from_row(&values) { - Ok(b) => b, - Err(e) => { - return Err(self.response_error( - task, - ErrorCode::Internal { - detail: format!("columnar insert: pk encode failed: {e}"), - }, - )); - } - } - }; - - let prior_row = self - .columnar_engines - .get(engine_key) - .and_then(|e| e.lookup_memtable_row_by_pk(&pk_bytes)) - .or_else(|| self.read_flushed_row_by_pk(engine_key, &pk_bytes)); + let pk_bytes = if merging || unique { + let engine = match self.columnar_engines.get(engine_key) { + Some(e) => e, + None => { + return Err(self.response_error( + task, + ErrorCode::Internal { + detail: "columnar engine vanished during insert".into(), + }, + )); + } + }; + match engine.encode_pk_from_row(&values) { + Ok(b) => b, + Err(e) => { + return Err(self.response_error( + task, + ErrorCode::Internal { + detail: format!("columnar insert: pk encode failed: {e}"), + }, + )); + } + } + } else { + Vec::new() + }; + // Resolve the actual row to write: merged for ON CONFLICT DO + // UPDATE, plain otherwise. + let final_values: Vec = match intent { + ColumnarInsertIntent::Put if merging => { + let prior_row = batch_rows.get(&pk_bytes).cloned().or_else(|| { + self.columnar_engines + .get(engine_key) + .and_then(|e| e.lookup_memtable_row_by_pk(&pk_bytes)) + .or_else(|| self.read_flushed_row_by_pk(engine_key, &pk_bytes)) + }); match prior_row { None => values, - Some(prior) => { - let existing_val = row_values_to_object(schema, &prior); - let excluded_val = row_values_to_object(schema, &values); - let merged = match apply_on_conflict_updates( - existing_val, - &excluded_val, - on_conflict_updates, - ) { - Ok(v) => v, - Err(e) => { - return Err(self.response_error(task, e)); - } - }; - let merged_obj = match merged { - nodedb_types::Value::Object(m) => m, - _ => { - return Err(self.response_error( - task, - ErrorCode::Internal { - detail: "merged ON CONFLICT value was not an object" - .into(), - }, - )); - } - }; - match schema - .columns - .iter() - .map(|col| { - ndb_field_to_value(merged_obj.get(&col.name), &col.column_type) - }) - .collect::, crate::Error>>() - { - Ok(v) => v, - Err(e) => { - return Err(self.response_error( - task, - ErrorCode::Internal { - detail: format!("columnar ON CONFLICT coercion: {e}"), - }, - )); - } - } - } + Some(prior) => self.merge_on_conflict( + task, + schema, + &prior, + &values, + on_conflict_updates, + )?, } } ColumnarInsertIntent::Put @@ -231,92 +276,221 @@ impl CoreLoop { return Err(self.response_error(task, error)); } - let engine = match self.columnar_engines.get_mut(engine_key) { - Some(e) => e, - None => { + if unique { + let taken = batch_rows.contains_key(&pk_bytes) + || self + .columnar_engines + .get(engine_key) + .is_some_and(|e| e.pk_index().contains(&pk_bytes)); + if taken { + let key_desc = schema + .columns + .iter() + .zip(final_values.iter()) + .filter(|(col, _)| col.primary_key) + .map(|(col, v)| format!("{}={v}", col.name)) + .collect::>() + .join(", "); return Err(self.response_error( task, - ErrorCode::Internal { - detail: "columnar engine vanished during insert".into(), + crate::Error::RejectedConstraint { + collection: engine_key.2.clone(), + constraint: "unique".to_string(), + detail: format!( + "duplicate key value '{key_desc}' violates primary-key \ + uniqueness on '{}'", + engine_key.2 + ), }, )); } - }; - let row_surrogate = surrogates.get(row_idx).copied(); - let result = match intent { - ColumnarInsertIntent::InsertIfAbsent => engine.insert_if_absent(&final_values), - ColumnarInsertIntent::InsertUnique => { - let pk_bytes = match engine.encode_pk_from_row(&final_values) { - Ok(b) => b, - Err(e) => { - return Err(self.response_error( - task, - ErrorCode::Internal { - detail: format!("columnar insert: pk encode failed: {e}"), - }, - )); - } - }; - if engine.pk_index().contains(&pk_bytes) { - let key_desc = schema - .columns - .iter() - .zip(final_values.iter()) - .filter(|(col, _)| col.primary_key) - .map(|(col, v)| format!("{}={v}", col.name)) - .collect::>() - .join(", "); - return Err(self.response_error( - task, - crate::Error::RejectedConstraint { - collection: engine_key.2.clone(), - constraint: "unique".to_string(), - detail: format!( - "duplicate key value '{key_desc}' violates primary-key \ - uniqueness on '{}'", - engine_key.2 - ), - }, - )); - } - match row_surrogate { - Some(s) => engine.insert_with_surrogate(&final_values, s), - None => engine.insert(&final_values), - } - } - ColumnarInsertIntent::Insert | ColumnarInsertIntent::Put => match row_surrogate { - Some(s) => engine.insert_with_surrogate(&final_values, s), - None => engine.insert(&final_values), - }, - }; + } - match result { - // An `insert_if_absent` that hit an existing key returns an - // EMPTY `wal_records` — that is the engine's documented no-op - // signal, and the only way to tell a skip from a write. Counting - // it reported an `INSERT 1` for a row that was never stored, and - // returning it would hand back a row that does not exist. - Ok(mutation) if mutation.wal_records.is_empty() => {} - Ok(_) => { - accepted += 1; - if collect_stored_rows { - stored_rows.push(final_values); - } - } - Err(e) => { - return Err(self.response_error( - task, - ErrorCode::Internal { - detail: format!("columnar insert failed: {e}"), - }, - )); - } + if merging { + batch_rows.insert(pk_bytes, final_values.clone()); + } else if unique { + batch_rows.insert(pk_bytes, Vec::new()); } + resolved.push(ResolvedRow { + values: final_values, + surrogate: surrogates.get(row_idx).copied(), + }); } + Ok(resolved) + } - Ok(RowIngestOutcome { - accepted, - stored_rows, - }) + /// Merge an incoming row into its prior row by the ON CONFLICT DO UPDATE + /// assignments. + fn merge_on_conflict( + &self, + task: &ExecutionTask, + schema: &ColumnarSchema, + prior: &[Value], + values: &[Value], + on_conflict_updates: &[(String, UpdateValue)], + ) -> Result, Response> { + let existing_val = row_values_to_object(schema, prior); + let excluded_val = row_values_to_object(schema, values); + let merged = + match apply_on_conflict_updates(existing_val, &excluded_val, on_conflict_updates) { + Ok(v) => v, + Err(e) => return Err(self.response_error(task, e)), + }; + let merged_obj = match merged { + nodedb_types::Value::Object(m) => m, + _ => { + return Err(self.response_error( + task, + ErrorCode::Internal { + detail: "merged ON CONFLICT value was not an object".into(), + }, + )); + } + }; + schema + .columns + .iter() + .map(|col| ndb_field_to_value(merged_obj.get(&col.name), &col.column_type)) + .collect::, crate::Error>>() + .map_err(|e| { + self.response_error( + task, + ErrorCode::Internal { + detail: format!("columnar ON CONFLICT coercion: {e}"), + }, + ) + }) + } +} + +/// One batch row resolved to the values it writes. +struct ResolvedRow { + values: Vec, + surrogate: Option, +} + +#[cfg(test)] +mod tests { + use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp}; + use nodedb_types::{RlsWriteCheck, Value}; + + use crate::bridge::envelope::{ErrorCode, Status}; + use crate::data::executor::core_loop::CoreLoop; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::data::executor::handlers::columnar_write::ColumnarInsertParams; + use crate::data::executor::task::ExecutionTask; + use crate::types::{DatabaseId, TenantId, VShardId}; + + const TID: u64 = 1; + const COLLECTION: &str = "unique_rows"; + + fn task() -> ExecutionTask { + CoreLoop::replay_task( + TenantId::new(TID), + DatabaseId::DEFAULT, + VShardId::new(0), + crate::bridge::envelope::PhysicalPlan::Columnar(ColumnarOp::Truncate { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLLECTION), + restart_identity: false, + }), + None, + ) + } + + fn row(id: &str) -> Value { + Value::Object(std::collections::HashMap::from([ + ("id".to_string(), Value::String(id.into())), + ("v".to_string(), Value::Integer(1)), + ])) + } + + fn schema_bytes() -> Vec { + use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; + let schema = ColumnarSchema::new(vec![ + ColumnDef::required("id", ColumnType::String).with_primary_key(), + ColumnDef::required("v", ColumnType::Int64), + ]) + .expect("valid schema"); + zerompk::to_msgpack_vec(&schema).expect("encode schema") + } + + fn insert( + core: &mut CoreLoop, + intent: ColumnarInsertIntent, + rows: Vec, + ) -> crate::bridge::envelope::Response { + let payload = nodedb_types::value_to_msgpack(&Value::Array(rows)).expect("encode rows"); + let schema = schema_bytes(); + core.execute_columnar_insert( + &task(), + ColumnarInsertParams { + collection: COLLECTION, + payload: &payload, + format: "msgpack", + intent, + on_conflict_updates: &[], + surrogates: &[], + schema_bytes: &schema, + provenance: None, + rls_write_check: &RlsWriteCheck::already_decided_elsewhere(), + returning: None, + rls_filters: &[], + spatial_undo: None, + }, + ) + } + + fn live_rows(core: &CoreLoop) -> usize { + core.columnar_engines + .get(&( + DatabaseId::DEFAULT, + TenantId::new(TID), + COLLECTION.to_string(), + )) + .map_or(0, |e| e.live_row_count()) + } + + /// The funnel cancels the batch's record on a unique refusal, so the + /// refusal must leave no row of the batch behind. + #[test] + fn a_duplicate_key_late_in_a_unique_batch_writes_no_row() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let seeded = insert(&mut core, ColumnarInsertIntent::Insert, vec![row("a")]); + assert_eq!(seeded.status, Status::Ok, "{:?}", seeded.error_code); + + let refused = insert( + &mut core, + ColumnarInsertIntent::InsertUnique, + vec![row("b"), row("a")], + ); + + assert!(matches!( + refused.error_code.as_deref(), + Some(ErrorCode::RejectedConstraint { .. }) + )); + assert_eq!( + live_rows(&core), + 1, + "the row before the duplicate is not written" + ); + } + + #[test] + fn a_key_repeated_inside_a_unique_batch_writes_no_row() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + + let refused = insert( + &mut core, + ColumnarInsertIntent::InsertUnique, + vec![row("c"), row("c")], + ); + + assert!(matches!( + refused.error_code.as_deref(), + Some(ErrorCode::RejectedConstraint { .. }) + )); + assert_eq!(live_rows(&core), 0); } } diff --git a/nodedb/src/data/executor/handlers/control/calvin/flush.rs b/nodedb/src/data/executor/handlers/control/calvin/flush.rs index 8c3126608..84d154a15 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/flush.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/flush.rs @@ -107,7 +107,9 @@ mod tests { use crate::types::TenantId; use super::super::shared::CalvinExecCtx; - use super::super::shared::test_support::{make_task, point_insert_plan}; + use super::super::shared::test_support::{ + bulk_delete_plan, make_task, point_insert_plan, seed_row, + }; #[test] fn calvin_flush_drops_synthetic_overlay() { @@ -138,4 +140,36 @@ mod tests { "flush must drop the synthetic overlay entry alongside commit_pending" ); } + + /// A staged plan keeps its OLLP prediction until the flush replays it. + /// A row that joins the predicate between stage and flush makes the + /// leader's flush answer `OllpRetryRequired`, although the verdict + /// already committed the transaction. + #[test] + fn a_staged_prediction_that_drifts_before_the_flush_answers_ollp_retry() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + seed_row(&mut core, "orders", 1); + + let task = make_task(); + let tenant_id = TenantId::new(1); + let plans = vec![bulk_delete_plan("orders", Some(vec![1]))]; + let ctx = CalvinExecCtx { + epoch: 1, + position: 0, + epoch_system_ms: 0, + is_group_leader: true, + }; + let staged = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); + assert_eq!(staged.status, Status::Ok, "{:?}", staged.error_code); + + seed_row(&mut core, "orders", 2); + let flushed = core.execute_calvin_flush(&task, 1, 0); + + assert_eq!(flushed.status, Status::Error); + assert_eq!( + flushed.error_code.as_deref(), + Some(&crate::bridge::envelope::ErrorCode::OllpRetryRequired) + ); + } } diff --git a/nodedb/src/data/executor/handlers/control/calvin/shared.rs b/nodedb/src/data/executor/handlers/control/calvin/shared.rs index c8b566932..8d83cd6a9 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/shared.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/shared.rs @@ -43,16 +43,6 @@ impl CoreLoop { } } -pub(super) fn calvin_panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String { - if let Some(message) = payload.downcast_ref::<&'static str>() { - (*message).to_owned() - } else if let Some(message) = payload.downcast_ref::() { - message.clone() - } else { - "".to_owned() - } -} - #[cfg(test)] pub(in crate::data::executor::handlers::control::calvin) mod test_support { use std::time::{Duration, Instant}; diff --git a/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs b/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs index cb73dc6b5..bb66435b2 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs @@ -16,8 +16,9 @@ use crate::types::TenantId; use nodedb_physical::physical_plan::PhysicalPlan; use crate::data::executor::handlers::control::calvin_txn_id::calvin_synthetic_txn_id; +use crate::data::panic_payload::panic_payload_to_string; -use super::shared::{CalvinExecCtx, calvin_panic_payload_to_string}; +use super::shared::CalvinExecCtx; impl CoreLoop { /// Validate a static-set Calvin transaction and stage it for commit. @@ -106,7 +107,7 @@ impl CoreLoop { ErrorCode::Internal { detail: format!( "panic while staging static Calvin transaction: {}", - calvin_panic_payload_to_string(payload.as_ref()) + panic_payload_to_string(payload.as_ref()) ), }, ); diff --git a/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs b/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs index 49bbef079..d05c69252 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs @@ -20,11 +20,9 @@ use super::params::{CRDT_PENDING_DEPENDENCIES, CRDT_SINGLE_DOCUMENT_DELTA, CrdtA /// Why a local apply produced no materializable row. enum LocalRefusal { - /// Permanent: the same bytes fail identically on a retry. - Terminal { - constraint: &'static str, - detail: String, - }, + /// Permanent: the delta fails to decode, fails its signature + /// check, or writes rows outside its frame target. Nothing is imported. + Malformed, /// Nothing applied, but the identical bytes apply once the missing causal /// history arrives. Retryable { detail: String }, @@ -56,6 +54,23 @@ impl CoreLoop { } } + // The engine accepts a delta only when every row it writes is the + // frame target. An empty target admits any row, and the import is + // installed before its write set is known. Refusing it here keeps + // the refusal ahead of every state change. + if document_id.is_empty() { + return self.response_error( + task, + ErrorCode::RejectedConstraint { + constraint: CRDT_SINGLE_DOCUMENT_DELTA.to_string(), + detail: format!( + "a CRDT apply into {collection} must name the one document its delta \ + writes; nothing was applied" + ), + }, + ); + } + // Borrow the engine in a nested block so the &mut borrow is dropped // before the sparse write below takes &self. On a Clean apply we read // the merged row back and encode it while the borrow is live, carrying @@ -82,25 +97,12 @@ impl CoreLoop { peer_id, ); match outcome { - ValidatedApplyOutcome::Clean { write_set, .. } => { + ValidatedApplyOutcome::Clean { .. } => { imported_authoritative = true; - // Enforce the one-document-per-delta contract before - // materializing: a delta that wrote rows other than the - // frame target has no surrogate for those rows, so - // materializing only `document_id` would silently drop the - // rest. - match Self::single_document_write_set(collection, document_id, &write_set) { - Ok(()) => { - if surrogate != Surrogate::ZERO { - Ok(Self::encode_crdt_row(engine, collection, document_id)) - } else { - Ok(None) - } - } - Err(detail) => Err(LocalRefusal::Terminal { - constraint: CRDT_SINGLE_DOCUMENT_DELTA, - detail, - }), + if surrogate != Surrogate::ZERO { + Ok(Self::encode_crdt_row(engine, collection, document_id)) + } else { + Ok(None) } } ValidatedApplyOutcome::Rejected(vt) => { @@ -110,10 +112,7 @@ impl CoreLoop { tracing::debug!(core = self.core_id, %collection, reason = %vt, "crdt apply violated constraint (DLQ)"); Ok(None) } - ValidatedApplyOutcome::Malformed => { - warn!(core = self.core_id, %collection, "crdt apply skipped malformed delta"); - Ok(None) - } + ValidatedApplyOutcome::Malformed => Err(LocalRefusal::Malformed), ValidatedApplyOutcome::PendingDependencies => { // Nothing was imported: the operations are buffered awaiting // predecessors this collection's document has never seen. @@ -155,22 +154,20 @@ impl CoreLoop { } Ok(None) => {} Err(refusal) => { - if imported_authoritative { - self.note_collection_write_lsn(task, collection); - } let code = match refusal { - LocalRefusal::Terminal { constraint, detail } => { + LocalRefusal::Malformed => { warn!( core = self.core_id, %collection, %document_id, - constraint, - detail = %detail, - "crdt apply rejected: delta could not be materialized" + "crdt apply refused a malformed delta" ); - ErrorCode::RejectedConstraint { - constraint: constraint.to_string(), - detail, + ErrorCode::RejectedPrevalidation { + reason: format!( + "delta for {collection}/{document_id} could not be decoded, \ + failed its signature check, or wrote rows outside \ + {document_id}; nothing was applied" + ), } } LocalRefusal::Retryable { detail } => { @@ -191,3 +188,95 @@ impl CoreLoop { self.response_ok(task) } } + +#[cfg(test)] +mod tests { + use loro::LoroValue; + use nodedb_types::Surrogate; + + use super::*; + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + + fn params<'a>(document_id: &'a str, delta: &'a [u8]) -> CrdtApplyParams<'a> { + CrdtApplyParams { + collection: "docs", + document_id, + delta, + surrogate: Surrogate::ZERO, + peer_id: 7, + provenance: None, + constraint_version_required: 0, + expected_frontier_digest: None, + auth_user_id: 0, + auth_device_id: 0, + auth_seq_no: 0, + delta_signature: [0; 32], + signing_required: false, + } + } + + fn two_row_delta() -> Vec { + let source = nodedb_crdt::CrdtState::new(42).expect("source state"); + for row in ["one", "two"] { + source + .upsert("docs", row, &[("value", LoroValue::String(row.into()))]) + .expect("source write"); + } + source.export_snapshot().expect("source snapshot") + } + + /// A delta with no named target is refused before the import. The + /// refusal code claims nothing applied, so no row can exist afterwards. + #[test] + fn a_delta_without_a_target_document_imports_no_row() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _request_tx, _response_rx) = make_core_with_dir(dir.path()); + let task = make_default_task(); + let delta = two_row_delta(); + + let response = core.apply_crdt_local(&task, params("", &delta)); + + assert_eq!(response.status, Status::Error); + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RejectedConstraint { .. }) + ), + "got {:?}", + response.error_code + ); + let key = (task.request.database_id, task.request.tenant_id); + let imported = core.crdt_engines.get(&key).is_some_and(|engine| { + engine.row_exists("docs", "one") || engine.row_exists("docs", "two") + }); + assert!(!imported, "a refused delta must not reach the CRDT state"); + } + + /// A delta that writes rows beyond its named target is refused as + /// malformed, and neither row is imported. + #[test] + fn a_delta_writing_a_foreign_row_imports_no_row() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _request_tx, _response_rx) = make_core_with_dir(dir.path()); + let task = make_default_task(); + let delta = two_row_delta(); + + let response = core.apply_crdt_local(&task, params("one", &delta)); + + assert_eq!(response.status, Status::Error); + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RejectedPrevalidation { .. }) + ), + "got {:?}", + response.error_code + ); + let key = (task.request.database_id, task.request.tenant_id); + let imported = core.crdt_engines.get(&key).is_some_and(|engine| { + engine.row_exists("docs", "one") || engine.row_exists("docs", "two") + }); + assert!(!imported, "a refused delta must not reach the CRDT state"); + } +} diff --git a/nodedb/src/data/executor/handlers/graph_edge_write/put_batch.rs b/nodedb/src/data/executor/handlers/graph_edge_write/put_batch.rs index 5a651ea32..dcaf26b5f 100644 --- a/nodedb/src/data/executor/handlers/graph_edge_write/put_batch.rs +++ b/nodedb/src/data/executor/handlers/graph_edge_write/put_batch.rs @@ -25,23 +25,17 @@ impl CoreLoop { ) -> Response { debug!(core = self.core_id, count = edges.len(), "edge put batch"); let database_id = task.request.database_id.as_u64(); + // Every endpoint is checked before any edge is written, so a dangling + // refusal applies nothing. + if let Some(missing_node) = edges.iter().find_map(|edge| { + [&edge.src_id, &edge.dst_id] + .into_iter() + .find(|node| self.is_node_deleted(database_id, tid, node)) + .cloned() + }) { + return self.response_error(task, ErrorCode::RejectedDanglingEdge { missing_node }); + } for (idx, edge) in edges.iter().enumerate() { - if self.is_node_deleted(database_id, tid, &edge.src_id) { - return self.response_error( - task, - ErrorCode::RejectedDanglingEdge { - missing_node: edge.src_id.clone(), - }, - ); - } - if self.is_node_deleted(database_id, tid, &edge.dst_id) { - return self.response_error( - task, - ErrorCode::RejectedDanglingEdge { - missing_node: edge.dst_id.clone(), - }, - ); - } let ord = self .active_graph_system_from .unwrap_or_else(|| self.hlc.next_ordinal()); @@ -122,3 +116,59 @@ impl CoreLoop { self.response_affected(task, edges.len() as u64) } } + +#[cfg(test)] +mod tests { + use crate::bridge::envelope::{ErrorCode, Status}; + use crate::types::TenantId; + use nodedb_physical::physical_plan::BatchEdge; + use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate}; + + use super::super::shared::test_support::{make_core, make_task_with_lsn}; + + fn edge(src: &str, dst: &str) -> BatchEdge { + BatchEdge { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "knows"), + src_id: src.to_string(), + label: "KNOWS".to_string(), + dst_id: dst.to_string(), + src_surrogate: Surrogate::new(1), + dst_surrogate: Surrogate::new(2), + } + } + + /// The funnel cancels the batch's record on a dangling refusal, so the + /// refusal must leave no edge of the batch behind. + #[test] + fn a_dangling_edge_late_in_the_batch_writes_no_edge() { + let mut h = make_core(); + h.core + .mark_node_deleted(DatabaseId::DEFAULT.as_u64(), 1, "gone"); + let task = make_task_with_lsn(9); + let edges = vec![edge("a", "b"), edge("c", "gone")]; + + let resp = h.core.execute_edge_put_batch(&task, 1, &edges); + + assert_eq!(resp.status, Status::Error); + assert!(matches!( + resp.error_code.as_deref(), + Some(ErrorCode::RejectedDanglingEdge { missing_node }) if missing_node == "gone" + )); + let first = h + .core + .edge_store + .get_edge( + DatabaseId::DEFAULT.as_u64(), + TenantId::new(1), + "knows", + "a", + "KNOWS", + "b", + ) + .expect("read edge"); + assert!( + first.is_none(), + "the edge before the dangling one is not written" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/mod.rs b/nodedb/src/data/executor/handlers/mod.rs index d87918471..582c9c2fa 100644 --- a/nodedb/src/data/executor/handlers/mod.rs +++ b/nodedb/src/data/executor/handlers/mod.rs @@ -40,6 +40,7 @@ pub mod kv; pub mod merge; pub(super) mod merge_helpers; pub(super) mod merge_orchestrated; +pub(super) mod partial_refusal; pub mod point; pub(super) mod provider_scan; pub(super) mod provider_scan_compute; diff --git a/nodedb/src/data/executor/handlers/partial_refusal.rs b/nodedb/src/data/executor/handlers/partial_refusal.rs new file mode 100644 index 000000000..7f506ef96 --- /dev/null +++ b/nodedb/src/data/executor/handlers/partial_refusal.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The answer a handler gives when it refuses after part of its write landed. +//! +//! A definite refusal code claims nothing applied. On one, the Control Plane +//! writes an abort marker for the request's WAL records, and recovery skips +//! them. A handler whose refusal follows a landed write must answer with a +//! code that keeps the records, or recovery drops the part that landed. + +use crate::bridge::envelope::ErrorCode; +use crate::control::server::dispatch_utils::write_definitely_not_applied; + +/// `code` as a refusal that follows a landed write. +/// +/// A definite code becomes `Internal`, which keeps the request's records +/// for replay. Any other code already keeps them and passes through. +pub(in crate::data::executor) fn refusal_after_partial_apply(code: ErrorCode) -> ErrorCode { + if write_definitely_not_applied(&code) { + ErrorCode::Internal { + detail: format!("refused after part of the write applied, which stays: {code:?}"), + } + } else { + code + } +} + +/// `code` from a handler that commits row by row, after `rows_landed` rows +/// committed. With no row landed, `code` passes through. +pub(in crate::data::executor) fn refusal_after_rows( + rows_landed: u64, + code: impl Into, +) -> ErrorCode { + let code = code.into(); + if rows_landed > 0 { + refusal_after_partial_apply(code) + } else { + code + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_definite_refusal_after_a_landed_write_keeps_the_records() { + let answered = refusal_after_partial_apply(ErrorCode::RejectedAuthz { + resource: "RLS write policy on 'orders' rejected the row".into(), + }); + assert!(matches!(answered, ErrorCode::Internal { .. })); + assert!(!write_definitely_not_applied(&answered)); + } + + #[test] + fn a_refusal_before_any_row_landed_stays_definite() { + let code = ErrorCode::PeriodLocked { + collection: "ledger".into(), + }; + assert_eq!(refusal_after_rows(0, code.clone()), code); + assert!(matches!( + refusal_after_rows(1, code), + ErrorCode::Internal { .. } + )); + } + + #[test] + fn a_code_that_keeps_the_records_passes_through() { + assert_eq!( + refusal_after_partial_apply(ErrorCode::DeadlineExceeded), + ErrorCode::DeadlineExceeded + ); + } +} diff --git a/nodedb/src/data/executor/handlers/timeseries/admission.rs b/nodedb/src/data/executor/handlers/timeseries/admission.rs index 2cdd1c91b..27a8300c5 100644 --- a/nodedb/src/data/executor/handlers/timeseries/admission.rs +++ b/nodedb/src/data/executor/handlers/timeseries/admission.rs @@ -96,6 +96,30 @@ pub(super) fn has_tag_headroom( true } +/// Whether some symbol column in `symbol_columns` receives more distinct +/// values from `lines` than `max_tag_cardinality` allows. +/// +/// Such a batch cannot fit in any memtable generation. A flush resets the +/// dictionaries, and some of its rows are rejected all the same. A batch no +/// longer than the ceiling cannot trip it, so that case costs nothing. +pub(super) fn exceeds_tag_ceiling( + symbol_columns: &[String], + lines: &[IlpLine<'_>], + max_tag_cardinality: u32, +) -> bool { + let ceiling = max_tag_cardinality as usize; + if lines.len() <= ceiling { + return false; + } + symbol_columns.iter().any(|col_name| { + let distinct: HashSet<&str> = lines + .iter() + .map(|line| symbol_value(line, col_name)) + .collect(); + distinct.len() > ceiling + }) +} + /// The value `ingest_batch_with_lvc` would resolve for symbol column /// `col_name` on `line`. /// diff --git a/nodedb/src/data/executor/handlers/timeseries/ingest.rs b/nodedb/src/data/executor/handlers/timeseries/ingest.rs index ae9949231..e369b8ca4 100644 --- a/nodedb/src/data/executor/handlers/timeseries/ingest.rs +++ b/nodedb/src/data/executor/handlers/timeseries/ingest.rs @@ -178,6 +178,34 @@ impl CoreLoop { return self.response_error(task, error); } + // A `RETURNING` ingest must take every row or none, because its + // row set has no place to report a rejected row. The tag ceiling is + // the one rejection a flush cannot clear, so it is decided here, + // before the first write. + if returning.is_some() + && admission::exceeds_tag_ceiling( + &self.ts_symbol_columns_for_ingest( + task.request.database_id, + tid, + collection, + &lines, + ), + &lines, + self.ts_tuning.max_tag_cardinality, + ) + { + return self.response_error( + task, + ErrorCode::RejectedPrevalidation { + reason: format!( + "timeseries ingest with RETURNING carries more distinct tag values \ + than the tag cardinality limit ({}) allows", + self.ts_tuning.max_tag_cardinality + ), + }, + ); + } + if mode == TimeseriesApplyMode::CommitDeferred && let Err(error) = self.prevalidate_deferred_ilp_ingest(task, tid, collection, &lines) { @@ -302,22 +330,20 @@ impl CoreLoop { ); } - // A rejected row is a FAILURE, not a requested skip, and the two answer - // shapes report it differently: the count response below carries - // `rejected`, so a client can see rows were dropped, but a `RETURNING` - // response is a row set with nowhere to put that number — a short row - // set is indistinguishable from a complete one. Rather than tell the - // client less than the truth, a projecting ingest fails outright and - // names the count and the first reason. The non-projecting path keeps - // its counts unchanged because it already reports them honestly. + // A rejected row is a FAILURE, not a requested skip. The count + // response below reports `rejected`, but a `RETURNING` row set has no + // place for that number. The tag-ceiling check above refuses every + // rejection it can foresee before any row lands. A rejection that + // still reaches here follows accepted rows, so the error is + // `Internal`: a refusal code would claim nothing applied. if returning.is_some() && rejected > 0 { let reason = outcome .first_rejection .unwrap_or_else(|| "no reason recorded".to_string()); return self.response_error( task, - ErrorCode::RejectedPrevalidation { - reason: format!( + ErrorCode::Internal { + detail: format!( "timeseries ingest with RETURNING rejected {rejected} of {} rows and \ cannot report them alongside a row set; first rejection: {reason}", accepted + rejected diff --git a/nodedb/src/data/executor/handlers/timeseries/ingest_schema.rs b/nodedb/src/data/executor/handlers/timeseries/ingest_schema.rs index 78068f716..2837d5c5c 100644 --- a/nodedb/src/data/executor/handlers/timeseries/ingest_schema.rs +++ b/nodedb/src/data/executor/handlers/timeseries/ingest_schema.rs @@ -7,6 +7,7 @@ //! rows, and the one that decides what every later read of the collection is //! shaped like. +use crate::engine::timeseries::columnar_memtable::ColumnType; use crate::engine::timeseries::ilp; use crate::engine::timeseries::ilp_ingest; @@ -83,4 +84,34 @@ impl CoreLoop { ); ilp_ingest::infer_schema(lines) } + + /// Names of the symbol columns `lines` resolve against once the ingest + /// has created or evolved the memtable. Reads live state only. + pub(super) fn ts_symbol_columns_for_ingest( + &self, + database_id: crate::types::DatabaseId, + tid: crate::types::TenantId, + collection: &str, + lines: &[ilp::IlpLine<'_>], + ) -> Vec { + let key = (database_id, tid, collection.to_string()); + let columns = match self.columnar_memtables.get(&key) { + Some(memtable) => { + let schema = memtable.schema(); + let mut columns = schema.columns.clone(); + columns.extend(ilp_ingest::new_columns(schema, lines)); + columns + } + None => { + self.declared_ts_memtable_schema(database_id, tid, collection) + .unwrap_or_else(|| ilp_ingest::infer_schema(lines)) + .columns + } + }; + columns + .into_iter() + .filter(|(_, col_type)| *col_type == ColumnType::Symbol) + .map(|(name, _)| name) + .collect() + } } diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs index c7d5baeeb..e8ddca9a3 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch.rs @@ -2,10 +2,11 @@ //! Transaction batch execution handler. //! -//! Executes a `PhysicalPlan::TransactionBatch` atomically: all sub-plans -//! succeed or all are rolled back. Write operations (PointPut, PointDelete, -//! VectorInsert, EdgePut, EdgeDelete) are tracked for rollback on failure. -//! CRDT deltas are accumulated in a scratch buffer and only applied on success. +//! Executes a `PhysicalPlan::TransactionBatch`: on a failure, every write an +//! engine-specific handler made is rolled back. A sub-plan with no such +//! handler runs through the ordinary dispatch path and records no undo, so +//! its write stays (see `batch_irreversible`). CRDT deltas are accumulated in +//! a scratch buffer and only applied on success. use std::panic::{AssertUnwindSafe, catch_unwind}; @@ -17,6 +18,9 @@ use crate::data::executor::task::ExecutionTask; use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_types::calvin::VersionedReadEntry; +use crate::data::panic_payload::panic_payload_to_string; + +use super::batch_irreversible::{applied_irreversibly, batch_failure_code, batch_failure_response}; use super::undo::UndoEntry; /// A CRDT delta buffered during a transaction batch: `(delta_bytes, id, @@ -97,7 +101,7 @@ impl CoreLoop { }; self.active_bitemporal_stamps.clear(); self.active_graph_system_from = None; - let (last_response, undo_log, crdt_deltas) = match sub { + let (last_response, undo_log, crdt_deltas, irreversible) = match sub { Ok(v) => v, Err(resp) => return resp, }; @@ -117,12 +121,13 @@ impl CoreLoop { "BALANCED constraint violated, rolling back {} operations", undo_log.len() ); - return self.rollback_transaction_failure( + let response = self.rollback_transaction_failure( task, tid, undo_log, self.response_error(task, error_code), ); + return batch_failure_response(response, irreversible); } Err(payload) => { let detail = panic_payload_to_string(payload.as_ref()); @@ -148,7 +153,7 @@ impl CoreLoop { let crdt_delta_count = crdt_deltas.len(); let undo_log = match self.apply_crdt_deltas_or_rollback(task, tid, undo_log, crdt_deltas) { Ok(undo_log) => undo_log, - Err(response) => return response, + Err(response) => return batch_failure_response(response, irreversible), }; if crdt_delta_count > 0 { // Match the direct CRDT apply path: successful transaction-batch @@ -230,10 +235,12 @@ impl CoreLoop { plans: &[PhysicalPlan], mut undo_log: Vec, mut crdt_deltas: Vec, - ) -> Result<(Response, Vec, Vec), Response> { + ) -> Result<(Response, Vec, Vec, bool), Response> { let mut last_response = self.response_ok(task); + let mut irreversible = false; for (i, plan) in plans.iter().enumerate() { + let undo_before = undo_log.len(); let user_roles = &task.request.user_roles; let outcome = catch_unwind(AssertUnwindSafe(|| { let r = self.execute_tx_sub_plan_from_batch( @@ -265,6 +272,7 @@ impl CoreLoop { match result { Ok(resp) => { + irreversible |= applied_irreversibly(plan, undo_before, undo_log.len()); last_response = resp; } Err(error_code) => { @@ -283,7 +291,7 @@ impl CoreLoop { let rollback_error_code = match catch_unwind(AssertUnwindSafe(|| { self.rollback_undo_log(task.request.database_id.as_u64(), tid, undo_log) })) { - Ok(Ok(())) => error_code, + Ok(Ok(())) => batch_failure_code(error_code, irreversible), Ok(Err((entry_index, detail))) => { error!( core = self.core_id, @@ -337,7 +345,7 @@ impl CoreLoop { } } - Ok((last_response, undo_log, crdt_deltas)) + Ok((last_response, undo_log, crdt_deltas, irreversible)) } /// Apply all buffered CRDT deltas only after every sub-plan and the @@ -479,19 +487,6 @@ impl CoreLoop { } } -/// Best-effort conversion of a panic payload to a human-readable string. -/// Tries the two common payload types (`&'static str` and `String`); falls -/// back to `""` for anything else. -pub(super) fn panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String { - if let Some(s) = payload.downcast_ref::<&'static str>() { - (*s).to_string() - } else if let Some(s) = payload.downcast_ref::() { - s.clone() - } else { - "".to_string() - } -} - #[cfg(test)] mod tests { use super::*; diff --git a/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs b/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs index 5bf629900..64f580b98 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs @@ -9,6 +9,7 @@ use tracing::error; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; +use crate::data::panic_payload::panic_payload_to_string; use crate::types::TenantId; use super::batch::CrdtDelta; @@ -88,7 +89,7 @@ impl CoreLoop { undo_len, format!( "panic during transaction rollback: {}", - super::batch::panic_payload_to_string(payload.as_ref()) + panic_payload_to_string(payload.as_ref()) ), )), }; @@ -141,7 +142,7 @@ impl CoreLoop { ErrorCode::Internal { detail: format!( "panic while capturing CRDT transaction pre-images: {}", - super::batch::panic_payload_to_string(payload.as_ref()) + panic_payload_to_string(payload.as_ref()) ), }, ); @@ -159,7 +160,7 @@ impl CoreLoop { ErrorCode::Internal { detail: format!( "panic during CRDT transaction gate: {}", - super::batch::panic_payload_to_string(payload.as_ref()) + panic_payload_to_string(payload.as_ref()) ), }, ), @@ -239,7 +240,7 @@ impl CoreLoop { collection: "".into(), detail: format!( "panic while restoring CRDT pre-images: {}", - super::batch::panic_payload_to_string(payload.as_ref()) + panic_payload_to_string(payload.as_ref()) ), }], }), @@ -321,7 +322,7 @@ impl CoreLoop { collection, detail: format!( "panic while restoring CRDT pre-image: {}", - super::batch::panic_payload_to_string(payload.as_ref()) + panic_payload_to_string(payload.as_ref()) ), }), } diff --git a/nodedb/src/data/executor/handlers/transaction/batch_irreversible.rs b/nodedb/src/data/executor/handlers/transaction/batch_irreversible.rs new file mode 100644 index 000000000..823c8222f --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/batch_irreversible.rs @@ -0,0 +1,146 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sub-plans a transaction batch cannot roll back. +//! +//! A sub-plan with no engine-specific transaction handler runs through the +//! ordinary dispatch path. That path records no undo entry, so a later +//! rollback leaves its write in place. A failed batch that ran such a write +//! must not answer with a code that claims nothing applied: the Control +//! Plane cancels every record of a batch refused with such a code, and +//! recovery would then drop the write that stayed. + +use nodedb_physical::physical_plan::PhysicalPlan; + +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::control::server::shared::write_admission::plan_is_write; +use crate::data::executor::handlers::partial_refusal::refusal_after_partial_apply; + +/// Whether `plan` wrote state the undo log does not cover. +/// +/// A tracked write that changed state always adds an undo entry. A write +/// plan that added none either ran untracked or changed nothing. Both count +/// here, which errs toward replaying the batch's records. +pub(super) fn applied_irreversibly( + plan: &PhysicalPlan, + undo_before: usize, + undo_after: usize, +) -> bool { + undo_after == undo_before && plan_is_write(plan) +} + +/// The code a failed batch answers with once its rollback finished. +/// +/// After an irreversible sub-plan applied, a definite refusal becomes +/// `Internal`, which keeps the batch's records for replay. +pub(super) fn batch_failure_code(code: ErrorCode, irreversible: bool) -> ErrorCode { + if irreversible { + refusal_after_partial_apply(code) + } else { + code + } +} + +/// [`batch_failure_code`] applied to a failed batch's response. +pub(super) fn batch_failure_response(mut response: Response, irreversible: bool) -> Response { + if let Some(code) = response.error_code.take() { + response.error_code = Some(Box::new(batch_failure_code(*code, irreversible))); + } + response +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::server::dispatch_utils::write_definitely_not_applied; + + #[test] + fn a_definite_refusal_after_an_irreversible_write_keeps_the_records() { + let code = ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }; + let answered = batch_failure_code(code, true); + assert!(matches!(answered, ErrorCode::Internal { .. })); + assert!(!write_definitely_not_applied(&answered)); + } + + #[test] + fn a_definite_refusal_with_every_write_rolled_back_stays_definite() { + let code = ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }; + assert_eq!(batch_failure_code(code.clone(), false), code); + } + + /// A predicate delete runs untracked. A later sub-plan refuses the batch, + /// the rollback leaves the delete in place, and the answer must not claim + /// that nothing applied. + #[test] + fn a_batch_refused_after_an_untracked_delete_keeps_its_records() { + use std::collections::HashMap; + + use nodedb_physical::physical_plan::{CrdtOp, KvOp}; + use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate, Value}; + + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let task = make_default_task(); + let did = task.request.database_id.as_u64(); + let tid = task.request.tenant_id.as_u64(); + let collection = QualifiedCollection::new(DatabaseId::DEFAULT, "items"); + let value = zerompk::to_msgpack_vec(&Value::Object(HashMap::from([( + "v".to_string(), + Value::Integer(1), + )]))) + .expect("encode value"); + let seed = PhysicalPlan::Kv(KvOp::Put { + collection: collection.clone(), + key: b"k1".to_vec(), + value, + ttl_ms: 0, + surrogate: Surrogate::new(1), + returning: None, + rls_filters: Vec::new(), + }); + let seeded = core.execute_transaction_batch(&task, tid, &[seed], &[], None); + assert_eq!(seeded.status, Status::Ok, "seed put must apply"); + + let delete_all = PhysicalPlan::Kv(KvOp::PredicateDelete { + collection: collection.clone(), + filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }); + let refused = PhysicalPlan::Crdt(CrdtOp::Apply { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), + document_id: "doc".into(), + delta: vec![1], + peer_id: 1, + mutation_id: 1, + surrogate: Surrogate::ZERO, + provenance: None, + constraint_version_required: 0, + expected_frontier_digest: None, + }); + let response = + core.execute_transaction_batch(&task, tid, &[delete_all, refused], &[], None); + + assert_eq!(response.status, Status::Error); + let code = response.error_code.map(|code| *code); + assert!( + matches!(code, Some(ErrorCode::Internal { .. })), + "got {code:?}" + ); + let now = crate::engine::kv::current_ms(); + assert_eq!( + core.kv_engine.get(did, tid, "items", b"k1", now), + None, + "the untracked delete stays after the rollback" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/mod.rs b/nodedb/src/data/executor/handlers/transaction/mod.rs index b95f108f1..e6982cfc4 100644 --- a/nodedb/src/data/executor/handlers/transaction/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/mod.rs @@ -2,6 +2,7 @@ mod batch; mod batch_crdt; +mod batch_irreversible; pub(in crate::data::executor) mod index_write_values; pub mod overlay; mod overlay_gauge; diff --git a/nodedb/src/data/executor/handlers/update_from_join_write.rs b/nodedb/src/data/executor/handlers/update_from_join_write.rs index cd5f57afd..fb2e3af71 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_write.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_write.rs @@ -10,6 +10,9 @@ use nodedb_types::columnar::StrictSchema; use crate::bridge::envelope::{ErrorCode, Response, WriteSetEntry}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::write_hook; +use crate::data::executor::handlers::partial_refusal::{ + refusal_after_partial_apply, refusal_after_rows, +}; use crate::data::executor::handlers::point::update_reindex_vector::UpdateVectorReindex; use crate::data::executor::handlers::returning_doc; use crate::data::executor::handlers::transaction::stage_write::stored_row_identity; @@ -79,6 +82,34 @@ impl CoreLoop { Vec::new() }; + // A closed period refuses an edit to a row it holds, and an edit that + // assigns the period column into it. Both images of every row are + // judged before the first row commits: each row below commits on its + // own, so a lock judged there refuses after earlier rows landed. + if let Some(lock) = self + .doc_configs + .get(&config_key) + .and_then(|config| config.enforcement.period_lock.as_ref()) + { + for row in &rows { + for image in [&row.old_body, &row.body] { + if let Err(e) = + crate::data::executor::enforcement::period_lock::check_period_lock( + &self.sparse, + database_id, + tid, + target_collection, + image, + lock, + resolved_sum_targets, + ) + { + return Err(self.response_error(task, e)); + } + } + } + } + for row in rows { let ResolvedUpdateRow { key: storage_key, @@ -87,43 +118,13 @@ impl CoreLoop { doc, } = row; - // Period lock, both images — matching `execute_point_update`: a - // closed period must reject an edit to a row it already holds, - // and must reject an edit that assigns the period column into it. - if let Some(config) = self.doc_configs.get(&config_key) - && let Some(ref pl) = config.enforcement.period_lock - { - if let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( - &self.sparse, - database_id, - tid, - target_collection, - &old_body, - pl, - resolved_sum_targets, - ) { - return Err(self.response_error(task, e)); - } - if let Err(e) = crate::data::executor::enforcement::period_lock::check_period_lock( - &self.sparse, - database_id, - tid, - target_collection, - &updated_bytes, - pl, - resolved_sum_targets, - ) { - return Err(self.response_error(task, e)); - } - } - // The row's body and the materialized-sum delta it owes share ONE // transaction. `ResolvedUpdateRow` already carries BOTH images — // `old_body` as stored and `body` as the post-image — so the fold // re-reads nothing; the struct was built to carry them. let row_txn = match self.sparse.begin_write() { Ok(txn) => txn, - Err(e) => return Err(self.response_error(task, e)), + Err(e) => return Err(self.response_error(task, refusal_after_rows(affected, e))), }; let stored = self.sparse.put_in_txn( &row_txn, @@ -158,14 +159,19 @@ impl CoreLoop { Ok(outcome) => outcome.target_writes, // Dropping `row_txn` un-committed reverses the row and every // target it had already moved. - Err(e) => return Err(self.response_error(task, e)), + Err(e) => { + return Err(self.response_error(task, refusal_after_rows(affected, e))); + } }; if let Err(e) = row_txn.commit() { return Err(self.response_error( task, - ErrorCode::Internal { - detail: format!("update-from-join commit: {e}"), - }, + refusal_after_rows( + affected, + ErrorCode::Internal { + detail: format!("update-from-join commit: {e}"), + }, + ), )); } // One durable redo entry per moved target row, naming the TARGET @@ -224,7 +230,10 @@ impl CoreLoop { is_strict, has_vectors, }) { - return Err(self.response_error(task, e)); + // The row's body already committed. + return Err( + self.response_error(task, refusal_after_partial_apply(e.into())) + ); } write_set.push(WriteSetEntry { surrogate: storage_key.surrogate().as_u32(), @@ -253,3 +262,118 @@ impl CoreLoop { }) } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::doc_format; + use crate::engine::document::store::CollectionConfig; + use crate::types::{DatabaseId, TenantId}; + use nodedb_physical::physical_plan::PeriodLockConfig; + use nodedb_types::{StorageKey, Surrogate}; + + const TID: u64 = 1; + const COLLECTION: &str = "journal"; + + fn resolved_row(surrogate: u32, old: serde_json::Value) -> ResolvedUpdateRow { + let mut new = old.clone(); + if let Some(fields) = new.as_object_mut() { + fields.insert("note".into(), serde_json::json!("new")); + } + ResolvedUpdateRow { + key: StorageKey::for_surrogate(Surrogate(surrogate)), + body: doc_format::encode_to_msgpack(&new), + old_body: doc_format::encode_to_msgpack(&old), + doc: new, + } + } + + /// A closed period holds one resolved row. The refusal code claims + /// nothing applied, so no row can be rewritten, including the rows ahead + /// of it. + #[test] + fn a_period_lock_on_any_resolved_row_rewrites_no_row() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let task = make_default_task(); + let database_id = task.request.database_id.as_u64(); + let mut config = CollectionConfig::new(COLLECTION); + config.enforcement.period_lock = Some(PeriodLockConfig { + period_column: "fiscal_period".into(), + ref_table: "fiscal_periods".into(), + ref_pk: "period_key".into(), + status_column: "status".into(), + allowed_statuses: vec!["OPEN".into()], + }); + core.doc_configs.insert( + ( + DatabaseId::new(database_id), + TenantId::new(TID), + COLLECTION.to_string(), + ), + config, + ); + let old_rows = [ + serde_json::json!({"note": "old"}), + // No reference row resolves this period, so the lock refuses it. + serde_json::json!({"note": "old", "fiscal_period": "2026-01"}), + serde_json::json!({"note": "old"}), + ]; + for (surrogate, row) in (1u32..).zip(old_rows.iter()) { + core.sparse + .put( + database_id, + TID, + COLLECTION, + &StorageKey::for_surrogate(Surrogate(surrogate)), + &doc_format::encode_to_msgpack(row), + ) + .expect("seed row"); + } + let rows: Vec = (1u32..) + .zip(old_rows.iter()) + .map(|(surrogate, row)| resolved_row(surrogate, row.clone())) + .collect(); + + let outcome = core.write_resolved_update_from_join_rows( + &task, + WriteResolvedRowsCtx { + tid: TID, + target_collection: COLLECTION, + resolved_sum_targets: &[], + has_vectors: false, + strict_schema: None, + declared_primary_key: None, + want_returning: false, + }, + rows, + ); + + let Err(response) = outcome else { + panic!("a locked period must refuse the statement"); + }; + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::PeriodLocked { .. }) + ), + "got {:?}", + response.error_code + ); + for surrogate in 1u32..=3 { + let stored = core + .sparse + .get( + database_id, + TID, + COLLECTION, + &StorageKey::for_surrogate(Surrogate(surrogate)), + ) + .expect("read row") + .expect("row exists"); + let doc = doc_format::decode_document(&stored).expect("row decodes"); + assert_eq!(doc.get("note"), Some(&serde_json::json!("old"))); + } + } +} diff --git a/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs b/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs index 49cd0a72b..006c2f210 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs @@ -85,8 +85,13 @@ impl CoreLoop { "vector resolved direct write" ); let database_id = task.request.database_id.as_u64(); - if let Err(e) = self.check_vector_resolved_preconditions(database_id, tid, index, mutations) - { + if let Err(e) = self.check_vector_resolved_preconditions( + database_id, + tid, + index, + mutations, + rls_write_check, + ) { return self.response_error(task, e); } let touched = match self.apply_vector_resolved_mutations( @@ -109,18 +114,62 @@ impl CoreLoop { /// Refuse the whole write when any row moved past what the resolve read. /// A vector that no longer fits the index is the same constraint error - /// the live write reports. + /// the live write reports. Every refusal the apply pass can answer is + /// decided here, before the first mutation lands. fn check_vector_resolved_preconditions( &self, database_id: u64, tid: u64, index: VectorResolvedIndexSpec<'_>, mutations: &[VectorResolvedMutation], + rls_write_check: &RlsWriteCheck, ) -> Result<(), ErrorCode> { let index_key = CoreLoop::vector_index_key(database_id, tid, index.collection, index.field); + // An absent index takes the width of the first vector written into + // it, so every vector in the write must share one width. + let mut width: Option = None; for mutation in mutations { if let Some(vector) = mutation.stored_vector() { self.check_vector_direct_index(database_id, tid, &index.with_dim(vector.len()))?; + if let Some(&declared) = self.declared_dims.get(&index_key) + && declared != 0 + && declared != vector.len() + { + return Err(ErrorCode::RejectedConstraint { + detail: String::new(), + constraint: format!( + "dimension mismatch: index declares {declared}, got {}", + vector.len() + ), + }); + } + match width { + Some(first) if first != vector.len() => { + return Err(ErrorCode::RejectedConstraint { + detail: String::new(), + constraint: format!( + "vector dimension mismatch: the write's first vector has {first}, got {}", + vector.len() + ), + }); + } + Some(_) => {} + None => width = Some(vector.len()), + } + } + match mutation { + VectorResolvedMutation::Update { merged_payload, .. } => { + decode_resolved_payload( + merged_payload, + rls_write_check, + tid, + index.collection, + )?; + } + VectorResolvedMutation::Upsert { payload, .. } => { + decode_resolved_payload(payload, rls_write_check, tid, index.collection)?; + } + VectorResolvedMutation::Delete { .. } => {} } let surrogate = mutation.surrogate(); let bound = self.vector_direct_node(&index_key, surrogate).is_some(); @@ -630,4 +679,44 @@ mod tests { "a payload-only update keeps the HNSW node" ); } + + /// Two upserts into an absent index carry different widths. The refusal + /// code claims nothing applied, so the first upsert does not land either. + #[test] + fn a_width_mismatch_late_in_the_write_applies_no_mutation() { + let mut h = make_core(); + let mutations = vec![ + VectorResolvedMutation::Upsert { + surrogate: Surrogate::new(1), + pk_bytes: Vec::new(), + vector: vec![1.0, 0.0], + payload: payload("alice"), + old_payload: None, + }, + VectorResolvedMutation::Upsert { + surrogate: Surrogate::new(2), + pk_bytes: Vec::new(), + vector: vec![1.0, 0.0, 0.0], + payload: payload("alice"), + old_payload: None, + }, + ]; + let outcome = VectorResolveOutcome { + mutations, + response_payload: Vec::new(), + }; + + let resp = apply(&mut h, &outcome); + + assert_eq!(resp.status, Status::Error); + assert!( + matches!( + resp.error_code.as_deref(), + Some(ErrorCode::RejectedConstraint { .. }) + ), + "got {:?}", + resp.error_code + ); + assert_eq!(stored_owner(&h, Surrogate::new(1)), None); + } } diff --git a/nodedb/src/data/executor/handlers/vector_write.rs b/nodedb/src/data/executor/handlers/vector_write.rs index 4758ee5d7..1d4abaf71 100644 --- a/nodedb/src/data/executor/handlers/vector_write.rs +++ b/nodedb/src/data/executor/handlers/vector_write.rs @@ -27,24 +27,26 @@ impl CoreLoop { ) -> Response { debug!(core = self.core_id, %collection, dim, count = vectors.len(), "vector batch insert"); let database_id = task.request.database_id.as_u64(); + // Every vector is checked before any is inserted, so a dimension + // refusal applies nothing. + if let Some(bad) = vectors.iter().find(|vector| vector.len() != dim) { + return self.response_error( + task, + ErrorCode::RejectedConstraint { + detail: String::new(), + constraint: format!( + "dimension mismatch in batch: expected {dim}, got {}", + bad.len() + ), + }, + ); + } let index_key = CoreLoop::vector_index_key(database_id, tid, collection, ""); // A committed-redo install seals once the whole record landed. let defer_seal = self.recording_redo_undo(); match self.get_or_create_vector_index(database_id, tid, collection, dim, "") { Ok(collection_ref) => { for (i, vector) in vectors.iter().enumerate() { - if vector.len() != dim { - return self.response_error( - task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "dimension mismatch in batch: expected {dim}, got {}", - vector.len() - ), - }, - ); - } let s = surrogates.get(i).copied().unwrap_or(Surrogate::ZERO); collection_ref.insert_with_surrogate(vector.clone(), s); } @@ -257,6 +259,28 @@ mod tests { }) } + /// The funnel cancels the batch's record on a dimension refusal, so the + /// refusal must leave no vector of the batch behind. + #[test] + fn a_wrong_dimension_late_in_the_batch_inserts_no_vector() { + let mut h = make_core(); + let task = make_task_with_lsn(12); + let vectors = vec![vec![1.0, 2.0], vec![3.0, 4.0, 5.0]]; + let surrogates = vec![Surrogate::new(1), Surrogate::new(2)]; + + let response = + h.core + .execute_vector_batch_insert(&task, 1, "docs", &vectors, 2, &surrogates); + + assert_eq!(response.status, Status::Error); + let key = CoreLoop::vector_index_key(0, 1, "docs", ""); + assert_eq!( + h.core.vector_collections.get(&key).map_or(0, |c| c.len()), + 0, + "the vector before the bad one is not inserted" + ); + } + #[test] fn vector_delete_populates_collection_floor_only_not_vector_id_as_surrogate() { let mut h = make_core(); diff --git a/nodedb/src/data/mod.rs b/nodedb/src/data/mod.rs index 56c42cee9..db784bd07 100644 --- a/nodedb/src/data/mod.rs +++ b/nodedb/src/data/mod.rs @@ -4,6 +4,7 @@ pub mod core_health; pub mod eventfd; pub mod executor; pub mod io; +pub(crate) mod panic_payload; pub mod runtime; pub mod snapshot; #[cfg(target_os = "linux")] diff --git a/nodedb/src/data/panic_payload.rs b/nodedb/src/data/panic_payload.rs new file mode 100644 index 000000000..6b08a15b9 --- /dev/null +++ b/nodedb/src/data/panic_payload.rs @@ -0,0 +1,16 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Rendering a caught panic's payload for an error report. + +/// Best-effort conversion of a panic payload to a human-readable string. +/// Tries the two common payload types (`&'static str` and `String`); falls +/// back to `""` for anything else. +pub(crate) fn panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String { + if let Some(s) = payload.downcast_ref::<&'static str>() { + (*s).to_string() + } else if let Some(s) = payload.downcast_ref::() { + s.clone() + } else { + "".to_string() + } +} diff --git a/nodedb/src/data/runtime/event_loop.rs b/nodedb/src/data/runtime/event_loop.rs index c88338a49..08ba80cc5 100644 --- a/nodedb/src/data/runtime/event_loop.rs +++ b/nodedb/src/data/runtime/event_loop.rs @@ -100,7 +100,8 @@ pub(super) fn run_event_loop( } Err(panic_payload) => { // Extract panic message for logging. - let msg = panic_message(&panic_payload); + let msg = + crate::data::panic_payload::panic_payload_to_string(panic_payload.as_ref()); error!( core_id, panic_count = watchdog.consecutive_panics + 1, @@ -236,14 +237,3 @@ fn heartbeat_interval_with_jitter() -> std::time::Duration { let jitter_ms = (x % 201) as i64 - 100; std::time::Duration::from_millis((1000 + jitter_ms) as u64) } - -/// Extract a human-readable message from a panic payload. -fn panic_message(payload: &Box) -> String { - if let Some(s) = payload.downcast_ref::<&str>() { - (*s).to_string() - } else if let Some(s) = payload.downcast_ref::() { - s.clone() - } else { - "non-string panic payload".to_string() - } -} diff --git a/nodedb/src/diag/context/mod.rs b/nodedb/src/diag/context/mod.rs index 664fd0962..2bb6ac718 100644 --- a/nodedb/src/diag/context/mod.rs +++ b/nodedb/src/diag/context/mod.rs @@ -10,6 +10,7 @@ mod catalog; mod crdt; mod data_plane; mod ingest; +mod outcome_floor; mod quota; mod recovery; mod retention; @@ -27,6 +28,7 @@ pub(in crate::diag) use data_plane::{ }; pub(in crate::diag) use ingest::IlpAcceptedLinesDropped; pub use ingest::IlpFlushOutcome; +pub(in crate::diag) use outcome_floor::{WriteWindowHeld, WriteWindowLeaked}; pub use quota::{DATABASE_SCOPE, TENANT_SCOPE}; pub(in crate::diag) use quota::{ QuotaRowNotInstalled, QuotaRowWriteFailed, QuotaScopePurgeIncomplete, QuotaScopeReplayAborted, diff --git a/nodedb/src/diag/context/outcome_floor.rs b/nodedb/src/diag/context/outcome_floor.rs new file mode 100644 index 000000000..045524745 --- /dev/null +++ b/nodedb/src/diag/context/outcome_floor.rs @@ -0,0 +1,83 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Forensic payload for a write window that leaked from the outcome floor. + +use faultbox::DomainContext; +use faultbox::serde_json::{Value, json}; + +/// A write window was dropped before its write's outcome was final. +pub(in crate::diag) struct WriteWindowLeaked { + /// Ticket of the leaked window. + pub ticket: u64, + /// The window's horizon: the outcome floor stays below it. + pub horizon: u64, + /// How long the window had been open when it leaked. + pub open_for_ms: u64, +} + +impl DomainContext for WriteWindowLeaked { + fn domain_kind(&self) -> &'static str { + "nodedb.write_window_leaked" + } + + fn grouping_key(&self) -> String { + // One bug class: a mint site dropped its window on a path with no + // settle. Ticket and horizon are the occurrence. + "write_window_leaked".to_string() + } + + fn to_json(&self) -> Value { + json!({ + "ticket": self.ticket, + "horizon": self.horizon, + "open_for_ms": self.open_for_ms, + "why_fatal": "the outcome floor bounds every engine watermark and every WAL \ + truncation. A leaked window holds the floor below its horizon \ + until the process restarts, so no checkpoint on this node \ + advances past it and the WAL grows without bound", + "operator_action": "restart the node to release the floor: restart replay \ + reaches the record the window held. Then find the mint \ + site whose early-return path drops its window without \ + settling or holding it; the backtrace names it", + }) + } +} + +/// A write window was held: its record has no final outcome in this process. +pub(in crate::diag) struct WriteWindowHeld { + /// `file:line` of the hold call. + pub site: String, + /// Ticket of the held window. + pub ticket: u64, + /// The window's horizon: the outcome floor stays below it. + pub horizon: u64, + /// How long the window had been open when it was held. + pub open_for_ms: u64, +} + +impl DomainContext for WriteWindowHeld { + fn domain_kind(&self) -> &'static str { + "nodedb.write_window_held" + } + + fn grouping_key(&self) -> String { + // One report per hold site. Ticket and horizon are the occurrence. + format!("write_window_held:{}", self.site) + } + + fn to_json(&self) -> Value { + json!({ + "site": self.site, + "ticket": self.ticket, + "horizon": self.horizon, + "open_for_ms": self.open_for_ms, + "impact": "the outcome floor bounds every engine watermark and every WAL \ + truncation. A held window keeps the floor below its horizon \ + until the process restarts, so no checkpoint on this node \ + advances past it and the WAL grows until then", + "operator_action": "restart the node once convenient: restart replay reaches \ + the held record and gives it a final outcome. The site \ + names the path that could not decide the outcome", + }) + } +} diff --git a/nodedb/src/diag/mod.rs b/nodedb/src/diag/mod.rs index 2eeb445f2..54dd99a67 100644 --- a/nodedb/src/diag/mod.rs +++ b/nodedb/src/diag/mod.rs @@ -19,4 +19,5 @@ pub use recording::{ quota_scope_replay_aborted, replay_record_unapplied, retention_autowire_orphaned, scope_quota_not_installed, strict_row_undecodable, synonym_group_not_applied, vector_index_not_applied, wal_archival_failed_truncation_held, write_acked_without_durability, + write_window_held, write_window_leaked, }; diff --git a/nodedb/src/diag/recording/mod.rs b/nodedb/src/diag/recording/mod.rs index 5590b940e..73afb8d47 100644 --- a/nodedb/src/diag/recording/mod.rs +++ b/nodedb/src/diag/recording/mod.rs @@ -12,6 +12,7 @@ mod catalog; mod crdt; mod data_plane; mod ingest; +mod outcome_floor; mod quota; mod recovery; mod retention; @@ -28,6 +29,7 @@ pub use data_plane::{ data_plane_response_lost, data_plane_responses_lost, }; pub use ingest::{ilp_invalid_utf8_drop, ilp_line_read_drop}; +pub use outcome_floor::{write_window_held, write_window_leaked}; pub use quota::{ quota_row_invalid, quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, quota_scope_replay_aborted, scope_quota_not_installed, diff --git a/nodedb/src/diag/recording/outcome_floor.rs b/nodedb/src/diag/recording/outcome_floor.rs new file mode 100644 index 000000000..a23e329c7 --- /dev/null +++ b/nodedb/src/diag/recording/outcome_floor.rs @@ -0,0 +1,49 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capture site for a write window that leaked from the outcome floor. + +use std::time::Duration; + +use faultbox::{Capture, EventKind}; + +use crate::diag::context; + +/// Report a write window dropped without a settle or a hold. Called from the +/// window's drop, the one site that detects the leak. +pub fn write_window_leaked(ticket: u64, horizon: u64, open_for: Duration) { + let ctx = context::WriteWindowLeaked { + ticket, + horizon, + open_for_ms: u64::try_from(open_for.as_millis()).unwrap_or(u64::MAX), + }; + let _ = Capture::new( + EventKind::InvariantViolation, + "write window leaked: the outcome floor is held until restart", + ) + .domain(&ctx) + .with_backtrace() + .emit(); +} + +/// Report a write window held until restart. Called from the hold, the one +/// site that decides it. +pub fn write_window_held( + site: &std::panic::Location<'_>, + ticket: u64, + horizon: u64, + open_for: Duration, +) { + let ctx = context::WriteWindowHeld { + site: format!("{}:{}", site.file(), site.line()), + ticket, + horizon, + open_for_ms: u64::try_from(open_for.as_millis()).unwrap_or(u64::MAX), + }; + let _ = Capture::new( + EventKind::Error, + "write window held: the outcome floor stays below it until restart", + ) + .domain(&ctx) + .with_backtrace() + .emit(); +} diff --git a/nodedb/src/engine/timeseries/ilp_ingest.rs b/nodedb/src/engine/timeseries/ilp_ingest.rs index 106983d0d..94a06c028 100644 --- a/nodedb/src/engine/timeseries/ilp_ingest.rs +++ b/nodedb/src/engine/timeseries/ilp_ingest.rs @@ -13,7 +13,7 @@ use super::ilp::{FieldValue, IlpLine}; use nodedb_types::columnar::schema::{TS_SYSTEM, TS_VALID_FROM, TS_VALID_UNTIL}; use nodedb_types::timeseries::{IngestResult, SeriesCatalog, SeriesKey}; -pub use super::ilp_schema::{ensure_bitemporal_columns, evolve_schema, infer_schema}; +pub use super::ilp_schema::{ensure_bitemporal_columns, evolve_schema, infer_schema, new_columns}; /// Bitemporal stamps applied per-row on ingest. `system_ms` is always /// engine-assigned (client-supplied values are ignored); the valid-time diff --git a/nodedb/src/engine/timeseries/ilp_schema.rs b/nodedb/src/engine/timeseries/ilp_schema.rs index 94d025f1e..730a822ad 100644 --- a/nodedb/src/engine/timeseries/ilp_schema.rs +++ b/nodedb/src/engine/timeseries/ilp_schema.rs @@ -96,12 +96,16 @@ pub fn ensure_bitemporal_columns(schema: &mut ColumnarSchema) { /// Must be called BEFORE `ingest_batch` so the batch can map values to /// the expanded schema. pub fn evolve_schema(memtable: &mut ColumnarMemtable, lines: &[IlpLine<'_>]) { - let existing: std::collections::HashSet = memtable - .schema() - .columns - .iter() - .map(|(n, _)| n.clone()) - .collect(); + for (name, col_type) in new_columns(memtable.schema(), lines) { + memtable.add_column(name, col_type); + } +} + +/// The columns `evolve_schema` adds to `schema` for `lines`, in the order +/// it adds them. The first line naming a key decides its type. +pub fn new_columns(schema: &ColumnarSchema, lines: &[IlpLine<'_>]) -> Vec<(String, ColumnType)> { + let existing: std::collections::HashSet<&str> = + schema.columns.iter().map(|(n, _)| n.as_str()).collect(); let mut new_columns: Vec<(String, ColumnType)> = Vec::new(); let mut seen: std::collections::HashSet = std::collections::HashSet::new(); @@ -124,8 +128,5 @@ pub fn evolve_schema(memtable: &mut ColumnarMemtable, lines: &[IlpLine<'_>]) { } } } - - for (name, col_type) in new_columns { - memtable.add_column(name, col_type); - } + new_columns } diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index c16acb085..9efae4486 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -23,7 +23,9 @@ use crate::bridge::envelope::{CounterFault, ErrorCode}; /// one. pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { match code { - ErrorCode::DeadlineExceeded => NodeDbError::deadline_exceeded(), + ErrorCode::DeadlineExceeded | ErrorCode::ExpiredBeforeExecution => { + NodeDbError::deadline_exceeded() + } // The Data Plane's `RejectedConstraint` carries no collection name, // only the constraint kind and detail — leave collection blank // rather than misreport the kind string as the collection. diff --git a/nodedb/src/wal/manager/appender.rs b/nodedb/src/wal/manager/appender.rs index 95259b5de..4e46790fc 100644 --- a/nodedb/src/wal/manager/appender.rs +++ b/nodedb/src/wal/manager/appender.rs @@ -9,6 +9,8 @@ //! takes an appender, so no append can read a key another caller set and no //! caller has a key to clear. +use std::sync::Mutex; + use nodedb_wal::RecordTarget; use nodedb_wal::record::RecordType; @@ -23,6 +25,8 @@ pub const NO_APPLY_KEY: u64 = 0; pub struct WalAppender<'a> { wal: &'a WalManager, apply_key: u64, + /// Collects the LSN of every record this appender writes, when set. + sink: Option<&'a Mutex>>, } impl WalManager { @@ -33,6 +37,21 @@ impl WalManager { WalAppender { wal: self, apply_key, + sink: None, + } + } + + /// An appender that also pushes the LSN of every record it writes onto + /// `sink`, in append order. + pub fn recording_appender<'a>( + &'a self, + apply_key: u64, + sink: &'a Mutex>, + ) -> WalAppender<'a> { + WalAppender { + wal: self, + apply_key, + sink: Some(sink), } } } @@ -65,7 +84,12 @@ impl WalAppender<'_> { self.apply_key, ) .map_err(crate::Error::Wal)?; - Ok(Lsn::new(lsn)) + drop(wal); + let lsn = Lsn::new(lsn); + if let Some(sink) = self.sink { + sink.lock().unwrap_or_else(|p| p.into_inner()).push(lsn); + } + Ok(lsn) } } @@ -97,4 +121,22 @@ mod tests { .collect(); assert_eq!(keys, vec![0xAB, NO_APPLY_KEY, 0xAB]); } + + #[test] + fn a_recording_appender_collects_every_lsn_it_writes() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); + let (t, v, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + let sink = Mutex::new(Vec::new()); + + let recording = wal.recording_appender(NO_APPLY_KEY, &sink); + let first = recording.append_put(t, v, db, b"a").expect("append"); + wal.appender(NO_APPLY_KEY) + .append_put(t, v, db, b"unrecorded") + .expect("append"); + let second = recording.append_put(t, v, db, b"b").expect("append"); + + let recorded = sink.lock().expect("sink").clone(); + assert_eq!(recorded, vec![first, second]); + } } diff --git a/nodedb/tests/inproc/cases/request_tracker_backpressure.rs b/nodedb/tests/inproc/cases/request_tracker_backpressure.rs index ffa33a352..706e8dca0 100644 --- a/nodedb/tests/inproc/cases/request_tracker_backpressure.rs +++ b/nodedb/tests/inproc/cases/request_tracker_backpressure.rs @@ -13,9 +13,8 @@ //! buffer is full rather than silently expanding forever. use nodedb::bridge::envelope::{Payload, Response, Status}; -use nodedb::control::request_tracker::RequestTracker; +use nodedb::control::request_tracker::{RequestTracker, ResponseReceiver}; use nodedb::types::{Lsn, RequestId}; -use tokio::sync::mpsc; fn partial(id: u64, data: &[u8]) -> Response { Response { @@ -33,12 +32,11 @@ fn partial(id: u64, data: &[u8]) -> Response { } #[test] -fn register_returns_bounded_receiver() { - // Compile-gate: the mpsc type must be the bounded `Receiver`, not - // `UnboundedReceiver`. This is the single largest guarantee — bounded - // type is what forces backpressure through the rest of the pipeline. +fn register_returns_a_response_receiver() { + // Compile-gate: partials reach the session through the receiver's + // bounded channel. The next test shows the bound. let tracker = RequestTracker::new(); - let _rx: mpsc::Receiver = tracker.register(RequestId::new(1)); + let _rx: ResponseReceiver = tracker.register(RequestId::new(1)); } #[test] diff --git a/nodedb/tests/inproc/cases/snapshot_round_trip.rs b/nodedb/tests/inproc/cases/snapshot_round_trip.rs index fc92721a4..26f0847c8 100644 --- a/nodedb/tests/inproc/cases/snapshot_round_trip.rs +++ b/nodedb/tests/inproc/cases/snapshot_round_trip.rs @@ -66,7 +66,7 @@ async fn snapshot_round_trip_builder_to_applier() { "INSERT INTO {COLL} (id, val) VALUES ('{pk}', 'v_{pk}')" )) .await - .unwrap_or_else(|e| panic!("INSERT {pk} on source: {e}")); + .unwrap_or_else(|e| panic!("INSERT {pk} on source: {e:?}")); } } From 9bee5549a1f365c1dd77539888f6d862207c31b1 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 16:02:34 +0800 Subject: [PATCH 21/64] test(pgwire-harness): install the production gateway and a single-voter read gate The harness built `SharedState` without installing the gateway or a Raft read gate, so remote-routed dispatch and linearizable reads behaved differently under test than in production. Every server-start path now calls the same `install_gateway` production boot runs, and a routed harness installs a `SingleVoterReadGate` that answers leadership and read index the way a group with one voter does, since the harness runs no Raft loop. A routed harness also derives its `node_id` from the routing table's sole leader instead of leaving it unset. --- nodedb-test-support/Cargo.toml | 1 + nodedb-test-support/src/pgwire_harness/mod.rs | 1 + .../src/pgwire_harness/multicore.rs | 3 + .../src/pgwire_harness/read_gate.rs | 73 +++++++++++++++++++ .../src/pgwire_harness/restart.rs | 3 + .../src/pgwire_harness/start.rs | 14 +++- .../src/pgwire_harness/support.rs | 18 +++++ 7 files changed, 112 insertions(+), 1 deletion(-) create mode 100644 nodedb-test-support/src/pgwire_harness/read_gate.rs diff --git a/nodedb-test-support/Cargo.toml b/nodedb-test-support/Cargo.toml index 716aec860..c93ed50f0 100644 --- a/nodedb-test-support/Cargo.toml +++ b/nodedb-test-support/Cargo.toml @@ -24,6 +24,7 @@ nodedb-types = { workspace = true } nodedb-wal = { workspace = true } # Runtime + wire deps used by the harness itself. +async-trait = { workspace = true } tokio = { workspace = true, features = ["test-util"] } tokio-postgres = { workspace = true } tokio-tungstenite = { workspace = true } diff --git a/nodedb-test-support/src/pgwire_harness/mod.rs b/nodedb-test-support/src/pgwire_harness/mod.rs index f38c81cd9..47aa90af5 100644 --- a/nodedb-test-support/src/pgwire_harness/mod.rs +++ b/nodedb-test-support/src/pgwire_harness/mod.rs @@ -8,6 +8,7 @@ mod multicore; mod query; pub mod raw_pgwire; +mod read_gate; mod restart; mod start; mod support; diff --git a/nodedb-test-support/src/pgwire_harness/multicore.rs b/nodedb-test-support/src/pgwire_harness/multicore.rs index 94421b3f8..0f71f6873 100644 --- a/nodedb-test-support/src/pgwire_harness/multicore.rs +++ b/nodedb-test-support/src/pgwire_harness/multicore.rs @@ -51,6 +51,9 @@ impl TestServer { s.governor = init_test_memory_governor(); } let shared = shared; + // The same gateway install production boot runs, after every + // `Arc::get_mut` above. + nodedb::bootstrap::state_wiring::install_gateway(&shared); let mut core_stop_txs = Vec::new(); let mut core_handles = Vec::new(); diff --git a/nodedb-test-support/src/pgwire_harness/read_gate.rs b/nodedb-test-support/src/pgwire_harness/read_gate.rs new file mode 100644 index 000000000..e501b7646 --- /dev/null +++ b/nodedb-test-support/src/pgwire_harness/read_gate.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The Raft read gate for a harness that runs as a single-node cluster. +//! +//! Production `start_raft` publishes a gate backed by the Raft loop. A +//! harness started with a routing table runs no Raft loop. Without a gate, +//! every linearizable read refuses with "no leader is currently serving +//! this range". This gate answers both questions the way a group with one +//! voter answers them. + +use std::sync::{Arc, RwLock}; +use std::time::Duration; + +use nodedb::control::cluster::{RaftReadGate, ReadIndexRefusal}; +use nodedb::control::state::SharedState; +use nodedb_cluster::RoutingTable; + +/// Read gate for groups whose only voter is this node. +struct SingleVoterReadGate { + node_id: u64, + routing: Arc>, +} + +impl SingleVoterReadGate { + /// Whether this node is the sole voter and the leader of `group_id`. + fn is_sole_leader(&self, group_id: u64) -> bool { + let routing = self.routing.read().unwrap_or_else(|p| p.into_inner()); + routing + .group_info(group_id) + .is_some_and(|info| info.leader == self.node_id && info.members == [self.node_id]) + } +} + +#[async_trait::async_trait] +impl RaftReadGate for SingleVoterReadGate { + /// A sole voter is its own quorum, so it confirms leadership at once. + /// + /// The harness keeps no Raft log, so the read index is `0`. The caller + /// serves the read from local state. + async fn confirm_leader( + &self, + group_id: u64, + _timeout: Duration, + ) -> Result { + if self.is_sole_leader(group_id) { + Ok(0) + } else { + Err(ReadIndexRefusal::NotLeader) + } + } + + /// A sole voter holds the only copy, so it is never behind. + fn within_staleness_bound(&self, group_id: u64, _max_staleness: Duration) -> bool { + self.is_sole_leader(group_id) + } +} + +/// Publish the single-voter read gate when `shared` carries a routing table. +/// +/// `raft_read_gate` is a `OnceLock`, so this runs once, after every +/// `Arc::get_mut` install. +pub(super) fn install_single_voter_read_gate(shared: &SharedState) { + let Some(routing) = shared.cluster_routing.as_ref() else { + return; + }; + let gate: Arc = Arc::new(SingleVoterReadGate { + node_id: shared.node_id, + routing: Arc::clone(routing), + }); + if shared.raft_read_gate.set(gate).is_err() { + panic!("harness raft_read_gate installed twice"); + } +} diff --git a/nodedb-test-support/src/pgwire_harness/restart.rs b/nodedb-test-support/src/pgwire_harness/restart.rs index cd93c0d26..25cfc6de6 100644 --- a/nodedb-test-support/src/pgwire_harness/restart.rs +++ b/nodedb-test-support/src/pgwire_harness/restart.rs @@ -200,6 +200,9 @@ impl TestServer { s.governor = init_test_memory_governor(); } let shared = shared; + // The same gateway install production boot runs, after every + // `Arc::get_mut` above. + nodedb::bootstrap::state_wiring::install_gateway(&shared); nodedb::bootstrap::credentials::replay_surrogate_wal( &shared, &wal_records, diff --git a/nodedb-test-support/src/pgwire_harness/start.rs b/nodedb-test-support/src/pgwire_harness/start.rs index 4dae6b47e..782b8a4a0 100644 --- a/nodedb-test-support/src/pgwire_harness/start.rs +++ b/nodedb-test-support/src/pgwire_harness/start.rs @@ -13,7 +13,10 @@ use nodedb::control::state::SharedState; use nodedb::event::{EventPlane, EventPlaneConfig, create_event_bus}; use nodedb::wal::WalManager; -use super::support::{bind_http_listener, bind_native_listener, init_test_memory_governor}; +use super::read_gate::install_single_voter_read_gate; +use super::support::{ + bind_http_listener, bind_native_listener, init_test_memory_governor, single_routing_leader, +}; use super::types::{TestClient, TestDataDir, TestServer}; /// Knobs for spawning a `TestServer`. `Default` reproduces the historical @@ -228,6 +231,10 @@ impl TestServer { s.backup_kek = Some(Arc::new([0x42u8; 32])); s.governor = init_test_memory_governor(); if let Some(routing) = cfg.routing { + // Production takes `node_id` from the cluster handle that owns + // the routing table. A single-node server runs as the one node + // that leads every group in it, so the gateway routes locally. + s.node_id = single_routing_leader(&routing); s.cluster_routing = Some(std::sync::Arc::new(std::sync::RwLock::new(routing))); } s.jwks_registry = cfg.jwks_registry; @@ -240,6 +247,11 @@ impl TestServer { ); } let shared = shared; + // The same gateway install production boot runs, after every + // `Arc::get_mut` above. + nodedb::bootstrap::state_wiring::install_gateway(&shared); + // Production `start_raft` publishes the read gate for a routed node. + install_single_voter_read_gate(&shared); // Data Plane core. Share the SharedState's array_catalog so DDL // mutations made by the SQL converter are visible to the handler diff --git a/nodedb-test-support/src/pgwire_harness/support.rs b/nodedb-test-support/src/pgwire_harness/support.rs index f150057f5..13f7de794 100644 --- a/nodedb-test-support/src/pgwire_harness/support.rs +++ b/nodedb-test-support/src/pgwire_harness/support.rs @@ -27,6 +27,24 @@ pub(super) fn init_test_memory_governor() -> Arc { nodedb::memory::init_governor(ceiling, &budgets).expect("harness governor config is valid") } +/// The one node that leads every group in a single-node routing table. +/// +/// Panics when the table names no leader, or more than one: a single-node +/// harness cannot serve a group another node leads. +pub(super) fn single_routing_leader(routing: &nodedb_cluster::RoutingTable) -> u64 { + let mut leaders: Vec = routing + .group_members() + .values() + .map(|group| group.leader) + .collect(); + leaders.sort_unstable(); + leaders.dedup(); + match leaders.as_slice() { + [leader] if *leader != 0 => *leader, + other => panic!("single-node harness routing must name one leader, got {other:?}"), + } +} + /// Bind a native (MessagePack) protocol listener on `127.0.0.1:0` and /// spawn its accept loop. Returns the listener's local port plus the /// handle to await on shutdown. From ce322d03e0844044c66afaca6590dda048a78a3f Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 17:38:52 +0800 Subject: [PATCH 22/64] feat(crdt): persist dead-letter entries across restart MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bind each rejected delta's dead-letter entry to the log position of the record that produced it, and store it in the sparse engine's redb table. A tenant's CRDT engine restores its entries from storage when created, so restart replay — which never reaches a rejected record — no longer loses them. Re-applying the same record binds to the same position, keeping one entry per record instead of accumulating duplicates. Route every rejection path (snapshot import, sync apply, local apply, transaction batch, WAL replay) through the same store-then-respond sequence, and report a store failure as part of the refusal instead of silently dropping it. Purge stored entries when a collection is dropped or a tenant is purged, reusing the bounded reclaim retry now extracted into its own module. --- nodedb-crdt/src/dead_letter.rs | 128 ++++++++- nodedb-crdt/src/lib.rs | 2 +- .../executor/core_loop/crdt_dead_letters.rs | 94 ++++++ nodedb/src/data/executor/core_loop/mod.rs | 2 + .../src/data/executor/core_loop/response.rs | 31 +- .../data/executor/handlers/control/crdt.rs | 22 +- .../handlers/control/crdt_apply/gated.rs | 14 + .../handlers/control/crdt_apply/local.rs | 171 ++++++++++- nodedb/src/data/executor/handlers/mod.rs | 1 + nodedb/src/data/executor/handlers/purge.rs | 12 + .../data/executor/handlers/reclaim_retry.rs | 61 ++++ .../executor/handlers/transaction/batch.rs | 7 +- .../handlers/transaction/batch_crdt.rs | 38 ++- .../handlers/unregister_collection.rs | 63 +--- nodedb/src/data/executor/wal_replay/crdt.rs | 91 ++++++ .../crdt/tenant_state/apply_validated.rs | 23 +- nodedb/src/engine/crdt/tenant_state/core.rs | 5 + .../engine/crdt/tenant_state/dead_letters.rs | 48 ++++ nodedb/src/engine/crdt/tenant_state/mod.rs | 1 + .../engine/sparse/btree/crdt_dead_letter.rs | 269 ++++++++++++++++++ nodedb/src/engine/sparse/btree/engine.rs | 5 + nodedb/src/engine/sparse/btree/mod.rs | 1 + 22 files changed, 986 insertions(+), 103 deletions(-) create mode 100644 nodedb/src/data/executor/core_loop/crdt_dead_letters.rs create mode 100644 nodedb/src/data/executor/handlers/reclaim_retry.rs create mode 100644 nodedb/src/engine/crdt/tenant_state/dead_letters.rs create mode 100644 nodedb/src/engine/sparse/btree/crdt_dead_letter.rs diff --git a/nodedb-crdt/src/dead_letter.rs b/nodedb-crdt/src/dead_letter.rs index 7d29f55ab..ce188d613 100644 --- a/nodedb-crdt/src/dead_letter.rs +++ b/nodedb-crdt/src/dead_letter.rs @@ -25,7 +25,16 @@ use crate::constraint::Constraint; use crate::error::{CrdtError, Result}; /// Suggested action the application should take to resolve a constraint violation. -#[derive(Debug, Clone, Serialize, Deserialize)] +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] pub enum CompensationHint { /// Retry with a different value for the conflicting field. /// Example: UNIQUE violation — suggest appending a suffix. @@ -58,7 +67,16 @@ pub enum CompensationHint { } /// A rejected delta with metadata for debugging and recovery. -#[derive(Debug, Clone, Serialize, Deserialize)] +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] pub struct DeadLetter { /// Unique ID for this dead letter entry. pub id: u64, @@ -103,6 +121,12 @@ pub struct DeadLetter { /// Number of times this delta has been retried. pub retry_count: u32, + + /// The log position of the record whose apply rejected this delta, once + /// the applier binds it. One record yields at most one entry, so a + /// re-applied record never adds a second. + #[serde(default)] + pub source_lsn: Option, } /// Parameters for [`DeadLetterQueue::enqueue`]. @@ -175,11 +199,60 @@ impl DeadLetterQueue { hint, rejected_at: now, retry_count: 0, + source_lsn: None, }); Ok(id) } + /// Bind entry `id` to the record at `source_lsn` that produced it. + /// + /// Returns the bound entry. Returns `None` when another entry already + /// names that record: entry `id` is a repeat of it and is removed. Also + /// `None` when no entry `id` exists. + pub fn bind_source(&mut self, id: u64, source_lsn: u64) -> Option<&DeadLetter> { + let repeat = self + .entries + .iter() + .any(|dl| dl.id != id && dl.source_lsn == Some(source_lsn)); + if repeat { + self.remove(id); + return None; + } + let entry = self.entries.iter_mut().find(|dl| dl.id == id)?; + entry.source_lsn = Some(source_lsn); + Some(&*entry) + } + + /// Put back an entry read from durable storage. + /// + /// An entry whose source record an existing entry already names is + /// skipped. Fresh ids stay above every restored id. + pub fn restore(&mut self, entry: DeadLetter) -> Result<()> { + let known = entry.source_lsn.is_some() + && self + .entries + .iter() + .any(|dl| dl.source_lsn == entry.source_lsn); + if known { + return Ok(()); + } + if self.entries.len() >= self.capacity { + return Err(CrdtError::DlqFull { + capacity: self.capacity, + pending: self.entries.len(), + }); + } + self.next_id = self.next_id.max(entry.id.saturating_add(1)); + self.entries.push_back(entry); + Ok(()) + } + + /// Every pending entry, oldest first. + pub fn iter(&self) -> impl Iterator { + self.entries.iter() + } + /// Peek at the oldest dead letter without removing it. pub fn peek(&self) -> Option<&DeadLetter> { self.entries.front() @@ -490,4 +563,55 @@ mod tests { assert_eq!(removed.reason, "a"); assert_eq!(dlq.len(), 1); } + + fn enqueue_one(dlq: &mut DeadLetterQueue, peer_id: u64) -> u64 { + dlq.enqueue(EnqueueDeadLetterArgs { + peer_id, + user_id: 0, + tenant_id: 0, + delta: b"delta".to_vec(), + constraint: &test_constraint(), + reason: "duplicate".into(), + hint: CompensationHint::ManualIntervention { + reason: "duplicate".into(), + }, + }) + .expect("enqueue") + } + + #[test] + fn a_record_bound_twice_keeps_one_entry() { + let mut dlq = DeadLetterQueue::new(10); + let first = enqueue_one(&mut dlq, 1); + assert!(dlq.bind_source(first, 7).is_some()); + let repeat = enqueue_one(&mut dlq, 1); + assert!(dlq.bind_source(repeat, 7).is_none()); + assert_eq!(dlq.len(), 1); + assert_eq!(dlq.peek().map(|dl| dl.source_lsn), Some(Some(7))); + } + + #[test] + fn a_restored_entry_is_kept_once_and_fresh_ids_stay_above_it() { + let mut source = DeadLetterQueue::new(10); + let id = enqueue_one(&mut source, 1); + let stored = source.bind_source(id, 9).cloned().expect("bound"); + + let mut restored = DeadLetterQueue::new(10); + restored.restore(stored.clone()).expect("restore"); + restored.restore(stored.clone()).expect("restore again"); + assert_eq!(restored.len(), 1); + assert_eq!(restored.peek(), Some(&stored)); + let fresh = enqueue_one(&mut restored, 2); + assert!(fresh > stored.id); + } + + #[test] + fn a_dead_letter_round_trips_through_messagepack() { + let mut dlq = DeadLetterQueue::new(10); + let id = enqueue_one(&mut dlq, 3); + let entry = dlq.bind_source(id, 11).cloned().expect("bound"); + let bytes = zerompk::to_msgpack_vec(&entry).expect("encode"); + let decoded: DeadLetter = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, entry); + } } diff --git a/nodedb-crdt/src/lib.rs b/nodedb-crdt/src/lib.rs index 643fd107f..d38961cb7 100644 --- a/nodedb-crdt/src/lib.rs +++ b/nodedb-crdt/src/lib.rs @@ -38,7 +38,7 @@ pub mod validator; pub use auth::CrdtAuthContext; pub use constraint::{Constraint, ConstraintKind, ConstraintSet}; -pub use dead_letter::{CompensationHint, DeadLetterQueue, EnqueueDeadLetterArgs}; +pub use dead_letter::{CompensationHint, DeadLetter, DeadLetterQueue, EnqueueDeadLetterArgs}; pub use deferred::DeferredQueue; pub use error::{CrdtError, Result}; pub use policy::{ diff --git a/nodedb/src/data/executor/core_loop/crdt_dead_letters.rs b/nodedb/src/data/executor/core_loop/crdt_dead_letters.rs new file mode 100644 index 000000000..303e6d1ac --- /dev/null +++ b/nodedb/src/data/executor/core_loop/crdt_dead_letters.rs @@ -0,0 +1,94 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Storing the dead-letter entry a rejected CRDT delta leaves, and the +//! refusal the writer receives. + +use nodedb_types::DatabaseId; +use nodedb_types::sync::violation::ViolationType; + +use crate::bridge::envelope::ErrorCode; +use crate::types::{Lsn, TenantId}; + +use super::CoreLoop; + +impl CoreLoop { + /// Bind the entry the latest validated apply on `(database_id, + /// tenant_id)` enqueued to the record at `source_lsn`, and store it. + /// + /// Call it after every validated apply that returned `Rejected`. A + /// record already bound keeps its one entry. A write with no record has + /// nothing to replay, so its entry stays in memory only. + /// + /// A store error is returned: the entry is then in memory only, and the + /// caller must not report the rejection as durable. + pub(in crate::data::executor) fn store_crdt_dead_letter( + &mut self, + database_id: DatabaseId, + tenant_id: TenantId, + source_lsn: Option, + ) -> crate::Result<()> { + let Some(source_lsn) = source_lsn.filter(|lsn| *lsn != Lsn::ZERO) else { + return Ok(()); + }; + let Some(engine) = self.crdt_engines.get_mut(&(database_id, tenant_id)) else { + return Ok(()); + }; + let Some(entry) = engine.bind_dead_letter_source(source_lsn.as_u64()) else { + return Ok(()); + }; + self.sparse.put_crdt_dead_letter( + database_id.as_u64(), + tenant_id.as_u64(), + source_lsn.as_u64(), + &entry, + ) + } +} + +impl CoreLoop { + /// Store the entry the rejection of the replayed record at `lsn` + /// produced. The live apply of the record stored the same entry, so this + /// keeps one. A store error is logged, and the record keeps the entry in + /// memory. + pub(in crate::data::executor) fn store_replayed_dead_letter( + &mut self, + database_id: DatabaseId, + tenant_id: TenantId, + lsn: u64, + ) { + if let Err(error) = self.store_crdt_dead_letter(database_id, tenant_id, Some(Lsn::new(lsn))) + { + tracing::error!( + core = self.core_id, + %database_id, + %tenant_id, + lsn, + %error, + "a replayed CRDT rejection's dead-letter entry could not be stored" + ); + } + } +} + +/// The refusal for a CRDT delta rejected by constraint `violation`. +/// +/// Nothing applied, so the write's record is cancelled. The rejected delta +/// stays in the dead-letter queue. +pub(in crate::data::executor) fn crdt_rejection( + collection: &str, + target: &str, + violation: &ViolationType, +) -> ErrorCode { + let text = violation.to_string(); + let constraint = text + .split_once(':') + .map_or(text.as_str(), |(kind, _)| kind) + .to_string(); + ErrorCode::RejectedConstraint { + constraint, + detail: format!( + "delta for {collection}/{target} violates {violation}; nothing was applied, \ + and the delta is in the dead-letter queue" + ), + } +} diff --git a/nodedb/src/data/executor/core_loop/mod.rs b/nodedb/src/data/executor/core_loop/mod.rs index 7e7bba79c..872e75ad6 100644 --- a/nodedb/src/data/executor/core_loop/mod.rs +++ b/nodedb/src/data/executor/core_loop/mod.rs @@ -5,6 +5,7 @@ mod bitemporal_time; pub(in crate::data::executor) mod checkpoint_floors; mod columnar_schema_seed; pub(in crate::data::executor) mod commit_pending; +mod crdt_dead_letters; mod decode_stored; pub(in crate::data::executor) mod deferred; mod doc_config_seed; @@ -27,6 +28,7 @@ mod vector_index_rebuild; mod vector_index_seed; pub(in crate::data::executor) mod write_index; +pub(in crate::data::executor) use crdt_dead_letters::crdt_rejection; pub use doc_config_seed::DocConfigSeedEntry; pub(in crate::data::executor) use segment_keks::SegmentKeks; pub use state::CoreLoop; diff --git a/nodedb/src/data/executor/core_loop/response.rs b/nodedb/src/data/executor/core_loop/response.rs index a73053ba7..b2ce5b24f 100644 --- a/nodedb/src/data/executor/core_loop/response.rs +++ b/nodedb/src/data/executor/core_loop/response.rs @@ -205,19 +205,26 @@ impl CoreLoop { database_id: DatabaseId, tenant_id: TenantId, ) -> crate::Result<&mut TenantCrdtEngine> { - let key = (database_id, tenant_id); - if !self.crdt_engines.contains_key(&key) { - tracing::debug!( - core = self.core_id, - %database_id, - %tenant_id, - "creating CRDT engine for database tenant" - ); - let engine = - TenantCrdtEngine::new(tenant_id, self.core_id as u64, ConstraintSet::new())?; - self.crdt_engines.insert(key, engine); + match self.crdt_engines.entry((database_id, tenant_id)) { + std::collections::hash_map::Entry::Occupied(engine) => Ok(engine.into_mut()), + std::collections::hash_map::Entry::Vacant(slot) => { + tracing::debug!( + core = self.core_id, + %database_id, + %tenant_id, + "creating CRDT engine for database tenant" + ); + let mut engine = + TenantCrdtEngine::new(tenant_id, self.core_id as u64, ConstraintSet::new())?; + // Rejected deltas leave no replayable record, so their + // dead-letter entries come back from storage. + engine.restore_dead_letters( + self.sparse + .load_crdt_dead_letters(database_id.as_u64(), tenant_id.as_u64())?, + ); + Ok(slot.insert(engine)) + } } - Ok(self.crdt_engines.get_mut(&key).expect("just inserted")) } /// Release the per-collection validation candidates every CRDT engine on diff --git a/nodedb/src/data/executor/handlers/control/crdt.rs b/nodedb/src/data/executor/handlers/control/crdt.rs index be3cf3fb9..673a58cb8 100644 --- a/nodedb/src/data/executor/handlers/control/crdt.rs +++ b/nodedb/src/data/executor/handlers/control/crdt.rs @@ -244,12 +244,24 @@ impl CoreLoop { } crate::engine::crdt::tenant_state::ValidatedApplyOutcome::Rejected(reason) => { warn!(core = self.core_id, %reason, "crdt snapshot rejected by constraints"); - self.response_error( - task, - ErrorCode::Internal { - detail: format!("CRDT snapshot violates constraints: {reason}"), + // Nothing applied, so the record is cancelled and replay never + // reaches this rejection. Its dead-letter entry is stored first. + let code = match self.store_crdt_dead_letter( + task.request.database_id, + tid, + task.wal_lsn(), + ) { + Ok(()) => crate::data::executor::core_loop::crdt_rejection( + collection, "snapshot", &reason, + ), + Err(error) => ErrorCode::Internal { + detail: format!( + "CRDT snapshot for {collection} violates {reason}, and its \ + dead-letter entry could not be stored: {error}" + ), }, - ) + }; + self.response_error(task, code) } crate::engine::crdt::tenant_state::ValidatedApplyOutcome::Malformed => { warn!(core = self.core_id, "crdt snapshot import was malformed"); diff --git a/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs b/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs index 0ea52f780..77582cfd1 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs @@ -282,6 +282,20 @@ impl CoreLoop { GateOutcome::Applied(ValidatedApplyOutcome::Rejected(vt)) => { imported_authoritative = true; self.checkpoint_coordinator.mark_dirty("crdt", 1); + // Replaying this record binds to the same log position, so + // the stored entry stays the only one. + if let Err(error) = + self.store_crdt_dead_letter(task.request.database_id, tenant_id, task.wal_lsn()) + { + warn!( + core = self.core_id, + %collection, + %document_id, + %error, + "crdt sync apply rejected a delta, and its dead-letter entry could not \ + be stored; replay of its record rebuilds it" + ); + } GateDisposition::Terminal(vt) } GateOutcome::Applied(ValidatedApplyOutcome::Malformed) => { diff --git a/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs b/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs index d05c69252..1e615c6e8 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs @@ -13,8 +13,10 @@ use nodedb_types::Surrogate; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::crdt_rejection; use crate::data::executor::task::ExecutionTask; use crate::engine::crdt::tenant_state::ValidatedApplyOutcome; +use nodedb_types::sync::violation::ViolationType; use super::params::{CRDT_PENDING_DEPENDENCIES, CRDT_SINGLE_DOCUMENT_DELTA, CrdtApplyParams}; @@ -26,6 +28,9 @@ enum LocalRefusal { /// Nothing applied, but the identical bytes apply once the missing causal /// history arrives. Retryable { detail: String }, + /// Permanent: a row the delta writes violates a constraint. Nothing + /// applied, and the delta is in the dead-letter queue. + Constraint(ViolationType), } impl CoreLoop { @@ -105,13 +110,7 @@ impl CoreLoop { Ok(None) } } - ValidatedApplyOutcome::Rejected(vt) => { - imported_authoritative = true; - // There is no client to answer here, so the validated - // outcome is observed only for its DLQ side effect. - tracing::debug!(core = self.core_id, %collection, reason = %vt, "crdt apply violated constraint (DLQ)"); - Ok(None) - } + ValidatedApplyOutcome::Rejected(vt) => Err(LocalRefusal::Constraint(vt)), ValidatedApplyOutcome::Malformed => Err(LocalRefusal::Malformed), ValidatedApplyOutcome::PendingDependencies => { // Nothing was imported: the operations are buffered awaiting @@ -128,8 +127,8 @@ impl CoreLoop { } } }; - // Engine borrow dropped here. A clean or constraint-rejected Loro - // import changed authoritative state; malformed bytes did not. + // Engine borrow dropped here. A clean import changed authoritative + // state. A refused one did not. if imported_authoritative { self.checkpoint_coordinator.mark_dirty("crdt", 1); } @@ -148,8 +147,8 @@ impl CoreLoop { } } Ok(None) if imported_authoritative => { - // Headless and constraint-rejected imports have no sparse - // projection, but still changed authoritative Loro state. + // A headless import has no sparse projection, but still + // changed authoritative Loro state. self.note_collection_write_lsn(task, collection); } Ok(None) => {} @@ -170,6 +169,31 @@ impl CoreLoop { ), } } + LocalRefusal::Constraint(violation) => { + tracing::debug!( + core = self.core_id, + %collection, + %document_id, + reason = %violation, + "crdt apply refused: constraint violated" + ); + // The record is cancelled, so replay never reaches + // this rejection. Its dead-letter entry is stored + // before the refusal is reported. + match self.store_crdt_dead_letter( + task.request.database_id, + tenant_id, + task.wal_lsn(), + ) { + Ok(()) => crdt_rejection(collection, document_id, &violation), + Err(error) => ErrorCode::Internal { + detail: format!( + "delta for {collection}/{document_id} violates {violation}, \ + and its dead-letter entry could not be stored: {error}" + ), + }, + } + } LocalRefusal::Retryable { detail } => { warn!( core = self.core_id, @@ -279,4 +303,129 @@ mod tests { }); assert!(!imported, "a refused delta must not reach the CRDT state"); } + + fn users_params<'a>(document_id: &'a str, delta: &'a [u8]) -> CrdtApplyParams<'a> { + CrdtApplyParams { + collection: "users", + ..params(document_id, delta) + } + } + + fn user_delta(peer: u64, row_id: &str, email: &str) -> Vec { + let source = nodedb_crdt::CrdtState::new(peer).expect("source state"); + source + .upsert( + "users", + row_id, + &[("email", LoroValue::String(email.into()))], + ) + .expect("source write"); + source.export_snapshot().expect("source snapshot") + } + + fn task_at(lsn: u64) -> ExecutionTask { + ExecutionTask::with_wal_lsn( + make_default_task().request, + Some(crate::types::Lsn::new(lsn)), + ) + } + + /// Install a UNIQUE email constraint under a strict policy, so a clash + /// is a rejection. + fn install_unique_email(core: &mut CoreLoop, task: &ExecutionTask) { + let engine = core + .get_crdt_engine(task.request.database_id, task.request.tenant_id) + .expect("engine"); + assert!(engine.set_collection_constraints( + "users", + 1, + vec![nodedb_crdt::Constraint { + name: "users_email_unique".into(), + collection: "users".into(), + field: "email".into(), + kind: nodedb_crdt::ConstraintKind::Unique, + }], + )); + engine + .set_collection_policy_typed("users", nodedb_crdt::policy::CollectionPolicy::strict()); + } + + /// Seed row `a`, then apply row `b` with the same email at `lsn`. + fn reject_duplicate_email(core: &mut CoreLoop, lsn: u64) -> Response { + let seed = task_at(lsn - 1); + install_unique_email(core, &seed); + let first = user_delta(2, "a", "x@y.com"); + assert_eq!( + core.apply_crdt_local(&seed, users_params("a", &first)) + .status, + Status::Ok + ); + let second = user_delta(3, "b", "x@y.com"); + core.apply_crdt_local(&task_at(lsn), users_params("b", &second)) + } + + fn dead_letters(core: &mut CoreLoop) -> Vec { + let task = make_default_task(); + core.get_crdt_engine(task.request.database_id, task.request.tenant_id) + .expect("engine") + .dead_letters() + .cloned() + .collect() + } + + /// A delta rejected by a constraint is refused with the constraint, the + /// row stays absent, and one dead-letter entry is stored for its record. + #[test] + fn a_constraint_rejection_is_refused_and_stores_its_dead_letter() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _request_tx, _response_rx) = make_core_with_dir(dir.path()); + + let response = reject_duplicate_email(&mut core, 11); + + assert_eq!(response.status, Status::Error); + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RejectedConstraint { constraint, .. }) if constraint == "unique" + ), + "got {:?}", + response.error_code + ); + let task = make_default_task(); + let db = task.request.database_id; + let tenant = task.request.tenant_id; + assert!( + core.crdt_engines + .get(&(db, tenant)) + .is_some_and(|engine| !engine.row_exists("users", "b")), + "a rejected delta must not reach the CRDT state" + ); + let live = dead_letters(&mut core); + assert_eq!(live.len(), 1); + assert_eq!(live[0].source_lsn, Some(11)); + assert_eq!( + core.sparse + .load_crdt_dead_letters(db.as_u64(), tenant.as_u64()) + .expect("load"), + live + ); + } + + /// After a restart the queue holds the entries the live path stored. A + /// record that rejects again keeps its one entry. + #[test] + fn a_restart_restores_the_live_dead_letters_without_duplicates() { + let dir = tempfile::tempdir().expect("tempdir"); + let live = { + let (mut core, _request_tx, _response_rx) = make_core_with_dir(dir.path()); + assert_eq!(reject_duplicate_email(&mut core, 11).status, Status::Error); + dead_letters(&mut core) + }; + + let (mut core, _request_tx, _response_rx) = make_core_with_dir(dir.path()); + assert_eq!(dead_letters(&mut core), live); + + assert_eq!(reject_duplicate_email(&mut core, 11).status, Status::Error); + assert_eq!(dead_letters(&mut core), live, "the record keeps one entry"); + } } diff --git a/nodedb/src/data/executor/handlers/mod.rs b/nodedb/src/data/executor/handlers/mod.rs index 582c9c2fa..1b91b8d76 100644 --- a/nodedb/src/data/executor/handlers/mod.rs +++ b/nodedb/src/data/executor/handlers/mod.rs @@ -47,6 +47,7 @@ pub(super) mod provider_scan_compute; pub mod purge; pub mod query_collection_size; pub mod reclaim; +pub(super) mod reclaim_retry; pub mod recursive; pub mod recursive_value; pub mod returning_doc; diff --git a/nodedb/src/data/executor/handlers/purge.rs b/nodedb/src/data/executor/handlers/purge.rs index 59dab9b83..aaa168499 100644 --- a/nodedb/src/data/executor/handlers/purge.rs +++ b/nodedb/src/data/executor/handlers/purge.rs @@ -176,6 +176,18 @@ impl CoreLoop { ); } + // Stored CRDT dead-letter entries: a tenant recreated under the same + // id must not restore them. + if let Err(e) = self.sparse.delete_crdt_dead_letters_for_tenant(tenant_id) { + warn!(tenant_id, error = %e, "sparse crdt dead-letter purge failed"); + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("sparse crdt dead-letter purge: {e}"), + }, + ); + } + // Sparse vector indexes: remove for this tenant (all databases). self.sparse_vector_indexes .retain(|(_, t, _, _), _| *t != tid_key); diff --git a/nodedb/src/data/executor/handlers/reclaim_retry.rs b/nodedb/src/data/executor/handlers/reclaim_retry.rs new file mode 100644 index 000000000..da3b9909d --- /dev/null +++ b/nodedb/src/data/executor/handlers/reclaim_retry.rs @@ -0,0 +1,61 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Bounded retry for collection-purge reclaim operations. + +use tracing::info; + +/// Bounded retry wrapper for collection-purge reclaim ops. +/// +/// Runs the op up to `MAX_ATTEMPTS` times; returns the first `Ok(T)` +/// it sees. No sleep between attempts — the Data Plane is single-threaded +/// per core and a `sleep` here would stall every other request on the +/// shard; an immediate retry still recovers the vast majority of +/// transient fs-level errors (momentary lock, inflight fsync race). +/// +/// **Fail-closed:** after exhausting all attempts this returns the last +/// error rather than swallowing it. The purge path must not warn-and- +/// continue: a partially-purged collection whose catalog row is then +/// removed leaves addressable storage rows that a re-CREATE of the same +/// name would resurrect. The caller propagates the error so the DROP +/// fails and the collection remains fully intact for the next attempt. +const L1_RECLAIM_MAX_ATTEMPTS: u32 = 3; + +pub(in crate::data::executor) fn retry_reclaim( + op_name: &str, + tenant_id: u64, + collection: &str, + mut op: F, +) -> crate::Result +where + F: FnMut() -> Result, + E: std::fmt::Display, +{ + let mut last_err: Option = None; + for attempt in 1..=L1_RECLAIM_MAX_ATTEMPTS { + match op() { + Ok(v) => { + if attempt > 1 { + info!( + tenant_id, + collection, + op = op_name, + attempt, + "collection-purge reclaim recovered after transient failure" + ); + } + return Ok(v); + } + Err(e) => { + last_err = Some(e.to_string()); + } + } + } + Err(crate::Error::Storage { + engine: "collection-purge".into(), + detail: format!( + "reclaim op '{op_name}' for tenant {tenant_id} collection '{collection}' \ + failed after {L1_RECLAIM_MAX_ATTEMPTS} attempts: {}", + last_err.as_deref().unwrap_or("(no detail)") + ), + }) +} diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs index e8ddca9a3..c19c8312e 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch.rs @@ -386,6 +386,7 @@ impl CoreLoop { ?outcome, "CRDT delta validation failed; transaction rollback required" ); + let code = self.crdt_batch_refusal(task, tenant_id, &collection, outcome); return Some(Response { request_id: task.request_id(), status: Status::Error, @@ -393,11 +394,7 @@ impl CoreLoop { partial: false, payload: crate::bridge::envelope::Payload::empty(), watermark_lsn: self.watermark, - error_code: Some(Box::new( - crate::bridge::envelope::ErrorCode::Internal { - detail: format!("CRDT delta validation failed: {outcome:?}"), - }, - )), + error_code: Some(Box::new(code)), read_set_valid: None, read_version_lsn: crate::types::Lsn::ZERO, write_set: Vec::new(), diff --git a/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs b/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs index 64f580b98..234a59fc3 100644 --- a/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs +++ b/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs @@ -7,14 +7,50 @@ use std::panic::{AssertUnwindSafe, catch_unwind}; use tracing::error; use crate::bridge::envelope::{ErrorCode, Response}; -use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::{CoreLoop, crdt_rejection}; use crate::data::executor::task::ExecutionTask; use crate::data::panic_payload::panic_payload_to_string; +use crate::engine::crdt::tenant_state::ValidatedApplyOutcome; use crate::types::TenantId; use super::batch::CrdtDelta; use super::undo::UndoEntry; +impl CoreLoop { + /// The refusal for a buffered CRDT delta that did not apply cleanly. + /// + /// A constraint rejection stores its dead-letter entry first, then + /// refuses with the constraint. The transaction rolls back either way. + pub(super) fn crdt_batch_refusal( + &mut self, + task: &ExecutionTask, + tenant_id: TenantId, + collection: &str, + outcome: ValidatedApplyOutcome, + ) -> ErrorCode { + match outcome { + ValidatedApplyOutcome::Rejected(violation) => match self.store_crdt_dead_letter( + task.request.database_id, + tenant_id, + task.wal_lsn(), + ) { + Ok(()) => crdt_rejection(collection, "transaction batch", &violation), + Err(error) => ErrorCode::Internal { + detail: format!( + "CRDT delta for {collection} violates {violation}, and its \ + dead-letter entry could not be stored: {error}" + ), + }, + }, + outcome @ (ValidatedApplyOutcome::Clean { .. } + | ValidatedApplyOutcome::Malformed + | ValidatedApplyOutcome::PendingDependencies) => ErrorCode::Internal { + detail: format!("CRDT delta validation failed: {outcome:?}"), + }, + } + } +} + /// A CRDT collection's state before a transaction starts importing its deltas. struct CrdtCollectionPreimage { collection: String, diff --git a/nodedb/src/data/executor/handlers/unregister_collection.rs b/nodedb/src/data/executor/handlers/unregister_collection.rs index 1df0a3e80..39d53dac6 100644 --- a/nodedb/src/data/executor/handlers/unregister_collection.rs +++ b/nodedb/src/data/executor/handlers/unregister_collection.rs @@ -36,6 +36,7 @@ use tracing::info; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::reclaim; +use crate::data::executor::handlers::reclaim_retry::retry_reclaim; use crate::data::executor::task::ExecutionTask; use crate::types::TenantId; @@ -59,62 +60,6 @@ pub(in crate::data::executor) struct ClearCollectionStats { pub l1: reclaim::ReclaimStats, } -/// Bounded retry wrapper for collection-purge reclaim ops. -/// -/// Runs the op up to `MAX_ATTEMPTS` times; returns the first `Ok(T)` -/// it sees. No sleep between attempts — the Data Plane is single-threaded -/// per core and a `sleep` here would stall every other request on the -/// shard; an immediate retry still recovers the vast majority of -/// transient fs-level errors (momentary lock, inflight fsync race). -/// -/// **Fail-closed:** after exhausting all attempts this returns the last -/// error rather than swallowing it. The purge path must not warn-and- -/// continue: a partially-purged collection whose catalog row is then -/// removed leaves addressable storage rows that a re-CREATE of the same -/// name would resurrect. The caller propagates the error so the DROP -/// fails and the collection remains fully intact for the next attempt. -const L1_RECLAIM_MAX_ATTEMPTS: u32 = 3; - -fn retry_reclaim( - op_name: &str, - tenant_id: u64, - collection: &str, - mut op: F, -) -> crate::Result -where - F: FnMut() -> Result, - E: std::fmt::Display, -{ - let mut last_err: Option = None; - for attempt in 1..=L1_RECLAIM_MAX_ATTEMPTS { - match op() { - Ok(v) => { - if attempt > 1 { - info!( - tenant_id, - collection, - op = op_name, - attempt, - "collection-purge reclaim recovered after transient failure" - ); - } - return Ok(v); - } - Err(e) => { - last_err = Some(e.to_string()); - } - } - } - Err(crate::Error::Storage { - engine: "collection-purge".into(), - detail: format!( - "reclaim op '{op_name}' for tenant {tenant_id} collection '{collection}' \ - failed after {L1_RECLAIM_MAX_ATTEMPTS} attempts: {}", - last_err.as_deref().unwrap_or("(no detail)") - ), - }) -} - impl CoreLoop { /// Purge collection-scoped data on this core. pub(in crate::data::executor) fn execute_unregister_collection( @@ -370,6 +315,12 @@ impl CoreLoop { })?, None => 0, }; + // Stored dead-letter entries go with it, or the next engine created + // for this tenant restores them. + retry_reclaim("crdt.dead_letters", tid_raw, collection, || { + self.sparse + .delete_crdt_dead_letters_for_collection(db_raw, tid_raw, collection) + })?; // Doc cache: evict entries for this collection. self.doc_cache.evict_collection(db_raw, tid_raw, collection); diff --git a/nodedb/src/data/executor/wal_replay/crdt.rs b/nodedb/src/data/executor/wal_replay/crdt.rs index bcf3463c7..7f8e82225 100644 --- a/nodedb/src/data/executor/wal_replay/crdt.rs +++ b/nodedb/src/data/executor/wal_replay/crdt.rs @@ -231,6 +231,7 @@ impl CoreLoop { // The committed record remains a deterministic no-op // whose collection floor advances on every replica. warn!(core = self.core_id, tenant = tid.as_u64(), %collection, %reason, "CRDT WAL delta rejected during replay"); + self.store_replayed_dead_letter(database_id, tid, record.header.lsn); None } crate::engine::crdt::tenant_state::ValidatedApplyOutcome::Malformed => { @@ -279,6 +280,7 @@ impl CoreLoop { reason, ) => { warn!(core = self.core_id, tenant = tid.as_u64(), %collection, %reason, "legacy CRDT WAL delta rejected during replay"); + self.store_replayed_dead_letter(database_id, tid, record.header.lsn); None } crate::engine::crdt::tenant_state::ValidatedApplyOutcome::Malformed => { @@ -735,4 +737,93 @@ mod crdt_replay_tests { "core 0 must not replay a record routed to core 1" ); } + + /// A `users` row record at `lsn` whose delta sets `email`. + fn email_record( + tid: TenantId, + row_id: &str, + email: &str, + peer: u64, + lsn: u64, + ) -> nodedb_wal::WalRecord { + let state = nodedb_crdt::state::CrdtState::new(peer).expect("state"); + state + .upsert( + "users", + row_id, + &[("email", LoroValue::String(email.into()))], + ) + .expect("upsert"); + let payload = crate::wal::CrdtDeltaWalPayload::new( + state.export_snapshot().expect("snapshot"), + Some("users".into()), + None, + None, + Some(row_id.to_owned()), + Some(0), + ); + nodedb_wal::WalRecord::new(nodedb_wal::WalRecordArgs { + record_type: RecordType::CrdtDelta as u32, + lsn, + tenant_id: tid.as_u64(), + vshard_id: 0, + database_id: DatabaseId::DEFAULT.as_u64(), + payload: payload.encode().expect("encode"), + encryption_key: None, + preamble_bytes: None, + }) + .expect("record") + } + + /// A record whose delta a constraint rejects leaves one stored entry, + /// however often it replays. + #[test] + fn a_replayed_rejection_stores_one_dead_letter_for_its_record() { + let tid = TenantId::new(7); + let mut h = make_core(0); + { + let engine = h + .core + .get_crdt_engine(DatabaseId::DEFAULT, tid) + .expect("engine"); + assert!(engine.set_collection_constraints( + "users", + 1, + vec![nodedb_crdt::Constraint { + name: "users_email_unique".into(), + collection: "users".into(), + field: "email".into(), + kind: nodedb_crdt::ConstraintKind::Unique, + }], + )); + engine.set_collection_policy_typed( + "users", + nodedb_crdt::policy::CollectionPolicy::strict(), + ); + } + let tombstones = nodedb_wal::TombstoneSet::new(); + let records = [ + email_record(tid, "a", "x@y.com", 2, 19), + email_record(tid, "b", "x@y.com", 3, 20), + ]; + h.core.replay_crdt_wal(&records, 1, &tombstones); + h.core.replay_crdt_wal(&records[1..], 1, &tombstones); + + let entries: Vec<_> = h + .core + .get_crdt_engine(DatabaseId::DEFAULT, tid) + .expect("engine") + .dead_letters() + .cloned() + .collect(); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].source_lsn, Some(20)); + assert_eq!( + h.core + .sparse + .load_crdt_dead_letters(DatabaseId::DEFAULT.as_u64(), tid.as_u64()) + .expect("load"), + entries + ); + } } diff --git a/nodedb/src/engine/crdt/tenant_state/apply_validated.rs b/nodedb/src/engine/crdt/tenant_state/apply_validated.rs index 56c6fcb83..f8c158f32 100644 --- a/nodedb/src/engine/crdt/tenant_state/apply_validated.rs +++ b/nodedb/src/engine/crdt/tenant_state/apply_validated.rs @@ -110,6 +110,7 @@ impl TenantCrdtEngine { peer_id: u64, admission: DeltaSigningAdmission, ) -> ValidatedApplyOutcome { + self.last_dead_letter = None; if admission.required && admission.auth.delta_signature == [0; 32] { return ValidatedApplyOutcome::Malformed; } @@ -313,7 +314,7 @@ impl TenantCrdtEngine { let reason = violation.reason.clone(); match constraint { Some(constraint) => { - if let Err(e) = + let enqueued = self.validator .dlq_mut() .enqueue(nodedb_crdt::EnqueueDeadLetterArgs { @@ -324,14 +325,15 @@ impl TenantCrdtEngine { constraint: &constraint, reason, hint: violation.hint.clone(), - }) - { - tracing::warn!( + }); + match enqueued { + Ok(id) => self.last_dead_letter = Some(id), + Err(e) => tracing::warn!( tenant = tenant_id, collection, error = %e, "crdt: failed to enqueue rejected delta to DLQ" - ); + ), } } None => { @@ -347,7 +349,7 @@ impl TenantCrdtEngine { let hint = nodedb_crdt::CompensationHint::ManualIntervention { reason: reason.clone(), }; - if let Err(e) = + let enqueued = self.validator .dlq_mut() .enqueue(nodedb_crdt::EnqueueDeadLetterArgs { @@ -358,14 +360,15 @@ impl TenantCrdtEngine { constraint: &fallback, reason, hint, - }) - { - tracing::warn!( + }); + match enqueued { + Ok(id) => self.last_dead_letter = Some(id), + Err(e) => tracing::warn!( tenant = tenant_id, collection, error = %e, "crdt: failed to enqueue rejected delta to DLQ (unresolved constraint)" - ); + ), } } } diff --git a/nodedb/src/engine/crdt/tenant_state/core.rs b/nodedb/src/engine/crdt/tenant_state/core.rs index fbd1c2703..ceba64912 100644 --- a/nodedb/src/engine/crdt/tenant_state/core.rs +++ b/nodedb/src/engine/crdt/tenant_state/core.rs @@ -102,6 +102,10 @@ pub struct TenantCrdtEngine { /// alive across a run of deltas and rebuilt only when a delta is refused. /// Cleared by `clear_apply_candidates` when the run ends. pub(super) apply_candidates: HashMap, + + /// The dead-letter entry the latest validated apply enqueued. The applier + /// binds it to the record that carried the delta. + pub(super) last_dead_letter: Option, } impl TenantCrdtEngine { @@ -119,6 +123,7 @@ impl TenantCrdtEngine { collections: HashMap::new(), constraint_versions: HashMap::new(), apply_candidates: HashMap::new(), + last_dead_letter: None, }) } diff --git a/nodedb/src/engine/crdt/tenant_state/dead_letters.rs b/nodedb/src/engine/crdt/tenant_state/dead_letters.rs new file mode 100644 index 000000000..84a49a5e6 --- /dev/null +++ b/nodedb/src/engine/crdt/tenant_state/dead_letters.rs @@ -0,0 +1,48 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Binding dead-letter entries to the records that produced them, and +//! restoring entries read from durable storage. +//! +//! A rejected delta leaves one entry per record. The applier binds the entry +//! to the record's log position and stores it. Replaying the same record +//! binds to the same position, so the queue keeps one entry for it. + +use nodedb_crdt::DeadLetter; + +use super::core::TenantCrdtEngine; + +impl TenantCrdtEngine { + /// Bind the entry the latest validated apply enqueued to the record at + /// `source_lsn`. + /// + /// Returns the entry to store. Returns `None` when that apply enqueued + /// nothing, or when an entry for the record already exists. + pub fn bind_dead_letter_source(&mut self, source_lsn: u64) -> Option { + let id = self.last_dead_letter.take()?; + self.validator + .dlq_mut() + .bind_source(id, source_lsn) + .cloned() + } + + /// Put back entries read from durable storage. An entry the queue cannot + /// hold is logged and skipped: storage keeps it. + pub fn restore_dead_letters(&mut self, entries: Vec) { + for entry in entries { + let source_lsn = entry.source_lsn; + if let Err(error) = self.validator.dlq_mut().restore(entry) { + tracing::warn!( + tenant = self.tenant_id.as_u64(), + ?source_lsn, + %error, + "crdt: a stored dead-letter entry does not fit the queue" + ); + } + } + } + + /// Every pending dead-letter entry, oldest first. + pub fn dead_letters(&self) -> impl Iterator { + self.validator.dlq().iter() + } +} diff --git a/nodedb/src/engine/crdt/tenant_state/mod.rs b/nodedb/src/engine/crdt/tenant_state/mod.rs index b058f5f7c..e39cd046c 100644 --- a/nodedb/src/engine/crdt/tenant_state/mod.rs +++ b/nodedb/src/engine/crdt/tenant_state/mod.rs @@ -9,6 +9,7 @@ pub mod apply; pub mod apply_validated; pub mod constraints; pub mod core; +pub mod dead_letters; pub mod doc_mutate; pub mod history; pub mod list_ops; diff --git a/nodedb/src/engine/sparse/btree/crdt_dead_letter.rs b/nodedb/src/engine/sparse/btree/crdt_dead_letter.rs new file mode 100644 index 000000000..2e852e312 --- /dev/null +++ b/nodedb/src/engine/sparse/btree/crdt_dead_letter.rs @@ -0,0 +1,269 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable dead-letter entries for rejected CRDT deltas. +//! +//! A rejected delta changes no CRDT state, and its record carries an abort +//! marker, so restart replay never reaches it. The entry it left is state of +//! its own. It is stored here when the rejection happens and read back when +//! the tenant's CRDT engine is created. +//! +//! Storage: the `crdt_dead_letters` redb table, keyed +//! `"{database_id}:{tenant_id}:{source_lsn:020}"`. The key names the record +//! that produced the entry, so storing an entry for a record twice keeps one. +//! The value is the entry as MessagePack. + +use nodedb_crdt::DeadLetter; +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::engine::SparseEngine; +use super::tables::redb_err; + +/// Table definition for stored dead-letter entries. +pub(super) const CRDT_DEAD_LETTERS: TableDefinition<&str, &[u8]> = + TableDefinition::new("crdt_dead_letters"); + +fn dead_letter_prefix(database_id: u64, tenant_id: u64) -> String { + format!("{database_id}:{tenant_id}:") +} + +fn entry_key(database_id: u64, tenant_id: u64, source_lsn: u64) -> String { + format!("{database_id}:{tenant_id}:{source_lsn:020}") +} + +/// The tenant a key belongs to. +fn key_tenant(key: &str) -> crate::Result { + key.split(':') + .nth(1) + .and_then(|tenant| tenant.parse::().ok()) + .ok_or_else(|| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("malformed crdt dead-letter key: {key}"), + }) +} + +fn decode(key: &str, bytes: &[u8]) -> crate::Result { + zerompk::from_msgpack(bytes).map_err(|e| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("decode crdt dead-letter {key}: {e}"), + }) +} + +impl SparseEngine { + /// Store `entry`, produced by the record at `source_lsn`. + pub fn put_crdt_dead_letter( + &self, + database_id: u64, + tenant_id: u64, + source_lsn: u64, + entry: &DeadLetter, + ) -> crate::Result<()> { + let bytes = zerompk::to_msgpack_vec(entry).map_err(|e| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("encode crdt dead-letter: {e}"), + })?; + let key = entry_key(database_id, tenant_id, source_lsn); + let txn = self + .db + .begin_write() + .map_err(|e| redb_err("write txn", e))?; + { + let mut table = txn + .open_table(CRDT_DEAD_LETTERS) + .map_err(|e| redb_err("open crdt dead letters", e))?; + table + .insert(key.as_str(), bytes.as_slice()) + .map_err(|e| redb_err("insert crdt dead letter", e))?; + } + txn.commit() + .map_err(|e| redb_err("commit crdt dead letter", e)) + } + + /// Every stored entry of one tenant in one database, in record order. + pub fn load_crdt_dead_letters( + &self, + database_id: u64, + tenant_id: u64, + ) -> crate::Result> { + let prefix = dead_letter_prefix(database_id, tenant_id); + let read_txn = self.db.begin_read().map_err(|e| redb_err("read txn", e))?; + let table = read_txn + .open_table(CRDT_DEAD_LETTERS) + .map_err(|e| redb_err("open crdt dead letters", e))?; + let mut out = Vec::new(); + for entry in table + .range(prefix.as_str()..) + .map_err(|e| redb_err("range crdt dead letters", e))? + { + let (k, v) = entry.map_err(|e| redb_err("scan crdt dead letter", e))?; + let key = k.value(); + if !key.starts_with(prefix.as_str()) { + break; + } + out.push(decode(key, v.value())?); + } + Ok(out) + } + + /// Remove every stored entry of one collection. Returns how many. + pub fn delete_crdt_dead_letters_for_collection( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + ) -> crate::Result { + let prefix = dead_letter_prefix(database_id, tenant_id); + self.delete_crdt_dead_letters_where(|key, entry| { + Ok(key.starts_with(prefix.as_str()) && entry.collection == collection) + }) + } + + /// Remove every stored entry of `tenant_id`, in every database. Returns + /// how many. + pub fn delete_crdt_dead_letters_for_tenant(&self, tenant_id: u64) -> crate::Result { + self.delete_crdt_dead_letters_where(|key, _| Ok(key_tenant(key)? == tenant_id)) + } + + fn delete_crdt_dead_letters_where( + &self, + doomed: impl Fn(&str, &DeadLetter) -> crate::Result, + ) -> crate::Result { + let keys: Vec = { + let read_txn = self.db.begin_read().map_err(|e| redb_err("read txn", e))?; + let table = read_txn + .open_table(CRDT_DEAD_LETTERS) + .map_err(|e| redb_err("open crdt dead letters", e))?; + let mut out = Vec::new(); + for entry in table + .iter() + .map_err(|e| redb_err("iter crdt dead letters", e))? + { + let (k, v) = entry.map_err(|e| redb_err("scan crdt dead letter", e))?; + let key = k.value(); + if doomed(key, &decode(key, v.value())?)? { + out.push(key.to_string()); + } + } + out + }; + if keys.is_empty() { + return Ok(0); + } + let txn = self + .db + .begin_write() + .map_err(|e| redb_err("write txn", e))?; + { + let mut table = txn + .open_table(CRDT_DEAD_LETTERS) + .map_err(|e| redb_err("open crdt dead letters", e))?; + for key in &keys { + table + .remove(key.as_str()) + .map_err(|e| redb_err("remove crdt dead letter", e))?; + } + } + txn.commit() + .map_err(|e| redb_err("commit crdt dead letter purge", e))?; + Ok(keys.len()) + } +} + +#[cfg(test)] +mod tests { + use nodedb_crdt::{ + CompensationHint, Constraint, ConstraintKind, DeadLetterQueue, EnqueueDeadLetterArgs, + }; + + use super::*; + + fn entry(collection: &str, source_lsn: u64) -> DeadLetter { + let mut dlq = DeadLetterQueue::new(4); + let id = dlq + .enqueue(EnqueueDeadLetterArgs { + peer_id: 1, + user_id: 0, + tenant_id: 3, + delta: b"delta".to_vec(), + constraint: &Constraint { + name: "email_unique".into(), + collection: collection.into(), + field: "email".into(), + kind: ConstraintKind::Unique, + }, + reason: "duplicate".into(), + hint: CompensationHint::ManualIntervention { + reason: "duplicate".into(), + }, + }) + .expect("enqueue"); + dlq.bind_source(id, source_lsn).cloned().expect("bound") + } + + fn open() -> (tempfile::TempDir, SparseEngine) { + let dir = tempfile::tempdir().expect("tempdir"); + let engine = SparseEngine::open(&dir.path().join("sparse.redb")).expect("open"); + (dir, engine) + } + + #[test] + fn stored_entries_survive_a_reopen_once_per_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let path = dir.path().join("sparse.redb"); + let first = entry("users", 5); + { + let engine = SparseEngine::open(&path).expect("open"); + engine.put_crdt_dead_letter(1, 3, 5, &first).expect("put"); + engine + .put_crdt_dead_letter(1, 3, 5, &first) + .expect("put again"); + engine + .put_crdt_dead_letter(1, 4, 6, &entry("users", 6)) + .expect("put other tenant"); + } + let engine = SparseEngine::open(&path).expect("reopen"); + assert_eq!( + engine.load_crdt_dead_letters(1, 3).expect("load"), + vec![first] + ); + } + + #[test] + fn a_purged_collection_or_tenant_leaves_no_entry() { + let (_dir, engine) = open(); + engine + .put_crdt_dead_letter(1, 3, 5, &entry("users", 5)) + .expect("put"); + engine + .put_crdt_dead_letter(1, 3, 6, &entry("orders", 6)) + .expect("put"); + engine + .put_crdt_dead_letter(2, 3, 7, &entry("users", 7)) + .expect("put"); + + assert_eq!( + engine + .delete_crdt_dead_letters_for_collection(1, 3, "users") + .expect("purge collection"), + 1 + ); + assert_eq!(engine.load_crdt_dead_letters(1, 3).expect("load").len(), 1); + assert_eq!( + engine + .delete_crdt_dead_letters_for_tenant(3) + .expect("purge tenant"), + 2 + ); + assert!( + engine + .load_crdt_dead_letters(1, 3) + .expect("load") + .is_empty() + ); + assert!( + engine + .load_crdt_dead_letters(2, 3) + .expect("load") + .is_empty() + ); + } +} diff --git a/nodedb/src/engine/sparse/btree/engine.rs b/nodedb/src/engine/sparse/btree/engine.rs index ad97de216..b4c947a9a 100644 --- a/nodedb/src/engine/sparse/btree/engine.rs +++ b/nodedb/src/engine/sparse/btree/engine.rs @@ -11,6 +11,7 @@ use redb::{Database, WriteTransaction}; use tracing::info; use super::chain_head::CHAIN_HEADS; +use super::crdt_dead_letter::CRDT_DEAD_LETTERS; use super::tables::{DOCUMENTS, INDEXES, redb_err}; /// redb-backed B-Tree storage engine for sparse/metadata queries. @@ -41,6 +42,10 @@ impl SparseEngine { let _ = write_txn .open_table(CHAIN_HEADS) .map_err(|e| redb_err("open chain heads table", e))?; + // Read transactions open it when a CRDT engine is created. + let _ = write_txn + .open_table(CRDT_DEAD_LETTERS) + .map_err(|e| redb_err("open crdt dead letters table", e))?; } write_txn.commit().map_err(|e| redb_err("commit", e))?; diff --git a/nodedb/src/engine/sparse/btree/mod.rs b/nodedb/src/engine/sparse/btree/mod.rs index 75c0e3aae..87213fee4 100644 --- a/nodedb/src/engine/sparse/btree/mod.rs +++ b/nodedb/src/engine/sparse/btree/mod.rs @@ -3,6 +3,7 @@ //! redb-backed B-Tree storage for the sparse engine's non-versioned tables. pub mod chain_head; +pub mod crdt_dead_letter; pub mod document; pub mod engine; pub mod keys; From 6ac1670d2e0ae48f668075bd4439f0b0b6943f5b Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 17:39:06 +0800 Subject: [PATCH 23/64] fix(bridge): cancel or hold minted records dropped before dispatch A caller dropped while its write waits for admission, or before a dispatch result arrives, left a set of minted WAL records with no path to settle their outcome-floor window: neither the writer's own cancel logic nor the eventual response ever ran. Give MintedRecords a Drop impl that cancels the records in place with a WriteAborted marker and settles the window when no core has been given the request yet, or holds the window and reports it when a core may already hold the records. Track which record a recording WAL appender wrote through RecordedAppend (LSN plus the tenant/vshard/database that placed it) instead of the bare LSN, so the drop path can append its own cancel markers without the caller re-deriving that context. Mark records as sent once a core has been handed the request, at every call site that dispatches to one, so a later drop knows to hold rather than cancel. Fold the health endpoint's outcome-floor-stuck body into its own function so it can be exercised directly by tests, and add a failpoint on the write-aborted WAL append to exercise the failure path where a cancel marker cannot be written. --- .../post_apply/async_dispatch/vector.rs | 4 + .../scheduler/driver/core/commit_redo.rs | 7 +- .../control/server/dispatch_utils/dispatch.rs | 98 ++++++ .../server/dispatch_utils/minted/owned.rs | 2 +- .../server/dispatch_utils/minted/records.rs | 311 +++++++++++++++--- .../server/dispatch_utils/minted/resolve.rs | 2 +- .../submit_write/funnel/driver.rs | 9 +- .../src/control/server/http/routes/health.rs | 103 ++++-- .../shared/ddl/sync_dispatch/dispatch.rs | 5 + .../server/sync/raft_dispatch/write.rs | 97 ++++++ nodedb/src/wal/manager/append_index.rs | 5 + nodedb/src/wal/manager/appender.rs | 36 +- nodedb/src/wal/manager/mod.rs | 2 +- 13 files changed, 597 insertions(+), 84 deletions(-) diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs index b4fa63e60..f4ed74724 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs @@ -162,6 +162,8 @@ async fn install_params( report(&error, "set_params_wal_append", &target); } + // The cores hold the record from here. It closes from their answers. + minted.mark_sent(); let set_params = fan_out(&shared, &fanout(&target), &plan).await; let stage = if set_params.pending.is_empty() { let refused = set_params.refused; @@ -304,6 +306,8 @@ async fn drop_index(name: IndexName, shared: Arc, ready: oneshot::S Err(error) => report(&error, "drop_index_wal_append", &target), } + // The cores hold the record from here. It closes from their answers. + minted.mark_sent(); let answers = fan_out(&shared, &fanout(&target), &plan).await; // Cores still working past the deadline answer later. The caller moves // on while this task waits for their final answers. diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index 34301b461..cdfb1c389 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -90,7 +90,12 @@ impl Scheduler { &redo, ); match appended { - Ok(lsn) => (Some(lsn), Some(records)), + Ok(lsn) => { + // The txn committed, so its redo record is never + // cancelled. The flush closes it from its outcome. + records.mark_sent(); + (Some(lsn), Some(records)) + } Err(e) => { // The txn stays pending and unapplied, and a failed append // leaves no record for restart replay to reach. diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index fa2828feb..84b588e65 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -744,4 +744,102 @@ mod tests { assert!(state.outcome_floor.floor() >= lsn); assert_eq!(state.outcome_floor.leaked_windows(), 0); } + + /// A point write the admission gate serializes on its key. + fn incr_plan() -> crate::bridge::envelope::PhysicalPlan { + crate::bridge::envelope::PhysicalPlan::Kv(nodedb_physical::physical_plan::KvOp::Incr { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), + key: b"k1".to_vec(), + delta: 1, + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::new(1), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + shape: nodedb_physical::physical_plan::KvCounterShape::Raw, + }) + } + + /// Wait until the floor passes `lsn`, routing responses meanwhile. + async fn floor_passes(state: &SharedState, lsn: Lsn) -> bool { + let deadline = Instant::now() + Duration::from_secs(5); + while state.outcome_floor.floor() < lsn && Instant::now() < deadline { + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + state.outcome_floor.floor() >= lsn + } + + #[tokio::test] + async fn a_caller_dropped_while_waiting_for_admission_cancels_its_records() { + let (state, _side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let plan = incr_plan(); + let (_, keys) = + crate::control::server::shared::write_admission::lock_keys::plan_lock_keys(&plan) + .expect("a point write has a lock key"); + let key = keys.into_iter().next().expect("one key"); + let held = state.write_order_locks.lock_owned(key).await; + let mut write = write_with(minted, lsn); + write.plan = plan; + + let waited = tokio::time::timeout( + Duration::from_millis(50), + super::dispatch_trusted_internal_write_to_data_plane(&state, write), + ) + .await; + drop(held); + + assert!(waited.is_err(), "the write waits behind the held key"); + assert!( + !replayed(&state).contains(&lsn.as_u64()), + "a marker names it" + ); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + assert_eq!(state.outcome_floor.held_windows(), 0); + } + + /// Drop the caller once its write is enqueued, then answer the write. + async fn drop_after_dispatch_then_answer( + status: Status, + code: Option, + ) -> (Arc, Lsn, tempfile::TempDir) { + let (state, side, directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let waited = tokio::time::timeout( + Duration::from_millis(50), + super::dispatch_trusted_internal_write_to_data_plane(&state, write_with(minted, lsn)), + ) + .await; + assert!(waited.is_err(), "no response arrived before the drop"); + assert!( + state.outcome_floor.floor() < lsn, + "the core holds the records" + ); + respond_once_with(Arc::clone(&state), side, status, code).await; + assert!( + floor_passes(&state, lsn).await, + "the final response closed the window" + ); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + (state, lsn, directory) + } + + #[tokio::test] + async fn a_caller_dropped_after_dispatch_settles_its_records_from_the_answer() { + let (state, lsn, _directory) = drop_after_dispatch_then_answer(Status::Ok, None).await; + assert!(replayed(&state).contains(&lsn.as_u64())); + } + + #[tokio::test] + async fn a_caller_dropped_after_dispatch_cancels_its_records_on_a_refusal() { + let (state, lsn, _directory) = drop_after_dispatch_then_answer( + Status::Error, + Some(crate::bridge::envelope::ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }), + ) + .await; + assert!(!replayed(&state).contains(&lsn.as_u64())); + } } diff --git a/nodedb/src/control/server/dispatch_utils/minted/owned.rs b/nodedb/src/control/server/dispatch_utils/minted/owned.rs index 023ef6ab2..e221f2825 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/owned.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/owned.rs @@ -153,7 +153,7 @@ mod tests { } } - fn minted_record(wal: &WalManager, floor: &Arc) -> (MintedRecords, Lsn) { + fn minted_record(wal: &Arc, floor: &Arc) -> (MintedRecords, Lsn) { let minted = MintedRecords::open(floor); let lsn = minted .appender(wal, NO_APPLY_KEY) diff --git a/nodedb/src/control/server/dispatch_utils/minted/records.rs b/nodedb/src/control/server/dispatch_utils/minted/records.rs index f36ea4a6d..13fe58186 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/records.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/records.rs @@ -17,15 +17,23 @@ //! Appends go through [`MintedRecords::appender`], which records the LSN of //! every record it writes. A plan that appends several records is cancelled //! whole. +//! +//! Records dropped without a close still close. Records never sent to a core +//! are cancelled in place: a `WriteAborted` marker names each one and the +//! window settles. Records a core can hold have no known outcome, so their +//! window leaks and files its report. A caller dropped at any await before +//! the dispatch therefore leaves nothing open. After the dispatch the +//! records belong to the task that waits for the final response. -use std::sync::{Arc, Mutex, MutexGuard}; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, MutexGuard, OnceLock}; use crate::bridge::dispatch::{OutcomeFloor, WriteWindow}; use crate::bridge::envelope::PhysicalPlan; use crate::control::server::wal_dispatch::{WalAppendOutcome, WalAppendRequest, wal_append}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::WalManager; -use crate::wal::manager::{NO_APPLY_KEY, WalAppender}; +use crate::wal::manager::{NO_APPLY_KEY, RecordedAppend, WalAppender}; /// Where a write's records live. The abort markers that cancel them carry it. #[derive(Debug, Clone, Copy)] @@ -37,24 +45,39 @@ pub(crate) struct RecordOwner { /// The records one write appended, and the window that holds the outcome /// floor below them. -#[derive(Debug)] #[must_use = "minted records hold the outcome floor until they settle, cancel, or hold"] pub(crate) struct MintedRecords { - window: WriteWindow, - lsns: Mutex>, - /// Whether this write appended the records. A resent record belongs to - /// the write that appended it, and only that write can cancel it. - appended_here: bool, + /// `None` once a close took it. + window: Option, + appended: Mutex>, + /// The existing record this set resends. A resent record belongs to the + /// write that appended it, and only that write can cancel it. + resent: Option, + /// The WAL the records went to. The first append stores it. + wal: OnceLock>, + /// Whether a core can hold the records. + sent: AtomicBool, +} + +impl std::fmt::Debug for MintedRecords { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("MintedRecords") + .field("open", &self.window.is_some()) + .field("appended", &*self.recorded()) + .field("resent", &self.resent) + .field("sent", &self.sent.load(Ordering::Acquire)) + .finish() + } } +/// A closed set's parts: the window, every appended record, and the +/// resent LSN. +type Parts = (WriteWindow, Vec, Option); + impl MintedRecords { /// Open the window. Call it before the first record is appended. pub(crate) fn open(floor: &Arc) -> Self { - Self { - window: floor.open_write(), - lsns: Mutex::new(Vec::new()), - appended_here: true, - } + Self::with_window(floor.open_write(), None) } /// Hold an existing record at `lsn` that is sent to a core again. `None` @@ -62,26 +85,37 @@ impl MintedRecords { /// apply would land below the floor. pub(crate) fn resend(floor: &Arc, lsn: Lsn) -> Option { let window = floor.open_existing(lsn)?; - Some(Self { - window, - lsns: Mutex::new(vec![lsn]), - appended_here: false, - }) + Some(Self::with_window(window, Some(lsn))) + } + + fn with_window(window: WriteWindow, resent: Option) -> Self { + Self { + window: Some(window), + appended: Mutex::new(Vec::new()), + resent, + wal: OnceLock::new(), + sent: AtomicBool::new(false), + } } - fn recorded(&self) -> MutexGuard<'_, Vec> { - self.lsns.lock().unwrap_or_else(|p| p.into_inner()) + fn recorded(&self) -> MutexGuard<'_, Vec> { + self.appended.lock().unwrap_or_else(|p| p.into_inner()) } /// An appender whose records carry `apply_key` and join this set. - pub(crate) fn appender<'a>(&'a self, wal: &'a WalManager, apply_key: u64) -> WalAppender<'a> { - wal.recording_appender(apply_key, &self.lsns) + pub(crate) fn appender<'a>( + &'a self, + wal: &'a Arc, + apply_key: u64, + ) -> WalAppender<'a> { + self.wal.get_or_init(|| Arc::clone(wal)); + wal.recording_appender(apply_key, &self.appended) } /// Append `plan`'s redo records under this window. pub(crate) fn append_plan( &self, - wal: &WalManager, + wal: &Arc, owner: RecordOwner, plan: &PhysicalPlan, ) -> crate::Result { @@ -96,42 +130,57 @@ impl MintedRecords { }) } - /// The highest appended LSN, or `None` when nothing was appended. + /// Mark the records as held by a core. Call it once the request carrying + /// them is enqueued, or once they are committed to a path that carries + /// them to their outcome. Dropped records are never cancelled after it. + pub(crate) fn mark_sent(&self) { + self.sent.store(true, Ordering::Release); + } + + /// The highest appended or resent LSN, or `None` when there is none. pub(crate) fn highest(&self) -> Option { - self.recorded().iter().copied().max() + self.recorded() + .iter() + .map(|record| record.lsn) + .chain(self.resent) + .max() } /// Every appended LSN, in append order. #[cfg(test)] pub(crate) fn lsns(&self) -> Vec { - self.recorded().clone() - } - - /// Note every recorded LSN on the window and take the list out. - fn into_parts(self) -> (WriteWindow, Vec, bool) { - let Self { - window, - lsns, - appended_here, - } = self; - let lsns = lsns.into_inner().unwrap_or_else(|p| p.into_inner()); - if let Some(highest) = lsns.iter().copied().max() { + self.recorded().iter().map(|record| record.lsn).collect() + } + + /// Take the window and the records out, and note the highest LSN on the + /// window. `None` when a close already took them. + fn take_parts(&mut self) -> Option { + let window = self.window.take()?; + let appended = std::mem::take(&mut *self.recorded()); + let highest = appended + .iter() + .map(|record| record.lsn) + .chain(self.resent) + .max(); + if let Some(highest) = highest { window.note_minted(highest); } - (window, lsns, appended_here) + Some((window, appended, self.resent)) } /// The outcome of every record is final. - pub(crate) fn settle(self) { - let (window, _, _) = self.into_parts(); - window.settle(); + pub(crate) fn settle(mut self) { + if let Some((window, _, _)) = self.take_parts() { + window.settle(); + } } /// The records have no final outcome in this process. #[track_caller] - pub(crate) fn hold(self) { - let (window, _, _) = self.into_parts(); - window.hold(); + pub(crate) fn hold(mut self) { + if let Some((window, _, _)) = self.take_parts() { + window.hold(); + } } /// Cancel every record with a `WriteAborted` marker that carries @@ -163,23 +212,25 @@ impl MintedRecords { } async fn cancel_in_place( - self, + mut self, wal: &WalManager, owner: RecordOwner, marker_key: u64, ) -> crate::Result<()> { - let (window, lsns, appended_here) = self.into_parts(); - if !appended_here { + let Some((window, appended, resent)) = self.take_parts() else { + return Ok(()); + }; + if resent.is_some() { window.settle(); return Ok(()); } let mut last_marker = None; - for lsn in &lsns { + for record in &appended { match wal.appender(marker_key).append_write_aborted( owner.tenant_id, owner.vshard_id, owner.database_id, - *lsn, + record.lsn, ) { Ok(marker) => last_marker = Some(marker), Err(error) => { @@ -195,7 +246,7 @@ impl MintedRecords { return Err(error); } tracing::debug!( - cancelled = lsns.len(), + cancelled = appended.len(), "refused write records cancelled in the WAL" ); window.settle(); @@ -228,6 +279,65 @@ impl MintedRecords { }), } } + + /// Close records dropped without a close. Runs in the dropping thread and + /// spawns nothing. + /// + /// Records no core holds are cancelled in place, and the window settles + /// once each marker is appended. The markers are not awaited: nothing + /// reported an outcome for these records, so a crash that loses a marker + /// leaves a write whose caller never learned its outcome. A marker that + /// fails to append holds the window. + /// + /// A resent record, or a set with nothing appended, settles. Records a + /// core can hold leak their window, which files its report. + fn close_dropped(&mut self) { + let sent = self.sent.load(Ordering::Acquire); + let Some((window, appended, resent)) = self.take_parts() else { + return; + }; + if sent { + drop(window); + return; + } + if resent.is_some() || appended.is_empty() { + window.settle(); + return; + } + let Some(wal) = self.wal.get() else { + // Only `appender` adds records, and it stores the WAL first. + window.hold(); + return; + }; + for record in &appended { + if let Err(error) = wal.appender(NO_APPLY_KEY).append_write_aborted( + record.tenant_id, + record.vshard_id, + record.database_id, + record.lsn, + ) { + tracing::error!( + %error, + lsn = record.lsn.as_u64(), + "records dropped before dispatch could not be cancelled; \ + their window is held until restart" + ); + window.hold(); + return; + } + } + tracing::debug!( + cancelled = appended.len(), + "records dropped before dispatch cancelled in the WAL" + ); + window.settle(); + } +} + +impl Drop for MintedRecords { + fn drop(&mut self) { + self.close_dropped(); + } } /// The task cancelling records another path superseded. @@ -257,7 +367,7 @@ mod tests { } } - fn append(wal: &WalManager, minted: &MintedRecords, body: &[u8]) -> Lsn { + fn append(wal: &Arc, minted: &MintedRecords, body: &[u8]) -> Lsn { minted .appender(wal, NO_APPLY_KEY) .append_put( @@ -269,10 +379,23 @@ mod tests { .expect("append") } + fn open_wal(dir: &tempfile::TempDir) -> Arc { + Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")) + } + + fn replayed(wal: &WalManager) -> Vec { + wal.sync().expect("sync"); + wal.replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect() + } + #[test] fn settled_records_release_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); - let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("wal"); + let wal = open_wal(&dir); let floor = OutcomeFloor::new(); let minted = MintedRecords::open(&floor); let lsn = append(&wal, &minted, b"a"); @@ -284,7 +407,7 @@ mod tests { #[test] fn held_records_keep_the_floor_below_them() { let dir = tempfile::tempdir().expect("tempdir"); - let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("wal"); + let wal = open_wal(&dir); let floor = OutcomeFloor::new(); let minted = MintedRecords::open(&floor); let lsn = append(&wal, &minted, b"a"); @@ -350,4 +473,86 @@ mod tests { assert!(!replayed.contains(&second.as_u64())); assert!(floor.floor() >= second, "the window settled"); } + + #[test] + fn records_dropped_before_dispatch_are_cancelled_and_release_the_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let first = append(&wal, &minted, b"a"); + let second = append(&wal, &minted, b"b"); + drop(minted); + let replayed = replayed(&wal); + assert!(!replayed.contains(&first.as_u64())); + assert!(!replayed.contains(&second.as_u64())); + assert!(floor.floor() >= second, "the window settled"); + assert_eq!(floor.leaked_windows(), 0); + assert_eq!(floor.held_windows(), 0); + } + + #[test] + fn records_dropped_after_dispatch_keep_the_floor_below_them() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + minted.mark_sent(); + drop(minted); + assert!(replayed(&wal).contains(&lsn.as_u64()), "no marker names it"); + assert!(floor.floor() < lsn); + assert_eq!( + floor.leaked_windows(), + 1, + "the dropped window files its leak" + ); + } + + #[test] + fn a_dropped_set_with_nothing_appended_settles() { + let floor = OutcomeFloor::new(); + drop(MintedRecords::open(&floor)); + assert_eq!(floor.leaked_windows(), 0); + assert_eq!(floor.held_windows(), 0); + } + + #[test] + fn a_dropped_resend_settles_without_a_marker() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let lsn = wal + .appender(NO_APPLY_KEY) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"a", + ) + .expect("append"); + drop(MintedRecords::resend(&floor, lsn).expect("the floor is below the record")); + assert!(replayed(&wal).contains(&lsn.as_u64()), "no marker names it"); + assert_eq!(floor.floor(), lsn); + assert_eq!(floor.leaked_windows(), 0); + } + + #[cfg(feature = "failpoints")] + #[test] + fn a_failed_drop_cancel_holds_the_window() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + { + let _fail = + crate::fail_point::FailGuard::fail("wal::append_write_aborted", "disk full"); + drop(minted); + } + assert!(replayed(&wal).contains(&lsn.as_u64()), "no marker names it"); + assert!(floor.floor() < lsn); + assert_eq!(floor.held_windows(), 1); + assert_eq!(floor.leaked_windows(), 0); + } } diff --git a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs index 26abbd17e..701df3a88 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs @@ -124,7 +124,7 @@ mod tests { ) } - fn minted_record(wal: &WalManager, floor: &Arc) -> (MintedRecords, Lsn) { + fn minted_record(wal: &Arc, floor: &Arc) -> (MintedRecords, Lsn) { let minted = MintedRecords::open(floor); let lsn = minted .appender(wal, NO_APPLY_KEY) diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index 0ebf8f2e9..46da61d60 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -185,7 +185,14 @@ pub(crate) async fn submit_write( post_apply.is_some(), ); let dispatch_outcome = match dispatched { - Ok(outcome) => outcome, + Ok(outcome) => { + // A core holds the request now. From here the records close + // from its final response, never from a drop. + if let Some(minted) = &minted { + minted.mark_sent(); + } + outcome + } Err(error) => { // The dispatcher refused the request, so no core applied it. A // dispatch refusal depends on this node's load, so the markers diff --git a/nodedb/src/control/server/http/routes/health.rs b/nodedb/src/control/server/http/routes/health.rs index d8dbc95e4..811282d4d 100644 --- a/nodedb/src/control/server/http/routes/health.rs +++ b/nodedb/src/control/server/http/routes/health.rs @@ -156,25 +156,10 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { return (status, axum::Json(body)); } - // A write window open past the longest statement deadline holds the - // outcome floor, so no checkpoint on this node advances past it. The node - // serves, so it reports degraded. - if let Some(stuck) = state - .shared - .outcome_floor - .stuck(outcome_floor_bound(&state)) - { - let body = json!({ - "status": "degraded", - "reason": "outcome_floor_stuck", - "node_id": state.shared.node_id, - "outcome_floor": stuck.floor.as_u64(), - "oldest_window_horizon": stuck.horizon.as_u64(), - "oldest_window_open_secs": stuck.open_for.as_secs(), - "open_windows": stuck.open_windows, - "leaked_windows": state.shared.outcome_floor.leaked_windows(), - "held_windows": state.shared.outcome_floor.held_windows(), - }); + // A write window open past the outcome-floor bound holds the floor, so + // no checkpoint on this node advances past it. The node serves, so it + // reports degraded. + if let Some(body) = outcome_floor_stuck_body(&state, outcome_floor_bound(&state)) { return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); } @@ -198,6 +183,27 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { (status, axum::Json(body)) } +/// The degraded body for an outcome floor held past `bound` by a window that +/// is not held, or `None` when no such window exists. +fn outcome_floor_stuck_body( + state: &AppState, + bound: std::time::Duration, +) -> Option { + let floor = &state.shared.outcome_floor; + let stuck = floor.stuck(bound)?; + Some(json!({ + "status": "degraded", + "reason": "outcome_floor_stuck", + "node_id": state.shared.node_id, + "outcome_floor": stuck.floor.as_u64(), + "oldest_window_horizon": stuck.horizon.as_u64(), + "oldest_window_open_secs": stuck.open_for.as_secs(), + "open_windows": stuck.open_windows, + "leaked_windows": floor.leaked_windows(), + "held_windows": floor.held_windows(), + })) +} + /// How long a write window can hold the outcome floor before readiness reports /// it: the longest path from a window's open to its final outcome. /// @@ -408,4 +414,63 @@ mod tests { assert_ne!(body["reason"], "calvin_apply_halted"); } + + /// The bound covers the longest statement deadline plus the apply wait, + /// and the vector install's two core dispatch deadlines. + #[tokio::test] + async fn the_outcome_floor_bound_covers_every_path_to_a_final_outcome() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + let network = &state.shared.tuning.network; + let statement = std::time::Duration::from_secs( + network + .default_deadline_secs + .max(network.copy_deadline_secs), + ) + crate::control::metadata_proposer::DEFAULT_PROPOSE_TIMEOUT; + let vector = crate::control::catalog_entry::post_apply::vector_install_longest_core_wait(); + + let bound = outcome_floor_bound(&state); + + assert!(bound >= statement); + assert!(bound >= vector); + assert_eq!(bound, statement.max(vector)); + } + + /// A window open past the bound degrades readiness and names the floor + /// it holds. + #[tokio::test] + async fn a_window_open_past_the_bound_reports_a_stuck_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + let window = state.shared.outcome_floor.open_write(); + window.note_minted(crate::types::Lsn::new(7)); + std::thread::sleep(std::time::Duration::from_millis(2)); + + let body = outcome_floor_stuck_body(&state, std::time::Duration::ZERO) + .expect("the window is older than a zero bound"); + + assert_eq!(body["reason"], "outcome_floor_stuck"); + assert_eq!(body["open_windows"], 1); + assert_eq!(body["oldest_window_horizon"], 1); + assert_eq!(body["held_windows"], 0); + window.settle(); + assert!(outcome_floor_stuck_body(&state, std::time::Duration::ZERO).is_none()); + } + + /// A held window never degrades readiness. The healthz body counts it. + #[tokio::test] + async fn a_held_window_is_reported_without_degrading_readiness() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + let window = state.shared.outcome_floor.open_write(); + window.note_minted(crate::types::Lsn::new(7)); + window.hold(); + std::thread::sleep(std::time::Duration::from_millis(2)); + + assert!(outcome_floor_stuck_body(&state, std::time::Duration::ZERO).is_none()); + let (_status, body) = healthz_body(state).await; + + assert_ne!(body["reason"], "outcome_floor_stuck"); + assert_eq!(body["held_windows"], 1); + } } diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs index 5dcce8feb..9dab6a3ab 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs @@ -217,6 +217,11 @@ async fn dispatch_plan(state: &SharedState, dispatch: PlanDispatch) -> crate::Re detail: error.to_string(), }); } + // A core holds the request now. From here the records close from its + // final response, never from a drop. + if let Some(minted) = &minted { + minted.mark_sent(); + } // Await to the same instant the envelope carries — yields the thread so the // response poller can run. Reaching that instant is the statement running diff --git a/nodedb/src/control/server/sync/raft_dispatch/write.rs b/nodedb/src/control/server/sync/raft_dispatch/write.rs index 038fd82b4..48d5864ef 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/write.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/write.rs @@ -333,4 +333,101 @@ mod tests { holder.abort(); } } + + /// A caller dropped while its frontier write waits for the sequencer + /// sent nothing to a core, so the drop cancels its records and their + /// window settles. + #[tokio::test] + async fn a_caller_dropped_in_the_sequencer_wait_cancels_its_records() { + use super::super::durability_test_support::{authorized_plan, vshard}; + + let (state, _side, _directory) = fixture(); + let (minted, lsn) = minted_buffered_record(&state); + let sequencer = Arc::clone(&state.vshard_admission_sequencer); + let holder = tokio::spawn(async move { + sequencer + .run(vshard(), std::future::pending::>) + .await + }); + for _ in 0..8 { + tokio::task::yield_now().await; + } + let authorized = authorized_plan( + &state, + crate::bridge::envelope::PhysicalPlan::Crdt( + nodedb_physical::physical_plan::CrdtOp::DocDelete { + collection: nodedb_types::QualifiedCollection::new( + crate::types::DatabaseId::DEFAULT, + COLLECTION, + ), + document_id: "d1".into(), + surrogate: nodedb_types::Surrogate::ZERO, + returning: None, + rls_filters: Vec::new(), + }, + ), + ); + + let waited = tokio::time::timeout( + Duration::from_millis(50), + dispatch_write_replicated( + &state, + COLLECTION, + authorized, + Duration::from_secs(5), + EventSource::CrdtSync, + Some(minted), + ), + ) + .await; + + assert!(waited.is_err(), "the write waits behind the running holder"); + state.wal.sync().expect("sync"); + let replayed: Vec = state + .wal + .replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect(); + assert!(!replayed.contains(&lsn.as_u64()), "a marker names it"); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + assert_eq!(state.outcome_floor.held_windows(), 0); + holder.abort(); + } + + /// The Raft entry carried the write, so its result stands although the + /// caller's records could not be cancelled. Their window is held, which + /// keeps the floor below them and reports it. + #[cfg(feature = "failpoints")] + #[tokio::test] + async fn a_failed_cancel_keeps_the_proposed_result_and_holds_the_window() { + let (state, _side, _directory) = fixture(); + let raw: Arc = + Arc::new(|_vshard, _key, _data| { + Box::pin(async { Ok((b"applied".to_vec(), crate::types::Lsn::ZERO)) }) + }); + crate::control::vshard_admission::install_async_raft_proposer(&state, raw) + .expect("install proposer"); + let (minted, lsn) = minted_buffered_record(&state); + let _fail = crate::fail_point::FailGuard::fail("wal::append_write_aborted", "disk full"); + let authorized = authorized_write(&state); + + let payload = dispatch_write_replicated( + &state, + COLLECTION, + authorized, + Duration::from_secs(5), + EventSource::CrdtSync, + Some(minted), + ) + .await + .expect("the proposal's result stands"); + + assert_eq!(payload, b"applied".to_vec()); + assert!(state.outcome_floor.floor() < lsn, "the window is held"); + assert_eq!(state.outcome_floor.held_windows(), 1); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + } } diff --git a/nodedb/src/wal/manager/append_index.rs b/nodedb/src/wal/manager/append_index.rs index 4271beb3a..d67a8af33 100644 --- a/nodedb/src/wal/manager/append_index.rs +++ b/nodedb/src/wal/manager/append_index.rs @@ -96,6 +96,11 @@ impl WalAppender<'_> { db: DatabaseId, aborted_lsn: Lsn, ) -> crate::Result { + crate::fail_point_err!("wal::append_write_aborted", |detail: String| { + crate::Error::Internal { + detail: format!("write-aborted append failed (failpoint): {detail}"), + } + }); let payload = nodedb_wal::WriteAbortedPayload::new(aborted_lsn.as_u64()).to_bytes(); self.append_record(RecordType::WriteAborted, tid, vs, db, &payload) } diff --git a/nodedb/src/wal/manager/appender.rs b/nodedb/src/wal/manager/appender.rs index 4e46790fc..3981fb451 100644 --- a/nodedb/src/wal/manager/appender.rs +++ b/nodedb/src/wal/manager/appender.rs @@ -20,13 +20,23 @@ use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; /// The apply key of a record no replicated proposal owns. pub const NO_APPLY_KEY: u64 = 0; +/// One record a recording appender wrote: its LSN and the header fields +/// that place it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RecordedAppend { + pub lsn: Lsn, + pub tenant_id: TenantId, + pub vshard_id: VShardId, + pub database_id: DatabaseId, +} + /// Appends WAL records that all carry one apply key. #[derive(Clone, Copy)] pub struct WalAppender<'a> { wal: &'a WalManager, apply_key: u64, - /// Collects the LSN of every record this appender writes, when set. - sink: Option<&'a Mutex>>, + /// Collects every record this appender writes, when set. + sink: Option<&'a Mutex>>, } impl WalManager { @@ -41,12 +51,12 @@ impl WalManager { } } - /// An appender that also pushes the LSN of every record it writes onto - /// `sink`, in append order. + /// An appender that also pushes every record it writes onto `sink`, in + /// append order. pub fn recording_appender<'a>( &'a self, apply_key: u64, - sink: &'a Mutex>, + sink: &'a Mutex>, ) -> WalAppender<'a> { WalAppender { wal: self, @@ -87,7 +97,14 @@ impl WalAppender<'_> { drop(wal); let lsn = Lsn::new(lsn); if let Some(sink) = self.sink { - sink.lock().unwrap_or_else(|p| p.into_inner()).push(lsn); + sink.lock() + .unwrap_or_else(|p| p.into_inner()) + .push(RecordedAppend { + lsn, + tenant_id, + vshard_id, + database_id, + }); } Ok(lsn) } @@ -136,7 +153,12 @@ mod tests { .expect("append"); let second = recording.append_put(t, v, db, b"b").expect("append"); - let recorded = sink.lock().expect("sink").clone(); + let recorded: Vec = sink + .lock() + .expect("sink") + .iter() + .map(|record| record.lsn) + .collect(); assert_eq!(recorded, vec![first, second]); } } diff --git a/nodedb/src/wal/manager/mod.rs b/nodedb/src/wal/manager/mod.rs index 7d7ef79e3..75438420e 100644 --- a/nodedb/src/wal/manager/mod.rs +++ b/nodedb/src/wal/manager/mod.rs @@ -15,5 +15,5 @@ pub mod encryption; pub mod ops; pub mod replay; -pub use appender::{NO_APPLY_KEY, WalAppender}; +pub use appender::{NO_APPLY_KEY, RecordedAppend, WalAppender}; pub use core::WalManager; From c2a851070bdd618cf0e952f2b7853af4d13b434d Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Thu, 24 Sep 2026 17:39:22 +0800 Subject: [PATCH 24/64] refactor(pgwire): drop unused user_id param on dispatch_task_no_wal Its one caller always passed None. --- nodedb/src/control/server/pgwire/handler/dispatch/local.rs | 3 +-- .../control/server/pgwire/handler/transaction_cmds/commit.rs | 2 +- 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs index acdcc6a27..62043d722 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs @@ -41,7 +41,6 @@ impl NodeDbPgHandler { pub(in crate::control::server::pgwire::handler) async fn dispatch_task_no_wal( &self, task: PhysicalTask, - user_id: Option>, ) -> crate::Result { // Without this, a transaction begun before the freeze could COMMIT mid-scan and // break the as-of contract. @@ -63,7 +62,7 @@ impl NodeDbPgHandler { vshard_id: task.vshard_id, database_id: task.database_id, plan: task.plan, - user_id, + user_id: None, txn_id, // No per-task TTL instant (see `flush_transaction_buffer`), so a TTL-bearing // KV write falls back to `epoch_system_ms` at apply time. diff --git a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs index d82e2f6f5..7a937d568 100644 --- a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs +++ b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs @@ -39,7 +39,7 @@ impl TxnDataPlane for PgwireTxnDp<'_> { &'a self, task: PhysicalTask, ) -> Pin> + Send + 'a>> { - Box::pin(self.handler.dispatch_task_no_wal(task, None)) + Box::pin(self.handler.dispatch_task_no_wal(task)) } fn event_source(&self) -> crate::event::EventSource { From 77b05d618c9069cc17d2a6cf29571cf275f7b40e Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 06:42:12 +0800 Subject: [PATCH 25/64] refactor(cluster): split shuffle-push streaming into its own module send.rs mixed the pooled single-attempt RPC path with the outbound streaming-shuffle helpers. Move send_shuffle_push, open_shuffle_push_stream, and ShufflePushStream into a dedicated shuffle_push module and re-export it from transport::client, leaving send.rs to the RPC send path plus the connection-target verification it now shares via a widened visibility. --- nodedb-cluster/src/transport/client/mod.rs | 3 +- nodedb-cluster/src/transport/client/send.rs | 153 ++---------------- .../src/transport/client/shuffle_push.rs | 152 +++++++++++++++++ 3 files changed, 164 insertions(+), 144 deletions(-) create mode 100644 nodedb-cluster/src/transport/client/shuffle_push.rs diff --git a/nodedb-cluster/src/transport/client/mod.rs b/nodedb-cluster/src/transport/client/mod.rs index ea7827884..8e1e8abf0 100644 --- a/nodedb-cluster/src/transport/client/mod.rs +++ b/nodedb-cluster/src/transport/client/mod.rs @@ -13,7 +13,8 @@ pub mod pool; pub mod raft_impl; pub mod send; pub mod serve; +pub mod shuffle_push; pub mod transport; -pub use send::ShufflePushStream; +pub use shuffle_push::ShufflePushStream; pub use transport::{NexarTransport, TransportPeerSnapshot}; diff --git a/nodedb-cluster/src/transport/client/send.rs b/nodedb-cluster/src/transport/client/send.rs index 4790e34de..7fe24d1c5 100644 --- a/nodedb-cluster/src/transport/client/send.rs +++ b/nodedb-cluster/src/transport/client/send.rs @@ -16,15 +16,9 @@ use futures::Stream; use rustls::pki_types::CertificateDer; use tracing::debug; -use std::sync::Arc; - use crate::circuit_breaker::RetryPolicy; use crate::error::{ClusterError, Result}; -use crate::rpc_codec::{ - self, RaftRpc, ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, TypedClusterError, - auth_envelope, -}; -use crate::transport::auth_context::AuthContext; +use crate::rpc_codec::{self, RaftRpc, TypedClusterError, auth_envelope}; use crate::transport::config::SNI_HOSTNAME; use crate::transport::peer_identity_verifier::{ VerifyOutcome, spki_pin_from_cert_der, verify_peer_identity, @@ -320,49 +314,6 @@ impl NexarTransport { }) } - /// Open a cross-node streaming-shuffle push to `target` (E1). - /// - /// Producer → receiver direction (the mirror of [`send_rpc_stream`], which - /// streams a response back): opens a bidi stream on the pooled connection, - /// writes the [`ShufflePushRequest`] envelope, then one [`ShufflePushChunk`] - /// envelope per pre-batched payload, then exactly one [`ShufflePushEnd`] - /// (clean EOF), and `finish()`es the send half. It does **not** read a - /// reply — the server deposits the chunks and never writes back, so the - /// helper fire-and-finishes. - /// - /// Each `batches` element is a standalone msgpack array of rows (the same - /// convention as `RowBatch.payload`). E4 — the planner-side caller — is - /// responsible for computing `partition_hash(row, keys) % num_parts` and - /// grouping rows into per-partition batches before calling this; E1's - /// helper takes already-partitioned payloads. - pub async fn send_shuffle_push( - &self, - target: u64, - req: ShufflePushRequest, - batches: Vec>, - ) -> Result<()> { - // Reimplemented on top of the incremental [`ShufflePushStream`] handle so - // the one-shot path and the E4a fanout sink share ONE wire encoder: open - // → push each pre-batched payload → finish with a clean EOF. - let mut stream = ShufflePushStream::open(self, target, req).await?; - for payload in batches { - stream.push_chunk(payload).await?; - } - stream.finish(None).await - } - - /// Open an incremental shuffle-push stream to `target`. - /// - /// Thin convenience wrapper around [`ShufflePushStream::open`] for callers - /// that hold `&self` (the E4a fanout sink opens one per part). - pub async fn open_shuffle_push_stream( - &self, - target: u64, - req: ShufflePushRequest, - ) -> Result { - ShufflePushStream::open(self, target, req).await - } - /// Single-attempt RPC send (no retry, no circuit breaker). async fn try_send_once( &self, @@ -433,7 +384,11 @@ impl NexarTransport { /// guard the second accept trips on its own first — the envelope /// was never replayed, the same window simply saw traffic from both /// directions for `peer_id == local_node_id`. - fn verify_connection_target(&self, conn: &quinn::Connection, target: u64) -> Result<()> { + pub(super) fn verify_connection_target( + &self, + conn: &quinn::Connection, + target: u64, + ) -> Result<()> { if !self.identity_store.enforces_peer_identity() || self.identity_store.bootstrap_window_open() { @@ -487,97 +442,6 @@ impl NexarTransport { } } -/// An incremental producer → receiver shuffle-push stream (E4a). -/// -/// The one-shot [`NexarTransport::send_shuffle_push`] writes every chunk up -/// front; the E4a fanout sink instead opens one of these per target part and -/// pushes chunks as the local scan produces rows, finishing the stream (clean -/// EOF or terminal error) only once the scan ends. It owns its `quinn::SendStream` -/// and an `Arc` clone of the transport's [`AuthContext`] so each frame is wrapped -/// with a fresh outbound `seq` exactly like the one-shot path — no borrow of the -/// transport is held for the stream's lifetime. -/// -/// Each `write_all` is awaited inline, so QUIC flow control back-pressures the -/// producer when the receiver falls behind — bounded memory, never a buffered -/// whole side. -pub struct ShufflePushStream { - send: quinn::SendStream, - auth: Arc, - target: u64, -} - -impl ShufflePushStream { - /// Open a bidi stream to `target` and write the opening - /// [`ShufflePushRequest`] envelope. The send half stays open for subsequent - /// [`push_chunk`](Self::push_chunk) calls; the receiver deposits frames and - /// never writes back. - pub async fn open( - transport: &NexarTransport, - target: u64, - req: ShufflePushRequest, - ) -> Result { - let conn = transport.get_or_connect(target).await?; - transport.verify_connection_target(&conn, target)?; - let (mut send, _recv) = conn.open_bi().await.map_err(|e| ClusterError::Transport { - detail: format!("open_bi (shuffle push) to node {target}: {e}"), - })?; - - let auth = Arc::clone(transport.auth()); - let req_envelope = wrap_with_auth(&auth, &RaftRpc::ShufflePushRequest(req))?; - send.write_all(&req_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle push request to node {target}: {e}"), - })?; - - Ok(Self { send, auth, target }) - } - - /// Write one [`ShufflePushChunk`] envelope (a standalone msgpack row array). - pub async fn push_chunk(&mut self, payload: Vec) -> Result<()> { - let chunk_envelope = wrap_with_auth( - &self.auth, - &RaftRpc::ShufflePushChunk(ShufflePushChunk { payload }), - )?; - self.send - .write_all(&chunk_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle push chunk to node {}: {e}", self.target), - }) - } - - /// Write the terminal [`ShufflePushEnd`] envelope (`error: None` for a clean - /// EOF, `Some(e)` to fail the receiver fast) and finish the send half. - pub async fn finish(mut self, error: Option) -> Result<()> { - let end_envelope = wrap_with_auth( - &self.auth, - &RaftRpc::ShufflePushEnd(ShufflePushEnd { error }), - )?; - self.send - .write_all(&end_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle push end to node {}: {e}", self.target), - })?; - self.send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish shuffle push to node {}: {e}", self.target), - })?; - Ok(()) - } -} - -/// Encode `rpc` and wrap it in an authenticated envelope with a fresh outbound -/// `seq` — the standalone form of [`NexarTransport::wrap_outbound`] for the -/// owned-[`AuthContext`] [`ShufflePushStream`]. -fn wrap_with_auth(auth: &AuthContext, rpc: &RaftRpc) -> Result> { - let inner = rpc_codec::encode(rpc, &auth.epoch)?; - let seq = auth.peer_seq_out.next(); - let mut out = Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + inner.len()); - auth_envelope::write_envelope(auth.local_node_id, seq, &inner, &auth.mac_key, &mut out)?; - Ok(out) -} - /// Map a terminal [`TypedClusterError`] carried by `ExecuteStreamEnd` into a /// [`ClusterError`] for the stream's `Err` item. /// @@ -586,5 +450,8 @@ fn wrap_with_auth(auth: &AuthContext, rpc: &RaftRpc) -> Result> { /// surfaced as-is. The original typed shape is preserved in the detail string. fn stream_terminal_error(err: TypedClusterError) -> ClusterError { let detail = format!("{err:?}"); - ClusterError::StreamTerminal { error: err, detail } + ClusterError::StreamTerminal { + error: Box::new(err), + detail, + } } diff --git a/nodedb-cluster/src/transport/client/shuffle_push.rs b/nodedb-cluster/src/transport/client/shuffle_push.rs new file mode 100644 index 000000000..413f57baa --- /dev/null +++ b/nodedb-cluster/src/transport/client/shuffle_push.rs @@ -0,0 +1,152 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Outbound streaming-shuffle push: a producer streams row batches to a +//! receiver over one QUIC bidi stream, each frame in an authenticated +//! envelope. + +use std::sync::Arc; + +use crate::error::{ClusterError, Result}; +use crate::rpc_codec::{ + self, RaftRpc, ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, TypedClusterError, + auth_envelope, +}; +use crate::transport::auth_context::AuthContext; + +use super::transport::NexarTransport; + +impl NexarTransport { + /// Open a cross-node streaming-shuffle push to `target`. + /// + /// Producer → receiver direction (the mirror of [`NexarTransport::send_rpc_stream`], which + /// streams a response back): opens a bidi stream on the pooled connection, + /// writes the [`ShufflePushRequest`] envelope, then one [`ShufflePushChunk`] + /// envelope per pre-batched payload, then exactly one [`ShufflePushEnd`] + /// (clean EOF), and `finish()`es the send half. It does **not** read a + /// reply — the server deposits the chunks and never writes back, so the + /// helper fire-and-finishes. + /// + /// Each `batches` element is a standalone msgpack array of rows (the same + /// convention as `RowBatch.payload`). The planner-side caller is + /// responsible for computing `partition_hash(row, keys) % num_parts` and + /// grouping rows into per-partition batches before calling this. This + /// helper takes already-partitioned payloads. + pub async fn send_shuffle_push( + &self, + target: u64, + req: ShufflePushRequest, + batches: Vec>, + ) -> Result<()> { + // Reimplemented on top of the incremental [`ShufflePushStream`] handle so + // the one-shot path and the fanout sink share one wire encoder: open + // → push each pre-batched payload → finish with a clean EOF. + let mut stream = ShufflePushStream::open(self, target, req).await?; + for payload in batches { + stream.push_chunk(payload).await?; + } + stream.finish(None).await + } + + /// Open an incremental shuffle-push stream to `target`. + /// + /// Thin convenience wrapper around [`ShufflePushStream::open`] for callers + /// that hold `&self` (the fanout sink opens one per part). + pub async fn open_shuffle_push_stream( + &self, + target: u64, + req: ShufflePushRequest, + ) -> Result { + ShufflePushStream::open(self, target, req).await + } +} + +/// An incremental producer → receiver shuffle-push stream. +/// +/// The one-shot [`NexarTransport::send_shuffle_push`] writes every chunk up +/// front. The fanout sink instead opens one of these per target part and +/// pushes chunks as the local scan produces rows, finishing the stream (clean +/// EOF or terminal error) only once the scan ends. It owns its `quinn::SendStream` +/// and an `Arc` clone of the transport's [`AuthContext`] so each frame is wrapped +/// with a fresh outbound `seq` exactly like the one-shot path — no borrow of the +/// transport is held for the stream's lifetime. +/// +/// Each `write_all` is awaited inline, so QUIC flow control back-pressures the +/// producer when the receiver falls behind — bounded memory, never a buffered +/// whole side. +pub struct ShufflePushStream { + send: quinn::SendStream, + auth: Arc, + target: u64, +} + +impl ShufflePushStream { + /// Open a bidi stream to `target` and write the opening + /// [`ShufflePushRequest`] envelope. The send half stays open for subsequent + /// [`push_chunk`](Self::push_chunk) calls; the receiver deposits frames and + /// never writes back. + pub async fn open( + transport: &NexarTransport, + target: u64, + req: ShufflePushRequest, + ) -> Result { + let conn = transport.get_or_connect(target).await?; + transport.verify_connection_target(&conn, target)?; + let (mut send, _recv) = conn.open_bi().await.map_err(|e| ClusterError::Transport { + detail: format!("open_bi (shuffle push) to node {target}: {e}"), + })?; + + let auth = Arc::clone(transport.auth()); + let req_envelope = wrap_with_auth(&auth, &RaftRpc::ShufflePushRequest(req))?; + send.write_all(&req_envelope) + .await + .map_err(|e| ClusterError::Transport { + detail: format!("write shuffle push request to node {target}: {e}"), + })?; + + Ok(Self { send, auth, target }) + } + + /// Write one [`ShufflePushChunk`] envelope (a standalone msgpack row array). + pub async fn push_chunk(&mut self, payload: Vec) -> Result<()> { + let chunk_envelope = wrap_with_auth( + &self.auth, + &RaftRpc::ShufflePushChunk(ShufflePushChunk { payload }), + )?; + self.send + .write_all(&chunk_envelope) + .await + .map_err(|e| ClusterError::Transport { + detail: format!("write shuffle push chunk to node {}: {e}", self.target), + }) + } + + /// Write the terminal [`ShufflePushEnd`] envelope (`error: None` for a clean + /// EOF, `Some(e)` to fail the receiver fast) and finish the send half. + pub async fn finish(mut self, error: Option) -> Result<()> { + let end_envelope = wrap_with_auth( + &self.auth, + &RaftRpc::ShufflePushEnd(ShufflePushEnd { error }), + )?; + self.send + .write_all(&end_envelope) + .await + .map_err(|e| ClusterError::Transport { + detail: format!("write shuffle push end to node {}: {e}", self.target), + })?; + self.send.finish().map_err(|e| ClusterError::Transport { + detail: format!("finish shuffle push to node {}: {e}", self.target), + })?; + Ok(()) + } +} + +/// Encode `rpc` and wrap it in an authenticated envelope with a fresh outbound +/// `seq` — the standalone form of [`NexarTransport::wrap_outbound`] for the +/// owned-[`AuthContext`] [`ShufflePushStream`]. +fn wrap_with_auth(auth: &AuthContext, rpc: &RaftRpc) -> Result> { + let inner = rpc_codec::encode(rpc, &auth.epoch)?; + let seq = auth.peer_seq_out.next(); + let mut out = Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + inner.len()); + auth_envelope::write_envelope(auth.local_node_id, seq, &inner, &auth.mac_key, &mut out)?; + Ok(out) +} From 8a988de360003e8b85ca699590fc832497911533 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 06:42:48 +0800 Subject: [PATCH 26/64] feat(physical): support staged-redo transactions and KV counter atomics Origin no longer installs a committed transaction by re-executing its sub-plans as a batch; it installs the transaction's redo record instead. Update MetaOp::TransactionBatch, CalvinExecute and StageWrite doc comments to describe the staged/redo-apply flow, and derive Eq on the document sum-target keys the redo path now compares. Add SortedIndexRead/SortedIndexSpec so a sorted-index read inside an explicit transaction can be answered from a transaction-local tree built from base rows with the transaction's staged writes folded in, routed as a new KvOp::SortedIndexTxnRead. Move KV counter/atomic-op computation (integer and float-text arithmetic, counter fault classification) out of the nodedb binary crate into a new nodedb-physical::kv_atomic module shared across engine code, pulling in rust_decimal for its float-text parsing. Add DataPlaneErrorCode::SyncRejected to carry a sync frame's rejection provenance across the wire, and box ClusterError::StreamTerminal's error payload now that TypedClusterError has grown larger variants. Add the active_sql_transaction SQLSTATE and rename the transaction-batch fail point to match the new Calvin overlay-stage code path. --- Cargo.lock | 1 + nodedb-cluster/src/error.rs | 2 +- .../src/rpc_codec/data_plane_error.rs | 13 +- nodedb-physical/Cargo.toml | 1 + nodedb-physical/src/kv_atomic/compute.rs | 543 ++++++++++++++++++ .../src/kv_atomic/counter_fault.rs | 73 +++ nodedb-physical/src/kv_atomic/error.rs | 23 + nodedb-physical/src/kv_atomic/float_text.rs | 224 ++++++++ nodedb-physical/src/kv_atomic/mod.rs | 11 + nodedb-physical/src/lib.rs | 1 + .../src/physical_plan/document/sum_target.rs | 2 + .../src/physical_plan/kv/collection.rs | 7 +- nodedb-physical/src/physical_plan/kv/mod.rs | 2 + nodedb-physical/src/physical_plan/kv/op.rs | 17 + .../src/physical_plan/kv/sorted_read.rs | 69 +++ nodedb-physical/src/physical_plan/meta.rs | 84 +-- nodedb-physical/src/physical_plan/mod.rs | 4 +- nodedb-types/src/error/sqlstate.rs | 4 + nodedb-types/src/fail_point.rs | 4 +- 19 files changed, 1035 insertions(+), 50 deletions(-) create mode 100644 nodedb-physical/src/kv_atomic/compute.rs create mode 100644 nodedb-physical/src/kv_atomic/counter_fault.rs create mode 100644 nodedb-physical/src/kv_atomic/error.rs create mode 100644 nodedb-physical/src/kv_atomic/float_text.rs create mode 100644 nodedb-physical/src/kv_atomic/mod.rs create mode 100644 nodedb-physical/src/physical_plan/kv/sorted_read.rs diff --git a/Cargo.lock b/Cargo.lock index 0c28306a1..466f5ce30 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4558,6 +4558,7 @@ dependencies = [ "nodedb-query", "nodedb-sql", "nodedb-types", + "rust_decimal", "serde", "thiserror 2.0.20", "zerompk", diff --git a/nodedb-cluster/src/error.rs b/nodedb-cluster/src/error.rs index 4650576f9..65c9e1f17 100644 --- a/nodedb-cluster/src/error.rs +++ b/nodedb-cluster/src/error.rs @@ -106,7 +106,7 @@ pub enum ClusterError { /// `dispatch_remote_stream`). `detail` is the `Debug` rendering for logs. #[error("streaming execution terminal error: {detail}")] StreamTerminal { - error: crate::rpc_codec::TypedClusterError, + error: Box, detail: String, }, diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index d84e13d86..8d8953135 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -130,9 +130,20 @@ pub enum DataPlaneErrorCode { /// The request's deadline passed before the core started it; nothing /// ran. ExpiredBeforeExecution, + /// A sync frame the validator refused for good. Nothing applied. The + /// four provenance fields name the stream position the high-water mark + /// advanced to. + SyncRejected { + violation: nodedb_types::sync::violation::ViolationType, + applied_seq: u64, + producer_id: u64, + epoch: u64, + stream_id: u64, + seq: u64, + }, } -/// Wire mirror of `nodedb::bridge::envelope::CounterFault`. +/// Wire mirror of `nodedb_physical::kv_atomic::CounterFault`. #[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub enum DataPlaneCounterFault { NotAnInteger, diff --git a/nodedb-physical/Cargo.toml b/nodedb-physical/Cargo.toml index 38af8ef5e..2dca5186c 100644 --- a/nodedb-physical/Cargo.toml +++ b/nodedb-physical/Cargo.toml @@ -16,6 +16,7 @@ nodedb-graph = { workspace = true } nodedb-query = { workspace = true } nodedb-sql = { workspace = true } nodedb-types = { workspace = true } +rust_decimal = { workspace = true } serde = { workspace = true } thiserror = { workspace = true } zerompk = { workspace = true } diff --git a/nodedb-physical/src/kv_atomic/compute.rs b/nodedb-physical/src/kv_atomic/compute.rs new file mode 100644 index 000000000..02a7e2250 --- /dev/null +++ b/nodedb-physical/src/kv_atomic/compute.rs @@ -0,0 +1,543 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Pure value computation for `INCR`/`INCR_FLOAT`/`CAS`/`GETSET`. +//! +//! Every executor computes a stored value with these functions. On Origin +//! that is the autocommit `KvEngine` methods, transaction staging, the +//! resolve handlers, and WAL replay. On Lite it is the KV write path. All of +//! them store the same bytes for the same op. +//! +//! A body has one of two shapes ([`kv_body_shape`]). A typed row (a msgpack +//! map) keeps its typed column semantics. A raw body (the single-`value` SQL +//! form, RESP `SET`) is a byte string. `INCR` and `INCR_FLOAT` read it as +//! decimal text by the Redis rules and store the result as decimal text. + +use std::collections::HashMap; + +use nodedb_query::msgpack_scan::{KvBodyShape, kv_body_shape, row_to_kv_body}; +use nodedb_types::Value; + +use super::counter_fault::CounterFault; +use super::error::AtomicComputeError; +use super::float_text; +use crate::physical_plan::KvCounterShape; + +/// The field of a typed row an atomic never targets. +const KEY_FIELD: &str = "key"; + +/// Decode a map-shaped body into its typed columns. Returns `Ok(None)` for a +/// raw body, and `TypeMismatch` for a map-shaped body that does not decode. +fn typed_row(bytes: &[u8]) -> Result>, AtomicComputeError> { + if kv_body_shape(bytes) != KvBodyShape::Map { + return Ok(None); + } + match nodedb_types::value_from_msgpack(bytes) { + Ok(Value::Object(map)) => Ok(Some(map)), + Ok(other) => Err(AtomicComputeError::TypeMismatch { + detail: format!("stored row is {}, not an object", other.type_name()), + }), + Err(e) => Err(AtomicComputeError::TypeMismatch { + detail: format!("stored row does not decode: {e}"), + }), + } +} + +/// Encode typed columns back into a map-shaped body. The fields are written +/// in key order, so every replica and every WAL replay stores the same +/// bytes. +fn encode_map(map: HashMap) -> Result, AtomicComputeError> { + row_to_kv_body(&Value::Object(map), KvBodyShape::Map).map_err(|e| AtomicComputeError::Encode { + detail: format!("typed row re-encode: {e}"), + }) +} + +/// The column an atomic reads and writes in a typed row: the first column +/// in key order that `pick` accepts, never the `key` column. +/// +/// Key order is the order the row is stored in. A `HashMap` iterates in a +/// per-process random order, so choosing by iteration order lets two +/// replicas move two different columns. +fn target_field( + map: &HashMap, + pick: impl Fn(&Value) -> Option, +) -> Option<(String, T)> { + let mut chosen: Option<(&String, T)> = None; + for (name, value) in map { + if name == KEY_FIELD { + continue; + } + if chosen.as_ref().is_some_and(|(best, _)| *best <= name) { + continue; + } + if let Some(picked) = pick(value) { + chosen = Some((name, picked)); + } + } + chosen.map(|(name, picked)| (name.clone(), picked)) +} + +/// The i64 an `INCR` reads from a typed column. +fn column_i64(value: &Value) -> Option { + match value { + Value::Integer(i) => Some(*i), + Value::Float(f) => integral_f64_to_i64(*f), + _ => None, + } +} + +/// The f64 an `INCR_FLOAT` reads from a typed column. +fn column_f64(value: &Value) -> Option { + match value { + Value::Float(f) => Some(*f), + Value::Integer(i) => Some(*i as f64), + _ => None, + } +} + +/// The string a `CAS` or `GETSET` addresses in a typed column. +fn column_string(value: &Value) -> Option { + match value { + Value::String(s) => Some(s.clone()), + _ => None, + } +} + +/// A whole `f64` inside the i64 range, as an i64. +fn integral_f64_to_i64(v: f64) -> Option { + (v.fract() == 0.0 && v >= i64::MIN as f64 && v <= i64::MAX as f64).then_some(v as i64) +} + +fn not_an_integer_column() -> AtomicComputeError { + AtomicComputeError::TypeMismatch { + detail: "row has no integer column".into(), + } +} + +fn not_a_numeric_column() -> AtomicComputeError { + AtomicComputeError::TypeMismatch { + detail: "row has no numeric column".into(), + } +} + +/// Read a raw body as a decimal i64 by the Redis rule. +fn parse_raw_i64(bytes: &[u8]) -> Result { + std::str::from_utf8(bytes) + .ok() + .filter(|text| is_canonical_integer(text)) + .and_then(|text| text.parse::().ok()) + .ok_or(AtomicComputeError::Counter(CounterFault::NotAnInteger)) +} + +/// The Redis integer grammar: `0`, or an optional `-` then digits with no +/// leading zero. A `+` sign, whitespace, and an empty body are refused. +fn is_canonical_integer(text: &str) -> bool { + let digits = text.strip_prefix('-').unwrap_or(text); + text == "0" + || (digits + .bytes() + .next() + .is_some_and(|b| (b'1'..=b'9').contains(&b)) + && digits.bytes().all(|b| b.is_ascii_digit())) +} + +/// The raw body for an integer: its decimal text, the text +/// `scalar_to_raw_bytes` writes for the same value. +fn raw_decimal(v: i64) -> Vec { + v.to_string().into_bytes() +} + +/// The row an absent key becomes under a typed [`KvCounterShape`]: the +/// template with `column` set to `value`. +fn fresh_typed_row( + column: &Option, + template: &[u8], + value: Value, + missing_column: AtomicComputeError, +) -> Result, AtomicComputeError> { + let column = column.as_ref().ok_or(missing_column)?; + let mut map = typed_row(template)?.ok_or(AtomicComputeError::TypeMismatch { + detail: "fresh row template is not a typed row".into(), + })?; + map.insert(column.clone(), value); + encode_map(map) +} + +/// Compute the new value for `INCR`, given the current body (if any). +/// Returns `(new_i64, new_bytes)`. +/// +/// A typed row keeps its shape: the integer column [`target_field`] picks +/// moves, and every other column stays. A raw body is decimal text in and +/// decimal text out. An absent key starts at 0 and takes `shape`. +pub fn incr( + current: Option<&[u8]>, + delta: i64, + shape: &KvCounterShape, +) -> Result<(i64, Vec), AtomicComputeError> { + let overflow = AtomicComputeError::Counter(CounterFault::IntegerOverflow); + let Some(bytes) = current else { + let written = match shape { + KvCounterShape::Raw => raw_decimal(delta), + KvCounterShape::Typed { column, template } => fresh_typed_row( + column, + template, + Value::Integer(delta), + not_an_integer_column(), + )?, + }; + return Ok((delta, written)); + }; + if let Some(mut map) = typed_row(bytes)? { + let (field, old_i64) = target_field(&map, column_i64).ok_or(not_an_integer_column())?; + let new_i64 = old_i64.checked_add(delta).ok_or(overflow)?; + map.insert(field, Value::Integer(new_i64)); + return Ok((new_i64, encode_map(map)?)); + } + let new_i64 = parse_raw_i64(bytes)?.checked_add(delta).ok_or(overflow)?; + Ok((new_i64, raw_decimal(new_i64))) +} + +/// Compute the new value for `INCR_FLOAT`. `delta` is the client's decimal +/// text. Returns `(new_f64, new_bytes)`. +/// +/// A typed row keeps its shape, as in [`incr`], and its column adds in +/// `f64`. A raw body is decimal text in and decimal text out, added exactly +/// by the Redis rules (see `float_text`). An absent key starts at 0 and takes +/// `shape`. +pub fn incr_float( + current: Option<&[u8]>, + delta: &str, + shape: &KvCounterShape, +) -> Result<(f64, Vec), AtomicComputeError> { + let Some(bytes) = current else { + return match shape { + KvCounterShape::Raw => float_text::fresh(delta), + KvCounterShape::Typed { column, template } => { + let value = float_text::delta_to_f64(delta)?; + let written = fresh_typed_row( + column, + template, + Value::Float(value), + not_a_numeric_column(), + )?; + Ok((value, written)) + } + }; + }; + let Some(mut map) = typed_row(bytes)? else { + return float_text::add(bytes, delta); + }; + let delta = float_text::delta_to_f64(delta)?; + let (field, old_f64) = target_field(&map, column_f64).ok_or(not_a_numeric_column())?; + let new_f64 = old_f64 + delta; + if !new_f64.is_finite() { + return Err(AtomicComputeError::Counter(CounterFault::NonFinite)); + } + map.insert(field, Value::Float(new_f64)); + Ok((new_f64, encode_map(map)?)) +} + +/// Write `new_value` into the string column of the typed row `row` and +/// encode it. `column` is the column [`target_field`] picked. +fn swap_string_column( + mut row: HashMap, + column: String, + new_value: &[u8], +) -> Result, AtomicComputeError> { + row.insert( + column, + Value::String(String::from_utf8_lossy(new_value).into_owned()), + ); + encode_map(row) +} + +/// A typed row and its string column, when `current` is a typed row with +/// one. [`cas`] and [`getset`] address the same column. +fn string_column(current: Option<&[u8]>) -> Option<(HashMap, String, String)> { + let row = typed_row(current?).ok().flatten()?; + let (column, text) = target_field(&row, column_string)?; + Some((row, column, text)) +} + +/// Compute the CAS outcome: whether `expected` matches the current value, +/// and the bytes to write when it does. +/// +/// The current value matches when its bytes equal `expected`, or when it is +/// a typed row whose string column holds `expected`. A typed row with a +/// string column keeps its shape: only that column is swapped. +pub fn cas( + current: Option<&[u8]>, + expected: &[u8], + new_value: &[u8], +) -> Result<(bool, Vec), AtomicComputeError> { + let Some(cur) = current else { + return Ok(if expected.is_empty() { + (true, new_value.to_vec()) + } else { + (false, Vec::new()) + }); + }; + let typed = string_column(current); + let column_matches = typed + .as_ref() + .is_some_and(|(_, _, text)| *text == String::from_utf8_lossy(expected)); + if cur != expected && !column_matches { + return Ok((false, Vec::new())); + } + let write_bytes = match typed { + Some((row, column, _)) => swap_string_column(row, column, new_value)?, + None => new_value.to_vec(), + }; + Ok((true, write_bytes)) +} + +/// Compute the bytes to write for `GETSET`: the string column of a typed +/// row swapped in place, or a plain overwrite. +pub fn getset(current: Option<&[u8]>, new_value: &[u8]) -> Result, AtomicComputeError> { + match string_column(current) { + Some((row, column, _)) => swap_string_column(row, column, new_value), + None => Ok(new_value.to_vec()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + static RAW: KvCounterShape = KvCounterShape::Raw; + + /// A typed shape moving `column`, with `rest` as the other stored columns. + fn typed_shape(column: Option<&str>, rest: &[(&str, Value)]) -> KvCounterShape { + KvCounterShape::Typed { + column: column.map(str::to_string), + template: row(rest), + } + } + + fn row(fields: &[(&str, Value)]) -> Vec { + let map: HashMap = fields + .iter() + .map(|(k, v)| ((*k).to_string(), v.clone())) + .collect(); + nodedb_types::value_to_msgpack(&Value::Object(map)).expect("encode row") + } + + fn columns(bytes: &[u8]) -> HashMap { + typed_row(bytes) + .expect("a typed row decodes") + .expect("a typed row stays a typed row") + } + + #[test] + fn incr_on_a_one_column_typed_row_keeps_the_row() { + let current = row(&[("n", Value::Integer(5))]); + let (new_i64, bytes) = incr(Some(¤t), 3, &RAW).expect("incr"); + assert_eq!(new_i64, 8); + assert_eq!(columns(&bytes).get("n"), Some(&Value::Integer(8))); + } + + #[test] + fn incr_moves_the_first_numeric_column_in_key_order() { + let current = row(&[ + ("b", Value::Integer(100)), + ("a", Value::Integer(1)), + ("label", Value::String("x".into())), + ]); + let (new_i64, bytes) = incr(Some(¤t), 1, &RAW).expect("incr"); + assert_eq!(new_i64, 2); + let cols = columns(&bytes); + assert_eq!(cols.get("a"), Some(&Value::Integer(2))); + assert_eq!(cols.get("b"), Some(&Value::Integer(100))); + assert_eq!(cols.get("label"), Some(&Value::String("x".into()))); + } + + #[test] + fn incr_on_a_typed_row_encodes_the_same_bytes_every_time() { + let current = row(&[ + ("a", Value::Integer(1)), + ("b", Value::Integer(2)), + ("c", Value::Integer(3)), + ]); + let (_, first) = incr(Some(¤t), 1, &RAW).expect("incr"); + for _ in 0..16 { + let (_, again) = incr(Some(¤t), 1, &RAW).expect("incr"); + assert_eq!(again, first); + } + } + + #[test] + fn incr_on_a_typed_row_without_a_numeric_column_is_a_type_mismatch() { + let current = row(&[("label", Value::String("x".into()))]); + assert!(matches!( + incr(Some(¤t), 1, &RAW), + Err(AtomicComputeError::TypeMismatch { .. }) + )); + } + + #[test] + fn incr_on_a_raw_body_reads_and_writes_decimal_text() { + let (new_i64, bytes) = incr(Some(b"5"), 1, &RAW).expect("incr"); + assert_eq!(new_i64, 6); + assert_eq!(bytes, b"6".to_vec()); + + let (new_i64, bytes) = incr(Some(b"-10"), 3, &RAW).expect("incr"); + assert_eq!(new_i64, -7); + assert_eq!(bytes, b"-7".to_vec()); + + let (fresh, bytes) = incr(None, 4, &RAW).expect("incr"); + assert_eq!(fresh, 4); + assert_eq!(bytes, b"4".to_vec()); + } + + #[test] + fn incr_on_non_integer_raw_text_is_not_an_integer() { + for body in [ + b"abc".as_slice(), + b"", + b"1.5", + b"+5", + b"05", + b"-0", + b" 5", + b"5 ", + b"99999999999999999999", + ] { + assert!( + matches!( + incr(Some(body), 1, &RAW), + Err(AtomicComputeError::Counter(CounterFault::NotAnInteger)) + ), + "{:?}", + String::from_utf8_lossy(body) + ); + } + } + + #[test] + fn incr_past_the_i64_range_is_an_overflow() { + let max = i64::MAX.to_string(); + assert!(matches!( + incr(Some(max.as_bytes()), 1, &RAW), + Err(AtomicComputeError::Counter(CounterFault::IntegerOverflow)) + )); + let min = i64::MIN.to_string(); + assert!(matches!( + incr(Some(min.as_bytes()), -1, &RAW), + Err(AtomicComputeError::Counter(CounterFault::IntegerOverflow)) + )); + let (value, bytes) = incr(Some(min.as_bytes()), 0, &RAW).expect("i64::MIN parses"); + assert_eq!(value, i64::MIN); + assert_eq!(bytes, min.into_bytes()); + } + + #[test] + fn incr_float_on_a_raw_body_reads_and_writes_decimal_text() { + let (new_f64, bytes) = incr_float(Some(b"1.5"), "1", &RAW).expect("incr_float"); + assert_eq!(new_f64, 2.5); + assert_eq!(bytes, b"2.5".to_vec()); + + let (new_f64, bytes) = incr_float(Some(b"10.5"), "0.5", &RAW).expect("incr_float"); + assert_eq!(new_f64, 11.0); + assert_eq!(bytes, b"11".to_vec()); + + let (_, bytes) = incr_float(Some(b"5"), "0.25", &RAW).expect("incr_float"); + assert_eq!(bytes, b"5.25".to_vec()); + + for (stored, delta, expected) in [ + ("0.1", "0.2", "0.3"), + ("10.5", "0.1", "10.6"), + ("5.0e3", "200", "5200"), + ("3.0", "0", "3"), + ("-1.5", "1.5", "0"), + ("1", "0.12345678901234567891", "1.12345678901234567891"), + ] { + let (_, bytes) = incr_float(Some(stored.as_bytes()), delta, &RAW).expect("incr_float"); + assert_eq!(bytes, expected.as_bytes().to_vec(), "{stored} + {delta}"); + } + } + + #[test] + fn incr_float_on_non_numeric_raw_text_is_not_a_float() { + for body in [b"abc".as_slice(), b"", b"NaN", b" 1.5"] { + assert!( + matches!( + incr_float(Some(body), "1", &RAW), + Err(AtomicComputeError::Counter(CounterFault::NotAFloat)) + ), + "{:?}", + String::from_utf8_lossy(body) + ); + } + } + + #[test] + fn incr_float_to_infinity_is_non_finite() { + let max = f64::MAX.to_string(); + assert!(matches!( + incr_float(Some(max.as_bytes()), &max, &RAW), + Err(AtomicComputeError::Counter(CounterFault::NonFinite)) + )); + } + + #[test] + fn incr_float_on_a_one_column_typed_row_keeps_the_row() { + let current = row(&[("score", Value::Float(1.5))]); + let (new_f64, bytes) = incr_float(Some(¤t), "1", &RAW).expect("incr_float"); + assert_eq!(new_f64, 2.5); + assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); + } + + #[test] + fn incr_on_an_absent_key_under_a_typed_shape_creates_the_typed_row() { + let shape = typed_shape(Some("n"), &[("status", Value::String("new".into()))]); + let (value, bytes) = incr(None, 7, &shape).expect("incr"); + assert_eq!(value, 7); + let cols = columns(&bytes); + assert_eq!(cols.get("n"), Some(&Value::Integer(7))); + assert_eq!(cols.get("status"), Some(&Value::String("new".into()))); + } + + #[test] + fn incr_float_on_an_absent_key_under_a_typed_shape_creates_the_typed_row() { + let shape = typed_shape(Some("score"), &[]); + let (value, bytes) = incr_float(None, "2.5", &shape).expect("incr_float"); + assert_eq!(value, 2.5); + assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); + } + + #[test] + fn an_absent_key_under_a_typed_shape_without_a_column_is_a_type_mismatch() { + let shape = typed_shape(None, &[]); + assert!(matches!( + incr(None, 1, &shape), + Err(AtomicComputeError::TypeMismatch { .. }) + )); + assert!(matches!( + incr_float(None, "1", &shape), + Err(AtomicComputeError::TypeMismatch { .. }) + )); + } + + #[test] + fn cas_on_a_one_column_typed_row_swaps_the_column() { + let current = row(&[("state", Value::String("idle".into()))]); + let (matched, bytes) = cas(Some(¤t), b"idle", b"busy").expect("cas"); + assert!(matched); + assert_eq!( + columns(&bytes).get("state"), + Some(&Value::String("busy".into())) + ); + let (matched, _) = cas(Some(¤t), b"busy", b"idle").expect("cas"); + assert!(!matched); + } + + #[test] + fn getset_on_a_one_column_typed_row_swaps_the_column() { + let current = row(&[("token", Value::String("old".into()))]); + let bytes = getset(Some(¤t), b"new").expect("getset"); + assert_eq!( + columns(&bytes).get("token"), + Some(&Value::String("new".into())) + ); + assert_eq!(getset(None, b"raw").expect("getset"), b"raw".to_vec()); + } +} diff --git a/nodedb-physical/src/kv_atomic/counter_fault.rs b/nodedb-physical/src/kv_atomic/counter_fault.rs new file mode 100644 index 000000000..4b950127c --- /dev/null +++ b/nodedb-physical/src/kv_atomic/counter_fault.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Why a KV counter atomic (`INCR`, `INCRBY`, `DECR`, `INCRBYFLOAT`) +//! computed no value. + +/// Why a KV counter atomic computed no value. +/// +/// Each fault has one client message, the text Redis answers with for the +/// same condition. RESP sends it after `ERR`. The SQL surfaces send it with +/// the SQLSTATE [`CounterFault::sqlstate`] names. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CounterFault { + /// The stored value is not a decimal integer in the `i64` range. + NotAnInteger, + /// The stored value is not a decimal float. + NotAFloat, + /// The integer result is outside the `i64` range. + IntegerOverflow, + /// The float result is NaN or infinite. + NonFinite, +} + +impl CounterFault { + /// The client message, without the RESP `ERR` prefix. + pub fn message(self) -> &'static str { + match self { + Self::NotAnInteger => "value is not an integer or out of range", + Self::NotAFloat => "value is not a valid float", + Self::IntegerOverflow => "increment or decrement would overflow", + Self::NonFinite => "increment would produce NaN or Infinity", + } + } + + /// Whether the fault is a result out of range. A fault that is not is a + /// stored value that does not parse. + pub fn is_out_of_range(self) -> bool { + matches!(self, Self::IntegerOverflow | Self::NonFinite) + } + + /// The SQLSTATE for the fault: `22P02` for a stored value that does not + /// parse, `22003` for a result out of range. + pub fn sqlstate(self) -> &'static str { + use nodedb_types::error::sqlstate; + if self.is_out_of_range() { + sqlstate::NUMERIC_VALUE_OUT_OF_RANGE + } else { + sqlstate::INVALID_TEXT_REPRESENTATION + } + } +} + +impl std::fmt::Display for CounterFault { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(self.message()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parse_faults_answer_invalid_text_representation() { + assert_eq!(CounterFault::NotAnInteger.sqlstate(), "22P02"); + assert_eq!(CounterFault::NotAFloat.sqlstate(), "22P02"); + } + + #[test] + fn range_faults_answer_numeric_value_out_of_range() { + assert_eq!(CounterFault::IntegerOverflow.sqlstate(), "22003"); + assert_eq!(CounterFault::NonFinite.sqlstate(), "22003"); + } +} diff --git a/nodedb-physical/src/kv_atomic/error.rs b/nodedb-physical/src/kv_atomic/error.rs new file mode 100644 index 000000000..60d66abf2 --- /dev/null +++ b/nodedb-physical/src/kv_atomic/error.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Why a KV atomic computed no stored value. + +use super::counter_fault::CounterFault; + +/// Why [`super::compute`] computed no stored value for a KV atomic. +/// +/// Each executor maps it into its own error. Origin maps it into the Data +/// Plane `ErrorCode`, and Lite maps it into `LiteError`. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum AtomicComputeError { + /// A typed row has no column of the type the atomic reads. + #[error("{detail}")] + TypeMismatch { detail: String }, + /// A counter atomic read a stored value it cannot parse as a number, or + /// computed a result out of range. + #[error("{0}")] + Counter(CounterFault), + /// The computed new value failed to re-encode as MessagePack. + #[error("{detail}")] + Encode { detail: String }, +} diff --git a/nodedb-physical/src/kv_atomic/float_text.rs b/nodedb-physical/src/kv_atomic/float_text.rs new file mode 100644 index 000000000..4592122d5 --- /dev/null +++ b/nodedb-physical/src/kv_atomic/float_text.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! `INCRBYFLOAT` on a raw KV body: decimal text in, decimal text out. +//! +//! Redis adds in `long double` and prints the sum with 17 fractional digits, +//! trailing zeros trimmed, so `"0.1"` plus `0.2` stores `"0.3"`. Rust has no +//! `long double`. Exact decimal addition gives the same text for every sum +//! that fits a [`Decimal`]: 28 significant digits, magnitude below 7.9e28. +//! An operand or a sum outside that range is added in `f64` instead. + +use std::str::FromStr; + +use rust_decimal::Decimal; +use rust_decimal::prelude::ToPrimitive; + +use super::counter_fault::CounterFault; +use super::error::AtomicComputeError; + +/// Add `delta` to the raw body `stored`. Returns the new value and the text +/// to store. +/// +/// `stored` and `delta` must both be decimal numbers (see +/// [`is_decimal_number`]). Anything else is `Counter(NotAFloat)`. A sum that +/// is not finite is `Counter(NonFinite)`. +pub(super) fn add(stored: &[u8], delta: &str) -> Result<(f64, Vec), AtomicComputeError> { + let text = std::str::from_utf8(stored) + .ok() + .filter(|text| is_decimal_number(text)) + .ok_or(AtomicComputeError::Counter(CounterFault::NotAFloat))?; + let delta_f64 = delta_to_f64(delta)?; + if let (Some(base), Some(step)) = (parse_decimal(text), parse_decimal(delta)) + && let Some(sum) = base.checked_add(step) + { + let sum = sum.normalize(); + let value = sum + .to_f64() + .ok_or(AtomicComputeError::Counter(CounterFault::NonFinite))?; + return Ok((value, sum.to_string().into_bytes())); + } + let base: f64 = text + .parse() + .map_err(|_| AtomicComputeError::Counter(CounterFault::NotAFloat))?; + let value = base + delta_f64; + if !value.is_finite() { + return Err(AtomicComputeError::Counter(CounterFault::NonFinite)); + } + Ok((value, float_text(value))) +} + +/// The text for a fresh float counter: `0` plus `delta`. +pub(super) fn fresh(delta: &str) -> Result<(f64, Vec), AtomicComputeError> { + add(b"0", delta) +} + +/// `delta` as a finite `f64`, for a typed column. A delta that is not a +/// decimal number is `Counter(NotAFloat)`. One outside the `f64` range is +/// `Counter(NonFinite)`. +pub fn delta_to_f64(delta: &str) -> Result { + if !is_decimal_number(delta) { + return Err(AtomicComputeError::Counter(CounterFault::NotAFloat)); + } + let value: f64 = delta + .parse() + .map_err(|_| AtomicComputeError::Counter(CounterFault::NotAFloat))?; + if value.is_finite() { + Ok(value) + } else { + Err(AtomicComputeError::Counter(CounterFault::NonFinite)) + } +} + +/// The exact value of `text`, or `None` when it does not fit a [`Decimal`]. +fn parse_decimal(text: &str) -> Option { + if text.contains(['e', 'E']) { + Decimal::from_scientific(text).ok() + } else { + Decimal::from_str(text).ok() + } +} + +/// The text for an `f64` sum outside the [`Decimal`] range: plain decimal +/// digits with no exponent, the form Redis prints. +fn float_text(value: f64) -> Vec { + let text = value.to_string(); + if text == "-0" { + b"0".to_vec() + } else { + text.into_bytes() + } +} + +/// The number grammar `INCRBYFLOAT` accepts, for a stored body and for the +/// client's increment: an optional sign, then digits with an optional point +/// (at least one digit), then an optional exponent `e` or `E` with an +/// optional sign and at least one digit. No whitespace, digit separators, +/// `inf`, or `nan`. +pub fn is_decimal_number(text: &str) -> bool { + let bytes = text.as_bytes(); + let mut i = 0; + if matches!(bytes.first(), Some(b'+' | b'-')) { + i += 1; + } + let int_digits = count_digits(&bytes[i..]); + i += int_digits; + let mut frac_digits = 0; + if bytes.get(i) == Some(&b'.') { + i += 1; + frac_digits = count_digits(&bytes[i..]); + i += frac_digits; + } + if int_digits + frac_digits == 0 { + return false; + } + if matches!(bytes.get(i), Some(b'e' | b'E')) { + i += 1; + if matches!(bytes.get(i), Some(b'+' | b'-')) { + i += 1; + } + let exp_digits = count_digits(&bytes[i..]); + if exp_digits == 0 { + return false; + } + i += exp_digits; + } + i == bytes.len() +} + +fn count_digits(bytes: &[u8]) -> usize { + bytes.iter().take_while(|b| b.is_ascii_digit()).count() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn text_of(stored: &str, delta: &str) -> String { + let (_, bytes) = add(stored.as_bytes(), delta).expect("add"); + String::from_utf8(bytes).expect("UTF-8") + } + + #[test] + fn decimal_text_adds_exactly_like_redis() { + assert_eq!(text_of("0.1", "0.2"), "0.3"); + assert_eq!(text_of("10.5", "0.1"), "10.6"); + assert_eq!(text_of("5.0e3", "200"), "5200"); + assert_eq!(text_of("3.0", "0"), "3"); + assert_eq!(text_of("-1.5", "1.5"), "0"); + assert_eq!(text_of("1.5", "1"), "2.5"); + assert_eq!(text_of("+2", "-0.5"), "1.5"); + assert_eq!(text_of("1E-2", "0"), "0.01"); + assert_eq!(text_of("1", "1e1"), "11"); + } + + #[test] + fn a_twenty_digit_delta_adds_exactly() { + assert_eq!( + text_of("1", "0.12345678901234567891"), + "1.12345678901234567891" + ); + assert_eq!(text_of("10000000000000000000", "1"), "10000000000000000001"); + } + + #[test] + fn the_returned_value_matches_the_stored_text() { + let (value, bytes) = add(b"0.1", "0.2").expect("add"); + assert_eq!(value, 0.3); + assert_eq!(bytes, b"0.3".to_vec()); + } + + #[test] + fn a_fresh_counter_stores_the_delta_text() { + assert_eq!(fresh("2.5").expect("fresh").1, b"2.5".to_vec()); + assert_eq!(fresh("0").expect("fresh").1, b"0".to_vec()); + assert_eq!(fresh("-0.0").expect("fresh").1, b"0".to_vec()); + } + + #[test] + fn a_sum_outside_the_decimal_range_adds_in_f64() { + let (value, bytes) = add(b"1e300", "1").expect("add"); + assert_eq!(value, 1e300); + assert_eq!(bytes, 1e300f64.to_string().into_bytes()); + } + + #[test] + fn text_that_is_not_a_number_is_not_a_float() { + for stored in [ + "abc", "", "NaN", "inf", " 1.5", "1.5 ", "1_000", ".", "1e", "e5", "0x10", + ] { + assert!( + matches!( + add(stored.as_bytes(), "1"), + Err(AtomicComputeError::Counter(CounterFault::NotAFloat)) + ), + "{stored:?}" + ); + } + } + + #[test] + fn a_non_finite_sum_is_refused() { + let max = f64::MAX.to_string(); + assert!(matches!( + add(max.as_bytes(), &max), + Err(AtomicComputeError::Counter(CounterFault::NonFinite)) + )); + assert!(matches!( + add(b"1", "1e400"), + Err(AtomicComputeError::Counter(CounterFault::NonFinite)) + )); + } + + #[test] + fn a_delta_that_is_not_a_number_is_not_a_float() { + for delta in ["abc", "", "inf", "NaN", " 1"] { + assert!( + matches!( + add(b"1", delta), + Err(AtomicComputeError::Counter(CounterFault::NotAFloat)) + ), + "{delta:?}" + ); + } + } +} diff --git a/nodedb-physical/src/kv_atomic/mod.rs b/nodedb-physical/src/kv_atomic/mod.rs new file mode 100644 index 000000000..dc4968b07 --- /dev/null +++ b/nodedb-physical/src/kv_atomic/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Value semantics of the KV atomics, shared by every executor. + +pub mod compute; +pub mod counter_fault; +pub mod error; +pub mod float_text; + +pub use counter_fault::CounterFault; +pub use error::AtomicComputeError; diff --git a/nodedb-physical/src/lib.rs b/nodedb-physical/src/lib.rs index 9b7489305..301925bb9 100644 --- a/nodedb-physical/src/lib.rs +++ b/nodedb-physical/src/lib.rs @@ -11,6 +11,7 @@ pub mod convert_context; pub mod error; +pub mod kv_atomic; pub mod physical_plan; pub mod physical_task; pub mod surrogate; diff --git a/nodedb-physical/src/physical_plan/document/sum_target.rs b/nodedb-physical/src/physical_plan/document/sum_target.rs index ac8e01179..bb619a42d 100644 --- a/nodedb-physical/src/physical_plan/document/sum_target.rs +++ b/nodedb-physical/src/physical_plan/document/sum_target.rs @@ -61,6 +61,7 @@ impl SumTargetKey { Debug, Clone, PartialEq, + Eq, serde::Serialize, serde::Deserialize, zerompk::ToMessagePack, @@ -132,6 +133,7 @@ impl ResolvedSumTarget { Debug, Clone, PartialEq, + Eq, serde::Serialize, serde::Deserialize, zerompk::ToMessagePack, diff --git a/nodedb-physical/src/physical_plan/kv/collection.rs b/nodedb-physical/src/physical_plan/kv/collection.rs index c309af459..2469250e8 100644 --- a/nodedb-physical/src/physical_plan/kv/collection.rs +++ b/nodedb-physical/src/physical_plan/kv/collection.rs @@ -8,8 +8,8 @@ use super::op::KvOp; impl KvOp { - /// The user collection this op targets, if any. Sorted-index ops (keyed - /// only by index name) and `ResolvedWrite` (mutations may span two + /// The user collection this op targets, if any. Sorted-index ops keyed + /// only by index name and `ResolvedWrite` (mutations may span two /// collections) return `None`. `TransferItem` reports its source. pub fn collection(&self) -> Option<&str> { match self { @@ -38,7 +38,8 @@ impl KvOp { | KvOp::RegisterSortedIndex { collection, .. } | KvOp::PredicateUpdate { collection, .. } | KvOp::PredicateDelete { collection, .. } - | KvOp::MaterializeScan { collection, .. } => Some(collection.as_str()), + | KvOp::MaterializeScan { collection, .. } + | KvOp::SortedIndexTxnRead { collection, .. } => Some(collection.as_str()), KvOp::TransferItem { source_collection, .. } => Some(source_collection.as_str()), diff --git a/nodedb-physical/src/physical_plan/kv/mod.rs b/nodedb-physical/src/physical_plan/kv/mod.rs index 7e18a7612..bbbdc278a 100644 --- a/nodedb-physical/src/physical_plan/kv/mod.rs +++ b/nodedb-physical/src/physical_plan/kv/mod.rs @@ -6,7 +6,9 @@ pub mod collection; pub mod counter_shape; pub mod op; pub mod resolved_mutation; +pub mod sorted_read; pub use counter_shape::KvCounterShape; pub use op::KvOp; pub use resolved_mutation::{KvResolveOutcome, KvResolvedMutation}; +pub use sorted_read::{SortedIndexRead, SortedIndexSpec}; diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index 1a73a5c99..c65772208 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -6,6 +6,7 @@ use nodedb_types::{QualifiedCollection, RlsWriteCheck, Surrogate}; use super::counter_shape::KvCounterShape; use super::resolved_mutation::KvResolvedMutation; +use super::sorted_read::{SortedIndexRead, SortedIndexSpec}; use crate::physical_plan::document::ReturningSpec; /// KV engine physical operations. @@ -475,6 +476,22 @@ pub enum KvOp { primary_key: Vec, }, + /// A sorted-index read inside an explicit transaction block. + /// + /// The Data Plane answers it from a transaction-local tree. It builds the + /// tree from the collection's base rows with the transaction's staged + /// writes folded in, so the read sees the transaction's own writes and an + /// index the transaction created. Routed by `collection`, the core that + /// holds the rows. + SortedIndexTxnRead { + collection: QualifiedCollection, + index_name: String, + /// The definition of an index this transaction created and has not + /// committed. `None` reads the definition registered on the core. + pending: Option, + read: SortedIndexRead, + }, + /// Cursor-paginated raw scan for the clone materializer. /// /// Unlike `Scan`, this returns raw `(key, value)` byte pairs **plus** the diff --git a/nodedb-physical/src/physical_plan/kv/sorted_read.rs b/nodedb-physical/src/physical_plan/kv/sorted_read.rs new file mode 100644 index 000000000..6bfb5b31c --- /dev/null +++ b/nodedb-physical/src/physical_plan/kv/sorted_read.rs @@ -0,0 +1,69 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The payload of a sorted-index read inside an explicit transaction. +//! +//! A transaction must see its own DDL and its own writes. A sorted index that +//! the transaction created has no tree before COMMIT, and a committed index's +//! tree holds none of the transaction's staged writes. The Data Plane +//! therefore answers such a read from a transaction-local tree. It builds that +//! tree from the collection's base rows with the transaction's staged writes +//! folded in. + +/// The definition `CREATE SORTED INDEX` registers, carried for an index the +/// transaction created and has not committed. +/// +/// The fields match `KvOp::RegisterSortedIndex`, so the Data Plane builds the +/// transaction-local definition with the same code that builds a registered one. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct SortedIndexSpec { + /// Sort columns: (column_name, direction "ASC"/"DESC"). + pub sort_columns: Vec<(String, String)>, + /// Primary key column name. + pub key_column: String, + /// Window type: "none", "daily", "weekly", "monthly", or "custom". + pub window_type: String, + /// Window timestamp column (empty if window_type == "none"). + pub window_timestamp_column: String, + /// Custom window start (ms since epoch, 0 if N/A). + pub window_start_ms: u64, + /// Custom window end (ms since epoch, 0 if N/A). + pub window_end_ms: u64, +} + +/// Which sorted-index read a transaction runs. +/// +/// Each arm answers exactly like its autocommit counterpart: +/// `SortedIndexRank`, `SortedIndexTopK`, `SortedIndexRange`, +/// `SortedIndexCount` and `SortedIndexScore`. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum SortedIndexRead { + /// The 1-based rank of one key. + Rank { primary_key: Vec }, + /// The top `k` entries. + TopK { k: u32 }, + /// The entries whose leading sort column lies in the score range. + Range { + score_min: Option>, + score_max: Option>, + }, + /// The number of entries. + Count, + /// The sort key of one key (ZSCORE equivalent). + Score { primary_key: Vec }, +} diff --git a/nodedb-physical/src/physical_plan/meta.rs b/nodedb-physical/src/physical_plan/meta.rs index 9b51156a3..5b2519435 100644 --- a/nodedb-physical/src/physical_plan/meta.rs +++ b/nodedb-physical/src/physical_plan/meta.rs @@ -36,15 +36,15 @@ pub enum MetaOp { target_request_id: nodedb_types::id::RequestId, }, - /// Atomic transaction batch: execute all sub-plans atomically. + /// A transaction's plans as one batch. An embedded (Lite) engine executes + /// the sub-plans atomically. An Origin core refuses it: a committed + /// transaction installs there only through its redo record + /// (`ApplyTransactionRedo`, `CalvinFlush`). /// /// `txn_id` identifies the committing session transaction whose staging - /// overlay holds the resolve-time bitemporal stamps this install must reuse - /// (so a `bitemporal=true` document put lands on the same version key the - /// redo carries, not a fresh one). `None` for install paths with no session - /// overlay to consult — Calvin (which threads its stamps in directly) and - /// procedural/test callers. Wire-additive: defaults to `None` on decode of - /// older entries. + /// overlay holds the resolve-time bitemporal stamps an install must reuse. + /// `None` for callers with no session overlay to consult. Wire-additive: + /// defaults to `None` on decode of older entries. TransactionBatch { plans: Vec, #[serde(default)] @@ -292,9 +292,10 @@ pub enum MetaOp { /// /// The Calvin scheduler dispatches this variant after lock acquisition for /// transactions whose read/write set is fully known at submission time (the - /// common case). The Data Plane handler executes `plans` atomically (same - /// semantics as `TransactionBatch`) and the scheduler writes a - /// `WalRecord::CalvinApplied` after a successful response. + /// common case). The Data Plane handler validates the read-set and stages + /// `plans` without mutating base. Once the global verdict is commit, the + /// scheduler resolves the staged plans into a redo record, appends it, and + /// flushes it (`CalvinResolve`, `CalvinFlush`). /// /// NOTE: This variant occupies the same msgpack positional tag as the /// original `CalvinExecute` variant it replaces, preserving wire @@ -450,9 +451,9 @@ pub enum MetaOp { /// (unique / primary-key) immediately, computes the real affected-row /// count, and records the resulting body (or tombstone) in the overlay so /// a subsequent same-transaction read-modify-write observes it. It does - /// NOT make the write durable — the buffered plan is still replayed - /// through the real apply path inside the COMMIT `TransactionBatch`, which - /// remains the sole durable apply. Keyed by the request's `txn_id`. + /// NOT make the write durable — COMMIT resolves the overlay into the + /// transaction's redo record, and the redo install remains the sole + /// durable apply. Keyed by the request's `txn_id`. StageWrite { plan: Box }, /// Drop the per-transaction staging overlay for a completed (committed @@ -493,40 +494,42 @@ pub enum MetaOp { array_marker: u64, }, - /// Record the per-key / per-collection write versions of a committed - /// Calvin transaction's locally-applied write plans. + /// Record the per-key write versions of a committed Calvin transaction's + /// locally-applied write plans. /// - /// A Calvin apply's committed WAL LSN is known only after the apply - /// succeeds, so the apply itself cannot advance the version index. The - /// scheduler stamps that LSN onto this op's `wal_lsn` and dispatches it back - /// to the same core, which funnels `plans` through the shared write-version - /// recorder at that LSN — landing in the same shard-local WAL-LSN space the + /// The scheduler stamps the transaction's committed LSN onto this op's + /// `wal_lsn` and dispatches it back to the same core once the flush + /// completes. The core funnels `plans` through the shared write-version + /// recorder at that LSN — the same shard-local WAL-LSN space the /// single-shard fast path and read watermarks use. Records only: no base - /// mutation, no WAL append, no event emission. Wire-additive (appended last) - /// so older log entries decode unchanged. + /// mutation, no WAL append, no event emission. RecordCalvinWriteVersions { /// Tenant scope for all plans. tenant_id: TenantId, /// The locally-applied write plans whose keys' versions are recorded. plans: Vec, - /// Calvin epoch of the applied transaction. With `position` and the - /// request's vShard, keys the index-value tuples the flush staged so the - /// core drains and records them at this op's applied LSN. - epoch: u64, - /// Calvin position within the epoch (see `epoch`). - position: u32, }, - /// Flush the staged writes of a Calvin transaction to base storage. + /// Install a committed Calvin transaction's redo record on base storage. + /// + /// `CalvinExecuteStatic` STAGES the transaction's plans without mutating + /// base, and `CalvinResolve` resolves them into one redo record, which the + /// scheduler appends to the WAL as a `TransactionRedo` record. This op + /// carries that record's bytes, and the request carries its LSN. The core + /// installs it through the same passes restart replay drives: validate, + /// install with undo, then settle and cover. It then drops the staged + /// state keyed by `(epoch, position)`. /// - /// `CalvinExecuteStatic` validates and STAGES the transaction's plans into - /// the per-core commit-pending buffer without mutating base. Once the local - /// commit vote resolves to commit, the scheduler dispatches this op back to - /// the same core, which pops the staged plans keyed by `(epoch, position)` - /// and replays them through the durable apply funnel (base + side effects + - /// version recording). Absent key (already flushed/dropped) is an idempotent - /// no-op, not an error. - CalvinFlush { epoch: u64, position: u32 }, + /// `redo` is empty when the transaction wrote nothing. `collections` names + /// every collection the transaction wrote. `sum_targets` is the + /// materialized-sum resolution its document writes fold into. + CalvinFlush { + epoch: u64, + position: u32, + redo: Vec, + collections: Vec, + sum_targets: Vec, + }, /// Discard the staged writes of a Calvin transaction. /// @@ -567,11 +570,8 @@ pub enum MetaOp { /// `commit_pending` under `(epoch, position, vshard)` and the per-core /// staging overlay written under the corresponding synthetic `TxnId` /// (see `calvin_synthetic_txn_id`). Dispatched by the scheduler once the - /// local commit vote resolves to commit, in place of (or ahead of) - /// `CalvinFlush` — the flush path mutates base directly, while resolve - /// produces a durable redo record for a later install phase instead. No - /// base engine is touched during resolve. Wire-additive: appended last - /// so older log entries decode unchanged. + /// global verdict is commit, ahead of `CalvinFlush`, which installs the + /// record this op returns. No base engine is touched during resolve. CalvinResolve { epoch: u64, position: u32 }, /// Apply one committed transaction's resolved redo record on the core that diff --git a/nodedb-physical/src/physical_plan/mod.rs b/nodedb-physical/src/physical_plan/mod.rs index 9a8dc4d03..644c10781 100644 --- a/nodedb-physical/src/physical_plan/mod.rs +++ b/nodedb-physical/src/physical_plan/mod.rs @@ -48,7 +48,9 @@ pub use exchange::{ExchangeMode, ExchangeOp}; pub use graph::{ BatchEdge, BspSuperstepPlan, BspSuperstepResult, GraphOp, WccSuperstepPlan, WccSuperstepResult, }; -pub use kv::{KvCounterShape, KvOp, KvResolveOutcome, KvResolvedMutation}; +pub use kv::{ + KvCounterShape, KvOp, KvResolveOutcome, KvResolvedMutation, SortedIndexRead, SortedIndexSpec, +}; pub use meta::{MetaOp, SAVEPOINT_MARKER_BYTES}; pub use plan::PhysicalPlan; pub use query::{AggregateSpec, GroupKeySpec, JoinProjection, QueryOp}; diff --git a/nodedb-types/src/error/sqlstate.rs b/nodedb-types/src/error/sqlstate.rs index a3287e26b..778b1c432 100644 --- a/nodedb-types/src/error/sqlstate.rs +++ b/nodedb-types/src/error/sqlstate.rs @@ -118,6 +118,10 @@ pub const PERIOD_LOCK_MISCONFIGURED: &str = "23609"; /// `28000` — `invalid_authorization_specification` (no valid credentials) pub const INVALID_AUTHORIZATION: &str = "28000"; +/// `active_sql_transaction`: the statement cannot run inside a transaction +/// block. +pub const ACTIVE_SQL_TRANSACTION: &str = "25001"; + // ── Class 3D — Invalid Catalog Name ────────────────────────────────────────── /// `3D000` — `invalid_catalog_name` (the selected database does not exist) diff --git a/nodedb-types/src/fail_point.rs b/nodedb-types/src/fail_point.rs index 1d8bab0a3..56e036fc2 100644 --- a/nodedb-types/src/fail_point.rs +++ b/nodedb-types/src/fail_point.rs @@ -194,10 +194,10 @@ pub use imp::{FAILPOINTS_ENV, FailAction, FailGuard, clear, eval, eval_fail, loo /// Inject a fail point. Expands to nothing without the `failpoints` feature. /// /// Usage in production code: -/// `nodedb_types::fail_point!("transaction_batch::between_subapply");` +/// `nodedb_types::fail_point!("calvin_static::during_overlay_stage");` /// /// Tests opt in by enabling the feature and installing actions: -/// `fail_point::set("transaction_batch::between_subapply", +/// `fail_point::set("calvin_static::during_overlay_stage", /// fail_point::FailAction::Panic);` #[macro_export] macro_rules! fail_point { From 7f0c312bd0e739eccec8590945ad702bbc6ba646 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 06:43:49 +0800 Subject: [PATCH 27/64] refactor(executor): install transactions from redo, not sub-plan replay MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A committed transaction used to re-execute its buffered plans as a 'sub-plan batch' at COMMIT — the sole durable apply path, with per-engine sub_plan_* modules translating each staged plan back into engine calls. Replace that with the staged/redo-apply pipeline already used for non-transactional writes: staged plans resolve into a redo record once, the record installs on every path (autocommit, explicit COMMIT, Calvin), and CalvinExecute now stages under lock instead of executing eagerly, resolving to redo only once the global verdict commits. Delete the sub_plan, sub_plan_doc, sub_plan_kv*, sub_plan_write, sub_request, batch/batch_crdt/batch_irreversible, index_write_values and resolve/vector_direct modules; extend redo_apply, resolve and stage_write to cover the cases they used to handle (document batches, timeseries ILP staging, columnar insert undo). Track write-outcome ownership end to end: add a bridge ClosedLsns range set and OutcomeFloor::ResendRefusal so a resend of an LSN whose outcome is already final, or one minted outside a live write window, is refused instead of replayed twice. Give Calvin's overlay-stage handlers their own calvin_reply module (stage, images, target, flush_read, reply) and split CoreLoop's Calvin and maintenance state into calvin_fence/calvin_state/ maintenance_state modules. Move KV counter and float-text arithmetic out of engine/kv into the new nodedb_physical::kv_atomic module, deleting engine_atomic_compute.rs and float_text.rs; give the sorted-index engine its own index module. Add transactional index DDL and sorted-index reads: index DDL statements inside an explicit transaction buffer their catalog entries and defer their engine side effects (secondary-index backfill, KV index build/drop, sorted-index tree build/drop, FTS analyzer bind) to COMMIT via a new DeferredDdlEffect/deferred_effects module, and a sorted-index read inside a transaction answers from a transaction-local tree via the new KvOp::SortedIndexTxnRead and kv/sorted_txn handler. Expose both as native opcodes (index_ddl_op, sorted_read_op) so the wire protocol reaches the same catalog and engine path as the SQL statements. Add ErrorCode::SyncRejected carrying the same rejection provenance as the cluster wire type. Update transaction, Calvin, WAL replication, undo, and crash/inproc/ native/wire test suites for the staged-redo install path, and add a single-core commit-plans test driver in nodedb-test-support. --- nodedb-test-support/src/lib.rs | 1 + .../src/native_harness/server.rs | 2 + nodedb-test-support/src/tx_batch_helpers.rs | 126 +-- nodedb-test-support/src/tx_commit.rs | 106 +++ nodedb/src/bridge/dispatch/closed_lsns.rs | 128 +++ nodedb/src/bridge/dispatch/mod.rs | 3 +- nodedb/src/bridge/dispatch/outcome_floor.rs | 133 ++- nodedb/src/bridge/envelope/counter_fault.rs | 60 -- nodedb/src/bridge/envelope/error_code.rs | 8 + nodedb/src/bridge/envelope/mod.rs | 3 +- .../src/control/array_sync/raft_apply/cell.rs | 6 +- .../src/control/array_sync/raft_apply/op.rs | 6 +- .../control/catalog_overlay/index_record.rs | 58 ++ nodedb/src/control/catalog_overlay/mod.rs | 4 +- .../catalog_overlay/vector_index_params.rs | 135 ++++ .../scheduler/driver/core/commit_redo.rs | 40 +- .../driver/core/commit_resolution_dispatch.rs | 50 +- .../driver/core/commit_resolve/apply_tail.rs | 168 +++- .../driver/core/commit_resolve/verdict.rs | 6 +- .../scheduler/driver/core/completion_route.rs | 6 +- .../driver/core/dispatch/active_dispatch.rs | 2 + .../driver/core/dispatch/static_dispatch.rs | 2 + .../calvin/scheduler/driver/core/routing.rs | 3 +- .../scheduler/driver/core/test_support.rs | 2 + .../driver/core/write_version_record.rs | 27 +- .../cluster/calvin/scheduler/driver/types.rs | 43 + .../cluster/calvin/scheduler/recovery.rs | 2 + .../control/cluster/data_plane_error_wire.rs | 64 +- .../apply_loop/write_dispatch.rs | 8 +- nodedb/src/control/event_action_error.rs | 1 + nodedb/src/control/gateway/dispatch_remote.rs | 2 +- .../control/gateway/version_set/plan_keys.rs | 3 +- .../control/planner/calvin/dependent_recon.rs | 7 +- .../control/planner/calvin/submit/local.rs | 7 +- .../src/control/planner/calvin/write_class.rs | 5 +- .../procedural/executor/core/dispatch.rs | 120 +-- .../planner/procedural/executor/core/state.rs | 1 - .../procedural/executor/transaction.rs | 36 +- .../src/control/planner/rls_injection/kv.rs | 8 + .../rls_injection/permission_tree/kv.rs | 8 + .../security/catalog/index_registry.rs | 17 +- .../security/catalog/vector_index_params.rs | 24 +- .../security/identity/plan_permission.rs | 1 + .../control/server/dispatch_utils/dispatch.rs | 190 +++-- .../server/dispatch_utils/minted/mod.rs | 1 - .../server/dispatch_utils/minted/records.rs | 84 +- .../server/dispatch_utils/minted/resolve.rs | 70 +- .../src/control/server/dispatch_utils/mod.rs | 4 +- .../submit_write/funnel/driver.rs | 22 +- .../dispatch_utils/submit_write/params.rs | 11 + .../server/dispatch_utils/write_abort.rs | 14 +- .../server/native/dispatch/index_ddl_op.rs | 393 +++++++++ .../src/control/server/native/dispatch/mod.rs | 4 + .../native/dispatch/plan_builder/dispatch.rs | 31 +- .../native/dispatch/plan_builder/document.rs | 189 +---- .../dispatch/plan_builder/document_bulk.rs | 135 ++++ .../server/native/dispatch/plan_builder/kv.rs | 162 ---- .../dispatch/plan_builder/kv_counter.rs | 2 +- .../native/dispatch/plan_builder/mod.rs | 1 + .../native/dispatch/plan_builder/vector.rs | 30 - .../server/native/dispatch/sorted_read_op.rs | 129 +++ .../server/native/dispatch/transaction.rs | 4 +- .../control/server/native/session/request.rs | 33 +- .../control/server/pgwire/types/error_map.rs | 3 + .../server/resp/handler_kv/counters.rs | 2 +- .../response_shape/types/plan_kind/kv.rs | 10 +- .../authorization/requirements/collect.rs | 2 + .../control/server/shared/ddl/engine_apply.rs | 52 ++ .../ddl/neutral/collection/index/build.rs | 160 ++++ .../ddl/neutral/collection/index/create.rs | 110 +-- .../ddl/neutral/collection/index/drop.rs | 18 +- .../ddl/neutral/collection/index/kv_index.rs | 195 +++++ .../ddl/neutral/collection/index/mod.rs | 2 + .../ddl/neutral/collection/index/teardown.rs | 78 +- .../shared/ddl/neutral/deferred_effects.rs | 145 ++++ .../shared/ddl/neutral/dsl/text_index.rs | 30 +- .../shared/ddl/neutral/dsl/vector_index.rs | 38 +- .../shared/ddl/neutral/kv_atomic/handlers.rs | 2 +- .../shared/ddl/neutral/kv_sorted_index/ddl.rs | 88 +- .../ddl/neutral/kv_sorted_index/dispatch.rs | 39 +- .../shared/ddl/neutral/kv_sorted_index/mod.rs | 2 + .../ddl/neutral/kv_sorted_index/query.rs | 168 ++-- .../ddl/neutral/kv_sorted_index/txn_read.rs | 212 +++++ .../control/server/shared/ddl/neutral/mod.rs | 1 + .../ddl/neutral/router/string_engine_ops.rs | 16 +- .../src/control/server/shared/ddl/sqlstate.rs | 5 + .../control/server/shared/returning/inject.rs | 1 + .../server/shared/session/commit/run.rs | 19 + .../server/shared/session/ddl_buffer.rs | 69 ++ .../server/shared/session/ddl_effect.rs | 79 ++ .../server/shared/session/ddl_rollback.rs | 6 +- .../src/control/server/shared/session/mod.rs | 1 + .../server/shared/sql/staging_predicates.rs | 1 + .../predicate/txn_buffering/classify.rs | 56 +- .../predicate/txn_buffering/mod.rs | 20 +- .../sync/async_dispatch/delta/outcome.rs | 38 + .../src/control/server/wal_dispatch/crdt.rs | 8 +- nodedb/src/control/server/wal_dispatch/mod.rs | 9 +- .../control/server/wal_dispatch/vector/mod.rs | 9 +- .../control/server/wal_dispatch_kv/append.rs | 1 + .../control/surrogate/assign/bind_plan/kv.rs | 1 + nodedb/src/control/system_txn/data_plane.rs | 4 +- nodedb/src/control/system_txn/mod.rs | 6 +- nodedb/src/control/system_txn/run.rs | 138 +++- nodedb/src/control/system_txn/tests.rs | 157 ++++ nodedb/src/control/wal_catchup.rs | 66 +- .../decode/transaction_redo.rs | 2 + .../wal_replication/encode/entry_kv.rs | 3 +- nodedb/src/control/write_resolve/kv.rs | 1 + .../src/data/executor/core_loop/accessors.rs | 10 + .../data/executor/core_loop/calvin_fence.rs | 340 ++++++++ .../data/executor/core_loop/calvin_state.rs | 53 ++ .../data/executor/core_loop/commit_pending.rs | 32 +- .../src/data/executor/core_loop/deferred.rs | 10 +- .../src/data/executor/core_loop/event_emit.rs | 4 +- .../data/executor/core_loop/maintenance.rs | 8 +- .../executor/core_loop/maintenance_state.rs | 51 ++ nodedb/src/data/executor/core_loop/mod.rs | 3 + nodedb/src/data/executor/core_loop/open.rs | 19 +- nodedb/src/data/executor/core_loop/state.rs | 111 +-- nodedb/src/data/executor/core_loop/tick.rs | 16 +- .../data/executor/core_loop/write_index.rs | 84 +- nodedb/src/data/executor/dispatch/meta.rs | 74 +- .../data/executor/enforcement/hash_chain.rs | 10 +- .../enforcement/materialized_sum/apply.rs | 57 +- .../materialized_sum/divergence.rs | 4 +- .../data/executor/enforcement/statement.rs | 25 +- .../executor/handlers/bulk_dml/admission.rs | 4 +- .../handlers/bulk_dml/delete_cascade.rs | 6 +- .../data/executor/handlers/compact/budget.rs | 2 +- .../executor/handlers/compact/maintenance.rs | 6 +- .../data/executor/handlers/compact/runner.rs | 2 +- .../executor/handlers/compact/segments.rs | 7 +- .../handlers/control/calvin/active_passive.rs | 111 +-- .../handlers/control/calvin/discard.rs | 5 + .../executor/handlers/control/calvin/flush.rs | 497 +++++++++--- .../executor/handlers/control/calvin/mod.rs | 30 +- .../handlers/control/calvin/shared.rs | 7 +- .../handlers/control/calvin/static_stage.rs | 54 +- .../handlers/control/calvin/test_commit.rs | 77 ++ .../handlers/control/calvin_active_verify.rs | 2 +- .../handlers/control/calvin_overlay_stage.rs | 183 ++++- .../control/calvin_overlay_stage_bulk.rs | 145 ++-- .../control/calvin_reply/flush_read.rs | 54 ++ .../handlers/control/calvin_reply/images.rs | 163 ++++ .../handlers/control/calvin_reply/mod.rs | 9 + .../handlers/control/calvin_reply/reply.rs | 66 ++ .../handlers/control/calvin_reply/stage.rs | 760 ++++++++++++++++++ .../handlers/control/calvin_reply/target.rs | 320 ++++++++ .../handlers/control/calvin_resolve.rs | 49 +- .../handlers/control/crdt_apply/gated.rs | 151 +++- .../handlers/control/crdt_apply/local.rs | 31 +- .../executor/handlers/control/crdt_doc.rs | 5 +- .../handlers/control/crdt_materialize.rs | 1 - .../src/data/executor/handlers/control/mod.rs | 3 + .../data/executor/handlers/control/reindex.rs | 13 +- .../src/data/executor/handlers/kv/dispatch.rs | 16 + .../executor/handlers/kv/field_compute.rs | 2 +- nodedb/src/data/executor/handlers/kv/mod.rs | 1 + .../handlers/kv/resolve/atomic_ops.rs | 27 +- .../executor/handlers/kv/resolve/dispatch.rs | 1 + .../src/data/executor/handlers/kv/sorted.rs | 78 +- .../data/executor/handlers/kv/sorted_txn.rs | 243 ++++++ .../executor/handlers/kv/transfer_compute.rs | 2 +- .../merge_orchestrated/delete_arms.rs | 4 +- .../executor/handlers/point/apply_delete.rs | 2 + .../executor/handlers/timeseries/flush.rs | 79 -- .../executor/handlers/timeseries/ingest.rs | 19 +- .../handlers/timeseries/ingest_dispatch.rs | 30 +- .../data/executor/handlers/timeseries/mod.rs | 1 + .../handlers/timeseries/returning_preview.rs | 82 ++ .../executor/handlers/transaction/batch.rs | 643 --------------- .../handlers/transaction/batch_crdt.rs | 703 ---------------- .../transaction/batch_irreversible.rs | 146 ---- .../transaction/index_write_values.rs | 323 -------- .../data/executor/handlers/transaction/mod.rs | 14 - .../overlay/array_staged/txn_overlay.rs | 6 +- .../overlay/graph_staged/txn_overlay.rs | 7 +- .../handlers/transaction/overlay/merge.rs | 3 +- .../handlers/transaction/overlay/mod.rs | 2 +- .../handlers/transaction/overlay/staged.rs | 94 ++- .../transaction/overlay/staged_sidecar.rs | 64 +- .../redo_apply/calvin_fold_tests.rs | 160 ++++ .../handlers/transaction/redo_apply/cover.rs | 16 +- .../transaction/redo_apply/document.rs | 50 +- .../handlers/transaction/redo_apply/entry.rs | 30 +- .../handlers/transaction/redo_apply/events.rs | 2 +- .../redo_apply/install_refusal_tests.rs | 534 ++++++++++++ .../handlers/transaction/redo_apply/mod.rs | 11 + .../handlers/transaction/redo_apply/passes.rs | 72 +- .../handlers/transaction/redo_apply/state.rs | 74 ++ .../transaction/redo_apply/test_commit.rs | 238 ++++++ .../handlers/transaction/resolve/classify.rs | 34 +- .../handlers/transaction/resolve/entry.rs | 163 ++-- .../handlers/transaction/resolve/kv.rs | 48 +- .../handlers/transaction/resolve/mod.rs | 3 - .../handlers/transaction/resolve/spatial.rs | 73 +- .../transaction/resolve/timeseries.rs | 29 +- .../handlers/transaction/resolve/vector.rs | 132 +-- .../transaction/resolve/vector_direct.rs | 174 ---- .../transaction/resolve/vector_primary.rs | 4 +- .../handlers/transaction/stage_write/mod.rs | 11 +- .../stage_write/stage_document_batch.rs | 398 +++++++++ .../transaction/stage_write/stage_kv.rs | 8 +- .../stage_write/stage_kv_atomic.rs | 17 +- .../stage_write/stage_kv_delete.rs | 3 +- .../stage_write/stage_kv_predicate.rs | 4 +- .../stage_write/stage_kv_transfer.rs | 2 +- .../transaction/stage_write/stage_kv_ttl.rs | 6 +- .../stage_write/stage_point_document.rs | 6 +- .../transaction/stage_write/stage_spatial.rs | 5 +- .../stage_write/stage_timeseries.rs | 45 +- .../stage_write/stage_timeseries_ilp.rs | 216 +++++ .../transaction/stage_write/stage_truncate.rs | 8 +- .../executor/handlers/transaction/sub_plan.rs | 460 ----------- .../handlers/transaction/sub_plan_columnar.rs | 144 ---- .../transaction/sub_plan_doc/delete.rs | 155 ---- .../handlers/transaction/sub_plan_doc/mod.rs | 7 - .../handlers/transaction/sub_plan_doc/put.rs | 324 -------- .../handlers/transaction/sub_plan_kv.rs | 267 ------ .../transaction/sub_plan_kv_atomics.rs | 196 ----- .../handlers/transaction/sub_plan_kv_ops.rs | 168 ---- .../transaction/sub_plan_kv_ttl_sorted.rs | 216 ----- .../transaction/sub_plan_kv_writes.rs | 430 ---------- .../handlers/transaction/sub_plan_write.rs | 353 -------- .../handlers/transaction/sub_request.rs | 114 --- .../handlers/transaction/undo/apply.rs | 91 +-- .../handlers/transaction/undo/balanced.rs | 53 -- .../transaction/undo/columnar_insert.rs | 58 ++ .../handlers/transaction/undo/document.rs | 13 - .../handlers/transaction/undo/document_fts.rs | 117 +-- .../transaction/undo/document_outcome.rs | 7 +- .../handlers/transaction/undo/entry.rs | 67 +- .../handlers/transaction/undo/graph_node.rs | 61 +- .../executor/handlers/transaction/undo/kv.rs | 533 +++--------- .../executor/handlers/transaction/undo/mod.rs | 2 +- .../handlers/transaction/undo/rollback.rs | 140 +--- .../handlers/transaction/write_version_kv.rs | 3 +- nodedb/src/data/executor/handlers/truncate.rs | 7 +- .../src/data/executor/handlers/write_batch.rs | 4 +- nodedb/src/data/executor/sync_gate.rs | 27 +- nodedb/src/data/executor/wal_replay/crdt.rs | 10 +- .../data/executor/wal_replay/crdt_ordered.rs | 17 + nodedb/src/data/executor/wal_replay/kv.rs | 55 +- .../data/executor/wal_replay_columnar_dml.rs | 14 +- .../src/data/executor/wal_replay_kv_expiry.rs | 58 +- .../data/executor/wal_replay_redo_document.rs | 52 +- nodedb/src/engine/kv/engine_atomic.rs | 12 +- nodedb/src/engine/kv/engine_atomic_compute.rs | 543 ------------- nodedb/src/engine/kv/engine_sorted.rs | 24 +- nodedb/src/engine/kv/float_text.rs | 224 ------ nodedb/src/engine/kv/mod.rs | 3 - .../src/engine/kv/sorted_index/checkpoint.rs | 3 +- nodedb/src/engine/kv/sorted_index/index.rs | 167 ++++ nodedb/src/engine/kv/sorted_index/manager.rs | 163 +--- nodedb/src/engine/kv/sorted_index/mod.rs | 2 + nodedb/src/error/types.rs | 5 + nodedb/src/error_classify.rs | 1 + nodedb/src/error_from_data_plane.rs | 19 +- nodedb/src/wal/crdt_payload.rs | 40 +- nodedb/src/wal/manager/appender.rs | 35 +- nodedb/src/wal/manager/mod.rs | 2 +- nodedb/src/wal/redo/record.rs | 17 +- nodedb/src/wal/redo/replay.rs | 30 +- nodedb/tests/crash_core_stall.rs | 18 +- nodedb/tests/crash_harness/boot.rs | 120 +++ nodedb/tests/crash_harness/mod.rs | 19 +- .../inproc/cases/calvin_executor_apply.rs | 190 +++-- .../cases/calvin_executor_panic_recovery.rs | 221 +++-- .../inproc/cases/calvin_two_phase_apply.rs | 208 +++-- nodedb/tests/inproc/cases/core_loop.rs | 2 - .../executor_tests/test_conditional_update.rs | 95 ++- .../cases/executor_tests/test_transaction.rs | 173 ++-- .../test_transaction_cross_engine.rs | 280 ++----- .../executor_tests/test_transaction_matrix.rs | 220 ++--- .../test_transaction_matrix_helpers.rs | 53 +- .../test_transaction_matrix_kv.rs | 371 ++++----- .../test_transaction_matrix_side_effects.rs | 208 ----- nodedb/tests/inproc/cases/mod.rs | 1 - .../inproc/cases/native_handshake_e2e.rs | 2 + .../cases/transaction_batch_cross_engine.rs | 129 +-- .../transaction_batch_cross_engine_crash.rs | 175 ---- .../transaction_batch_cross_engine_mixed.rs | 240 +++--- nodedb/tests/native/cases/mod.rs | 1 + .../native/cases/native_index_ddl_opcodes.rs | 371 +++++++++ nodedb/tests/wire/cases/mod.rs | 3 + .../wire/cases/native_index_ddl_restart.rs | 106 +++ .../sql_transactions_columnar_overlay.rs | 5 +- ..._transactions_graph_edge_delete_overlay.rs | 6 +- .../sql_transactions_timeseries_overlay.rs | 5 +- .../cases/transactional_ddl_index_families.rs | 329 ++++++++ .../cases/transactional_sorted_index_reads.rs | 164 ++++ 292 files changed, 12373 insertions(+), 10721 deletions(-) create mode 100644 nodedb-test-support/src/tx_commit.rs create mode 100644 nodedb/src/bridge/dispatch/closed_lsns.rs delete mode 100644 nodedb/src/bridge/envelope/counter_fault.rs create mode 100644 nodedb/src/control/catalog_overlay/vector_index_params.rs create mode 100644 nodedb/src/control/server/native/dispatch/index_ddl_op.rs create mode 100644 nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs create mode 100644 nodedb/src/control/server/native/dispatch/sorted_read_op.rs create mode 100644 nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs create mode 100644 nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs create mode 100644 nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs create mode 100644 nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/txn_read.rs create mode 100644 nodedb/src/control/server/shared/session/ddl_effect.rs create mode 100644 nodedb/src/control/system_txn/tests.rs create mode 100644 nodedb/src/data/executor/core_loop/calvin_fence.rs create mode 100644 nodedb/src/data/executor/core_loop/calvin_state.rs create mode 100644 nodedb/src/data/executor/core_loop/maintenance_state.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin/test_commit.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin_reply/flush_read.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin_reply/images.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin_reply/mod.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin_reply/reply.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin_reply/stage.rs create mode 100644 nodedb/src/data/executor/handlers/control/calvin_reply/target.rs create mode 100644 nodedb/src/data/executor/handlers/kv/sorted_txn.rs create mode 100644 nodedb/src/data/executor/handlers/timeseries/returning_preview.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/batch.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/batch_crdt.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/batch_irreversible.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/index_write_values.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/calvin_fold_tests.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/install_refusal_tests.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/test_commit.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/resolve/vector_direct.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/stage_write/stage_document_batch.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries_ilp.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_columnar.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_doc/mod.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ops.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ttl_sorted.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/sub_request.rs delete mode 100644 nodedb/src/data/executor/handlers/transaction/undo/balanced.rs create mode 100644 nodedb/src/data/executor/handlers/transaction/undo/columnar_insert.rs delete mode 100644 nodedb/src/engine/kv/engine_atomic_compute.rs delete mode 100644 nodedb/src/engine/kv/float_text.rs create mode 100644 nodedb/src/engine/kv/sorted_index/index.rs create mode 100644 nodedb/tests/crash_harness/boot.rs delete mode 100644 nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_side_effects.rs delete mode 100644 nodedb/tests/inproc/cases/transaction_batch_cross_engine_crash.rs create mode 100644 nodedb/tests/native/cases/native_index_ddl_opcodes.rs create mode 100644 nodedb/tests/wire/cases/native_index_ddl_restart.rs create mode 100644 nodedb/tests/wire/cases/transactional_ddl_index_families.rs create mode 100644 nodedb/tests/wire/cases/transactional_sorted_index_reads.rs diff --git a/nodedb-test-support/src/lib.rs b/nodedb-test-support/src/lib.rs index 16e370094..240cbfbec 100644 --- a/nodedb-test-support/src/lib.rs +++ b/nodedb-test-support/src/lib.rs @@ -15,6 +15,7 @@ pub mod pgwire_harness; pub mod sync_client; pub mod test_tracing; pub mod tx_batch_helpers; +pub mod tx_commit; use nodedb::event::cdc::event::CdcEvent; use nodedb_types::DatabaseId; diff --git a/nodedb-test-support/src/native_harness/server.rs b/nodedb-test-support/src/native_harness/server.rs index cda0d1d71..3bf526426 100644 --- a/nodedb-test-support/src/native_harness/server.rs +++ b/nodedb-test-support/src/native_harness/server.rs @@ -77,6 +77,8 @@ impl NativeTestServer { let shared = SharedState::new_with_credentials(dispatcher, Arc::clone(&wal), credentials, false) .expect("build shared state"); + // The same gateway install production boot runs. + nodedb::bootstrap::state_wiring::install_gateway(&shared); let data_side = data_sides.into_iter().next().expect("data side"); let core_dir = dir.path().to_path_buf(); diff --git a/nodedb-test-support/src/tx_batch_helpers.rs b/nodedb-test-support/src/tx_batch_helpers.rs index b9ab4bec3..e53437dd1 100644 --- a/nodedb-test-support/src/tx_batch_helpers.rs +++ b/nodedb-test-support/src/tx_batch_helpers.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Plan-builder helpers shared by transaction batch cross-engine tests. +//! Plan builders and a single-core commit driver shared by the transaction +//! cross-engine tests. #![allow(dead_code)] use std::sync::Arc; @@ -15,6 +16,8 @@ use nodedb_physical::physical_plan::{ AggregateSpec, ColumnarInsertIntent, ColumnarOp, CrdtOp, DocumentOp, GraphOp, KvOp, PhysicalPlan, QueryOp, TimeseriesOp, VectorOp, }; + +pub use crate::tx_commit::commit_plans; use nodedb_types::OrdinalClock; // ── Core setup ────────────────────────────────────────────────────────────── @@ -106,58 +109,6 @@ fn qualify(collection: &str) -> nodedb_types::QualifiedCollection { nodedb_types::QualifiedCollection::new(nodedb_types::DatabaseId::DEFAULT, collection) } -pub fn vector_set_params(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::SetParams { - collection: qualify(collection), - field_name: String::new(), - dim: 0, - m: 16, - ef_construction: 200, - metric: "cosine".into(), - index_type: String::new(), - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - }) -} - -pub fn vector_seed(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::Insert { - collection: qualify(collection), - vector: vec![1.0, 2.0, 3.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }) -} - -pub fn vector_insert_ok(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::Insert { - collection: qualify(collection), - vector: vec![0.5, 0.5, 0.5], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::new(101), - pk_bytes: None, - provenance: None, - }) -} - -/// Always fails: dim mismatch (index expects dim=3). -pub fn vector_fail(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::Insert { - collection: qualify(collection), - vector: vec![1.0, 2.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }) -} - pub fn doc_put(collection: &str, doc_id: &str, val: &[u8]) -> PhysicalPlan { PhysicalPlan::Document(DocumentOp::PointPut { collection: qualify(collection), @@ -250,7 +201,9 @@ pub fn columnar_insert(collection: &str, id: &str, val: i64) -> PhysicalPlan { format: "msgpack".into(), intent: ColumnarInsertIntent::Insert, on_conflict_updates: Vec::new(), - surrogates: Vec::new(), + // One surrogate per row, as the planner binds it: a transaction + // stages each columnar row under its surrogate. + surrogates: vec![row_surrogate(id)], schema_bytes: Vec::new(), provenance: None, wal_lsn: None, @@ -262,6 +215,15 @@ pub fn columnar_insert(collection: &str, id: &str, val: i64) -> PhysicalPlan { }) } +/// A stable surrogate for the row whose primary key is `id`. Never zero, and +/// distinct from the small fixed surrogates the other helpers use. +fn row_surrogate(id: &str) -> nodedb_types::Surrogate { + let hash = id.bytes().fold(2_166_136_261u32, |h, b| { + (h ^ u32::from(b)).wrapping_mul(16_777_619) + }); + nodedb_types::Surrogate::new(hash | 0x8000_0000) +} + pub fn columnar_count(collection: &str) -> PhysicalPlan { PhysicalPlan::Query(QueryOp::Aggregate { collection: qualify(collection), @@ -332,6 +294,62 @@ pub fn crdt_apply(collection: &str, doc_id: &str) -> PhysicalPlan { }) } +/// A vector-primary direct insert of a 3-dimensional vector. +pub fn vector_direct_insert(collection: &str, surrogate: u32) -> PhysicalPlan { + let mut payload = std::collections::HashMap::new(); + payload.insert( + "id".to_string(), + nodedb_types::Value::String(format!("r{surrogate}")), + ); + PhysicalPlan::Vector(VectorOp::DirectInsert { + collection: qualify(collection), + field: "vec".into(), + surrogate: nodedb_types::Surrogate::new(surrogate), + pk_bytes: format!("r{surrogate}").into_bytes(), + vector: vec![0.5, 0.5, 0.5], + payload: zerompk::to_msgpack_vec(&payload).unwrap(), + quantization: nodedb_types::VectorQuantization::None, + storage_dtype: nodedb_types::VectorStorageDtype::F32, + payload_indexes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + }) +} + +/// A CRDT row write of `{"title": title}`. +pub fn crdt_upsert(collection: &str, doc_id: &str, surrogate: u32) -> PhysicalPlan { + PhysicalPlan::Crdt(CrdtOp::DocUpsert { + collection: qualify(collection), + document_id: doc_id.into(), + fields_json: r#"{"title":"staged"}"#.into(), + surrogate: nodedb_types::Surrogate::new(surrogate), + partial: false, + verb: nodedb_physical::physical_plan::CrdtWriteVerb::Insert, + returning: None, + rls_filters: Vec::new(), + }) +} + +/// `plans` followed by two inserts of one key: the transaction's staging +/// refuses the second as a unique violation, so the transaction commits +/// none of `plans`. +pub fn with_unique_refusal(mut plans: Vec) -> Vec { + let insert = PhysicalPlan::Document(DocumentOp::PointInsert { + collection: qualify("refusal_probe"), + document_id: "dup".into(), + value: nodedb_types::json_to_msgpack(&serde_json::json!({"n": 1})).unwrap(), + surrogate: nodedb_types::Surrogate::new(9_001), + if_absent: false, + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + plans.push(insert.clone()); + plans.push(insert); + plans +} + // ── Assertion helpers ───────────────────────────────────────────────────────── /// Assert that a KV key is absent (NotFound or empty payload). diff --git a/nodedb-test-support/src/tx_commit.rs b/nodedb-test-support/src/tx_commit.rs new file mode 100644 index 000000000..bab25f2fa --- /dev/null +++ b/nodedb-test-support/src/tx_commit.rs @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A single-core commit driver: stage, resolve, and install a transaction. + +use nodedb::bridge::dispatch::{BridgeRequest, BridgeResponse}; +use nodedb::bridge::envelope::{Request, Response, Status}; +use nodedb::control::wal_replication::transaction_redo::collections::written_collections; +use nodedb::control::wal_replication::transaction_redo::sum_targets::redo_sum_targets; +use nodedb::data::executor::core_loop::CoreLoop; +use nodedb::types::Lsn; +use nodedb_bridge::buffer::{Consumer, Producer}; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; + +use crate::tx_batch_helpers::make_request; + +fn send_request( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + request: Request, +) -> Response { + tx.try_push(BridgeRequest::unfloored(request)).unwrap(); + core.tick(); + rx.try_pop().unwrap().inner +} + +/// Commit `plans` as one transaction the way a session COMMIT does on its +/// core: stage each plan under the transaction `lsn` names, resolve the +/// transaction into its redo record, and install the record at `lsn`. +/// +/// Returns the install's response, or the first staging or resolve refusal. +/// The staging overlay is released on every path. +pub fn commit_plans( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + plans: Vec, + lsn: u64, +) -> Response { + let txn_id = nodedb_types::id::TxnId::new(lsn); + let in_txn = |plan: PhysicalPlan| { + let mut request = make_request(plan); + request.txn_id = Some(txn_id); + request + }; + let mut refusal = None; + for plan in &plans { + let staged = send_request( + core, + tx, + rx, + in_txn(PhysicalPlan::Meta(MetaOp::StageWrite { + plan: Box::new(plan.clone()), + })), + ); + if staged.status != Status::Ok { + refusal = Some(staged); + break; + } + } + let resolved = match refusal { + Some(refusal) => Err(refusal), + None => { + let resolved = send_request( + core, + tx, + rx, + in_txn(PhysicalPlan::Meta(MetaOp::ResolveTxn { + txn_id, + plans: plans.clone(), + })), + ); + if resolved.status == Status::Ok { + Ok(resolved.payload.to_vec()) + } else { + Err(resolved) + } + } + }; + let response = match resolved { + Ok(redo) => { + let mut install = make_request(PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo, + collections: written_collections(&plans), + sum_targets: redo_sum_targets(&plans), + })); + install.wal_lsn = Some(Lsn::new(lsn)); + send_request(core, tx, rx, install) + } + Err(refusal) => refusal, + }; + // A session COMMIT releases the overlay once the install answered. + let dropped = send_request( + core, + tx, + rx, + in_txn(PhysicalPlan::Meta(MetaOp::DropTxnOverlay { txn_id })), + ); + assert_eq!( + dropped.status, + Status::Ok, + "drop overlay: {:?}", + dropped.error_code + ); + response +} diff --git a/nodedb/src/bridge/dispatch/closed_lsns.rs b/nodedb/src/bridge/dispatch/closed_lsns.rs new file mode 100644 index 000000000..bc21c9757 --- /dev/null +++ b/nodedb/src/bridge/dispatch/closed_lsns.rs @@ -0,0 +1,128 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The LSNs above the outcome floor whose last owner closed, kept as +//! disjoint ranges. +//! +//! A closed LSN's outcome is final, so a resend of it is refused. The floor +//! prunes every range at or below it. A held window keeps the floor down for +//! the rest of the process, so the set must stay bounded without the floor. +//! +//! A range grows across a gap of LSNs no live window owns. Such an LSN is +//! either closed already, or was never owned: its record was minted outside +//! a write window and must never reach a core, so a resend of it is refused +//! too. The ranges are therefore split only at LSNs a live window owns, and +//! their count is at most one more than the number of live owned LSNs. + +use std::collections::BTreeMap; + +/// Closed LSNs as `start -> end` ranges, both ends inclusive. +#[derive(Debug, Default)] +pub(super) struct ClosedLsns { + ranges: BTreeMap, +} + +impl ClosedLsns { + /// Record that `lsn`'s last owner closed. `owners` holds every LSN a live + /// window owns. + pub(super) fn insert(&mut self, lsn: u64, owners: &BTreeMap) { + if self.contains(lsn) { + return; + } + let mut start = lsn; + let mut end = lsn; + if let Some((&prev_start, &prev_end)) = self.ranges.range(..lsn).next_back() + && owners + .range(prev_end.saturating_add(1)..lsn) + .next() + .is_none() + { + self.ranges.remove(&prev_start); + start = prev_start; + } + if let Some((&next_start, &next_end)) = self.ranges.range(lsn.saturating_add(1)..).next() + && owners + .range(lsn.saturating_add(1)..next_start) + .next() + .is_none() + { + self.ranges.remove(&next_start); + end = next_end; + } + self.ranges.insert(start, end); + } + + /// Whether `lsn` lies in a closed range. + pub(super) fn contains(&self, lsn: u64) -> bool { + self.ranges + .range(..=lsn) + .next_back() + .is_some_and(|(_, end)| lsn <= *end) + } + + /// Drop every LSN at or below `floor`: the floor refuses those itself. + pub(super) fn prune_through(&mut self, floor: u64) { + let above = floor.saturating_add(1); + let straddling = self + .ranges + .range(..above) + .next_back() + .map(|(_, end)| *end) + .filter(|end| *end >= above); + self.ranges = self.ranges.split_off(&above); + if let Some(end) = straddling { + self.ranges.insert(above, end); + } + } +} + +#[cfg(test)] +impl ClosedLsns { + /// Number of stored ranges. + pub(super) fn range_count(&self) -> usize { + self.ranges.len() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn adjacent_and_unowned_gaps_merge_into_one_range() { + let owners = BTreeMap::new(); + let mut closed = ClosedLsns::default(); + for lsn in [5, 7, 6, 10] { + closed.insert(lsn, &owners); + } + assert_eq!(closed.range_count(), 1); + assert!( + closed.contains(8), + "an unowned gap LSN is refused like a closed one" + ); + assert!(!closed.contains(11)); + } + + #[test] + fn a_live_owned_lsn_splits_the_ranges() { + let owners = BTreeMap::from([(8, 1)]); + let mut closed = ClosedLsns::default(); + closed.insert(5, &owners); + closed.insert(10, &owners); + assert_eq!(closed.range_count(), 2); + assert!(!closed.contains(8)); + } + + #[test] + fn pruning_keeps_only_lsns_above_the_floor() { + let owners = BTreeMap::new(); + let mut closed = ClosedLsns::default(); + closed.insert(5, &owners); + closed.insert(9, &owners); + closed.prune_through(7); + assert!(!closed.contains(7)); + assert!(closed.contains(8)); + assert!(closed.contains(9)); + closed.prune_through(9); + assert_eq!(closed.range_count(), 0); + } +} diff --git a/nodedb/src/bridge/dispatch/mod.rs b/nodedb/src/bridge/dispatch/mod.rs index 8c5d6e55f..2bd5862d2 100644 --- a/nodedb/src/bridge/dispatch/mod.rs +++ b/nodedb/src/bridge/dispatch/mod.rs @@ -1,5 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 +mod closed_lsns; mod core_channel; mod dispatched_lsns; mod dispatcher; @@ -17,5 +18,5 @@ pub use dispatcher::{ DefaultPriorityResolver, Dispatcher, }; pub use drain::CorePending; -pub use outcome_floor::{OutcomeFloor, StuckFloor, WriteWindow}; +pub use outcome_floor::{OutcomeFloor, ResendRefusal, StuckFloor, WriteWindow}; pub use refusal::DispatchRefusal; diff --git a/nodedb/src/bridge/dispatch/outcome_floor.rs b/nodedb/src/bridge/dispatch/outcome_floor.rs index ff8049fdb..a47c3eca2 100644 --- a/nodedb/src/bridge/dispatch/outcome_floor.rs +++ b/nodedb/src/bridge/dispatch/outcome_floor.rs @@ -42,6 +42,13 @@ //! back. Its record was minted outside a window, and the floor passed it //! before it reached the dispatcher. The open logs that record. //! +//! ## Owned records +//! +//! A window owns every LSN it records with [`WriteWindow::own`]. A record +//! sent to a core again through [`OutcomeFloor::open_existing`] must have no +//! owner and no final outcome: an owner carries its record to an outcome, and +//! a closed owner already did. Both refuse the resend, as the floor does. +//! //! ## Closing a window //! //! [`WriteWindow::settle`] states that the outcome is final. @@ -57,6 +64,7 @@ use std::time::{Duration, Instant}; use tracing::{error, warn}; +use super::closed_lsns::ClosedLsns; use crate::types::Lsn; /// The node's registry of open write windows. @@ -73,6 +81,8 @@ struct OpenWindow { opened_at: Instant, /// Held until restart: the floor stays below it by design. held: bool, + /// LSNs this window owns. + lsns: Vec, } #[derive(Debug, Default)] @@ -89,6 +99,11 @@ struct Windows { published: u64, /// Number of held windows. held: usize, + /// Open windows owning each LSN. + owners: BTreeMap, + /// LSNs above the published floor whose last owner closed. Bounded by + /// the number of live owned LSNs, whatever the floor does. + closed: ClosedLsns, } impl Windows { @@ -101,6 +116,7 @@ impl Windows { horizon, opened_at: Instant::now(), held: false, + lsns: Vec::new(), }, ); *self.horizons.entry(horizon).or_insert(0) += 1; @@ -117,6 +133,43 @@ impl Windows { self.horizons.remove(&window.horizon); } } + let mut released = Vec::new(); + for lsn in window.lsns { + if let Some(count) = self.owners.get_mut(&lsn) { + *count -= 1; + if *count == 0 { + self.owners.remove(&lsn); + released.push(lsn); + } + } + } + for lsn in released { + self.closed.insert(lsn, &self.owners); + } + } + + /// Record that window `ticket` owns `lsn`. + fn own(&mut self, ticket: u64, lsn: u64) { + self.note(lsn); + let Some(window) = self.open.get_mut(&ticket) else { + return; + }; + if !window.lsns.contains(&lsn) { + window.lsns.push(lsn); + *self.owners.entry(lsn).or_insert(0) += 1; + } + } + + /// Why `lsn` is claimed: a live window owns it, or its last owner + /// closed. `None` when it is free. + fn claim(&self, lsn: u64) -> Option { + if self.owners.contains_key(&lsn) { + Some(ResendRefusal::Owned) + } else if self.closed.contains(lsn) { + Some(ResendRefusal::Closed) + } else { + None + } } /// Mark a window held. Returns its horizon and age. @@ -144,10 +197,23 @@ impl Windows { None => self.max_noted, }; self.published = self.published.max(computed); + // A closed LSN at or below the floor is refused by the floor itself. + self.closed.prune_through(self.published); self.published } } +/// Why [`OutcomeFloor::open_existing`] refused to send a record again. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ResendRefusal { + /// The floor passed the record: its outcome is final. + BelowFloor, + /// A live or held window owns the record and carries it to its outcome. + Owned, + /// The record's last owner closed: its outcome is final. + Closed, +} + /// The oldest window that holds the floor, and how long it has held it. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct StuckFloor { @@ -194,25 +260,33 @@ impl OutcomeFloor { it was minted outside a write window" ); } - windows.note(lsn.as_u64()); - windows.open(lsn.as_u64()) + let ticket = windows.open(lsn.as_u64()); + windows.own(ticket, lsn.as_u64()); + ticket }; WriteWindow::new(Arc::clone(self), ticket) } /// Open a window for an existing record at `lsn` that is sent to a core - /// again. `None` when the floor already passed `lsn`: the record's - /// outcome is final, and a second apply would land below the floor. - pub fn open_existing(self: &Arc, lsn: Lsn) -> Option { + /// again. Refused, with the reason, when the record must not be sent: + /// + /// - the floor passed `lsn`, so its outcome is final; + /// - a live or held window owns it, and carries it to its outcome; + /// - a window that owned it closed, so its outcome is final. + pub fn open_existing(self: &Arc, lsn: Lsn) -> Result { let ticket = { let mut windows = self.lock(); if lsn.as_u64() <= windows.floor() { - return None; + return Err(ResendRefusal::BelowFloor); + } + if let Some(refusal) = windows.claim(lsn.as_u64()) { + return Err(refusal); } - windows.note(lsn.as_u64()); - windows.open(lsn.as_u64()) + let ticket = windows.open(lsn.as_u64()); + windows.own(ticket, lsn.as_u64()); + ticket }; - Some(WriteWindow::new(Arc::clone(self), ticket)) + Ok(WriteWindow::new(Arc::clone(self), ticket)) } /// The current floor. Never lower than a value returned before. @@ -305,6 +379,12 @@ impl WriteWindow { self.owner.lock().note(lsn.as_u64()); } + /// Record that this window owns the record at `lsn`. Call it when the + /// record is appended. + pub fn own(&self, lsn: Lsn) { + self.owner.lock().own(self.ticket, lsn.as_u64()); + } + /// Close the window: the write's outcome is final. pub fn settle(mut self) { self.closed = true; @@ -492,7 +572,10 @@ mod tests { let window = floor.open_write(); window.note_minted(Lsn::new(8)); window.settle(); - assert!(floor.open_existing(Lsn::new(8)).is_none()); + assert_eq!( + floor.open_existing(Lsn::new(8)).err(), + Some(ResendRefusal::BelowFloor) + ); let resent = floor .open_existing(Lsn::new(9)) .expect("the floor has not passed 9"); @@ -501,6 +584,36 @@ mod tests { assert_eq!(floor.floor(), Lsn::new(9)); } + /// A held window keeps the floor down for the rest of the process. The + /// closed LSNs above it stay a bounded number of ranges however many + /// records close, and each stays refused. + #[test] + fn closed_lsns_stay_bounded_while_a_window_is_held() { + let floor = OutcomeFloor::new(); + floor.open_dispatched(Lsn::new(10)).hold(); + for lsn in 11..5_011u64 { + floor.open_dispatched(Lsn::new(lsn)).settle(); + } + assert_eq!( + floor.floor(), + Lsn::new(9), + "the held window keeps the floor down" + ); + assert!( + floor.lock().closed.range_count() <= 2, + "closed LSNs must stay bounded while a window is held" + ); + assert_eq!( + floor.open_existing(Lsn::new(2_500)).err(), + Some(ResendRefusal::Closed) + ); + assert_eq!( + floor.open_existing(Lsn::new(10)).err(), + Some(ResendRefusal::Owned), + "the held window still owns its record" + ); + } + #[test] fn a_dispatched_window_holds_the_floor_below_its_lsn() { let floor = OutcomeFloor::new(); diff --git a/nodedb/src/bridge/envelope/counter_fault.rs b/nodedb/src/bridge/envelope/counter_fault.rs deleted file mode 100644 index 3bf5a2d18..000000000 --- a/nodedb/src/bridge/envelope/counter_fault.rs +++ /dev/null @@ -1,60 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Why a KV counter atomic (`INCR`, `INCRBY`, `DECR`, `INCRBYFLOAT`) -//! computed no value. - -/// Why a KV counter atomic computed no value. -/// -/// Each fault has one client message, the text Redis answers with for the -/// same condition. RESP sends it after `ERR`. The SQL surfaces send it with -/// the SQLSTATE [`CounterFault::sqlstate`] names. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum CounterFault { - /// The stored value is not a decimal integer in the `i64` range. - NotAnInteger, - /// The stored value is not a decimal float. - NotAFloat, - /// The integer result is outside the `i64` range. - IntegerOverflow, - /// The float result is NaN or infinite. - NonFinite, -} - -impl CounterFault { - /// The client message, without the RESP `ERR` prefix. - pub fn message(self) -> &'static str { - match self { - Self::NotAnInteger => "value is not an integer or out of range", - Self::NotAFloat => "value is not a valid float", - Self::IntegerOverflow => "increment or decrement would overflow", - Self::NonFinite => "increment would produce NaN or Infinity", - } - } - - /// The SQLSTATE for the fault: `22P02` for a stored value that does not - /// parse, `22003` for a result out of range. - pub fn sqlstate(self) -> &'static str { - use nodedb_types::error::sqlstate; - match self { - Self::NotAnInteger | Self::NotAFloat => sqlstate::INVALID_TEXT_REPRESENTATION, - Self::IntegerOverflow | Self::NonFinite => sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn parse_faults_answer_invalid_text_representation() { - assert_eq!(CounterFault::NotAnInteger.sqlstate(), "22P02"); - assert_eq!(CounterFault::NotAFloat.sqlstate(), "22P02"); - } - - #[test] - fn range_faults_answer_numeric_value_out_of_range() { - assert_eq!(CounterFault::IntegerOverflow.sqlstate(), "22003"); - assert_eq!(CounterFault::NonFinite.sqlstate(), "22003"); - } -} diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index 812cc01da..150b6ce0a 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -25,6 +25,14 @@ pub enum ErrorCode { /// a retry channel can tell a retry apart from a permanent refusal instead /// of collapsing both into a terminal rejection. RetryableRefusal { reason: String }, + /// A sync frame the validator refused for good. Nothing applied. The + /// stream's high-water mark advanced to `provenance`, so the frame is + /// never admitted again, and `applied_seq` is the mark after the refusal. + SyncRejected { + violation: nodedb_types::sync::violation::ViolationType, + applied_seq: u64, + provenance: nodedb_types::sync::wire::SyncProvenance, + }, /// Document/collection not found. NotFound, /// Authorization failure. diff --git a/nodedb/src/bridge/envelope/mod.rs b/nodedb/src/bridge/envelope/mod.rs index ec09e4a9a..0a92b0870 100644 --- a/nodedb/src/bridge/envelope/mod.rs +++ b/nodedb/src/bridge/envelope/mod.rs @@ -2,15 +2,14 @@ //! Request/response envelopes exchanged over the SPSC bridge. -pub mod counter_fault; pub mod error_code; pub mod payload; pub mod request; pub mod response; pub mod status; -pub use counter_fault::CounterFault; pub use error_code::ErrorCode; +pub use nodedb_physical::kv_atomic::CounterFault; pub use nodedb_physical::physical_plan::PhysicalPlan; pub use payload::Payload; pub use request::{Admission, ExemptReason, Request}; diff --git a/nodedb/src/control/array_sync/raft_apply/cell.rs b/nodedb/src/control/array_sync/raft_apply/cell.rs index 10ffd5eea..905b9f9c9 100644 --- a/nodedb/src/control/array_sync/raft_apply/cell.rs +++ b/nodedb/src/control/array_sync/raft_apply/cell.rs @@ -129,7 +129,11 @@ pub(crate) async fn apply_array_cell_write( "apply_array_cell_write: apply failed" ); } - let applied_ok = result.is_ok(); + // A final refusal is the entry's outcome: its marker carries the key. + let applied_ok = result.is_ok() + || result + .as_ref() + .is_err_and(crate::control::server::dispatch_utils::error_is_final_refusal); tracker.complete(group_id, log_index, applied_key, result); applied_ok } diff --git a/nodedb/src/control/array_sync/raft_apply/op.rs b/nodedb/src/control/array_sync/raft_apply/op.rs index 4e553060c..f61675c19 100644 --- a/nodedb/src/control/array_sync/raft_apply/op.rs +++ b/nodedb/src/control/array_sync/raft_apply/op.rs @@ -226,8 +226,12 @@ pub(crate) async fn apply_array_op( group_id, index = log_index, array = %op.header.array, error = %e, "apply_array_op: apply failed" ); + // A final refusal is the entry's outcome: its marker carries the + // key, so a redelivered copy is never applied. + let refused_finally = + crate::control::server::dispatch_utils::error_is_final_refusal(&e); tracker.complete(group_id, log_index, applied_key, Err(e)); - false + refused_finally } } } diff --git a/nodedb/src/control/catalog_overlay/index_record.rs b/nodedb/src/control/catalog_overlay/index_record.rs index 037bddec7..8377662c7 100644 --- a/nodedb/src/control/catalog_overlay/index_record.rs +++ b/nodedb/src/control/catalog_overlay/index_record.rs @@ -51,6 +51,49 @@ pub fn resolve_index_record( ) } +/// Every index record of one `(database, tenant)`, with this connection's +/// uncommitted DDL replayed over the committed list in statement order. A +/// buffered create is listed, a buffered drop is not, and the result stays +/// in name order. +pub fn resolve_index_records( + database_id: u64, + tenant_id: u64, + committed: Vec, +) -> Vec { + let replayed = crate::control::server::shared::session::ddl_buffer::with_buffered(|buffered| { + let mut by_name: std::collections::BTreeMap = committed + .iter() + .map(|record| (record.name.clone(), record.clone())) + .collect(); + let mut touched = false; + for item in buffered { + match &item.entry { + CatalogEntry::PutIndexRecord(stored) + if stored.database_id == database_id && stored.tenant_id == tenant_id => + { + by_name.insert(stored.name.clone(), (**stored).clone()); + touched = true; + } + CatalogEntry::DeleteIndexRecord { + database_id: entry_db, + tenant_id: entry_tenant, + name, + .. + } if *entry_db == database_id && *entry_tenant == tenant_id => { + by_name.remove(name); + touched = true; + } + _ => {} + } + } + touched.then(|| by_name.into_values().collect::>()) + }); + match replayed { + Some(Some(records)) => records, + Some(None) | None => committed, + } +} + #[cfg(test)] mod tests { use super::*; @@ -108,6 +151,21 @@ mod tests { .await; } + #[tokio::test] + async fn the_listing_shows_a_buffered_create_and_hides_a_buffered_drop() { + conn_scope::scoped(async { + ddl_buffer::activate(); + ddl_buffer::try_buffer(put("idx_new")); + ddl_buffer::try_buffer(delete("idx_old")); + let names: Vec = resolve_index_records(0, 1, vec![stored("idx_old")]) + .into_iter() + .map(|record| record.name) + .collect(); + assert_eq!(names, vec!["idx_new".to_string()]); + }) + .await; + } + #[tokio::test] async fn outside_a_transaction_the_committed_row_wins() { conn_scope::scoped(async { diff --git a/nodedb/src/control/catalog_overlay/mod.rs b/nodedb/src/control/catalog_overlay/mod.rs index 744ee3c7d..f09f42da5 100644 --- a/nodedb/src/control/catalog_overlay/mod.rs +++ b/nodedb/src/control/catalog_overlay/mod.rs @@ -26,10 +26,12 @@ mod index_record; mod materialized_view; mod procedure; mod trigger; +mod vector_index_params; pub use self::collection::{resolve_collection, resolve_tenant_collections}; pub use self::function::resolve_function; -pub use self::index_record::resolve_index_record; +pub use self::index_record::{resolve_index_record, resolve_index_records}; pub use self::materialized_view::resolve_materialized_view; pub use self::procedure::resolve_procedure; pub use self::trigger::resolve_trigger; +pub use self::vector_index_params::resolve_vector_index_params; diff --git a/nodedb/src/control/catalog_overlay/vector_index_params.rs b/nodedb/src/control/catalog_overlay/vector_index_params.rs new file mode 100644 index 000000000..791cf3323 --- /dev/null +++ b/nodedb/src/control/catalog_overlay/vector_index_params.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Uncommitted-DDL overlay for vector index build parameters. +//! +//! See [`super::collection`] for the mechanism this mirrors: a `CREATE +//! VECTOR INDEX` buffered inside an open transaction must be visible to a +//! later `ALTER VECTOR INDEX` or duplicate check in that same transaction, +//! before COMMIT writes the row. + +use nodedb_types::StoredVectorIndexParams; + +use crate::control::catalog_entry::CatalogEntry; + +/// The vector index one `(database, tenant, collection, field)` names. +struct Target<'a> { + database_id: u64, + tenant_id: u64, + collection: &'a str, + field_name: &'a str, +} + +/// True when `entry` mutates the vector index `target` names. +fn targets(entry: &CatalogEntry, target: &Target<'_>) -> bool { + match entry { + CatalogEntry::PutVectorIndexParams(stored) => { + stored.database_id == target.database_id + && stored.tenant_id == target.tenant_id + && stored.collection == target.collection + && stored.field_name == target.field_name + } + CatalogEntry::DeleteVectorIndexParams { + database_id, + tenant_id, + collection, + field_name, + } => { + *database_id == target.database_id + && *tenant_id == target.tenant_id + && collection == target.collection + && field_name == target.field_name + } + _ => false, + } +} + +/// Replay one buffered entry over the state resolved so far. +fn step( + current: Option, + entry: &CatalogEntry, +) -> Option { + match entry { + CatalogEntry::PutVectorIndexParams(stored) => Some((**stored).clone()), + CatalogEntry::DeleteVectorIndexParams { .. } => None, + _ => current, + } +} + +/// Resolve one vector index's build parameters through this connection's +/// uncommitted DDL. +pub fn resolve_vector_index_params( + database_id: u64, + tenant_id: u64, + collection: &str, + field_name: &str, + committed: Option, +) -> Option { + let target = Target { + database_id, + tenant_id, + collection, + field_name, + }; + super::core::resolve(committed, |entry| targets(entry, &target), step) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::server::shared::session::{conn_scope, ddl_buffer}; + + fn params(m: usize) -> StoredVectorIndexParams { + StoredVectorIndexParams { + database_id: 0, + tenant_id: 1, + collection: "docs".to_owned(), + field_name: "emb".to_owned(), + dim: 3, + metric: "cosine".to_owned(), + m, + ef_construction: 200, + index_type: "hnsw".to_owned(), + pq_m: 0, + ivf_cells: 0, + ivf_nprobe: 0, + } + } + + fn resolve(committed: Option) -> Option { + resolve_vector_index_params(0, 1, "docs", "emb", committed) + } + + #[tokio::test] + async fn a_buffered_create_then_alter_resolves_to_the_altered_row() { + conn_scope::scoped(async { + ddl_buffer::activate(); + ddl_buffer::try_buffer(CatalogEntry::PutVectorIndexParams(Box::new(params(16)))); + ddl_buffer::try_buffer(CatalogEntry::PutVectorIndexParams(Box::new(params(32)))); + assert_eq!(resolve(None).map(|p| p.m), Some(32)); + }) + .await; + } + + #[tokio::test] + async fn a_buffered_drop_hides_the_committed_row() { + conn_scope::scoped(async { + ddl_buffer::activate(); + ddl_buffer::try_buffer(CatalogEntry::DeleteVectorIndexParams { + database_id: 0, + tenant_id: 1, + collection: "docs".to_owned(), + field_name: "emb".to_owned(), + }); + assert!(resolve(Some(params(16))).is_none()); + }) + .await; + } + + #[tokio::test] + async fn outside_a_transaction_the_committed_row_wins() { + conn_scope::scoped(async { + assert_eq!(resolve(Some(params(16))).map(|p| p.m), Some(16)); + }) + .await; + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index cdfb1c389..5f08146b9 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -12,6 +12,7 @@ //! `finish_resolved_commit` / `commit_apply_tail` complete. use super::super::types::CommitState; +use super::commit_resolution_dispatch::CommitResolution; use super::deferred::{DispatchOutcome, DispatchStep}; use super::halt::{HaltReason, HaltStep, error_response_text}; use super::scheduler::Scheduler; @@ -26,7 +27,7 @@ use nodedb_physical::physical_plan::meta::MetaOp; impl Scheduler { /// Handle the `MetaOp::CalvinResolve` response: decode the resolved /// `RedoRecord`, WAL-append it (unless its op set is empty), then dispatch - /// the flush stamped with that record's LSN. + /// the flush that installs it, stamped with that record's LSN. /// /// The verdict is already COMMIT, so a skipped resolve would tear the /// committed txn on this replica. A non-`Ok` response, a decode failure, @@ -59,12 +60,6 @@ impl Scheduler { return; } }; - redo.calvin_stamp = Some(CalvinStamp { - epoch: txn_id.epoch, - position: txn_id.position, - vshard_id: self.vshard_id, - }); - let Some(pending) = self.pending.get(&txn_id) else { // Txn state was reclaimed out from under us (should not happen — // locks are held until `on_txn_complete`); complete defensively. @@ -74,6 +69,34 @@ impl Scheduler { }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; + // The stamp carries what the slice folds, so the live install and + // restart replay fold at this record's LSN. + redo.calvin_stamp = Some(CalvinStamp { + epoch: txn_id.epoch, + position: txn_id.position, + vshard_id: self.vshard_id, + collections: pending.flush_scope.collections.clone(), + sum_targets: pending.flush_scope.sum_targets.clone(), + }); + + // The flush installs these exact bytes, the payload of the record + // appended below. + let redo_bytes = if redo.ops.is_empty() { + Vec::new() + } else { + match redo.to_bytes() { + Ok(bytes) => bytes, + Err(e) => { + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + format!("CalvinResolve redo record encode failed: {e}"), + ); + return; + } + } + }; // The record's outcome-floor window opens before the append. It stays // with the pending txn until the flush completes. @@ -112,12 +135,13 @@ impl Scheduler { }; if let Some(pending) = self.pending.get_mut(&txn_id) { pending.redo_records = redo_records; + pending.flush_scope.redo = redo_bytes; } // A flush refused at capacity is parked for re-send. The txn awaits its // flush response either way, so the state below is the same. if let DispatchOutcome::Failed(error) = - self.dispatch_commit_resolution(txn_id, true, redo_lsn) + self.dispatch_commit_resolution(txn_id, CommitResolution::Flush { redo_lsn }) { self.fail_dispatch_step(txn_id, DispatchStep::Flush, error); return; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs index 362cef90b..3eacfeceb 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs @@ -9,36 +9,62 @@ use super::commit_redo::missing_pending_error; use super::deferred::{DispatchOutcome, DispatchStep}; use super::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::types::Lsn; + +/// How a staged transaction resolves on this vShard. +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum CommitResolution { + /// Install the committed redo record the scheduler appended at + /// `redo_lsn`. The flush scope holds its bytes. Both are empty when the + /// transaction wrote nothing on this vShard. + Flush { redo_lsn: Option }, + /// Discard the staged state under an abort verdict. + Drop, +} impl Scheduler { /// Dispatch a flush or drop of a staged transaction's commit-pending buffer. /// + /// A flush carries the redo record, the collections the local plans + /// write, and their materialized-sum targets, so the Data Plane installs + /// the record the way every committed transaction installs. + /// /// A capacity refusal returns [`DispatchOutcome::Deferred`]: the flush or /// drop is parked for re-send and the txn stays in flight. A txn with no - /// `pending` entry returns [`DispatchOutcome::Failed`]. + /// `pending` entry returns [`DispatchOutcome::Failed`]. The flush takes its + /// collections and sum targets from the scope derived at stage time. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_commit_resolution( &mut self, txn_id: TxnId, - committed: bool, - wal_lsn: Option, + resolution: CommitResolution, ) -> DispatchOutcome { - let Some(pending) = self.pending.get(&txn_id) else { + let Some(pending) = self.pending.get_mut(&txn_id) else { return DispatchOutcome::Failed(missing_pending_error(txn_id)); }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; let epoch = txn_id.epoch; let position = txn_id.position; - let (plan, step) = if committed { - ( - PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch, position }), - DispatchStep::Flush, - ) - } else { - ( + let (plan, step, wal_lsn) = match resolution { + CommitResolution::Flush { redo_lsn } => { + pending.flush_scope.sends = pending.flush_scope.sends.saturating_add(1); + let scope = &pending.flush_scope; + ( + PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch, + position, + redo: scope.redo.clone(), + collections: scope.collections.clone(), + sum_targets: scope.sum_targets.clone(), + }), + DispatchStep::Flush, + redo_lsn, + ) + } + CommitResolution::Drop => ( PhysicalPlan::Meta(MetaOp::CalvinDrop { epoch, position }), DispatchStep::Drop, - ) + None, + ), }; let request_id = self.next_request_id(); let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, wal_lsn); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs index 33ea34afa..a01dc5da2 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -4,7 +4,11 @@ //! applied result, marking the apply durable, recording write versions, and //! proposing the `CompletionAck`. -use crate::bridge::envelope::{Response, Status}; +use crate::bridge::envelope::{ErrorCode, Response, Status}; +use crate::control::cluster::calvin::scheduler::driver::core::commit_resolution_dispatch::CommitResolution; +use crate::control::cluster::calvin::scheduler::driver::core::deferred::{ + DispatchOutcome, DispatchStep, +}; use crate::control::cluster::calvin::scheduler::driver::core::halt::{ HaltReason, HaltStep, error_response_text, }; @@ -13,6 +17,11 @@ use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Schedul use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::control::cluster::calvin::scheduler::metrics::infra_abort_reason; +/// The most flushes one committed txn sends. Each refused install rolled +/// every write back, so a resend is safe. A refusal that outlasts the bound +/// halts the scheduler. +pub(in crate::control::cluster::calvin::scheduler::driver::core) const MAX_FLUSH_SENDS: u32 = 8; + impl Scheduler { /// Run the commit tail once a flush/drop response has returned. /// @@ -23,10 +32,12 @@ impl Scheduler { /// written so there is no result to deposit, no apply LSN, and no versions /// to record. /// - /// A non-`Ok` flush halts the scheduler. The flush handler removes the - /// staged buffer before it applies, so a second flush applies nothing, and - /// a skipped flush tears the committed txn on this replica. A non-`Ok` drop - /// completes the txn: under an abort verdict no replica writes anything. + /// A flush refused with `RetryableRefusal` rolled its install back and + /// kept the staged buffer, so the scheduler sends the same flush again, up + /// to [`MAX_FLUSH_SENDS`] sends. Any other non-`Ok` flush, or a refusal + /// past the bound, halts the scheduler: a skipped flush tears the + /// committed txn on this replica. A non-`Ok` drop completes the txn: + /// under an abort verdict no replica writes anything. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn finish_resolved_commit( &mut self, txn_id: TxnId, @@ -35,6 +46,9 @@ impl Scheduler { redo_lsn: Option, ) { if response.status != Status::Ok { + if committed && self.resend_refused_flush(txn_id, &response, redo_lsn) { + return; + } if committed { self.halt_apply( txn_id, @@ -73,6 +87,44 @@ impl Scheduler { } } + /// Send the flush of `txn_id` again when the install refused it as + /// retryable and the send bound allows another. Returns whether the + /// refusal is handled: the flush went out again, or its dispatch failed + /// and the step's terminal handling ran. + fn resend_refused_flush( + &mut self, + txn_id: TxnId, + response: &Response, + redo_lsn: Option, + ) -> bool { + if !matches!( + response.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ) { + return false; + } + let sends = self + .pending + .get(&txn_id) + .map_or(MAX_FLUSH_SENDS, |pending| pending.flush_scope.sends); + if sends >= MAX_FLUSH_SENDS { + return false; + } + tracing::warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + sends, + "calvin: the flush install was refused as retryable; sending it again" + ); + if let DispatchOutcome::Failed(error) = + self.dispatch_commit_resolution(txn_id, CommitResolution::Flush { redo_lsn }) + { + self.fail_dispatch_step(txn_id, DispatchStep::Flush, error); + } + true + } + /// Deposit the applied result, durably mark the apply, record the apply's /// write versions, and propose the `CompletionAck`. /// @@ -96,6 +148,9 @@ impl Scheduler { response: Response, redo_lsn: Option, ) -> bool { + // The install folded materialized sums into target rows no redo + // sub-record names. The redo record's stamp carries the sum targets, + // so restart replay folds at the same LSN. // Deposit the FULL applied Response (affected-count + watermark + any // RETURNING rows) into the local sidecar BEFORE proposing the replicated // CompletionAck. The ack fires the coordinator's completion oneshot on @@ -122,6 +177,8 @@ impl Scheduler { if has_primary_write { use std::collections::hash_map::Entry; + let response = statement_reply(response); + use crate::control::state::CalvinApplyResult; let key = nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position); @@ -247,10 +304,20 @@ impl Scheduler { } } +/// The response the statement drains. A flush whose install succeeded but +/// whose reply failed to render answers `Ok` with the render error in +/// `error_code`. The statement reports that error. +fn statement_reply(mut response: Response) -> Response { + if response.status == Status::Ok && response.error_code.is_some() { + response.status = Status::Error; + response.payload = crate::bridge::envelope::Payload::empty(); + } + response +} + #[cfg(test)] mod tests { use super::*; - use crate::bridge::envelope::ErrorCode; use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ error_response, scheduler_with_pending, }; @@ -305,4 +372,93 @@ mod tests { assert!(!scheduler.pending.contains_key(&txn_id)); assert!(!scheduler.is_apply_halted()); } + fn retryable_refusal() -> Response { + error_response(ErrorCode::RetryableRefusal { + reason: "install rolled back".to_string(), + }) + } + + /// A flush refused as retryable reaches the Data Plane again with the + /// same redo record, and the txn stays pending and unhalted. + #[tokio::test] + async fn retryable_flush_refusal_resends_the_same_flush() { + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + await_data_plane_request, build_test_scheduler_with_data_side, make_sequenced_txn, + staged_pending, + }; + use nodedb_physical::physical_plan::PhysicalPlan; + use nodedb_physical::physical_plan::meta::MetaOp; + + let txn_id = TxnId::new(9, 2); + let registry = nodedb_cluster::calvin::CalvinCompletionRegistry::new_detached(); + let (mut scheduler, _dir, mut data_side) = build_test_scheduler_with_data_side(7, registry); + let mut pending = staged_pending(make_sequenced_txn(9, 2), txn_id); + pending.commit_state = Some(CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }); + pending.flush_scope.redo = vec![7, 7, 7]; + pending.flush_scope.sends = 1; + scheduler.pending.insert(txn_id, pending); + + scheduler.finish_resolved_commit(txn_id, retryable_refusal(), true, None); + + assert!(!scheduler.is_apply_halted()); + assert!(!scheduler.applied.is_applied(9, 2)); + assert_eq!( + scheduler.pending.get(&txn_id).map(|p| p.flush_scope.sends), + Some(2) + ); + assert!( + await_data_plane_request(&mut data_side, |plan| matches!( + plan, + PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch: 9, position: 2, redo, .. }) + if redo == &vec![7, 7, 7] + )) + .await, + "the resent flush carries the same redo record" + ); + } + + /// A retryable refusal past the send bound halts like any flush error. + #[tokio::test] + async fn retryable_flush_refusal_past_the_bound_halts() { + let txn_id = TxnId::new(9, 2); + let (mut scheduler, _dir) = scheduler_with_pending( + txn_id, + CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }, + ); + if let Some(pending) = scheduler.pending.get_mut(&txn_id) { + pending.flush_scope.sends = MAX_FLUSH_SENDS; + } + + scheduler.finish_resolved_commit(txn_id, retryable_refusal(), true, None); + + assert!(scheduler.pending.contains_key(&txn_id)); + assert_eq!( + scheduler.apply_halt().map(|h| h.reason), + Some(HaltReason::FlushFailed) + ); + } + + /// A render error on an installed flush reaches the statement as a typed + /// error, and the txn completes. + #[test] + fn a_render_error_on_an_installed_flush_becomes_the_statement_error() { + let mut response = error_response(ErrorCode::Internal { + detail: "render".to_string(), + }); + response.status = Status::Ok; + + let reply = statement_reply(response); + + assert_eq!(reply.status, Status::Error); + assert!(matches!( + reply.error_code.as_deref(), + Some(ErrorCode::Internal { .. }) + )); + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs index f9807529c..4ad2b3468 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs @@ -8,6 +8,7 @@ use std::time::Instant; use nodedb_cluster::calvin::VerdictSignal; +use crate::control::cluster::calvin::scheduler::driver::core::commit_resolution_dispatch::CommitResolution; use crate::control::cluster::calvin::scheduler::driver::core::deferred::{ DispatchOutcome, DispatchStep, }; @@ -74,7 +75,7 @@ impl Scheduler { (self.dispatch_calvin_resolve(txn_id), DispatchStep::Resolve) } else { ( - self.dispatch_commit_resolution(txn_id, false, None), + self.dispatch_commit_resolution(txn_id, CommitResolution::Drop), DispatchStep::Drop, ) }; @@ -365,7 +366,8 @@ mod tests { plan, PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch: 14, - position: 2 + position: 2, + .. }) ) }) diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs index 3e376bac0..3e9733646 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs @@ -57,9 +57,9 @@ impl Scheduler { self.metrics.record_executor_txn_duration_ms(elapsed_ms); // A staged transaction resolves through its commit-barrier state, OLLP - // answer included. A flush under a commit verdict that answers - // `OllpRetryRequired` wrote nothing on this replica while the others - // applied, so it halts and holds. It never settles as a retry. + // answer included. The flush installs a resolved redo record and runs + // no OLLP check, so an OLLP answer under a commit state is a failed + // step that halts and holds. It never settles as a retry. let commit_state = self.pending.get(&txn_id).and_then(|p| p.commit_state); // OLLP mismatch: the active executor detected predicate drift and returned diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs index bae382334..2726df689 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs @@ -98,6 +98,7 @@ impl Scheduler { let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); let has_returning = plans_have_returning(&plans); let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); + let flush_scope = super::super::super::types::FlushScope::of_plans(&plans); let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteActive { epoch, position, @@ -137,6 +138,7 @@ impl Scheduler { stage_error: None, // Set once a committed txn appends its redo record. redo_records: None, + flush_scope, }, ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs index 6a574abc6..db07d33dd 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs @@ -242,6 +242,7 @@ impl Scheduler { let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); let has_returning = plans_have_returning(&plans); let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); + let flush_scope = super::super::super::types::FlushScope::of_plans(&plans); let database_id = txn.tx_class.database_id; let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteStatic { epoch, @@ -280,6 +281,7 @@ impl Scheduler { stage_error: None, // Set once a committed txn appends its redo record. redo_records: None, + flush_scope, }, ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs index af8e2f59b..0ada4416a 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs @@ -196,7 +196,8 @@ fn kv_routing(op: &KvOp, database_id: DatabaseId) -> PlanRouting { | KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } - | KvOp::SortedIndexScore { .. } => PlanRouting::NotAWrite, + | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } => PlanRouting::NotAWrite, } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs index d3ff23762..327790df4 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -429,6 +429,8 @@ pub(super) fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { verdict_deadline: None, stage_error: None, redo_records: None, + flush_scope: crate::control::cluster::calvin::scheduler::driver::types::FlushScope::default( + ), } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs index 42b314570..0d27cb9e8 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/write_version_record.rs @@ -2,15 +2,14 @@ //! Post-apply write-version recording for committed Calvin transactions. //! -//! A Calvin apply's committed WAL LSN is allocated only AFTER the apply -//! succeeds — the `CalvinApplied` WAL record is appended on the Control Plane -//! once the executor response returns — so the apply itself carries no -//! committed LSN and the per-core write-version index cannot be advanced in -//! place (the dispatch stamps `wal_lsn: None`). Once the scheduler has that -//! LSN it dispatches a one-way, record-only op back to the same core, which -//! funnels the transaction's locally-applied write plans through the shared -//! write-version recorder at that LSN — the same shard-local WAL-LSN space the -//! single-shard fast path and read watermarks use. +//! A Calvin apply's committed WAL LSN is the LSN of its `TransactionRedo` +//! record, or of the `CalvinApplied` marker a transaction that wrote nothing +//! here appends. The install records the collection floors and index-value +//! versions of the record at its LSN. The per-key versions of the local write +//! plans are recorded here: the scheduler dispatches a one-way, record-only op +//! back to the same core, which funnels the plans through the shared +//! write-version recorder at that LSN — the same shard-local WAL-LSN space +//! the single-shard fast path and read watermarks use. use super::deferred::{DispatchOutcome, DispatchStep}; use super::scheduler::Scheduler; @@ -21,7 +20,7 @@ use nodedb_physical::physical_plan::meta::MetaOp; impl Scheduler { /// Record the per-key write versions of a just-committed Calvin - /// transaction's locally-applied write plans at its CalvinApplied WAL + /// transaction's locally-applied write plans at its committed WAL /// `applied_lsn`. /// /// Dispatches a record-only [`MetaOp::RecordCalvinWriteVersions`] op back to @@ -88,8 +87,6 @@ impl Scheduler { let plan = PhysicalPlan::Meta(MetaOp::RecordCalvinWriteVersions { tenant_id, plans: local, - epoch, - position, }); // The committed write-LSN for this Calvin apply — recorded against // every key the plans wrote, in the same WAL-LSN space as fast-path. @@ -142,11 +139,7 @@ mod tests { let arrived = await_data_plane_request(&mut data_side, |plan| { matches!( plan, - PhysicalPlan::Meta(MetaOp::RecordCalvinWriteVersions { - epoch: 21, - position: 0, - .. - }) + PhysicalPlan::Meta(MetaOp::RecordCalvinWriteVersions { .. }) ) }) .await; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs index 914e97a9c..cc0610ee4 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs @@ -75,6 +75,49 @@ pub(super) struct PendingTxn { /// completes, and holds when the scheduler halts or stops with the txn /// still pending. pub redo_records: Option, + /// What a committed flush names beside its redo record, derived once + /// from this vShard's slice when it stages. + pub flush_scope: FlushScope, +} + +/// What a vShard's committed flush carries: the collections its slice +/// writes, their materialized-sum targets, and the redo record. The Data +/// Plane installs the record the way every committed transaction installs. +/// +/// The collections and sum targets are derived when the slice stages, from +/// the plans the stage sends. A flush that follows a COMMIT verdict +/// therefore never decodes or routes a plan, so it cannot fail after the +/// verdict is durable. +#[derive(Debug, Clone, Default)] +pub(super) struct FlushScope { + pub collections: Vec, + pub sum_targets: Vec, + /// The encoded redo record the resolve appended. Empty until the resolve + /// answers, and empty when the slice wrote nothing on this vShard. A + /// resent flush carries the same bytes. + pub redo: Vec, + /// Flushes sent for this txn. A refused install resends the flush until + /// the count reaches its bound. + pub sends: u32, +} + +impl FlushScope { + /// The flush scope of the local plans `plans`. + pub(super) fn of_plans(plans: &[nodedb_physical::physical_plan::PhysicalPlan]) -> Self { + Self { + collections: + crate::control::wal_replication::transaction_redo::collections::written_collections( + plans, + ), + sum_targets: + crate::control::wal_replication::transaction_redo::sum_targets::redo_sum_targets( + plans, + ), + // The resolve fills these once it answers. + redo: Vec::new(), + sends: 0, + } + } } /// Commit-resolution state of a staged static Calvin transaction. diff --git a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs index c2d2ee2aa..30c67abfa 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs @@ -252,6 +252,8 @@ mod tests { epoch: 1, position: 1, vshard_id: vshard, + collections: Vec::new(), + sum_targets: Vec::new(), }), }; wal.appender(crate::wal::manager::NO_APPLY_KEY) diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index fb2b04d73..80d849642 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -77,6 +77,18 @@ impl From for DataPlaneErrorCode { } ErrorCode::RejectedPrevalidation { reason } => Self::RejectedPrevalidation { reason }, ErrorCode::RetryableRefusal { reason } => Self::RetryableRefusal { reason }, + ErrorCode::SyncRejected { + violation, + applied_seq, + provenance, + } => Self::SyncRejected { + violation, + applied_seq, + producer_id: provenance.producer_id, + epoch: provenance.epoch, + stream_id: provenance.stream_id, + seq: provenance.seq, + }, ErrorCode::NotFound => Self::NotFound, ErrorCode::RejectedAuthz { resource } => Self::RejectedAuthz { resource }, ErrorCode::ConflictRetry => Self::ConflictRetry, @@ -123,7 +135,7 @@ impl From for DataPlaneErrorCode { } ErrorCode::CounterFault { collection, fault } => Self::CounterFault { collection, - fault: fault.into(), + fault: counter_fault_to_wire(fault), }, ErrorCode::InsufficientBalance { collection, detail } => { Self::InsufficientBalance { collection, detail } @@ -175,6 +187,23 @@ impl From for ErrorCode { Self::RejectedPrevalidation { reason } } DataPlaneErrorCode::RetryableRefusal { reason } => Self::RetryableRefusal { reason }, + DataPlaneErrorCode::SyncRejected { + violation, + applied_seq, + producer_id, + epoch, + stream_id, + seq, + } => Self::SyncRejected { + violation, + applied_seq, + provenance: nodedb_types::sync::wire::SyncProvenance { + producer_id, + epoch, + stream_id, + seq, + }, + }, DataPlaneErrorCode::NotFound => Self::NotFound, DataPlaneErrorCode::RejectedAuthz { resource } => Self::RejectedAuthz { resource }, DataPlaneErrorCode::ConflictRetry => Self::ConflictRetry, @@ -225,7 +254,7 @@ impl From for ErrorCode { } DataPlaneErrorCode::CounterFault { collection, fault } => Self::CounterFault { collection, - fault: fault.into(), + fault: counter_fault_from_wire(fault), }, DataPlaneErrorCode::InsufficientBalance { collection, detail } => { Self::InsufficientBalance { collection, detail } @@ -270,25 +299,24 @@ impl From for ErrorCode { } } -impl From for DataPlaneCounterFault { - fn from(fault: CounterFault) -> Self { - match fault { - CounterFault::NotAnInteger => Self::NotAnInteger, - CounterFault::NotAFloat => Self::NotAFloat, - CounterFault::IntegerOverflow => Self::IntegerOverflow, - CounterFault::NonFinite => Self::NonFinite, - } +/// The wire form of a counter fault. Both types live in other crates, so the +/// mapping is a function, not a `From` impl. +fn counter_fault_to_wire(fault: CounterFault) -> DataPlaneCounterFault { + match fault { + CounterFault::NotAnInteger => DataPlaneCounterFault::NotAnInteger, + CounterFault::NotAFloat => DataPlaneCounterFault::NotAFloat, + CounterFault::IntegerOverflow => DataPlaneCounterFault::IntegerOverflow, + CounterFault::NonFinite => DataPlaneCounterFault::NonFinite, } } -impl From for CounterFault { - fn from(fault: DataPlaneCounterFault) -> Self { - match fault { - DataPlaneCounterFault::NotAnInteger => Self::NotAnInteger, - DataPlaneCounterFault::NotAFloat => Self::NotAFloat, - DataPlaneCounterFault::IntegerOverflow => Self::IntegerOverflow, - DataPlaneCounterFault::NonFinite => Self::NonFinite, - } +/// The counter fault a wire form carries. +fn counter_fault_from_wire(fault: DataPlaneCounterFault) -> CounterFault { + match fault { + DataPlaneCounterFault::NotAnInteger => CounterFault::NotAnInteger, + DataPlaneCounterFault::NotAFloat => CounterFault::NotAFloat, + DataPlaneCounterFault::IntegerOverflow => CounterFault::IntegerOverflow, + DataPlaneCounterFault::NonFinite => CounterFault::NonFinite, } } diff --git a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs index a8230426f..a80239efa 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs @@ -17,7 +17,8 @@ use crate::control::array_sync::raft_apply::{ }; use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; use crate::control::server::dispatch_utils::{ - ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, + ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, error_is_final_refusal, + submit_write, }; use crate::control::state::SharedState; use crate::control::wal_replication::from_replicated_entry; @@ -195,7 +196,10 @@ pub(super) async fn apply_generic_entry( } }; - let applied_ok = result.is_ok() || deterministic_crdt_fence_noop(&result); + // A final refusal is the entry's outcome: its marker carries the key. + let applied_ok = result.is_ok() + || deterministic_crdt_fence_noop(&result) + || result.as_ref().is_err_and(error_is_final_refusal); let applied = ledger_outcome(&result); tracker.complete(group_id, entry.index, applied_key, result); diff --git a/nodedb/src/control/event_action_error.rs b/nodedb/src/control/event_action_error.rs index 4f976f4e4..943e1b81a 100644 --- a/nodedb/src/control/event_action_error.rs +++ b/nodedb/src/control/event_action_error.rs @@ -108,6 +108,7 @@ mod tests { TriggerActionError::Transaction { source: SystemTxnError::Commit { detail: "serialization failure against a concurrent write".to_owned(), + code: None, }, } } diff --git a/nodedb/src/control/gateway/dispatch_remote.rs b/nodedb/src/control/gateway/dispatch_remote.rs index d86038f50..99aeaa999 100644 --- a/nodedb/src/control/gateway/dispatch_remote.rs +++ b/nodedb/src/control/gateway/dispatch_remote.rs @@ -356,7 +356,7 @@ pub(super) async fn dispatch_remote_stream( fn map_stream_cluster_error(err: nodedb_cluster::ClusterError, vshard_id: u64) -> Error { match err { nodedb_cluster::ClusterError::StreamTerminal { error, .. } => { - map_typed_cluster_error(error, vshard_id) + map_typed_cluster_error(*error, vshard_id) } other => Error::NotLeader { vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), diff --git a/nodedb/src/control/gateway/version_set/plan_keys.rs b/nodedb/src/control/gateway/version_set/plan_keys.rs index 9bd354e41..1a61717db 100644 --- a/nodedb/src/control/gateway/version_set/plan_keys.rs +++ b/nodedb/src/control/gateway/version_set/plan_keys.rs @@ -37,7 +37,8 @@ fn kv_touched_collections(op: &nodedb_physical::physical_plan::KvOp, out: &mut V | RegisterSortedIndex { collection, .. } | PredicateUpdate { collection, .. } | PredicateDelete { collection, .. } - | MaterializeScan { collection, .. } => out.push(collection.as_str().to_owned()), + | MaterializeScan { collection, .. } + | SortedIndexTxnRead { collection, .. } => out.push(collection.as_str().to_owned()), // TransferItem touches two collections. TransferItem { diff --git a/nodedb/src/control/planner/calvin/dependent_recon.rs b/nodedb/src/control/planner/calvin/dependent_recon.rs index e29fa429d..caed1c7e5 100644 --- a/nodedb/src/control/planner/calvin/dependent_recon.rs +++ b/nodedb/src/control/planner/calvin/dependent_recon.rs @@ -412,7 +412,12 @@ async fn dispatch_dependent_edge_recon_inner( .unwrap_or_else(|p| p.into_inner()) .remove(&completed_txn); let apply_result = match drained { - Some(CalvinApplyResult::Single { response, .. }) => Some(response), + Some(CalvinApplyResult::Single { response, .. }) => { + // An installed txn whose reply failed to render deposits it as an + // error for the statement. + crate::control::server::dispatch_utils::reject_data_plane_error(&response)?; + Some(response) + } Some(CalvinApplyResult::Conflict) => { return Err(Error::Internal { detail: "multi-participant cross-shard RETURNING not supported".to_owned(), diff --git a/nodedb/src/control/planner/calvin/submit/local.rs b/nodedb/src/control/planner/calvin/submit/local.rs index 57ea00b5a..84e8f496a 100644 --- a/nodedb/src/control/planner/calvin/submit/local.rs +++ b/nodedb/src/control/planner/calvin/submit/local.rs @@ -155,7 +155,12 @@ pub async fn submit_and_await_calvin_with_timeout( .unwrap_or_else(|p| p.into_inner()) .remove(&TxnId::new(epoch, position)); match drained { - Some(CalvinApplyResult::Single { response, .. }) => Ok(Some(response)), + Some(CalvinApplyResult::Single { response, .. }) => { + // An installed txn whose reply failed to render deposits it as an + // error for the statement. + crate::control::server::dispatch_utils::reject_data_plane_error(&response)?; + Ok(Some(response)) + } Some(CalvinApplyResult::Conflict) => Err(Error::Internal { detail: "multi-participant cross-shard RETURNING not supported".to_owned(), }), diff --git a/nodedb/src/control/planner/calvin/write_class.rs b/nodedb/src/control/planner/calvin/write_class.rs index d5a9940d8..26fcb832b 100644 --- a/nodedb/src/control/planner/calvin/write_class.rs +++ b/nodedb/src/control/planner/calvin/write_class.rs @@ -139,13 +139,14 @@ fn kv_is_write(op: &KvOp) -> bool { | KvOp::BatchGet { .. } | KvOp::FieldGet { .. } | KvOp::MaterializeScan { .. } - // `SortedIndexRank`/`TopK`/`Range`/`Count`/`Score` are `Permission::Read` - // (query-only) despite the `SortedIndex*` naming. + // `SortedIndexRank`/`TopK`/`Range`/`Count`/`Score`/`TxnRead` are + // `Permission::Read` (query-only) despite the `SortedIndex*` naming. | KvOp::SortedIndexRank { .. } | KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } // Read-only: reports what a governed write would apply, mutates // nothing, and is `NotAWrite` in `plan_vshard`. | KvOp::ResolveWrite(_) diff --git a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs index 9ab86cf58..cca6675aa 100644 --- a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs +++ b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs @@ -306,110 +306,40 @@ impl<'a> StatementExecutor<'a> { } } - /// Flush the procedure transaction buffer: WAL append + dispatch as batch. + /// Commit the procedure transaction buffer as one system transaction. + /// + /// Every statement's tasks stage through the same path a client + /// transaction takes, and COMMIT resolves them into one redo record that + /// installs all of them or none. Restart replay installs that same + /// record. Each statement's descriptor leases stay on the tasks it + /// buffered until COMMIT has checked them. pub(super) async fn flush_transaction_buffer(&self) -> crate::Result<()> { - let (tasks, _lease_scopes) = if let Some(ref tx_ctx) = self.tx_ctx { + let statements = if let Some(ref tx_ctx) = self.tx_ctx { let mut guard = tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); - guard.take_buffered() + guard.take_statements() } else { return Ok(()); }; - - // `_lease_scopes` owns every statement's descriptor admission through - // all WAL appends and the complete batch dispatch below. It is dropped - // only after this function returns, including on an execution error. - if tasks.is_empty() { + if statements.iter().all(|(tasks, _)| tasks.is_empty()) { return Ok(()); } - - // Each task's WAL record has its own LSN; the batch dispatch below - // carries the highest so the Data Plane's write-version floor advances - // past every write it applies. Same approximation for the resolved TTL - // instant: a single scalar can't represent one-per-task resolved - // instants for a heterogeneous multi-statement batch, so it is only - // threaded through when the buffer holds exactly one task (below); - // resolving that properly for N>1 would need `MetaOp::TransactionBatch` - // to carry a per-plan `Vec>`, a separate, wider change to - // the procedural batch-flush path, not this KV-write fix. - // - // Every record the loop appends is held under one outcome-floor - // window, which the funnel closes from the batch's outcome. - let owner = RecordOwner { - tenant_id: tasks[0].tenant_id, - database_id: tasks[0].database_id, - vshard_id: tasks[0].vshard_id, - }; - let minted = MintedRecords::open(&self.state.outcome_floor); - let mut single_task_resolved_now_ms: Option = None; - for task in &tasks { - let task_owner = RecordOwner { - tenant_id: task.tenant_id, - database_id: task.database_id, - vshard_id: task.vshard_id, - }; - let outcome = match minted.append_plan(&self.state.wal, task_owner, &task.plan) { - Ok(outcome) => outcome, - Err(error) => { - // The records appended so far never reach a core. - minted.cancel(&self.state.wal, owner, 0).await?; - return Err(error); - } - }; - single_task_resolved_now_ms = outcome.resolved_now_ms; - } - let max_wal_lsn = minted.highest(); - - if tasks.len() == 1 { - if let Some(task) = tasks.into_iter().next() { - crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( - self.state, - crate::control::server::dispatch_utils::WriteDispatch { - tenant_id: task.tenant_id, - database_id: task.database_id, - vshard_id: task.vshard_id, - plan: task.plan, - trace_id: TraceId::ZERO, - event_source: self.event_source, - txn_id: None, - wal_lsn: max_wal_lsn, - resolved_now_ms: single_task_resolved_now_ms, - minted: Some(minted), - }, - ) - .await?; - } - } else { - let tenant_id = tasks[0].tenant_id; - let database_id = tasks[0].database_id; - let vshard_id = tasks[0].vshard_id; - let plans: Vec<_> = tasks.into_iter().map(|t| t.plan).collect(); - let batch_plan = crate::bridge::envelope::PhysicalPlan::Meta( - nodedb_physical::physical_plan::MetaOp::TransactionBatch { - plans, - txn_id: None, - }, - ); - crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( - self.state, - crate::control::server::dispatch_utils::WriteDispatch { - tenant_id, - database_id, - vshard_id, - plan: batch_plan, - trace_id: TraceId::ZERO, - event_source: self.event_source, - txn_id: None, - wal_lsn: max_wal_lsn, - // N>1 batch: no single instant represents every task's - // resolved TTL — see the comment above the WAL-append loop. - resolved_now_ms: None, - minted: Some(minted), + let statements = statements + .into_iter() + .map( + |(tasks, lease_scope)| crate::control::system_txn::SystemTxnStatement { + tasks, + lease_scope: std::sync::Arc::new(lease_scope), }, ) - .await?; - } - - Ok(()) + .collect(); + crate::control::system_txn::run_statements_atomically( + self.state, + &self.identity_for_dispatch(), + statements, + self.event_source, + ) + .await + .map_err(crate::Error::from) } } diff --git a/nodedb/src/control/planner/procedural/executor/core/state.rs b/nodedb/src/control/planner/procedural/executor/core/state.rs index 6b5138c15..fa20841a8 100644 --- a/nodedb/src/control/planner/procedural/executor/core/state.rs +++ b/nodedb/src/control/planner/procedural/executor/core/state.rs @@ -35,7 +35,6 @@ pub struct CrossShardOrigin { /// Statement executor: steps through procedural SQL blocks with DML. pub struct StatementExecutor<'a> { pub(super) state: &'a SharedState, - #[allow(dead_code)] pub(super) identity: AuthenticatedIdentity, pub(super) tenant_id: TenantId, /// Database scope fixed for this executor's lifetime. diff --git a/nodedb/src/control/planner/procedural/executor/transaction.rs b/nodedb/src/control/planner/procedural/executor/transaction.rs index 0e67ed356..ae2ffad7f 100644 --- a/nodedb/src/control/planner/procedural/executor/transaction.rs +++ b/nodedb/src/control/planner/procedural/executor/transaction.rs @@ -23,7 +23,7 @@ struct BufferedStatementScope { /// Buffered transaction context for stored procedure execution. /// /// DML statements inside a procedure body are collected here until -/// an explicit COMMIT flushes them as a TransactionBatch, or ROLLBACK +/// an explicit COMMIT flushes them as one system transaction, or ROLLBACK /// discards them. An implicit COMMIT occurs at the end of the procedure. #[derive(Default)] pub struct ProcedureTransactionCtx { @@ -74,6 +74,26 @@ impl ProcedureTransactionCtx { (std::mem::take(&mut self.buffer), scopes) } + /// Take every buffered statement with its lease scope, in buffer order + /// (on COMMIT). Clears the savepoint stack. + pub fn take_statements(&mut self) -> Vec<(Vec, QueryLeaseScope)> { + self.savepoints.clear(); + let mut tasks = std::mem::take(&mut self.buffer); + let scopes = std::mem::take(&mut self.statement_scopes); + let mut statements = Vec::with_capacity(scopes.len() + 1); + // Each statement owns the tasks from its start to the next start. + for statement in scopes.into_iter().rev() { + let own = tasks.split_off(statement.task_start.min(tasks.len())); + statements.push((own, statement.scope)); + } + // Tasks buffered ahead of the first statement carry no lease. + if !tasks.is_empty() { + statements.push((tasks, QueryLeaseScope::empty())); + } + statements.reverse(); + statements + } + /// Take tasks only. Kept for existing task-oriented tests; it intentionally /// drops the associated scopes when the returned tasks are taken. pub fn take_buffered_tasks(&mut self) -> Vec { @@ -186,6 +206,20 @@ mod tests { assert!(ctx.take_buffered_tasks().is_empty()); } + #[test] + fn commit_takes_each_statement_with_its_own_tasks() { + let mut ctx = ProcedureTransactionCtx::new(); + ctx.buffer_statement( + vec![dummy_task("a"), dummy_task("b")], + QueryLeaseScope::empty(), + ); + ctx.buffer_statement(vec![dummy_task("c")], QueryLeaseScope::empty()); + let statements = ctx.take_statements(); + let sizes: Vec = statements.iter().map(|(tasks, _)| tasks.len()).collect(); + assert_eq!(sizes, vec![2, 1]); + assert!(ctx.take_statements().is_empty()); + } + #[test] fn rollback_and_commit_take_owned_statement_scopes() { let mut ctx = ProcedureTransactionCtx::new(); diff --git a/nodedb/src/control/planner/rls_injection/kv.rs b/nodedb/src/control/planner/rls_injection/kv.rs index 5091d6335..36c074b78 100644 --- a/nodedb/src/control/planner/rls_injection/kv.rs +++ b/nodedb/src/control/planner/rls_injection/kv.rs @@ -63,6 +63,14 @@ pub(super) fn inject_kv(ctx: &RlsCtx<'_>, op: &mut KvOp) -> crate::Result<()> { and the plan names only the index", ), + // Refuse: the reply is ranked keys, a rank, or a count, with no row + // body to filter. The plan names the owning collection. + KvOp::SortedIndexTxnRead { collection, .. } => ctx.refuse_if_policy( + collection, + "a sorted-index read returns ranked keys, a rank, or a count taken from stored rows, \ + so the row filter cannot be evaluated", + ), + // Admit now: a single-scalar `value` write has no field to name, // so it fails the same evaluation rather than a carve-out. KvOp::Put { diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs b/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs index 7d60dc9be..a889f13a2 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/kv.rs @@ -100,6 +100,14 @@ pub(super) fn apply_kv(ctx: &PermCtx<'_>, op: &mut KvOp) -> crate::Result<()> { and the plan names only the index", ), + // Refuse: the reply is ranked keys, a rank, or a count, with no row + // body to filter. The plan names the owning collection. + KvOp::SortedIndexTxnRead { collection, .. } => ctx.refuse_if_tree( + collection, + "a sorted-index read returns ranked keys, a rank, or a count taken from stored rows, \ + so the subtree filter cannot be evaluated", + ), + // Resolve against the wrapped op: it is the intercepted write // verbatim, so it authorizes at exactly the level that write does. KvOp::ResolveWrite(inner) => apply_kv(ctx, inner), diff --git a/nodedb/src/control/security/catalog/index_registry.rs b/nodedb/src/control/security/catalog/index_registry.rs index 7c88b21cc..d62a3cb72 100644 --- a/nodedb/src/control/security/catalog/index_registry.rs +++ b/nodedb/src/control/security/catalog/index_registry.rs @@ -146,11 +146,26 @@ impl SystemCatalog { Ok(removed) } - /// Every index record of one (database, tenant), in key order. + /// Every index record of one (database, tenant), in key order, with the + /// calling connection's buffered transactional DDL merged in. pub fn list_index_records( &self, database_id: u64, tenant_id: u64, + ) -> crate::Result> { + let committed = self.list_committed_index_records(database_id, tenant_id)?; + Ok(crate::control::catalog_overlay::resolve_index_records( + database_id, + tenant_id, + committed, + )) + } + + /// Committed-only listing, bypassing the transaction DDL overlay. + pub fn list_committed_index_records( + &self, + database_id: u64, + tenant_id: u64, ) -> crate::Result> { let prefix = tenant_prefix(database_id, tenant_id); let read_txn = self diff --git a/nodedb/src/control/security/catalog/vector_index_params.rs b/nodedb/src/control/security/catalog/vector_index_params.rs index 54cca2f89..0e9c34145 100644 --- a/nodedb/src/control/security/catalog/vector_index_params.rs +++ b/nodedb/src/control/security/catalog/vector_index_params.rs @@ -37,13 +37,35 @@ impl SystemCatalog { write_txn.commit().map_err(|e| catalog_err("commit", e)) } - /// Load vector index parameters for a specific collection/field. + /// Load vector index parameters for a specific collection/field, with + /// this connection's uncommitted transactional DDL merged in. pub fn get_vector_index_params( &self, database_id: u64, tenant_id: u64, collection: &str, field_name: &str, + ) -> crate::Result> { + let committed = + self.get_committed_vector_index_params(database_id, tenant_id, collection, field_name)?; + Ok( + crate::control::catalog_overlay::resolve_vector_index_params( + database_id, + tenant_id, + collection, + field_name, + committed, + ), + ) + } + + /// Committed-only read, bypassing the transaction DDL overlay. + pub fn get_committed_vector_index_params( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + field_name: &str, ) -> crate::Result> { let key = vector_index_params_key(database_id, tenant_id, collection, field_name); let read_txn = self diff --git a/nodedb/src/control/security/identity/plan_permission.rs b/nodedb/src/control/security/identity/plan_permission.rs index ef66cd9ad..9f43425b5 100644 --- a/nodedb/src/control/security/identity/plan_permission.rs +++ b/nodedb/src/control/security/identity/plan_permission.rs @@ -288,6 +288,7 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } // Read-only: reports what a governed write would apply; that write is authorized separately. | KvOp::ResolveWrite(_), ) => Permission::Read, diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index 84b588e65..bd91df676 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -9,7 +9,7 @@ use crate::control::server::shared::clone_write::CloneCheckedTask; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; -use super::minted::{MintedRecords, RecordOwner, resolve_on_response}; +use super::minted::{MintedRecords, RecordOwner}; use super::submit_write::{ ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; @@ -228,9 +228,7 @@ pub(crate) async fn dispatch_trusted_internal_write_to_data_plane( trace_id, event_source, txn_id, - // Caller pre-appended and supplied `wal_lsn` (e.g. the procedural - // batch-flush path whose dispatched plan is a `TransactionBatch` - // whose per-task records were appended upstream): the funnel must not + // Caller pre-appended and supplied `wal_lsn`: the funnel must not // append again. durability: WalDurability::CallerSupplied { wal_lsn, @@ -342,58 +340,51 @@ async fn dispatch_to_data_plane_inner( database_id, vshard_id, }; - // Resolve any Exchange data-movement nodes before dispatch: a root-level - // Gather fans the child to all cores and returns the merged response here; - // a Broadcast join child is gathered and embedded so the plan reaching a - // core is self-contained. Safe no-op for the many non-Exchange callers - // (writes, metrics, triggers). Catalog materialization is identity-scoped - // and already done upstream on the pgwire/native paths. - // Internal funnel (COPY, cursors, materialized-view refresh, constraint - // subqueries): not session-transaction-scoped, so `None`. - let resolved = crate::control::server::exchange::resolve_exchange_in_plan( - shared, - database_id, - tenant_id, - plan, - trace_id, - None, - ) - .await; - // A plan that never reaches the funnel closes the caller's records here. - let plan = match resolved { - Ok(crate::control::server::exchange::Resolved::Plan(p)) => *p, - Ok(crate::control::server::exchange::Resolved::Gathered( - resp, - _shard_watermarks, - _shuffle_reads, - )) => { + // A write that carries its own records is never a query. Only a query + // plan holds Exchange nodes, and resolving one fans it out to the cores, + // so a record-carrying query would reach the cores before any close. + let plan = if durability.has_minted() { + if matches!(plan, PhysicalPlan::Query(_)) { if let Some(minted) = durability.take_minted() { - resolve_on_response(&shared.wal, owner, 0, &resp, minted).await?; - } - return Ok(resp); - } - // Internal funnel callers want a fully-collected Response, not a lazy - // stream: materialize the stream into one merged-array Response, - // preserving the prior gather-then-return behaviour on this path. - Ok(crate::control::server::exchange::Resolved::Stream(s)) => { - let collected = crate::control::server::exchange::gather::stream_to_response(s).await; - if let Some(minted) = durability.take_minted() { - match &collected { - Ok(resp) => resolve_on_response(&shared.wal, owner, 0, resp, minted).await?, - // The gather failed part way: what reached the cores is - // unknown here. - Err(_) => minted.hold(), - } + minted.cancel(&shared.wal, owner, 0).await?; } - return collected; + return Err(crate::Error::Internal { + detail: "a write carrying WAL records reached the funnel as a query plan; \ + nothing was dispatched" + .into(), + }); } - Err(error) => { - // A write plan carries no exchange node, so a failed resolution - // dispatched none of it. - if let Some(minted) = durability.take_minted() { - minted.cancel(&shared.wal, owner, 0).await?; + plan + } else { + // Resolve any Exchange data-movement nodes before dispatch: a + // root-level Gather fans the child to all cores and returns the merged + // response here. A Broadcast join child is gathered and embedded so + // the plan reaching a core is self-contained. Plans with no Exchange + // node pass through unchanged. Catalog materialization is + // identity-scoped and already done upstream on the pgwire and native + // paths. The internal funnel is not session-transaction-scoped, so the + // transaction id is `None`. + let resolved = crate::control::server::exchange::resolve_exchange_in_plan( + shared, + database_id, + tenant_id, + plan, + trace_id, + None, + ) + .await?; + match resolved { + crate::control::server::exchange::Resolved::Plan(p) => *p, + crate::control::server::exchange::Resolved::Gathered( + resp, + _shard_watermarks, + _shuffle_reads, + ) => return Ok(resp), + // Internal funnel callers want one merged `Response`, not a lazy + // stream. + crate::control::server::exchange::Resolved::Stream(s) => { + return crate::control::server::exchange::gather::stream_to_response(s).await; } - return Err(error); } }; @@ -842,4 +833,99 @@ mod tests { .await; assert!(!replayed(&state).contains(&lsn.as_u64())); } + + /// A record-carrying write never resolves an Exchange, so nothing of it + /// reaches a core before its records close. + #[tokio::test] + async fn a_record_carrying_query_is_refused_before_any_fan_out() { + let (state, mut side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let mut write = write_with(minted, lsn); + write.plan = crate::bridge::envelope::PhysicalPlan::Query( + nodedb_physical::physical_plan::QueryOp::Exchange( + nodedb_physical::physical_plan::ExchangeOp { + child: Box::new(point_get_plan()), + mode: nodedb_physical::physical_plan::ExchangeMode::Gather { + as_aggregate: false, + }, + }, + ), + ); + + let result = super::dispatch_trusted_internal_write_to_data_plane(&state, write).await; + + assert!(result.is_err(), "a query plan cannot carry records"); + assert!( + side.request_rx.try_pop().is_err(), + "no request reached a core" + ); + assert!( + !replayed(&state).contains(&lsn.as_u64()), + "a marker names it" + ); + assert!(state.outcome_floor.floor() >= lsn); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + } + + /// A committed proposal refused for good is refused on every replica, so + /// its abort marker carries the proposal key and a redelivered copy finds + /// the refusal in the ledger rebuilt after a restart. + #[tokio::test] + async fn a_final_refusal_of_a_keyed_proposal_is_its_ledger_outcome() { + const KEY: u64 = 0xC0FF_EE01; + let (state, side, _directory) = fixture(); + let plan = + crate::bridge::envelope::PhysicalPlan::Kv(nodedb_physical::physical_plan::KvOp::Put { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), + key: b"k1".to_vec(), + value: b"v1".to_vec(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::new(1), + returning: None, + rls_filters: Vec::new(), + }); + let responder = tokio::spawn(respond_once_with( + Arc::clone(&state), + side, + Status::Error, + Some(crate::bridge::envelope::ErrorCode::RejectedConstraint { + constraint: "unique".into(), + detail: "duplicate key".into(), + }), + )); + + let outcome = super::submit_write( + &state, + super::SubmitWrite { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan, + trace_id: crate::types::TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + user_id: None, + durability: super::WalDurability::AppendHere { + now_override: None, + apply_key: KEY, + }, + ordering: super::WriteOrdering::AlreadyOrdered, + change_feed: super::ChangeFeedOwner::Unowned, + }, + ) + .await + .expect("the refusal is a response"); + responder.await.expect("responder completes"); + assert_eq!(outcome.response.status, Status::Error); + + state.wal.sync().expect("sync"); + let ledger = crate::control::distributed_applier::ProposalLedger::from_records( + &state.wal.replay().expect("replay"), + 8, + ); + assert!( + ledger.prior(KEY).is_some(), + "the refusal's marker names the proposal" + ); + } } diff --git a/nodedb/src/control/server/dispatch_utils/minted/mod.rs b/nodedb/src/control/server/dispatch_utils/minted/mod.rs index 3b8fb9b5e..847da8f72 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/mod.rs @@ -9,4 +9,3 @@ mod resolve; pub(crate) use owned::{Collect, OwnedResponse, OwnedWait, await_response_owned}; pub(crate) use records::{MintedRecords, RecordOwner}; -pub(crate) use resolve::resolve_on_response; diff --git a/nodedb/src/control/server/dispatch_utils/minted/records.rs b/nodedb/src/control/server/dispatch_utils/minted/records.rs index 13fe58186..af54d63e3 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/records.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/records.rs @@ -28,12 +28,12 @@ use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, Mutex, MutexGuard, OnceLock}; -use crate::bridge::dispatch::{OutcomeFloor, WriteWindow}; +use crate::bridge::dispatch::{OutcomeFloor, ResendRefusal, WriteWindow}; use crate::bridge::envelope::PhysicalPlan; use crate::control::server::wal_dispatch::{WalAppendOutcome, WalAppendRequest, wal_append}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::WalManager; -use crate::wal::manager::{NO_APPLY_KEY, RecordedAppend, WalAppender}; +use crate::wal::manager::{AppendSink, NO_APPLY_KEY, RecordedAppend, WalAppender}; /// Where a write's records live. The abort markers that cancel them carry it. #[derive(Debug, Clone, Copy)] @@ -80,12 +80,13 @@ impl MintedRecords { Self::with_window(floor.open_write(), None) } - /// Hold an existing record at `lsn` that is sent to a core again. `None` - /// when the floor already passed it: its outcome is final, and a second - /// apply would land below the floor. - pub(crate) fn resend(floor: &Arc, lsn: Lsn) -> Option { + /// Hold an existing record at `lsn` that is sent to a core again. + /// Refused, with the reason, when its outcome is final or a live or held + /// window carries it to one: a second apply would land below the floor, + /// or apply the record twice. + pub(crate) fn resend(floor: &Arc, lsn: Lsn) -> Result { let window = floor.open_existing(lsn)?; - Some(Self::with_window(window, Some(lsn))) + Ok(Self::with_window(window, Some(lsn))) } fn with_window(window: WriteWindow, resent: Option) -> Self { @@ -109,7 +110,7 @@ impl MintedRecords { apply_key: u64, ) -> WalAppender<'a> { self.wal.get_or_init(|| Arc::clone(wal)); - wal.recording_appender(apply_key, &self.appended) + wal.recording_appender(apply_key, self) } /// Append `plan`'s redo records under this window. @@ -152,19 +153,11 @@ impl MintedRecords { self.recorded().iter().map(|record| record.lsn).collect() } - /// Take the window and the records out, and note the highest LSN on the - /// window. `None` when a close already took them. + /// Take the window and the records out. `None` when a close already + /// took them. fn take_parts(&mut self) -> Option { let window = self.window.take()?; let appended = std::mem::take(&mut *self.recorded()); - let highest = appended - .iter() - .map(|record| record.lsn) - .chain(self.resent) - .max(); - if let Some(highest) = highest { - window.note_minted(highest); - } Some((window, appended, self.resent)) } @@ -334,6 +327,17 @@ impl MintedRecords { } } +/// Each append joins the set, and the window owns its LSN from the moment +/// the record exists. A resend of the record is refused while it is owned. +impl AppendSink for MintedRecords { + fn record(&self, append: RecordedAppend) { + if let Some(window) = &self.window { + window.own(append.lsn); + } + self.recorded().push(append); + } +} + impl Drop for MintedRecords { fn drop(&mut self) { self.close_dropped(); @@ -443,7 +447,7 @@ mod tests { assert!(replayed.contains(&lsn.as_u64()), "no marker names it"); assert_eq!(floor.floor(), lsn); assert!( - MintedRecords::resend(&floor, lsn).is_none(), + MintedRecords::resend(&floor, lsn).is_err(), "the floor passed the record" ); } @@ -555,4 +559,46 @@ mod tests { assert_eq!(floor.held_windows(), 1); assert_eq!(floor.leaked_windows(), 0); } + + /// A writer that appended a record and has not sent it yet owns it, so a + /// resend cannot race it to a core. + #[test] + fn a_resend_is_refused_while_a_live_window_owns_the_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let minted = MintedRecords::open(&floor); + let lsn = append(&wal, &minted, b"a"); + + assert!(floor.floor() < lsn); + assert!( + MintedRecords::resend(&floor, lsn).is_err(), + "the writer owns it" + ); + + minted.settle(); + } + + /// A record whose owner closed has a final outcome, even while an older + /// window keeps the floor below it. + #[test] + fn a_resend_is_refused_after_the_owner_closed_above_the_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = open_wal(&dir); + let floor = OutcomeFloor::new(); + let older = MintedRecords::open(&floor); + append(&wal, &older, b"a"); + let newer = MintedRecords::open(&floor); + let lsn = append(&wal, &newer, b"b"); + newer.settle(); + + assert!(floor.floor() < lsn, "the older window holds the floor"); + assert!( + MintedRecords::resend(&floor, lsn).is_err(), + "its outcome is final" + ); + + older.settle(); + assert!(floor.floor() >= lsn); + } } diff --git a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs index 701df3a88..ae74f51d3 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs @@ -8,7 +8,7 @@ use std::sync::Arc; -use crate::bridge::envelope::{Response, Status}; +use crate::bridge::envelope::{ErrorCode, Response, Status}; use crate::wal::WalManager; use super::super::write_abort::{refusal_is_final, write_definitely_not_applied}; @@ -17,8 +17,8 @@ use super::records::{MintedRecords, RecordOwner}; /// Close `minted` from the core's final `response`. /// /// `final_refusal_key` is the proposal key a final refusal's marker carries, -/// `0` when this write's refusals are never final. A failed cancel returns -/// the error and holds the window. +/// `0` when no proposal carries this write. A failed cancel returns the error +/// and holds the window. pub(crate) async fn resolve_on_response( wal: &Arc, owner: RecordOwner, @@ -37,6 +37,20 @@ pub(crate) async fn resolve_on_response( } else { 0 }; + // A refused sync frame advanced its stream's high-water mark. The + // frame's record is cancelled, so the mark gets a record of its + // own, durable with the markers. + if let ErrorCode::SyncRejected { provenance, .. } = code + && let Err(error) = wal.appender(marker_key).append_sync_seq_advance( + provenance.producer_id, + provenance.epoch, + provenance.stream_id, + provenance.seq, + ) + { + minted.hold(); + return Err(error); + } minted.cancel(wal, owner, marker_key).await } None => { @@ -238,4 +252,54 @@ mod tests { assert!(floor.floor() < lsn); assert_eq!(floor.leaked_windows(), 0); } + + /// A sync frame the gate refused for good is cancelled, and the + /// high-water mark it advanced is journalled on its own. Restart replay + /// then restores the mark, never applies the frame, and counts the + /// refusal as the proposal's outcome. + #[tokio::test] + async fn a_refused_sync_frame_journals_its_mark_and_cancels_its_record() { + const KEY: u64 = 0xAB; + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let rejected = response( + Status::Error, + Some(ErrorCode::SyncRejected { + violation: nodedb_types::sync::violation::ViolationType::PermissionDenied, + applied_seq: 4, + provenance: nodedb_types::sync::wire::SyncProvenance { + producer_id: 9, + epoch: 2, + stream_id: 1, + seq: 4, + }, + }), + false, + ); + + resolve_on_response(&wal, owner(), KEY, &rejected, minted) + .await + .expect("resolve"); + + wal.sync().expect("sync"); + let records = wal.replay().expect("replay"); + assert!( + !records + .iter() + .any(|record| record.header.lsn == lsn.as_u64()), + "the refused frame never replays" + ); + let (maps, _) = + crate::wal::replay::replay_sync_hwm_records(&records).expect("replay marks"); + assert_eq!(maps.sync_hwm.get(&(9, 1)), Some(&4)); + assert_eq!(maps.producer_epoch_floor.get(&9), Some(&2)); + let ledger = crate::control::distributed_applier::ProposalLedger::from_records(&records, 8); + assert!( + ledger.prior(KEY).is_some(), + "the refusal is the proposal's outcome" + ); + assert!(floor.floor() >= lsn); + } } diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index 5821fb7c8..e1b77515a 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -34,4 +34,6 @@ pub(crate) use submit_write::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; pub(crate) use types::{AutocommitWrite, WriteDispatch}; -pub(crate) use write_abort::{refusal_is_final, write_definitely_not_applied}; +pub(crate) use write_abort::{ + error_is_final_refusal, refusal_is_final, write_definitely_not_applied, +}; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index 46da61d60..b53a387e4 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -76,21 +76,13 @@ pub(crate) async fn submit_write( WalDurability::AppendHere { apply_key, .. } => *apply_key, WalDurability::CallerSupplied { .. } => 0, }; - // A transaction redo's refusal is final: every replica reaches it at the - // same log position against the same state. Its abort marker carries the - // entry's key, so the proposal ledger counts the refusal as the entry's - // outcome after a restart. Any other refused write keeps its entry - // replayable, so its abort marker carries no key. - let final_refusal_key = if matches!( - plan, - nodedb_physical::physical_plan::PhysicalPlan::Meta( - nodedb_physical::physical_plan::MetaOp::ApplyTransactionRedo { .. } - ) - ) { - apply_key - } else { - 0 - }; + // A committed proposal applies in log order against the same state on + // every replica, so a final refusal is its outcome everywhere. The abort + // marker of a final refusal carries the proposal's key, so the proposal + // ledger counts the refusal as the entry's outcome after a restart and a + // redelivered copy is never applied. A write no proposal carries has key + // `0`. + let final_refusal_key = apply_key; // Durable-at-ack obligation, also computed before `plan` moves. `Some` only // for a write whose redo record THIS funnel is required to mint; a caller diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs index 0468a3630..1ca820a09 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs @@ -49,6 +49,17 @@ pub(crate) enum WalDurability { } impl WalDurability { + /// Whether the caller supplied records it appended for this write. + pub(crate) fn has_minted(&self) -> bool { + matches!( + self, + Self::CallerSupplied { + minted: Some(_), + .. + } + ) + } + /// Take the caller's minted records out, leaving `None` in their place. pub(crate) fn take_minted(&mut self) -> Option { match self { diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index 746251ebf..f96c73818 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -22,9 +22,9 @@ //! The match is exhaustive on purpose. A new [`ErrorCode`] must be classified by //! whoever adds it, not silently inherit either answer. //! -//! [`resolve_on_response`](super::minted::resolve_on_response) is the one -//! place that acts on that verdict. Every write that mints a record for a -//! Data-Plane dispatch resolves its records there. +//! `minted::resolve_on_response` is the one place that acts on that +//! verdict. Every write that mints a record for a Data-Plane dispatch +//! resolves its records there. use crate::bridge::envelope::ErrorCode; @@ -53,6 +53,12 @@ pub(crate) fn refusal_is_final(code: &ErrorCode) -> bool { ) } +/// Whether a committed proposal's apply `error` is a final refusal: the +/// entry's outcome, which a redelivery must answer with and never apply. +pub(crate) fn error_is_final_refusal(error: &crate::Error) -> bool { + matches!(error, crate::Error::DataPlane(code) if refusal_is_final(code)) +} + /// Whether `code` proves the write was refused without applying anything. /// /// `false` means "not established", not "the write applied". @@ -63,6 +69,8 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { // refusal instead of installing the write. ErrorCode::RejectedConstraint { .. } | ErrorCode::RejectedPrevalidation { .. } + // The sync gate refused the frame before its delta installed. + | ErrorCode::SyncRejected { .. } | ErrorCode::RejectedAuthz { .. } | ErrorCode::RejectedDanglingEdge { .. } | ErrorCode::AppendOnlyViolation { .. } diff --git a/nodedb/src/control/server/native/dispatch/index_ddl_op.rs b/nodedb/src/control/server/native/dispatch/index_ddl_op.rs new file mode 100644 index 000000000..5b21a87d3 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/index_ddl_op.rs @@ -0,0 +1,393 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native index-DDL opcodes run the SQL DDL statements they name. +//! +//! | opcode | SQL | +//! |-------------------------|--------------------------------------------------| +//! | `KvRegisterSortedIndex` | `CREATE SORTED INDEX` | +//! | `KvDropSortedIndex` | `DROP SORTED INDEX` | +//! | `VectorSetParams` | `ALTER VECTOR INDEX`, or `CREATE VECTOR INDEX` | +//! | `KvRegisterIndex` | `CREATE INDEX IF NOT EXISTS ON c (field)` | +//! | `KvDropIndex` | `DROP INDEX` of the index on the field | +//! | `DocumentDropIndex` | `DROP INDEX` of the index on the field | +//! | `DocumentRegister` | `CREATE COLLECTION IF NOT EXISTS`, then one `CREATE INDEX IF NOT EXISTS` per index path | +//! +//! The opcode then reaches the same catalog entries, engine steps and +//! per-connection transaction buffer as the SQL statements. The index is in +//! the catalog, survives a restart, and is listed by `SHOW INDEXES`. Inside +//! an explicit transaction it is visible to later statements, applies at +//! COMMIT and is discarded on ROLLBACK. +//! +//! `DocumentRegister` runs its statements in order and stops at the first +//! error. Each statement is `IF NOT EXISTS`, so a retry after a partial run +//! completes it. +//! +//! The opcode answers like any other opcode: `Ok` with no status row, count +//! or verb on success, the first failing statement's error on failure. +//! +//! Every rendered identifier passes the SQL identifier rules and is a plain +//! token of ASCII letters, digits and underscores, so no field value can +//! change the statement's shape. + +use nodedb_types::protocol::{NativeResponse, OpCode, TextFields}; + +use super::{DispatchCtx, error_to_native_with_sqlstate, handle_sql}; + +/// Run the index-DDL opcode `op` as the SQL statement it names. +pub(crate) async fn handle_index_ddl_op( + ctx: &DispatchCtx<'_>, + seq: u64, + op: OpCode, + fields: &TextFields, +) -> NativeResponse { + let collection = fields + .collection + .as_deref() + .unwrap_or("default") + .to_lowercase(); + let statements = match index_ddl_statements(ctx, op, fields, &collection) { + Ok(statements) => statements, + Err(error) => return error_to_native_with_sqlstate(seq, "42601", &error), + }; + for sql in &statements { + let response = handle_sql(ctx, seq, sql, None).await; + // A failure keeps the statement's error. + if response.status != nodedb_types::protocol::ResponseStatus::Ok { + return response; + } + } + // An opcode answers as every opcode does: success carries no DDL status + // row, no count and no verb. + NativeResponse::ok(seq) +} + +/// The DDL statements opcode `op` names, in the order they run. +fn index_ddl_statements( + ctx: &DispatchCtx<'_>, + op: OpCode, + fields: &TextFields, + collection: &str, +) -> crate::Result> { + match op { + OpCode::KvRegisterSortedIndex => Ok(vec![create_sorted_index(fields, collection)?]), + OpCode::KvDropSortedIndex => { + let name = ident(required(fields.index_name.as_deref(), "index_name")?)?; + Ok(vec![format!("DROP SORTED INDEX {name}")]) + } + OpCode::VectorSetParams => Ok(vec![vector_set_params(ctx, fields, collection)?]), + OpCode::KvRegisterIndex => Ok(vec![kv_register_index(fields, collection)?]), + OpCode::KvDropIndex | OpCode::DocumentDropIndex => { + Ok(vec![drop_index_on_field(ctx, fields, collection)?]) + } + OpCode::DocumentRegister => document_register(fields, collection), + other => Err(bad_request(format!( + "opcode {other:?} is not an index DDL opcode" + ))), + } +} + +/// `KvRegisterIndex` indexes one field of the collection, backfilled from +/// every row it holds. +/// +/// `backfill = false` is refused: a catalog index answers lookups for every +/// row, and an index that skipped the existing rows would miss them. +fn kv_register_index(fields: &TextFields, collection: &str) -> crate::Result { + if fields.backfill == Some(false) { + return Err(bad_request( + "KvRegisterIndex with backfill = false is not supported: an index covers every \ + row, so it is built from the rows the collection already holds", + )); + } + let collection = ident(collection)?; + let field = index_field(required(fields.field.as_deref(), "field")?)?; + Ok(format!( + "CREATE INDEX IF NOT EXISTS ON {collection} ({field})" + )) +} + +/// `DocumentRegister` makes sure the document collection exists, then +/// indexes each of its index paths. +fn document_register(fields: &TextFields, collection: &str) -> crate::Result> { + let collection = ident(collection)?; + let mut statements = vec![format!( + "CREATE COLLECTION IF NOT EXISTS {collection} WITH (engine='document_schemaless')" + )]; + for path in fields.index_paths.as_deref().unwrap_or_default() { + let field = index_field(path)?; + statements.push(format!( + "CREATE INDEX IF NOT EXISTS ON {collection} ({field})" + )); + } + Ok(statements) +} + +/// A top-level field named bare or as `$.field`. +fn index_field(raw: &str) -> crate::Result { + ident(raw.strip_prefix("$.").unwrap_or(raw)) +} + +fn create_sorted_index(fields: &TextFields, collection: &str) -> crate::Result { + let name = ident(required(fields.index_name.as_deref(), "index_name")?)?; + let collection = ident(collection)?; + let columns = fields + .sort_columns + .as_deref() + .filter(|columns| !columns.is_empty()) + .ok_or_else(|| bad_request("missing 'sort_columns'"))? + .iter() + .map(|(column, direction)| { + let direction = direction.to_ascii_uppercase(); + if direction != "ASC" && direction != "DESC" { + return Err(bad_request(format!( + "invalid sort direction '{direction}', expected ASC or DESC" + ))); + } + Ok(format!("{} {direction}", ident(column)?)) + }) + .collect::>>()? + .join(", "); + let key = ident(required(fields.key_column.as_deref(), "key_column")?)?; + let mut sql = format!("CREATE SORTED INDEX {name} ON {collection} ({columns}) KEY {key}"); + match fields.window_type.as_deref().map(str::to_ascii_lowercase) { + None => {} + Some(window) if window == "none" => {} + Some(window) if matches!(window.as_str(), "daily" | "weekly" | "monthly") => { + sql.push_str(&format!(" WINDOW {}", window.to_ascii_uppercase())); + if let Some(column) = fields.window_timestamp_column.as_deref() { + sql.push_str(&format!(" ON {}", ident(column)?)); + } + } + Some(window) if window == "custom" => { + sql.push_str(&format!( + " WINDOW CUSTOM START {} END {}", + fields.window_start_ms.unwrap_or(0), + fields.window_end_ms.unwrap_or(0) + )); + if let Some(column) = fields.window_timestamp_column.as_deref() { + sql.push_str(&format!(" ON {}", ident(column)?)); + } + } + Some(window) => { + return Err(bad_request(format!( + "invalid window type '{window}', expected none, daily, weekly, monthly or \ + custom" + ))); + } + } + Ok(sql) +} + +/// `VectorSetParams` alters the vector index the column already carries, or +/// creates one when it has none. +fn vector_set_params( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let collection = ident(collection)?; + let column = match fields.field_name.as_deref().filter(|f| !f.is_empty()) { + Some(column) => Some(ident(column)?), + None => None, + }; + let index_type = fields.index_type.as_deref().map(plain_token).transpose()?; + let metric = fields.metric.as_deref().map(plain_token).transpose()?; + let existing = ctx.state.credentials.catalog().get_vector_index_params( + ctx.database_id().as_u64(), + ctx.tenant_id().as_u64(), + &collection, + column.as_deref().unwrap_or(""), + )?; + + if let Some(existing) = existing { + if metric.as_deref().is_some_and(|m| m != existing.metric) { + return Err(bad_request(format!( + "the vector index on '{collection}' uses metric '{}'; a metric change needs \ + a new index", + existing.metric + ))); + } + let mut set = Vec::new(); + if let Some(m) = fields.m { + set.push(format!("m = {m}")); + } + if let Some(ef) = fields.ef_construction { + set.push(format!("ef_construction = {ef}")); + } + if let Some(index_type) = &index_type { + set.push(format!("index_type = {index_type}")); + } + if set.is_empty() { + return Err(bad_request( + "VectorSetParams on an existing index needs m, ef_construction or index_type", + )); + } + let target = match &column { + Some(column) => format!("{collection}.{column}"), + None => collection.clone(), + }; + return Ok(format!( + "ALTER VECTOR INDEX ON {target} SET ({})", + set.join(", ") + )); + } + + let name = match fields.index_name.as_deref() { + Some(name) => ident(name)?, + None => match &column { + Some(column) => ident(&format!("vec_{collection}_{column}"))?, + None => ident(&format!("vec_{collection}"))?, + }, + }; + let mut sql = format!("CREATE VECTOR INDEX {name} ON {collection}"); + if let Some(column) = &column { + sql.push_str(&format!(" ({column})")); + } + sql.push_str(&format!(" DIM {}", fields.vector_dim.unwrap_or(0))); + if let Some(metric) = &metric { + sql.push_str(&format!(" METRIC {metric}")); + } + if let Some(m) = fields.m { + sql.push_str(&format!(" M {m}")); + } + if let Some(ef) = fields.ef_construction { + sql.push_str(&format!(" EF_CONSTRUCTION {ef}")); + } + if let Some(index_type) = &index_type { + sql.push_str(&format!(" INDEX_TYPE {index_type}")); + } + Ok(sql) +} + +/// `KvDropIndex` and `DocumentDropIndex` name a field; the SQL statement +/// names the index on it. +fn drop_index_on_field( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let field = required(fields.field.as_deref(), "field")?; + let path = if field.starts_with('$') { + field.to_string() + } else { + format!("$.{field}") + }; + let stored = ctx + .state + .credentials + .catalog() + .get_collection(ctx.database_id(), ctx.tenant_id().as_u64(), collection)? + .ok_or_else(|| crate::Error::CollectionNotFound { + tenant_id: ctx.tenant_id(), + collection: collection.to_string(), + })?; + let index = stored + .indexes + .iter() + .find(|index| index.field == path) + .ok_or_else(|| bad_request(format!("no index on field '{field}' of '{collection}'")))?; + Ok(format!("DROP INDEX {}", ident(&index.name)?)) +} + +fn required<'a>(value: Option<&'a str>, name: &str) -> crate::Result<&'a str> { + value + .filter(|v| !v.is_empty()) + .ok_or_else(|| bad_request(format!("missing '{name}'"))) +} + +/// A SQL identifier that renders without quoting. +fn ident(raw: &str) -> crate::Result { + let normalized = + nodedb_sql::reserved::check_identifier(raw).map_err(|e| bad_request(e.to_string()))?; + plain_token(&normalized) +} + +/// A token of ASCII letters, digits and underscores. +fn plain_token(raw: &str) -> crate::Result { + let plain = !raw.is_empty() && raw.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'_'); + if plain { + Ok(raw.to_string()) + } else { + Err(bad_request(format!( + "'{raw}' must be letters, digits and underscores" + ))) + } +} + +fn bad_request(detail: impl Into) -> crate::Error { + crate::Error::BadRequest { + detail: detail.into(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_sorted_index_opcode_renders_its_statement() { + let fields = TextFields { + index_name: Some("lb".into()), + sort_columns: Some(vec![("score".into(), "desc".into())]), + key_column: Some("player".into()), + window_type: Some("daily".into()), + window_timestamp_column: Some("ts".into()), + ..TextFields::default() + }; + assert_eq!( + create_sorted_index(&fields, "scores").expect("renders"), + "CREATE SORTED INDEX lb ON scores (score DESC) KEY player WINDOW DAILY ON ts" + ); + } + + #[test] + fn a_kv_index_opcode_renders_an_idempotent_create() { + let fields = TextFields { + field: Some("$.region".into()), + ..TextFields::default() + }; + assert_eq!( + kv_register_index(&fields, "sessions").expect("renders"), + "CREATE INDEX IF NOT EXISTS ON sessions (region)" + ); + let skip_backfill = TextFields { + field: Some("region".into()), + backfill: Some(false), + ..TextFields::default() + }; + assert!(kv_register_index(&skip_backfill, "sessions").is_err()); + } + + #[test] + fn a_register_opcode_creates_the_collection_then_each_index() { + let fields = TextFields { + index_paths: Some(vec!["$.region".into(), "status".into()]), + ..TextFields::default() + }; + assert_eq!( + document_register(&fields, "orders").expect("renders"), + vec![ + "CREATE COLLECTION IF NOT EXISTS orders WITH (engine='document_schemaless')" + .to_string(), + "CREATE INDEX IF NOT EXISTS ON orders (region)".to_string(), + "CREATE INDEX IF NOT EXISTS ON orders (status)".to_string(), + ] + ); + let nested = TextFields { + index_paths: Some(vec!["$.a.b".into()]), + ..TextFields::default() + }; + assert!(document_register(&nested, "orders").is_err()); + } + + #[test] + fn a_field_that_would_change_the_statement_is_refused() { + let fields = TextFields { + index_name: Some("lb; DROP".into()), + sort_columns: Some(vec![("score".into(), "DESC".into())]), + key_column: Some("player".into()), + ..TextFields::default() + }; + assert!(create_sorted_index(&fields, "scores").is_err()); + assert!(plain_token("cosine) KEY x").is_err()); + } +} diff --git a/nodedb/src/control/server/native/dispatch/mod.rs b/nodedb/src/control/server/native/dispatch/mod.rs index 344088f22..273dc5815 100644 --- a/nodedb/src/control/server/native/dispatch/mod.rs +++ b/nodedb/src/control/server/native/dispatch/mod.rs @@ -10,12 +10,14 @@ mod ctx; mod direct_ops; mod edge_recon_gate; mod graph_match; +mod index_ddl_op; mod limits; mod plan_builder; pub(crate) mod raw_dispatch; pub(crate) mod response; mod session_ops; mod single_task; +mod sorted_read_op; mod sql; mod sql_admin; mod sql_dispatch_task; @@ -35,7 +37,9 @@ pub(crate) use conversion::{ pub(crate) use ctx::DispatchCtx; pub(crate) use direct_ops::handle_direct_op; pub(crate) use graph_match::handle_graph_match; +pub(crate) use index_ddl_op::handle_index_ddl_op; pub(crate) use session_ops::{handle_reset, handle_set, handle_show, show_all}; +pub(crate) use sorted_read_op::handle_sorted_read_op; pub(crate) use sql::{handle_sql, handle_sql_streaming}; pub(crate) use streaming::{SqlOutcome, SqlStream}; pub(crate) use transaction::{NativeTxnDp, handle_begin, handle_commit, handle_rollback}; diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs b/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs index bc5b4ecbf..f86701665 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs @@ -9,7 +9,8 @@ use crate::bridge::envelope::PhysicalPlan; use super::super::DispatchCtx; use super::{ - columnar, crdt, document, graph, kv, kv_counter, query, spatial, text, timeseries, vector, + columnar, crdt, document, document_bulk, graph, kv, kv_counter, query, spatial, text, + timeseries, vector, }; /// Build a PhysicalPlan from an opcode and request fields. @@ -29,8 +30,8 @@ pub(crate) fn build_plan( OpCode::DocumentUpdate => document::build_update(ctx, fields, collection), OpCode::DocumentScan => document::build_scan(ctx, fields, collection), OpCode::DocumentUpsert => document::build_upsert(ctx, fields, collection), - OpCode::DocumentBulkUpdate => document::build_bulk_update(ctx, fields, collection), - OpCode::DocumentBulkDelete => document::build_bulk_delete(ctx, fields, collection), + OpCode::DocumentBulkUpdate => document_bulk::build_bulk_update(ctx, fields, collection), + OpCode::DocumentBulkDelete => document_bulk::build_bulk_delete(ctx, fields, collection), // Vector. OpCode::VectorSearch => vector::build_search(ctx, fields, collection), OpCode::VectorBatchInsert => vector::build_batch_insert(ctx, fields, collection), @@ -76,30 +77,18 @@ pub(crate) fn build_plan( OpCode::GraphAlgo => graph::build_algo(fields, collection), OpCode::GraphMatch => graph::build_match(fields, collection), // Document DDL. - OpCode::DocumentTruncate => document::build_truncate(ctx, collection), - OpCode::DocumentEstimateCount => document::build_estimate_count(ctx, fields, collection), - OpCode::DocumentInsertSelect => document::build_insert_select(ctx, fields, collection), - OpCode::DocumentRegister => document::build_register(ctx, fields, collection), - OpCode::DocumentDropIndex => document::build_drop_index(ctx, fields, collection), - // KV DDL. - OpCode::KvRegisterIndex => kv::build_register_index(ctx, fields, collection), - OpCode::KvDropIndex => kv::build_drop_index(ctx, fields, collection), + OpCode::DocumentTruncate => document_bulk::build_truncate(ctx, collection), + OpCode::DocumentEstimateCount => { + document_bulk::build_estimate_count(ctx, fields, collection) + } + OpCode::DocumentInsertSelect => document_bulk::build_insert_select(ctx, fields, collection), + // KV truncate. OpCode::KvTruncate => kv::build_truncate(ctx, collection), // KV atomic operations. OpCode::KvIncr => kv_counter::build_incr(ctx, collection, fields), OpCode::KvIncrFloat => kv_counter::build_incr_float(ctx, collection, fields), OpCode::KvCas => kv::build_cas(ctx, collection, fields), OpCode::KvGetSet => kv::build_getset(ctx, collection, fields), - // KV sorted index operations. - OpCode::KvRegisterSortedIndex => kv::build_register_sorted_index(ctx, collection, fields), - OpCode::KvDropSortedIndex => kv::build_drop_sorted_index(fields), - OpCode::KvSortedIndexRank => kv::build_sorted_index_rank(fields), - OpCode::KvSortedIndexTopK => kv::build_sorted_index_top_k(fields), - OpCode::KvSortedIndexRange => kv::build_sorted_index_range(fields), - OpCode::KvSortedIndexCount => kv::build_sorted_index_count(fields), - OpCode::KvSortedIndexScore => kv::build_sorted_index_score(fields), - // Vector DDL. - OpCode::VectorSetParams => vector::build_set_params(ctx, fields, collection), // Query. OpCode::RecursiveScan => query::build_recursive_scan(ctx, fields, collection), _ => Err(crate::Error::BadRequest { diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index 0c2b6b3a3..5c0c0431e 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -88,12 +88,19 @@ pub(crate) fn build_point_put( Some(CollectionType::Columnar(ColumnarProfile::Timeseries { .. })) => { let json_str = String::from_utf8_lossy(&value); let ilp_line = format!("{collection} value={json_str}\n"); + // The line's own surrogate keys its staged row, so a read later in + // the same transaction observes it. + let (surrogate, _identity) = ctx.state.surrogate_assigner.assign_fresh( + ctx.database_id(), + ctx.tenant_id(), + collection, + )?; Ok(PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: QualifiedCollection::new(ctx.database_id(), collection), payload: ilp_line.into_bytes(), format: "ilp".to_string(), wal_lsn: None, - surrogates: Vec::new(), + surrogates: vec![surrogate], provenance: None, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, @@ -353,183 +360,3 @@ pub(crate) fn build_upsert( resolved_sum_targets: Vec::new(), })) } - -pub(crate) fn build_bulk_update( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let filters = fields - .filters - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'filters'".to_string(), - })? - .clone(); - let updates: Vec<(String, nodedb_physical::physical_plan::UpdateValue)> = fields - .updates - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'updates'".to_string(), - })? - .iter() - .map(|(f, b)| { - ( - f.clone(), - nodedb_physical::physical_plan::UpdateValue::Literal(b.clone()), - ) - }) - .collect(); - Ok(PhysicalPlan::Document(DocumentOp::BulkUpdate { - collection: QualifiedCollection::new(ctx.database_id(), collection), - filters, - updates, - returning: None, - ollp_predicted_surrogates: None, - ollp_predicted_edges: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - // Filled in by the materialized-sum resolution pass. - resolved_sum_targets: Vec::new(), - // See `build_update`: reads the declared PRIMARY KEY from the catalog. - declared_primary_key: declared_primary_key(ctx, collection)?, - })) -} - -pub(crate) fn build_bulk_delete( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let filters = fields - .filters - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'filters'".to_string(), - })? - .clone(); - Ok(PhysicalPlan::Document(DocumentOp::BulkDelete { - collection: QualifiedCollection::new(ctx.database_id(), collection), - filters, - returning: None, - ollp_predicted_surrogates: None, - ollp_predicted_edges: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - // Filled in by the materialized-sum resolution pass. - resolved_sum_targets: Vec::new(), - // See `build_update`: reads the declared PRIMARY KEY from the catalog. - declared_primary_key: declared_primary_key(ctx, collection)?, - })) -} - -pub(crate) fn build_truncate( - ctx: &DispatchCtx<'_>, - collection: &str, -) -> crate::Result { - Ok(PhysicalPlan::Document(DocumentOp::Truncate { - collection: QualifiedCollection::new(ctx.database_id(), collection), - restart_identity: false, - // Filled in by the materialized-sum resolution pass. - resolved_sum_targets: Vec::new(), - // See `build_update`: reads the declared PRIMARY KEY from the catalog. - declared_primary_key: declared_primary_key(ctx, collection)?, - })) -} - -pub(crate) fn build_estimate_count( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let field = fields.field.as_deref().unwrap_or("id").to_string(); - - Ok(PhysicalPlan::Document(DocumentOp::EstimateCount { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field, - })) -} - -pub(crate) fn build_insert_select( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let source = fields - .source_collection - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'source_collection'".to_string(), - })? - .clone(); - let filters = fields.filters.clone().unwrap_or_default(); - let limit = fields.limit.unwrap_or(10_000) as usize; - - Ok(PhysicalPlan::Document(DocumentOp::InsertSelect { - target_collection: QualifiedCollection::new(ctx.database_id(), collection), - source_collection: QualifiedCollection::new(ctx.database_id(), &source), - source_filters: filters, - source_limit: limit, - // The native text-field form names no projection, so every source row - // copies unchanged. - column_map: Vec::new(), - })) -} - -pub(crate) fn build_register( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - // Native protocol exposes only a legacy `index_paths` text list — promote - // each entry to a `Ready`, non-unique `RegisteredIndex` named after the - // path. UNIQUE / COLLATE / build-state come from SQL DDL only. - let indexes = fields - .index_paths - .clone() - .unwrap_or_default() - .into_iter() - .map(|path| nodedb_physical::physical_plan::RegisteredIndex { - name: path.clone(), - path, - unique: false, - case_insensitive: false, - state: nodedb_physical::physical_plan::RegisteredIndexState::Ready, - predicate: None, - }) - .collect(); - - Ok(PhysicalPlan::Document(DocumentOp::Register { - collection: QualifiedCollection::new(ctx.database_id(), collection), - indexes, - crdt_enabled: false, - storage_mode: nodedb_physical::physical_plan::StorageMode::Schemaless, - enforcement: Box::new(nodedb_physical::physical_plan::EnforcementOptions::default()), - bitemporal: false, - conflict_policy: None, - timeseries: None, - // The native protocol's register frame carries only index paths; a - // vector-primary collection is created through SQL DDL, which goes - // through the catalog-sourced builder instead. - vector_primary: None, - })) -} - -pub(crate) fn build_drop_index( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let field = fields - .field - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'field'".to_string(), - })? - .clone(); - - Ok(PhysicalPlan::Document(DocumentOp::DropIndex { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field, - })) -} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs new file mode 100644 index 000000000..45681ae25 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Document engine plan builders for set-level writes: predicate update and +//! delete, truncate, insert-select, and the row-count estimate. + +use nodedb_types::QualifiedCollection; +use nodedb_types::protocol::TextFields; + +use crate::bridge::envelope::PhysicalPlan; +use nodedb_physical::physical_plan::DocumentOp; + +use super::super::DispatchCtx; +use super::declared_primary_key; + +pub(crate) fn build_bulk_update( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let filters = fields + .filters + .as_ref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'filters'".to_string(), + })? + .clone(); + let updates: Vec<(String, nodedb_physical::physical_plan::UpdateValue)> = fields + .updates + .as_ref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'updates'".to_string(), + })? + .iter() + .map(|(f, b)| { + ( + f.clone(), + nodedb_physical::physical_plan::UpdateValue::Literal(b.clone()), + ) + }) + .collect(); + Ok(PhysicalPlan::Document(DocumentOp::BulkUpdate { + collection: QualifiedCollection::new(ctx.database_id(), collection), + filters, + updates, + returning: None, + ollp_predicted_surrogates: None, + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Filled in by the materialized-sum resolution pass. + resolved_sum_targets: Vec::new(), + // See `build_update`: reads the declared PRIMARY KEY from the catalog. + declared_primary_key: declared_primary_key(ctx, collection)?, + })) +} + +pub(crate) fn build_bulk_delete( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let filters = fields + .filters + .as_ref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'filters'".to_string(), + })? + .clone(); + Ok(PhysicalPlan::Document(DocumentOp::BulkDelete { + collection: QualifiedCollection::new(ctx.database_id(), collection), + filters, + returning: None, + ollp_predicted_surrogates: None, + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Filled in by the materialized-sum resolution pass. + resolved_sum_targets: Vec::new(), + // See `build_update`: reads the declared PRIMARY KEY from the catalog. + declared_primary_key: declared_primary_key(ctx, collection)?, + })) +} + +pub(crate) fn build_truncate( + ctx: &DispatchCtx<'_>, + collection: &str, +) -> crate::Result { + Ok(PhysicalPlan::Document(DocumentOp::Truncate { + collection: QualifiedCollection::new(ctx.database_id(), collection), + restart_identity: false, + // Filled in by the materialized-sum resolution pass. + resolved_sum_targets: Vec::new(), + // See `build_update`: reads the declared PRIMARY KEY from the catalog. + declared_primary_key: declared_primary_key(ctx, collection)?, + })) +} + +pub(crate) fn build_estimate_count( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let field = fields.field.as_deref().unwrap_or("id").to_string(); + + Ok(PhysicalPlan::Document(DocumentOp::EstimateCount { + collection: QualifiedCollection::new(ctx.database_id(), collection), + field, + })) +} + +pub(crate) fn build_insert_select( + ctx: &DispatchCtx<'_>, + fields: &TextFields, + collection: &str, +) -> crate::Result { + let source = fields + .source_collection + .as_ref() + .ok_or_else(|| crate::Error::BadRequest { + detail: "missing 'source_collection'".to_string(), + })? + .clone(); + let filters = fields.filters.clone().unwrap_or_default(); + let limit = fields.limit.unwrap_or(10_000) as usize; + + Ok(PhysicalPlan::Document(DocumentOp::InsertSelect { + target_collection: QualifiedCollection::new(ctx.database_id(), collection), + source_collection: QualifiedCollection::new(ctx.database_id(), &source), + source_filters: filters, + source_limit: limit, + // The native text-field form names no projection, so every source row + // copies unchanged. + column_map: Vec::new(), + })) +} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index 720edea80..b3e7c33fa 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -207,48 +207,6 @@ fn require_key_bytes(fields: &TextFields) -> crate::Result> { }) } -pub(crate) fn build_register_index( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let field = fields - .field - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'field'".to_string(), - })? - .clone(); - let field_position = fields.field_position.unwrap_or(0) as usize; - let backfill = fields.backfill.unwrap_or(true); - - Ok(PhysicalPlan::Kv(KvOp::RegisterIndex { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field, - field_position, - backfill, - })) -} - -pub(crate) fn build_drop_index( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let field = fields - .field - .as_ref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'field'".to_string(), - })? - .clone(); - - Ok(PhysicalPlan::Kv(KvOp::DropIndex { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field, - })) -} - pub(crate) fn build_truncate( ctx: &DispatchCtx<'_>, collection: &str, @@ -330,123 +288,3 @@ pub(crate) fn build_getset( rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), })) } - -pub(crate) fn build_register_sorted_index( - ctx: &DispatchCtx<'_>, - collection: &str, - fields: &TextFields, -) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - let sort_columns = fields.sort_columns.clone().unwrap_or_default(); - let key_column = fields.key_column.clone().unwrap_or_default(); - let window_type = fields.window_type.clone().unwrap_or_else(|| "none".into()); - let window_timestamp_column = fields.window_timestamp_column.clone().unwrap_or_default(); - let window_start_ms = fields.window_start_ms.unwrap_or(0); - let window_end_ms = fields.window_end_ms.unwrap_or(0); - - Ok(PhysicalPlan::Kv(KvOp::RegisterSortedIndex { - collection: QualifiedCollection::new(ctx.database_id(), collection), - index_name: index_name.to_string(), - sort_columns, - key_column, - window_type, - window_timestamp_column, - window_start_ms, - window_end_ms, - })) -} - -pub(crate) fn build_drop_sorted_index(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::DropSortedIndex { - index_name: index_name.to_string(), - })) -} - -pub(crate) fn build_sorted_index_rank(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - let key = fields - .key - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'key'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::SortedIndexRank { - index_name: index_name.to_string(), - primary_key: key.as_bytes().to_vec(), - })) -} - -pub(crate) fn build_sorted_index_top_k(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - let k = fields.top_k_count.unwrap_or(10); - Ok(PhysicalPlan::Kv(KvOp::SortedIndexTopK { - index_name: index_name.to_string(), - k, - })) -} - -pub(crate) fn build_sorted_index_range(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::SortedIndexRange { - index_name: index_name.to_string(), - score_min: fields.score_min.clone(), - score_max: fields.score_max.clone(), - })) -} - -pub(crate) fn build_sorted_index_count(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::SortedIndexCount { - index_name: index_name.to_string(), - })) -} - -pub(crate) fn build_sorted_index_score(fields: &TextFields) -> crate::Result { - let index_name = fields - .index_name - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'index_name'".into(), - })?; - let key = fields - .key - .as_deref() - .ok_or_else(|| crate::Error::BadRequest { - detail: "missing 'key'".into(), - })?; - Ok(PhysicalPlan::Kv(KvOp::SortedIndexScore { - index_name: index_name.to_string(), - primary_key: key.as_bytes().to_vec(), - })) -} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs index dabf6960f..017b9cc1a 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs @@ -60,7 +60,7 @@ pub(crate) fn build_incr_float( // The delta stays the client's decimal text, so the engine adds every // digit the client sent. let delta = fields.incr_float_delta.as_deref().unwrap_or("1"); - if !crate::engine::kv::float_text::is_decimal_number(delta) { + if !nodedb_physical::kv_atomic::float_text::is_decimal_number(delta) { return Err(crate::Error::BadRequest { detail: format!("KvIncrFloat: delta must be a decimal number, got '{delta}'"), }); diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs b/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs index 61b4cf4e0..0d5555db5 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/mod.rs @@ -9,6 +9,7 @@ pub(crate) mod columnar; pub(crate) mod crdt; mod dispatch; pub(crate) mod document; +pub(crate) mod document_bulk; pub(crate) mod graph; mod helpers; pub(crate) mod kv; diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs b/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs index cea66a91c..dabd2eafa 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs @@ -176,33 +176,3 @@ pub(crate) fn build_delete( vector_id, })) } - -pub(crate) fn build_set_params( - ctx: &DispatchCtx<'_>, - fields: &TextFields, - collection: &str, -) -> crate::Result { - let m = fields.m.unwrap_or(16) as usize; - let ef_construction = fields.ef_construction.unwrap_or(200) as usize; - let metric = fields - .metric - .clone() - .unwrap_or_else(|| "cosine".to_string()); - let index_type = fields - .index_type - .clone() - .unwrap_or_else(|| "hnsw".to_string()); - - Ok(PhysicalPlan::Vector(VectorOp::SetParams { - collection: QualifiedCollection::new(ctx.database_id(), collection), - field_name: fields.field_name.clone().unwrap_or_default(), - dim: fields.vector_dim.unwrap_or(0) as usize, - m, - ef_construction, - metric, - index_type, - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - })) -} diff --git a/nodedb/src/control/server/native/dispatch/sorted_read_op.rs b/nodedb/src/control/server/native/dispatch/sorted_read_op.rs new file mode 100644 index 000000000..e38d5e755 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/sorted_read_op.rs @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native sorted-index read opcodes. +//! +//! `KvSortedIndexRank`, `KvSortedIndexTopK`, `KvSortedIndexRange`, +//! `KvSortedIndexCount` and `KvSortedIndexScore` name only an index. They run +//! through the same gated read as the SQL functions `RANK`, `TOPK`, `RANGE` +//! and `SORTED_COUNT`: +//! +//! - The caller must hold `Read` on the collection the index covers. The +//! index registry names that collection. +//! - The read goes to the core that holds the collection's rows. +//! - Inside an explicit transaction the read sees the transaction's own +//! writes and an index the transaction created. +//! +//! The reply keeps the Data Plane shape each opcode has always answered with. + +use nodedb_physical::physical_plan::SortedIndexRead; +use nodedb_types::protocol::{NativeResponse, OpCode, TextFields}; + +use crate::control::server::shared::ddl::neutral::kv_sorted_index::run_read; +use crate::control::server::shared::session::DmlTxnCtx; + +use super::response::data_plane_response_to_native; +use super::{DispatchCtx, ddl_result_to_native, error_to_native, error_to_native_with_sqlstate}; + +/// Run the sorted-index read opcode `op`. +pub(crate) async fn handle_sorted_read_op( + ctx: &DispatchCtx<'_>, + seq: u64, + op: OpCode, + fields: &TextFields, +) -> NativeResponse { + // Per-operation caps (top_k and the like), as every direct read has. + if let Err(e) = super::limits::check_op_limits(ctx.state, fields) { + return error_to_native_with_sqlstate(seq, "0A000", &e); + } + if let Err(e) = ctx.state.check_tenant_quota(ctx.tenant_id()) { + return error_to_native(seq, &e); + } + let (index_name, read) = match sorted_read(op, fields) { + Ok(parsed) => parsed, + Err(e) => return error_to_native_with_sqlstate(seq, "42601", &e), + }; + let txn_ctx = DmlTxnCtx { + sessions: ctx.sessions, + session_id: ctx.peer_addr.into(), + }; + match run_read( + ctx.state, + ctx.identity, + ctx.database_id(), + &txn_ctx, + &index_name, + read, + ) + .await + { + Ok((plan, response)) => data_plane_response_to_native(ctx, seq, &plan, &response), + Err(error) => ddl_result_to_native(seq, Err(error)), + } +} + +/// The index an opcode names and the read it asks for. +fn sorted_read(op: OpCode, fields: &TextFields) -> crate::Result<(String, SortedIndexRead)> { + let index_name = required(fields.index_name.as_deref(), "index_name")?.to_string(); + let key = || required(fields.key.as_deref(), "key").map(|key| key.as_bytes().to_vec()); + let read = match op { + OpCode::KvSortedIndexRank => SortedIndexRead::Rank { + primary_key: key()?, + }, + OpCode::KvSortedIndexTopK => SortedIndexRead::TopK { + k: fields.top_k_count.unwrap_or(10), + }, + OpCode::KvSortedIndexRange => SortedIndexRead::Range { + score_min: fields.score_min.clone(), + score_max: fields.score_max.clone(), + }, + OpCode::KvSortedIndexCount => SortedIndexRead::Count, + OpCode::KvSortedIndexScore => SortedIndexRead::Score { + primary_key: key()?, + }, + other => { + return Err(crate::Error::BadRequest { + detail: format!("opcode {other:?} is not a sorted-index read"), + }); + } + }; + Ok((index_name, read)) +} + +fn required<'a>(value: Option<&'a str>, name: &str) -> crate::Result<&'a str> { + value.ok_or_else(|| crate::Error::BadRequest { + detail: format!("missing '{name}'"), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn each_read_opcode_names_its_read() { + let fields = TextFields { + index_name: Some("lb".into()), + key: Some("p1".into()), + top_k_count: Some(3), + ..TextFields::default() + }; + assert_eq!( + sorted_read(OpCode::KvSortedIndexTopK, &fields).expect("top k"), + ("lb".to_string(), SortedIndexRead::TopK { k: 3 }) + ); + assert_eq!( + sorted_read(OpCode::KvSortedIndexScore, &fields) + .expect("score") + .1, + SortedIndexRead::Score { + primary_key: b"p1".to_vec() + } + ); + let no_key = TextFields { + index_name: Some("lb".into()), + ..TextFields::default() + }; + assert!(sorted_read(OpCode::KvSortedIndexRank, &no_key).is_err()); + assert!(sorted_read(OpCode::KvSortedIndexCount, &no_key).is_ok()); + } +} diff --git a/nodedb/src/control/server/native/dispatch/transaction.rs b/nodedb/src/control/server/native/dispatch/transaction.rs index 597cf3766..a0e5448b0 100644 --- a/nodedb/src/control/server/native/dispatch/transaction.rs +++ b/nodedb/src/control/server/native/dispatch/transaction.rs @@ -31,9 +31,9 @@ use super::DispatchCtx; /// Always dispatches through the direct SPSC write path using the task's /// pre-classified `vshard_id`, mirroring pgwire's `dispatch_task_no_wal`. /// The gateway must NOT be used here: commit-time tasks carry `MetaOp` plans -/// (`ResolveTxn`, `TransactionBatch`) with no named collection, so the +/// (`ResolveTxn`, `ApplyTransactionRedo`) with no named collection, so the /// gateway's router cannot derive a route for them and falls back to -/// vShard 0 — durably applying the commit batch on the wrong core. +/// vShard 0 — durably applying the commit on the wrong core. pub(crate) struct NativeTxnDp<'a> { pub(crate) state: &'a SharedState, } diff --git a/nodedb/src/control/server/native/session/request.rs b/nodedb/src/control/server/native/session/request.rs index 5c4f3eefe..0b228108a 100644 --- a/nodedb/src/control/server/native/session/request.rs +++ b/nodedb/src/control/server/native/session/request.rs @@ -325,6 +325,27 @@ impl NativeSession { dispatch::handle_sql(&ctx, seq, &format!("EXPLAIN {sql}"), None).await } + // Sorted-index reads name only an index: gated on its owning + // collection and run in the caller's transaction, like the SQL + // sorted-index functions. + OpCode::KvSortedIndexRank + | OpCode::KvSortedIndexTopK + | OpCode::KvSortedIndexRange + | OpCode::KvSortedIndexCount + | OpCode::KvSortedIndexScore => { + dispatch::handle_sorted_read_op(&ctx, seq, op, fields).await + } + + // Index DDL runs as the SQL statement it names, so it reaches the + // catalog and the transaction's DDL buffer. + OpCode::KvRegisterSortedIndex + | OpCode::KvDropSortedIndex + | OpCode::VectorSetParams + | OpCode::DocumentDropIndex + | OpCode::DocumentRegister + | OpCode::KvRegisterIndex + | OpCode::KvDropIndex => dispatch::handle_index_ddl_op(&ctx, seq, op, fields).await, + // Direct Data Plane operations. OpCode::PointGet | OpCode::PointPut @@ -369,23 +390,11 @@ impl NativeSession { | OpCode::DocumentTruncate | OpCode::DocumentEstimateCount | OpCode::DocumentInsertSelect - | OpCode::DocumentRegister - | OpCode::DocumentDropIndex - | OpCode::KvRegisterIndex - | OpCode::KvDropIndex | OpCode::KvTruncate - | OpCode::VectorSetParams | OpCode::KvIncr | OpCode::KvIncrFloat | OpCode::KvCas | OpCode::KvGetSet - | OpCode::KvRegisterSortedIndex - | OpCode::KvDropSortedIndex - | OpCode::KvSortedIndexRank - | OpCode::KvSortedIndexTopK - | OpCode::KvSortedIndexRange - | OpCode::KvSortedIndexCount - | OpCode::KvSortedIndexScore | OpCode::CrdtListInsert | OpCode::CrdtListDelete | OpCode::CrdtListMove => dispatch::handle_direct_op(&ctx, seq, op, fields).await, diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index fb0793152..0a9784b26 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -75,6 +75,9 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::FeatureNotSupported { detail } => { ("ERROR", sqlstate::FEATURE_NOT_SUPPORTED, detail.clone()) } + crate::Error::NotInTransactionBlock { .. } => { + ("ERROR", sqlstate::ACTIVE_SQL_TRANSACTION, err.to_string()) + } crate::Error::UndefinedFunction { name } => ( "ERROR", sqlstate::UNDEFINED_FUNCTION, diff --git a/nodedb/src/control/server/resp/handler_kv/counters.rs b/nodedb/src/control/server/resp/handler_kv/counters.rs index 4ea2d7916..a61bf86fc 100644 --- a/nodedb/src/control/server/resp/handler_kv/counters.rs +++ b/nodedb/src/control/server/resp/handler_kv/counters.rs @@ -136,7 +136,7 @@ pub(in crate::control::server::resp) async fn handle_incrbyfloat( // The delta stays the client's decimal text, so the engine adds every // digit the client sent. let delta = match cmd.arg_str(1) { - Some(s) if crate::engine::kv::float_text::is_decimal_number(s) => s.to_string(), + Some(s) if nodedb_physical::kv_atomic::float_text::is_decimal_number(s) => s.to_string(), _ => return RespValue::err("ERR value is not a valid float"), }; diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs b/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs index 12a3baf03..ee4cedf51 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind/kv.rs @@ -2,7 +2,7 @@ //! `KvOp` classification. -use nodedb_physical::physical_plan::KvOp; +use nodedb_physical::physical_plan::{KvOp, SortedIndexRead}; use super::kind::PlanKind; @@ -72,6 +72,14 @@ pub(super) fn describe_kv(op: &KvOp) -> PlanKind { // One row per sorted-index entry. KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } => PlanKind::MultiRow, + // Shaped like the autocommit read it stands for. + KvOp::SortedIndexTxnRead { read, .. } => match read { + SortedIndexRead::Rank { .. } + | SortedIndexRead::Count + | SortedIndexRead::Score { .. } => PlanKind::SingleDocument, + SortedIndexRead::TopK { .. } | SortedIndexRead::Range { .. } => PlanKind::MultiRow, + }, + // TTL metadata mutations: no row count. KvOp::Expire { .. } | KvOp::Persist { .. } diff --git a/nodedb/src/control/server/shared/authorization/requirements/collect.rs b/nodedb/src/control/server/shared/authorization/requirements/collect.rs index 1ef47efbe..656e3001c 100644 --- a/nodedb/src/control/server/shared/authorization/requirements/collect.rs +++ b/nodedb/src/control/server/shared/authorization/requirements/collect.rs @@ -224,6 +224,7 @@ fn collect_requirements(plan: &PhysicalPlan, out: &mut Vec Vec<&str> { | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } | KvOp::PredicateUpdate { .. } | KvOp::PredicateDelete { .. } | KvOp::MaterializeScan { .. }) => other.collection().into_iter().collect(), diff --git a/nodedb/src/control/server/shared/ddl/engine_apply.rs b/nodedb/src/control/server/shared/ddl/engine_apply.rs index 3c39f3752..b0728e087 100644 --- a/nodedb/src/control/server/shared/ddl/engine_apply.rs +++ b/nodedb/src/control/server/shared/ddl/engine_apply.rs @@ -43,3 +43,55 @@ pub(crate) async fn apply_in_engine( .map(|_| ()) .map_err(|e| DdlError::new(sqlstate, format!("{context}: {e}"))) } + +/// Refuse a vector index definition the engine would refuse, without +/// changing engine state. +/// +/// `VectorOp::SetParams` refuses a core whose index already materialized. +/// Inside an explicit transaction the parameters install at COMMIT, so the +/// statement probes with the read-only `VectorOp::QueryStats` instead: an +/// index that answers has materialized, and `NotFound` means the parameters +/// will install. +pub(crate) async fn refuse_materialized_vector_index( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + field_name: &str, + sqlstate: &str, + context: &str, +) -> Result<(), DdlError> { + let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); + let plan = PhysicalPlan::Vector(nodedb_physical::physical_plan::VectorOp::QueryStats { + collection: nodedb_types::QualifiedCollection::new(database_id, collection), + field_name: field_name.to_string(), + }); + let response = super::sync_dispatch::dispatch_system_response_with_source( + state, + SystemTask::new( + SystemReason::DdlApply, + tenant_id, + database_id, + collection, + plan, + ), + timeout, + crate::event::EventSource::User, + ) + .await + .map_err(|e| DdlError::new("XX000", format!("{context}: {e}")))?; + match (response.status, response.error_code.as_deref()) { + (crate::bridge::envelope::Status::Ok, _) => Err(DdlError::new( + sqlstate, + format!( + "{context}: cannot change index params after creation; drop and recreate \ + the collection" + ), + )), + (_, Some(crate::bridge::envelope::ErrorCode::NotFound)) => Ok(()), + (_, code) => Err(DdlError::new( + "XX000", + format!("{context}: vector index probe failed: {code:?}"), + )), + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs new file mode 100644 index 000000000..8a6914853 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs @@ -0,0 +1,160 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Build a secondary index committed in the `Building` state: backfill it on +//! every node, then mark it `Ready`. +//! +//! An autocommit `CREATE INDEX` runs this at once. Inside an explicit +//! transaction the statement buffers the `Building` entry and defers this to +//! COMMIT, so a ROLLBACK leaves no index entry behind. + +use crate::control::security::catalog::IndexBuildState; +use crate::control::server::shared::session::ddl_effect::SecondaryIndexBuild; +use crate::control::state::SharedState; +use crate::types::TraceId; + +use super::super::super::super::result::DdlError; +use super::commit::{commit_collection_mutation, err}; + +/// Backfill `build` on every node and flip it to `Ready`. +/// +/// A collection that no longer carries the index by name is skipped: the +/// same transaction dropped it before COMMIT. UNIQUE violations surface as +/// SQLSTATE 23505 and leave the index `Building`, so a later `DROP INDEX` +/// removes it after the data is fixed. +pub(crate) async fn build_secondary_index( + state: &SharedState, + build: &SecondaryIndexBuild, +) -> Result<(), DdlError> { + let SecondaryIndexBuild { + tenant_id, + database_id, + collection, + index_name, + extraction_path, + is_array, + unique, + case_insensitive, + predicate, + } = build; + let (tenant_id, database_id) = (*tenant_id, *database_id); + let catalog = state.credentials.catalog(); + let Some(coll) = catalog + .get_collection(database_id, tenant_id.as_u64(), collection) + .map_err(|e| err("XX000", e.to_string()))? + .filter(|coll| coll.indexes.iter().any(|i| &i.name == index_name)) + else { + return Ok(()); + }; + + // A key-value collection's rows live in the KV engine, which keeps its + // own secondary indexes: the index is built there, with a backfill. + if coll.collection_type.is_kv() { + let field = super::kv_index::kv_field( + collection, + &build_path(extraction_path, *is_array), + &super::kv_index::KvIndexOptions { + unique: *unique, + case_insensitive: *case_insensitive, + predicate: predicate.as_deref(), + }, + )?; + super::kv_index::register_kv_index(state, tenant_id, database_id, &coll, &field).await?; + return mark_ready(state, build).await; + } + + // Register the Building index on this node before the backfill scans, + // so a write that lands after the scan maintains it. The post-apply + // register of a transaction's buffered entry runs asynchronously. + super::super::dispatch_register_from_stored(state, &coll) + .await + .map_err(|e| err("XX000", e.to_string()))?; + + // The backfill runs on the local Data Plane (single node) or the leader + // (cluster), vShard-local per core. + let vshard = crate::types::VShardId::from_collection_in_database(database_id, collection); + let backfill_plan = crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::BackfillIndex { + collection: nodedb_types::QualifiedCollection::new(database_id, collection), + path: extraction_path.clone(), + is_array: *is_array, + unique: *unique, + case_insensitive: *case_insensitive, + predicate: predicate.clone(), + }, + ); + let backfill_resp = crate::control::server::dispatch_utils::dispatch_to_data_plane( + state, + tenant_id, + database_id, + vshard, + backfill_plan, + TraceId::ZERO, + ) + .await + .map_err(|e| err("XX000", e.to_string()))?; + + if backfill_resp.status == crate::bridge::envelope::Status::Error { + let detail = match backfill_resp.error_code.as_deref() { + Some(crate::bridge::envelope::ErrorCode::Internal { detail, .. }) => detail.clone(), + Some(other) => format!("{other:?}"), + None => String::from_utf8_lossy(&backfill_resp.payload).into_owned(), + }; + let code = if detail.to_lowercase().contains("unique") { + "23505" + } else { + "XX000" + }; + return Err(err(code, detail)); + } + + // Every other node backfills the rows it hosts. Single-node and peerless + // clusters return at once. + super::super::index_fanout::backfill_on_peers( + state, + super::super::index_fanout::PeerBackfill { + tenant_id, + database_id, + collection, + path: extraction_path, + is_array: *is_array, + unique: *unique, + case_insensitive: *case_insensitive, + predicate: predicate.as_deref(), + }, + ) + .await?; + + mark_ready(state, build).await +} + +/// The path as the statement named it: an array path keeps its `[]` suffix. +fn build_path(extraction_path: &str, is_array: bool) -> String { + if is_array { + format!("{extraction_path}[]") + } else { + extraction_path.to_string() + } +} + +/// Flip the built index to `Ready` on a fresh read, so a concurrent mutation +/// of the collection is folded in before the index vector is rewritten. A +/// collection dropped meanwhile has no index left to flip. +async fn mark_ready(state: &SharedState, build: &SecondaryIndexBuild) -> Result<(), DdlError> { + let catalog = state.credentials.catalog(); + if let Some(mut ready_coll) = catalog + .get_collection( + build.database_id, + build.tenant_id.as_u64(), + &build.collection, + ) + .map_err(|e| err("XX000", e.to_string()))? + { + for idx in ready_coll.indexes.iter_mut() { + if idx.name == build.index_name { + idx.state = IndexBuildState::Ready; + } + } + commit_collection_mutation(state, &ready_coll, build.database_id).await?; + } + Ok(()) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs index d327bb818..6fc00a172 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs @@ -21,11 +21,13 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::index_registry::{ IndexRegistration, propose_index_record, }; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::{DeferredDdlEffect, SecondaryIndexBuild}; use crate::control::state::SharedState; use crate::types::DatabaseId; -use crate::types::TraceId; use super::super::super::super::result::{DdlError, DdlResult}; +use super::build::build_secondary_index; use super::commit::{commit_collection_mutation, err}; /// Normalize a user-supplied field reference into the canonical JSON path @@ -170,6 +172,21 @@ pub async fn create_index( .unwrap_or(&canonical_field) .to_string(); + // A key-value collection's index lives in the KV engine, which indexes + // one top-level field by equality. Refuse what it cannot build before + // any catalog entry is written. + if coll.collection_type.is_kv() { + super::kv_index::kv_field( + collection, + &canonical_field, + &super::kv_index::KvIndexOptions { + unique: is_unique, + case_insensitive, + predicate: where_condition.as_deref(), + }, + )?; + } + // Two-phase Building→Ready pipeline. Phase 1: stamp `Building` and // commit — readers skip the index (planner filters to Ready), writers // dual-write (extraction iterates every registered path regardless of @@ -189,85 +206,22 @@ pub async fn create_index( commit_collection_mutation(state, &coll, database_id).await?; - // Phase 2: dispatch the backfill op. This runs on the local Data - // Plane (single-node) or the leader (cluster — distributed backfill - // across vShards is handled inside the handler by the existing scan - // primitive, which is vShard-local per core). UNIQUE violations here - // surface as a Data Plane error; we propagate as SQLSTATE 23505 and - // leave the index in `Building` so a subsequent retry can DROP + try - // with a wider data fix. - let vshard = crate::types::VShardId::from_collection_in_database(database_id, collection); - let backfill_plan = crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::BackfillIndex { - collection: nodedb_types::QualifiedCollection::new(database_id, collection), - path: extraction_path.clone(), - is_array, - unique: is_unique, - case_insensitive, - predicate: where_condition.clone(), - }, - ); - let backfill_resp = crate::control::server::dispatch_utils::dispatch_to_data_plane( - state, + // Phase 2 and 3: backfill on every node, then flip to Ready. Inside an + // explicit transaction the build waits for COMMIT, after the Building + // entry lands; a ROLLBACK discards it with the entry. + let build = SecondaryIndexBuild { tenant_id, database_id, - vshard, - backfill_plan, - TraceId::ZERO, - ) - .await - .map_err(|e| err("XX000", e.to_string()))?; - - if backfill_resp.status == crate::bridge::envelope::Status::Error { - let detail = match backfill_resp.error_code.as_deref() { - Some(crate::bridge::envelope::ErrorCode::Internal { detail, .. }) => detail.clone(), - Some(other) => format!("{other:?}"), - None => String::from_utf8_lossy(&backfill_resp.payload).into_owned(), - }; - let code = if detail.to_lowercase().contains("unique") { - "23505" - } else { - "XX000" - }; - return Err(err(code, detail)); - } - - // Phase 2b: fan the same backfill op to every other cluster node. - // `execute_backfill_index` is vShard-local per core, so without - // this step non-coordinator nodes never populate the index for - // the rows they host — the silent-miss bug. Single-node and - // peerless clusters short-circuit inside the helper. - super::super::index_fanout::backfill_on_peers( - state, - super::super::index_fanout::PeerBackfill { - tenant_id, - database_id, - collection, - path: &extraction_path, - is_array, - unique: is_unique, - case_insensitive, - predicate: where_condition.as_deref(), - }, - ) - .await?; - - // Phase 3: flip to Ready. Re-read the collection so any concurrent - // mutation (e.g. another DDL on the same collection — blocked by - // descriptor drain in cluster mode, serialized by pgwire session in - // single-node) is folded in before we rewrite the index vector. - if let Some(latest) = catalog - .get_collection(database_id, tenant_id.as_u64(), collection) - .ok() - .flatten() - { - let mut ready_coll = latest; - for idx in ready_coll.indexes.iter_mut() { - if idx.name == index_name { - idx.state = IndexBuildState::Ready; - } - } - commit_collection_mutation(state, &ready_coll, database_id).await?; + collection: collection.to_string(), + index_name: index_name.clone(), + extraction_path: extraction_path.clone(), + is_array, + unique: is_unique, + case_insensitive, + predicate: where_condition.clone(), + }; + if !ddl_buffer::defer_effect(DeferredDdlEffect::SecondaryIndexBuild(build.clone())) { + build_secondary_index(state, &build).await?; } // Identity record: what SHOW INDEXES lists and DROP INDEX resolves. diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs index 4e2639745..3d23952ea 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs @@ -20,6 +20,7 @@ use crate::control::security::audit::AuditEvent; use crate::control::security::catalog::{IndexKind, StoredIndexRecord}; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::session::ddl_buffer; use crate::control::state::SharedState; use crate::types::DatabaseId; @@ -118,9 +119,16 @@ pub async fn drop_index( )); } - // Engine + kind-specific catalog state first: if any of it survives, the - // identity record must survive with it so the drop can be retried. - super::teardown::teardown(state, &record, database_id, tenant_id).await?; + // Autocommit: engine and kind-specific catalog state first. If any of it + // survives, the identity record survives with it so the drop can be + // retried. Inside an explicit transaction every entry is buffered and + // the engine teardown waits for COMMIT, so the records are buffered + // first and the teardown's deferred effects ride on this statement's + // own entries: a ROLLBACK TO SAVEPOINT then discards them together. + let in_transaction = ddl_buffer::is_active(); + if !in_transaction { + super::teardown::teardown(state, &record, database_id, tenant_id).await?; + } super::super::super::super::index_registry::propose_delete_index_record( state, @@ -138,6 +146,10 @@ pub async fn drop_index( index_name, )?; + if in_transaction { + super::teardown::teardown(state, &record, database_id, tenant_id).await?; + } + state.audit_record( AuditEvent::AdminAction, Some(tenant_id), diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs new file mode 100644 index 000000000..681c7b36c --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The key-value engine side of a secondary index. +//! +//! A KV collection keeps its rows in the `KvEngine`, which maintains its own +//! per-field secondary indexes on every write. `CREATE INDEX` on a KV +//! collection therefore builds the index there, and `DROP INDEX` removes it +//! there. Both run through the autocommit write funnel, which appends the +//! `kv_register_index` / `kv_drop_index` WAL record. That record and the KV +//! checkpoint carry the index across a restart. +//! +//! The KV engine indexes one top-level field by equality. It has no UNIQUE +//! check, no COLLATE NOCASE fold, no partial predicate and no array or nested +//! path, so `CREATE INDEX` refuses those forms on a KV collection before any +//! catalog entry is written. + +use crate::control::security::catalog::StoredCollection; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; +use nodedb_types::QualifiedCollection; + +use super::super::super::super::result::DdlError; +use super::commit::err; + +/// The index options a `CREATE INDEX` statement asks for. +pub(super) struct KvIndexOptions<'a> { + pub unique: bool, + pub case_insensitive: bool, + pub predicate: Option<&'a str>, +} + +/// The KV field a canonical index path names. +/// +/// Refuses the options and paths the KV engine cannot index, with SQLSTATE +/// 0A000. +pub(super) fn kv_field( + collection: &str, + canonical_field: &str, + options: &KvIndexOptions<'_>, +) -> Result { + let refuse = |what: &str| { + err( + "0A000", + format!( + "{what} is not supported on key-value collection '{collection}': \ + its index matches one top-level field by equality" + ), + ) + }; + if options.unique { + return Err(refuse("a UNIQUE index")); + } + if options.case_insensitive { + return Err(refuse("COLLATE NOCASE")); + } + if options.predicate.is_some() { + return Err(refuse("a partial index (WHERE)")); + } + let field = canonical_field + .strip_prefix("$.") + .unwrap_or(canonical_field); + if field.is_empty() || field.contains('.') || field.contains('[') || field.starts_with('$') { + return Err(refuse(&format!("the index path '{canonical_field}'"))); + } + Ok(field.to_string()) +} + +/// Build the index on the core that holds the collection's rows, backfilled +/// from every row it already holds. +pub(super) async fn register_kv_index( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + coll: &StoredCollection, + field: &str, +) -> Result<(), DdlError> { + // The engine keeps a field's schema position with the index for the + // checkpoint. An undeclared field has none, and extraction is by name. + let field_position = coll + .fields + .iter() + .position(|(name, _)| name == field) + .unwrap_or(0); + let plan = PhysicalPlan::Kv(KvOp::RegisterIndex { + collection: QualifiedCollection::new(database_id, &coll.name), + field: field.to_string(), + field_position, + backfill: true, + }); + dispatch_durable(state, tenant_id, database_id, &coll.name, plan, "build").await +} + +/// Remove the index from the core that holds the collection's rows. A field +/// with no index is already in the state the drop asks for. +pub(crate) async fn drop_kv_index( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + field: &str, +) -> Result<(), DdlError> { + let plan = PhysicalPlan::Kv(KvOp::DropIndex { + collection: QualifiedCollection::new(database_id, collection), + field: field.to_string(), + }); + dispatch_durable(state, tenant_id, database_id, collection, plan, "drop").await +} + +/// Dispatch a KV index plan through the autocommit write funnel, which +/// appends its WAL record, and fail on a refused reply. +async fn dispatch_durable( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + plan: PhysicalPlan, + step: &str, +) -> Result<(), DdlError> { + let response = crate::control::server::dispatch_utils::dispatch_autocommit_write( + state, + crate::control::server::dispatch_utils::AutocommitWrite { + tenant_id, + database_id, + vshard_id: VShardId::from_collection_in_database(database_id, collection), + plan, + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + }, + ) + .await + .map_err(|e| { + err( + "XX000", + format!("key-value index {step} on '{collection}': {e}"), + ) + })?; + + if response.status == crate::bridge::envelope::Status::Error { + let detail = match response.error_code.as_deref() { + Some(code) => format!("{code:?}"), + None => String::from_utf8_lossy(&response.payload).into_owned(), + }; + return Err(err( + "XX000", + format!("key-value index {step} on '{collection}' was refused: {detail}"), + )); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn plain() -> KvIndexOptions<'static> { + KvIndexOptions { + unique: false, + case_insensitive: false, + predicate: None, + } + } + + #[test] + fn a_top_level_field_is_indexed_by_its_name() { + assert_eq!( + kv_field("sessions", "$.region", &plain()).expect("plain field"), + "region" + ); + } + + #[test] + fn options_and_paths_the_engine_cannot_index_are_refused() { + for path in ["$.a.b", "$.tags[]", "$"] { + let error = kv_field("sessions", path, &plain()).expect_err(path); + assert_eq!(error.sqlstate, "0A000", "{path}"); + } + let unique = KvIndexOptions { + unique: true, + ..plain() + }; + assert_eq!( + kv_field("sessions", "$.region", &unique) + .expect_err("unique") + .sqlstate, + "0A000" + ); + let partial = KvIndexOptions { + predicate: Some("region = 'eu'"), + ..plain() + }; + assert!(kv_field("sessions", "$.region", &partial).is_err()); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/mod.rs index 52b48651f..9ffb03097 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/mod.rs @@ -2,9 +2,11 @@ //! Protocol-neutral index DDL: CREATE INDEX, DROP INDEX. +pub mod build; pub mod commit; pub mod create; pub mod drop; +pub mod kv_index; pub mod teardown; pub use create::{CreateIndexRequest, create_index}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs index f1d36d72d..4ab02d78d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs @@ -7,7 +7,7 @@ //! //! | kind | durable state created | //! |-----------|----------------------------------------------------| -//! | secondary | `StoredCollection.indexes` entry + sparse-engine index entries | +//! | secondary | `StoredCollection.indexes` entry + sparse-engine index entries, or the KV engine's field index on a key-value collection | //! | vector | `_system.vector_index_params` row + Data Plane index + checkpoint | //! | fulltext | the collection's analyzer / fuzzy binding (per collection) | //! | spatial | none beyond the registry + ownership rows | @@ -28,6 +28,9 @@ use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId}; use super::super::super::super::result::DdlError; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; + use super::commit::{commit_collection_mutation, err}; /// Remove every piece of engine and catalog state belonging to `record`, @@ -54,6 +57,15 @@ pub(super) async fn teardown( // write — so its removal goes through the same route // `DROP SORTED INDEX` uses. IndexKind::Sorted => { + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::SortedIndexDrop { + tenant_id, + database_id, + collection: record.collection.clone(), + index_name: record.name.clone(), + }); + if deferred { + return Ok(()); + } super::super::super::kv_sorted_index::drop_in_engine( state, &super::super::super::kv_sorted_index::SortedIndexTarget { @@ -69,7 +81,8 @@ pub(super) async fn teardown( } /// Drop the `StoredIndex` entry from the owning collection and purge the -/// sparse engine's entries for the indexed path. +/// indexed path's entries: from the sparse engine, or from the KV engine on +/// a key-value collection. async fn secondary( state: &SharedState, record: &StoredIndexRecord, @@ -100,21 +113,34 @@ async fn secondary( let Some(field) = dropped_field.or_else(|| record.fields.first().cloned()) else { return Ok(()); }; + // A key-value collection's index lives in the KV engine. + if coll.collection_type.is_kv() { + let field = field.strip_prefix("$.").unwrap_or(&field).to_string(); + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::KvIndexDrop { + tenant_id, + database_id, + collection: record.collection.clone(), + field: field.clone(), + }); + if deferred { + return Ok(()); + } + return super::kv_index::drop_kv_index( + state, + tenant_id, + database_id, + &record.collection, + &field, + ) + .await; + } let plan = crate::bridge::envelope::PhysicalPlan::Document( nodedb_physical::physical_plan::DocumentOp::DropIndex { collection: nodedb_types::QualifiedCollection::new(database_id, &record.collection), field, }, ); - dispatch( - state, - tenant_id, - database_id, - &record.collection, - plan, - None, - ) - .await + teardown_now_or_at_commit(state, tenant_id, database_id, &record.collection, plan).await } /// Remove the vector index's durable build parameters and its Data Plane @@ -239,21 +265,35 @@ async fn fulltext( fuzzy_default: Some(false), }, ); - dispatch( - state, + teardown_now_or_at_commit(state, tenant_id, database_id, &record.collection, plan).await +} + +/// Run one teardown plan now, or at COMMIT inside an explicit transaction: +/// the catalog entry that drops the index is buffered, so its engine state +/// must survive a ROLLBACK. +async fn teardown_now_or_at_commit( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + plan: crate::bridge::envelope::PhysicalPlan, +) -> Result<(), DdlError> { + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::IndexTeardown { tenant_id, database_id, - &record.collection, - plan, - None, - ) - .await + collection: collection.to_string(), + plan: plan.clone(), + }); + if deferred { + return Ok(()); + } + dispatch(state, tenant_id, database_id, collection, plan, None).await } /// Dispatch one teardown plan to the Data Plane, surfacing both transport and /// handler-side failures. `minted` holds the record appended for the plan; /// the funnel closes its outcome-floor window from the plan's outcome. -async fn dispatch( +pub(crate) async fn dispatch( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, diff --git a/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs b/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs new file mode 100644 index 000000000..8b86fdf89 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs @@ -0,0 +1,145 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Run the engine side effects a transaction's index DDL deferred to COMMIT. +//! +//! COMMIT calls [`run_deferred_effects`] once the buffered catalog entries +//! landed, in statement order. Each effect runs through the same function its +//! statement runs in autocommit, so a transactional index ends in the same +//! engine state as an autocommit one. + +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; +use crate::control::state::SharedState; + +use super::super::result::DdlError; +use super::collection::index::build::build_secondary_index; +use super::collection::index::kv_index::drop_kv_index; +use super::collection::index::teardown; +use super::kv_sorted_index::SortedIndexTarget; +use super::kv_sorted_index::dispatch::register_in_engine; +use super::kv_sorted_index::drop_in_engine; + +/// Run every effect in order. The first failure stops the run and returns +/// its error: the effects before it applied, and the ones after it did not. +pub(crate) async fn run_deferred_effects( + state: &SharedState, + effects: Vec, +) -> crate::Result<()> { + for effect in effects { + let collection = effect_collection(&effect).to_string(); + run_one(state, effect) + .await + .map_err(|error| effect_error(&collection, error))?; + } + Ok(()) +} + +/// The collection an effect changes. +fn effect_collection(effect: &DeferredDdlEffect) -> &str { + match effect { + DeferredDdlEffect::SecondaryIndexBuild(build) => &build.collection, + DeferredDdlEffect::EngineApply { collection, .. } + | DeferredDdlEffect::IndexTeardown { collection, .. } + | DeferredDdlEffect::SortedIndexRegister { collection, .. } + | DeferredDdlEffect::SortedIndexDrop { collection, .. } + | DeferredDdlEffect::KvIndexDrop { collection, .. } => collection, + } +} + +async fn run_one(state: &SharedState, effect: DeferredDdlEffect) -> Result<(), DdlError> { + match effect { + DeferredDdlEffect::SecondaryIndexBuild(build) => build_secondary_index(state, &build).await, + DeferredDdlEffect::EngineApply { + tenant_id, + database_id, + collection, + plan, + sqlstate, + context, + } => { + crate::control::server::shared::ddl::engine_apply::apply_in_engine( + state, + tenant_id, + database_id, + &collection, + plan, + &sqlstate, + &context, + ) + .await + } + DeferredDdlEffect::IndexTeardown { + tenant_id, + database_id, + collection, + plan, + } => teardown::dispatch(state, tenant_id, database_id, &collection, plan, None).await, + DeferredDdlEffect::SortedIndexRegister { + tenant_id, + database_id, + collection, + plan, + } => { + let target = SortedIndexTarget { + tenant_id, + database_id, + collection: &collection, + }; + register_in_engine(state, &target, plan, "CREATE SORTED INDEX") + .await + .map(|_| ()) + } + DeferredDdlEffect::SortedIndexDrop { + tenant_id, + database_id, + collection, + index_name, + } => { + let target = SortedIndexTarget { + tenant_id, + database_id, + collection: &collection, + }; + drop_in_engine(state, &target, &index_name).await + } + DeferredDdlEffect::KvIndexDrop { + tenant_id, + database_id, + collection, + field, + } => drop_kv_index(state, tenant_id, database_id, &collection, &field).await, + } +} + +/// The COMMIT error for a failed effect. A UNIQUE violation keeps its class, +/// so the client sees SQLSTATE 23505 as an autocommit `CREATE INDEX` does. +fn effect_error(collection: &str, error: DdlError) -> crate::Error { + if error.sqlstate == "23505" { + return crate::Error::RejectedConstraint { + collection: collection.to_string(), + constraint: "unique".to_string(), + detail: error.message, + }; + } + crate::Error::Internal { + detail: format!( + "index DDL committed but its engine step failed (SQLSTATE {}): {}", + error.sqlstate, error.message + ), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_unique_violation_keeps_its_class() { + let error = effect_error("users", DdlError::new("23505", "duplicate 'a'")); + assert!(matches!( + error, + crate::Error::RejectedConstraint { ref constraint, .. } if constraint == "unique" + )); + let other = effect_error("users", DdlError::new("XX000", "core gone")); + assert!(matches!(other, crate::Error::Internal { .. })); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs index 0a551c4ff..45290802f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs @@ -15,6 +15,8 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::index_registry::{ IndexRegistration, propose_index_record, }; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; use crate::control::state::SharedState; use crate::types::DatabaseId; use nodedb_physical::physical_plan::TextOp; @@ -183,16 +185,28 @@ async fn create_text_index( analyzer_name: analyzer_name.clone(), fuzzy_default, }); - crate::control::server::shared::ddl::engine_apply::apply_in_engine( - state, + // Inside an explicit transaction the binding waits for COMMIT, after + // the buffered index record lands. + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::EngineApply { tenant_id, database_id, - &collection, - set_config_plan, - "58000", - command, - ) - .await?; + collection: collection.clone(), + plan: set_config_plan.clone(), + sqlstate: "58000".to_string(), + context: command.to_string(), + }); + if !deferred { + crate::control::server::shared::ddl::engine_apply::apply_in_engine( + state, + tenant_id, + database_id, + &collection, + set_config_plan, + "58000", + command, + ) + .await?; + } state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs index 74899fa31..3e4134293 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs @@ -20,6 +20,7 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::index_registry::{ IndexRegistration, propose_index_record, }; +use crate::control::server::shared::session::ddl_buffer; use crate::control::state::SharedState; use crate::types::DatabaseId; use nodedb_physical::physical_plan::VectorOp; @@ -183,16 +184,33 @@ pub async fn create_vector_index( // client — this pre-flight is the only place the statement can fail // closed. The post-apply dispatch that follows re-installs the same // parameters on this node, which is a no-op on an unmaterialized index. - crate::control::server::shared::ddl::engine_apply::apply_in_engine( - state, - tenant_id, - database_id, - collection, - set_params_plan.clone(), - "42P16", - CONTEXT, - ) - .await?; + // + // Inside an explicit transaction the parameters install at COMMIT from + // the buffered catalog row, so the pre-flight changes nothing: it probes + // for a materialized index and refuses the same way. + if ddl_buffer::is_active() { + crate::control::server::shared::ddl::engine_apply::refuse_materialized_vector_index( + state, + tenant_id, + database_id, + collection, + &field_name, + "42P16", + CONTEXT, + ) + .await?; + } else { + crate::control::server::shared::ddl::engine_apply::apply_in_engine( + state, + tenant_id, + database_id, + collection, + set_params_plan, + "42P16", + CONTEXT, + ) + .await?; + } // Only now make it durable. The replicated catalog row re-registers the // index at boot via `seed_vector_index_params`, and each node's post-apply diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs index eb449c407..ed7cf5a72 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs @@ -110,7 +110,7 @@ pub async fn kv_incr_float( // The delta stays the client's decimal text, so the engine adds every // digit the client wrote. let delta = unquote(&args[2]).trim().to_string(); - if !crate::engine::kv::float_text::is_decimal_number(&delta) { + if !nodedb_physical::kv_atomic::float_text::is_decimal_number(&delta) { return Err(ddl_err( "42601", format!( diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs index e59dea581..3f7da3752 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs @@ -14,6 +14,8 @@ use crate::control::security::audit::AuditEvent; use crate::control::security::catalog::IndexKind; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; use crate::control::state::SharedState; use crate::types::DatabaseId; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; @@ -108,22 +110,33 @@ pub async fn create_sorted_index( window_end_ms: window_end, }); + // Inside an explicit transaction the records are buffered first and the + // tree is built at COMMIT, so a ROLLBACK leaves no tree behind. + let in_transaction = ddl_buffer::is_active(); + // Routed by the collection, not by the index name: the backfill this plan // performs reads the collection's rows out of the `KvEngine` of whichever // core executes it, and every later write that must keep the tree current // lands on the collection's own core. Registering anywhere else builds an // empty tree that no write ever updates. - let response = register_in_engine( - state, - &SortedIndexTarget { - tenant_id, - database_id, - collection: &collection, - }, - plan, - "CREATE SORTED INDEX", - ) - .await?; + let response = if in_transaction { + vec![DdlResult::Status { + command: "CREATE SORTED INDEX".to_string(), + rows_affected: None, + }] + } else { + register_in_engine( + state, + &SortedIndexTarget { + tenant_id, + database_id, + collection: &collection, + }, + plan.clone(), + "CREATE SORTED INDEX", + ) + .await? + }; // Identity record: what resolves the index's owning collection on every // later read, what SHOW INDEXES lists, and what DROP INDEX resolves. @@ -149,6 +162,22 @@ pub async fn create_sorted_index( &identity.username, )?; + if in_transaction { + let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::SortedIndexRegister { + tenant_id, + database_id, + collection: collection.clone(), + plan, + }); + if !deferred { + return Err(ddl_err( + "XX000", + "CREATE SORTED INDEX: the transaction buffer took no entry to defer the \ + index build on", + )); + } + } + state.audit_record( AuditEvent::AdminAction, Some(tenant_id), @@ -207,16 +236,18 @@ pub async fn drop_sorted_index( )); } - drop_in_engine( - state, - &SortedIndexTarget { - tenant_id, - database_id, - collection: &collection, - }, - &index_name, - ) - .await?; + // Autocommit drops the tree first, so a failed drop keeps the records a + // retry resolves through. Inside an explicit transaction the records are + // buffered first and the tree is dropped at COMMIT. + let target = SortedIndexTarget { + tenant_id, + database_id, + collection: &collection, + }; + let in_transaction = ddl_buffer::is_active(); + if !in_transaction { + drop_in_engine(state, &target, &index_name).await?; + } propose_delete_index_record(state, database_id, tenant_id, &index_name, &collection)?; crate::control::server::shared::ddl::owner::propose_delete_owner( @@ -227,6 +258,21 @@ pub async fn drop_sorted_index( &index_name, )?; + if in_transaction + && !ddl_buffer::defer_effect(DeferredDdlEffect::SortedIndexDrop { + tenant_id, + database_id, + collection: collection.clone(), + index_name: index_name.clone(), + }) + { + return Err(ddl_err( + "XX000", + "DROP SORTED INDEX: the transaction buffer took no entry to defer the index \ + drop on", + )); + } + Ok(vec![DdlResult::Status { command: "DROP SORTED INDEX".to_string(), rows_affected: None, diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs index 66a91a2d2..a2ed87aa8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs @@ -17,6 +17,7 @@ use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use super::super::super::result::{DdlError, DdlResult}; use super::parse::ddl_err; +use super::txn_read::SortedRead; /// Where one sorted index's Data Plane state lives. /// @@ -79,19 +80,22 @@ fn refusal(target: &SortedIndexTarget<'_>, resp: &Response) -> Option } /// Dispatch a sorted-index read (`RANK` / `TOPK` / `RANGE` / `SORTED_COUNT` / -/// `ZSCORE`), which mints no durable record. -async fn dispatch_read( +/// score), which mints no durable record. A read inside a transaction carries +/// its transaction id, so the Data Plane folds in that transaction's staged +/// writes. +pub(super) async fn dispatch_read( state: &SharedState, target: &SortedIndexTarget<'_>, - plan: PhysicalPlan, + read: SortedRead, ) -> Result { - let resp = crate::control::server::dispatch_utils::dispatch_to_data_plane( + let resp = crate::control::server::dispatch_utils::dispatch_to_data_plane_with_txn( state, target.tenant_id, target.database_id, target.vshard(), - plan, + read.plan, TraceId::ZERO, + read.txn_id, ) .await .map_err(|e| ddl_err("XX000", e.to_string()))?; @@ -152,7 +156,7 @@ fn decode_rows(payload: &[u8]) -> Result, DdlError> { /// an apply that did not happen files a record for an index that exists /// nowhere, and every read of it then answers from an index that was never /// built. -pub(super) async fn register_in_engine( +pub(crate) async fn register_in_engine( state: &SharedState, target: &SortedIndexTarget<'_>, plan: PhysicalPlan, @@ -194,30 +198,19 @@ pub async fn drop_in_engine( Ok(()) } -/// Dispatch plan and return a single-row JSON response. -pub(super) async fn dispatch_and_respond_json( - state: &SharedState, - target: &SortedIndexTarget<'_>, - plan: PhysicalPlan, - col_name: &str, -) -> Result, DdlError> { - let resp = dispatch_read(state, target, plan).await?; +/// A read's reply as a single-row JSON response. +pub(super) fn respond_json(resp: &Response, col_name: &str) -> Vec { let payload_text = crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); let mut row = Map::new(); row.insert(col_name.to_string(), JsonValue::String(payload_text)); - Ok(vec![DdlResult::Rows(ShapedRows::text_rows( + vec![DdlResult::Rows(ShapedRows::text_rows( vec![col_name.to_string()], vec![row], - ))]) + ))] } -/// Dispatch plan and return multi-row response (for TOPK, RANGE). -pub(super) async fn dispatch_and_respond_rows( - state: &SharedState, - target: &SortedIndexTarget<'_>, - plan: PhysicalPlan, -) -> Result, DdlError> { - let resp = dispatch_read(state, target, plan).await?; +/// A read's reply as a multi-row response (for TOPK, RANGE). +pub(super) fn respond_rows(resp: &Response) -> Result, DdlError> { let rows_json = decode_rows(&resp.payload)?; let mut rows = Vec::with_capacity(rows_json.len()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/mod.rs index 2c9582dff..729be04d7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/mod.rs @@ -18,7 +18,9 @@ pub mod dispatch; pub mod gate; pub mod parse; pub mod query; +mod txn_read; pub use ddl::{create_sorted_index, drop_sorted_index}; pub use dispatch::{SortedIndexTarget, drop_in_engine}; +pub(crate) use query::run_read; pub use query::{select_range, select_rank, select_sorted_count, select_topk}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/query.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/query.rs index 24e4f6947..97f22a1c7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/query.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/query.rs @@ -4,24 +4,81 @@ //! //! Each names an index and returns keys, ranks, or counts drawn from the //! collection it was built over, so each resolves that collection and gates on -//! it before a plan is built (see [`super::gate`]). +//! it before a plan is built (see [`super::gate`]). The native sorted-index +//! read opcodes run through [`run_read`] too, so both protocols gate, route and +//! see the caller's transaction the same way. +use crate::bridge::envelope::Response; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; use crate::types::DatabaseId; -use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; +use nodedb_physical::physical_plan::{PhysicalPlan, SortedIndexRead}; use super::super::super::result::{DdlError, DdlResult}; -use super::dispatch::{SortedIndexTarget, dispatch_and_respond_json, dispatch_and_respond_rows}; +use super::dispatch::{SortedIndexTarget, dispatch_read, respond_json, respond_rows}; use super::gate::gate_read; use super::parse::{ddl_err, parse_function_args, parse_score_arg, unquote}; +use super::txn_read::{ReadScope, plan_read}; + +/// What a read delivers instead of row bodies, for the refusal message. +fn what(read: &SortedIndexRead) -> &'static str { + match read { + SortedIndexRead::Rank { .. } => { + "RANK(), which returns a position in the sorted index rather than rows" + } + SortedIndexRead::TopK { .. } => { + "TOPK(), which returns the sorted index's ranked keys rather than rows" + } + SortedIndexRead::Range { .. } => { + "RANGE(), which returns the sorted index's ranked keys rather than rows" + } + SortedIndexRead::Count => { + "SORTED_COUNT(), which returns a count over the sorted index rather than rows" + } + SortedIndexRead::Score { .. } => { + "a sorted-index score read, which returns a sort key rather than rows" + } + } +} -/// What each function delivers instead of row bodies, for the refusal message. -const RANK_WHAT: &str = "RANK(), which returns a position in the sorted index rather than rows"; -const TOPK_WHAT: &str = "TOPK(), which returns the sorted index's ranked keys rather than rows"; -const RANGE_WHAT: &str = "RANGE(), which returns the sorted index's ranked keys rather than rows"; -const COUNT_WHAT: &str = - "SORTED_COUNT(), which returns a count over the sorted index rather than rows"; +/// Gate, plan and dispatch one sorted-index read. +/// +/// Gated on the index's owning collection, routed to the core that holds its +/// rows, and run in the caller's transaction when one is open. Returns the +/// plan that ran and the Data Plane reply. +pub(crate) async fn run_read( + state: &SharedState, + identity: &AuthenticatedIdentity, + database_id: DatabaseId, + txn_ctx: &DmlTxnCtx<'_>, + index_name: &str, + read: SortedIndexRead, +) -> Result<(PhysicalPlan, Response), DdlError> { + let collection = gate_read(state, identity, database_id, index_name, what(&read))?; + let sorted = plan_read( + &ReadScope { + txn_ctx, + tenant_id: identity.tenant_id, + database_id, + collection: &collection, + index_name, + }, + read, + ); + let plan = sorted.plan.clone(); + let response = dispatch_read( + state, + &SortedIndexTarget { + tenant_id: identity.tenant_id, + database_id, + collection: &collection, + }, + sorted, + ) + .await?; + Ok((plan, response)) +} /// Handle `SELECT RANK(index_name, 'key_value')` pub async fn select_rank( @@ -29,6 +86,7 @@ pub async fn select_rank( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { let args = parse_function_args(sql)?; if args.len() < 2 { @@ -41,26 +99,11 @@ pub async fn select_rank( // A string literal is data: the index name resolves exactly as written, // with no case folding. let index_name = unquote(&args[0]); - let key_value = unquote(&args[1]); - - let collection = gate_read(state, identity, database_id, &index_name, RANK_WHAT)?; + let primary_key = unquote(&args[1]).into_bytes(); - let plan = PhysicalPlan::Kv(KvOp::SortedIndexRank { - index_name, - primary_key: key_value.into_bytes(), - }); - - dispatch_and_respond_json( - state, - &SortedIndexTarget { - tenant_id: identity.tenant_id, - database_id, - collection: &collection, - }, - plan, - "rank", - ) - .await + let read = SortedIndexRead::Rank { primary_key }; + let (_, response) = run_read(state, identity, database_id, txn_ctx, &index_name, read).await?; + Ok(respond_json(&response, "rank")) } /// Handle `SELECT * FROM TOPK(index_name, k)` or `SELECT TOPK(index_name, k)` @@ -69,6 +112,7 @@ pub async fn select_topk( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { let args = parse_function_args(sql)?; if args.len() < 2 { @@ -88,20 +132,9 @@ pub async fn select_topk( ) })?; - let collection = gate_read(state, identity, database_id, &index_name, TOPK_WHAT)?; - - let plan = PhysicalPlan::Kv(KvOp::SortedIndexTopK { index_name, k }); - - dispatch_and_respond_rows( - state, - &SortedIndexTarget { - tenant_id: identity.tenant_id, - database_id, - collection: &collection, - }, - plan, - ) - .await + let read = SortedIndexRead::TopK { k }; + let (_, response) = run_read(state, identity, database_id, txn_ctx, &index_name, read).await?; + respond_rows(&response) } /// Handle `SELECT * FROM RANGE(index_name, score_min, score_max)` @@ -110,6 +143,7 @@ pub async fn select_range( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { let args = parse_function_args(sql)?; if args.len() < 3 { @@ -122,27 +156,12 @@ pub async fn select_range( // A string literal is data: the index name resolves exactly as written, // with no case folding. let index_name = unquote(&args[0]); - let score_min = parse_score_arg(&args[1]); - let score_max = parse_score_arg(&args[2]); - - let collection = gate_read(state, identity, database_id, &index_name, RANGE_WHAT)?; - - let plan = PhysicalPlan::Kv(KvOp::SortedIndexRange { - index_name, - score_min, - score_max, - }); - - dispatch_and_respond_rows( - state, - &SortedIndexTarget { - tenant_id: identity.tenant_id, - database_id, - collection: &collection, - }, - plan, - ) - .await + let read = SortedIndexRead::Range { + score_min: parse_score_arg(&args[1]), + score_max: parse_score_arg(&args[2]), + }; + let (_, response) = run_read(state, identity, database_id, txn_ctx, &index_name, read).await?; + respond_rows(&response) } /// Handle `SELECT SORTED_COUNT(index_name)` @@ -151,6 +170,7 @@ pub async fn select_sorted_count( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + txn_ctx: &DmlTxnCtx<'_>, ) -> Result, DdlError> { let args = parse_function_args(sql)?; if args.is_empty() { @@ -163,20 +183,14 @@ pub async fn select_sorted_count( // A string literal is data: the index name resolves exactly as written, // with no case folding. let index_name = unquote(&args[0]); - - let collection = gate_read(state, identity, database_id, &index_name, COUNT_WHAT)?; - - let plan = PhysicalPlan::Kv(KvOp::SortedIndexCount { index_name }); - - dispatch_and_respond_json( + let (_, response) = run_read( state, - &SortedIndexTarget { - tenant_id: identity.tenant_id, - database_id, - collection: &collection, - }, - plan, - "sorted_count", + identity, + database_id, + txn_ctx, + &index_name, + SortedIndexRead::Count, ) - .await + .await?; + Ok(respond_json(&response, "sorted_count")) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/txn_read.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/txn_read.rs new file mode 100644 index 000000000..619285984 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/txn_read.rs @@ -0,0 +1,212 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The plan a sorted-index read runs, and the transaction it runs in. +//! +//! Outside a transaction block a read asks the index's registered tree. +//! Inside one it becomes `KvOp::SortedIndexTxnRead`. The Data Plane answers +//! that from a transaction-local tree over the collection's base rows and the +//! transaction's staged writes. An index this transaction created has no tree +//! before COMMIT, so its definition travels with the read. The definition +//! comes from the index build that the transaction deferred to COMMIT. + +use nodedb_physical::physical_plan::{KvOp, PhysicalPlan, SortedIndexRead, SortedIndexSpec}; +use nodedb_types::QualifiedCollection; + +use crate::control::server::shared::session::ddl_buffer; +use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; +use crate::control::server::shared::session::{DmlTxnCtx, TransactionState}; +use crate::types::{DatabaseId, TenantId, TxnId}; + +/// One sorted-index read: the plan and the transaction whose overlay it reads. +pub(super) struct SortedRead { + pub plan: PhysicalPlan, + pub txn_id: Option, +} + +/// Where a read runs and which index it names. +pub(super) struct ReadScope<'a> { + pub txn_ctx: &'a DmlTxnCtx<'a>, + pub tenant_id: TenantId, + pub database_id: DatabaseId, + /// The collection the index covers, resolved by the read gate. + pub collection: &'a str, + pub index_name: &'a str, +} + +/// The plan for `read`. +pub(super) fn plan_read(scope: &ReadScope<'_>, read: SortedIndexRead) -> SortedRead { + let txn_ctx = scope.txn_ctx; + if txn_ctx.sessions.transaction_state(txn_ctx.session_id) != TransactionState::InBlock { + return SortedRead { + plan: PhysicalPlan::Kv(autocommit_op(scope.index_name, read)), + txn_id: None, + }; + } + SortedRead { + plan: PhysicalPlan::Kv(KvOp::SortedIndexTxnRead { + collection: QualifiedCollection::new(scope.database_id, scope.collection), + index_name: scope.index_name.to_string(), + pending: pending_definition(scope.tenant_id, scope.database_id, scope.index_name), + read, + }), + txn_id: txn_ctx.sessions.tx_id(txn_ctx.session_id), + } +} + +/// The read outside a transaction block, against the registered tree. +fn autocommit_op(index_name: &str, read: SortedIndexRead) -> KvOp { + let index_name = index_name.to_string(); + match read { + SortedIndexRead::Rank { primary_key } => KvOp::SortedIndexRank { + index_name, + primary_key, + }, + SortedIndexRead::TopK { k } => KvOp::SortedIndexTopK { index_name, k }, + SortedIndexRead::Range { + score_min, + score_max, + } => KvOp::SortedIndexRange { + index_name, + score_min, + score_max, + }, + SortedIndexRead::Count => KvOp::SortedIndexCount { index_name }, + SortedIndexRead::Score { primary_key } => KvOp::SortedIndexScore { + index_name, + primary_key, + }, + } +} + +/// The definition of `index_name` when this transaction created it and has +/// not committed it. `None` for a committed index. +/// +/// Replays the deferred effects in statement order: a later drop of the same +/// name cancels an earlier create. +fn pending_definition( + tenant_id: TenantId, + database_id: DatabaseId, + index_name: &str, +) -> Option { + ddl_buffer::with_buffered(|items| { + let mut pending = None; + for effect in items.iter().flat_map(|item| item.effects.iter()) { + pending = step(pending, effect, tenant_id, database_id, index_name); + } + pending + }) + .flatten() +} + +/// Apply one deferred effect to the pending definition resolved so far. +fn step( + current: Option, + effect: &DeferredDdlEffect, + tenant_id: TenantId, + database_id: DatabaseId, + index_name: &str, +) -> Option { + match effect { + DeferredDdlEffect::SortedIndexRegister { + tenant_id: effect_tenant, + database_id: effect_db, + plan: + PhysicalPlan::Kv(KvOp::RegisterSortedIndex { + index_name: effect_index, + sort_columns, + key_column, + window_type, + window_timestamp_column, + window_start_ms, + window_end_ms, + .. + }), + .. + } if *effect_tenant == tenant_id + && *effect_db == database_id + && effect_index == index_name => + { + Some(SortedIndexSpec { + sort_columns: sort_columns.clone(), + key_column: key_column.clone(), + window_type: window_type.clone(), + window_timestamp_column: window_timestamp_column.clone(), + window_start_ms: *window_start_ms, + window_end_ms: *window_end_ms, + }) + } + DeferredDdlEffect::SortedIndexDrop { + tenant_id: effect_tenant, + database_id: effect_db, + index_name: effect_index, + .. + } if *effect_tenant == tenant_id + && *effect_db == database_id + && effect_index == index_name => + { + None + } + _ => current, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn register(index_name: &str) -> DeferredDdlEffect { + DeferredDdlEffect::SortedIndexRegister { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + collection: "board".to_string(), + plan: PhysicalPlan::Kv(KvOp::RegisterSortedIndex { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "board"), + index_name: index_name.to_string(), + sort_columns: vec![("score".to_string(), "DESC".to_string())], + key_column: "id".to_string(), + window_type: "none".to_string(), + window_timestamp_column: String::new(), + window_start_ms: 0, + window_end_ms: 0, + }), + } + } + + fn drop_effect(index_name: &str) -> DeferredDdlEffect { + DeferredDdlEffect::SortedIndexDrop { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + collection: "board".to_string(), + index_name: index_name.to_string(), + } + } + + fn replay(effects: &[DeferredDdlEffect], index_name: &str) -> Option { + effects.iter().fold(None, |current, effect| { + step( + current, + effect, + TenantId::new(1), + DatabaseId::DEFAULT, + index_name, + ) + }) + } + + #[test] + fn a_buffered_create_carries_its_definition() { + let spec = replay(&[register("lb")], "lb").expect("the create is pending"); + assert_eq!(spec.key_column, "id"); + assert_eq!( + spec.sort_columns, + vec![("score".to_string(), "DESC".to_string())] + ); + } + + #[test] + fn a_later_drop_cancels_the_create_and_other_names_are_ignored() { + assert!(replay(&[register("lb"), drop_effect("lb")], "lb").is_none()); + assert!(replay(&[register("other")], "lb").is_none()); + assert!(replay(&[drop_effect("lb"), register("lb")], "lb").is_some()); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/mod.rs index 7d6d82e97..0022b6393 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/mod.rs @@ -27,6 +27,7 @@ pub mod convert; pub mod crdt_ops; pub mod custom_type; pub mod database; +pub mod deferred_effects; pub mod dsl; pub mod emergency_ddl; pub mod estimate_count; diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs index cbcb01808..afca89e3f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs @@ -96,23 +96,31 @@ pub(super) async fn try_string( // doc-object UPSERT body, string literal, or comment carrying the token // can never reach these arms. if upper.starts_with("SELECT RANK(") || upper.starts_with("SELECT RANK (") { - return Some(kv_sorted_index::select_rank(state, identity, database_id, sql).await); + return Some( + kv_sorted_index::select_rank(state, identity, database_id, sql, txn_ctx).await, + ); } if upper.starts_with("SELECT TOPK(") || upper.starts_with("SELECT TOPK (") || upper.starts_with("SELECT * FROM TOPK(") || upper.starts_with("SELECT * FROM TOPK (") { - return Some(kv_sorted_index::select_topk(state, identity, database_id, sql).await); + return Some( + kv_sorted_index::select_topk(state, identity, database_id, sql, txn_ctx).await, + ); } if upper.starts_with("SELECT SORTED_COUNT(") || upper.starts_with("SELECT SORTED_COUNT (") { - return Some(kv_sorted_index::select_sorted_count(state, identity, database_id, sql).await); + return Some( + kv_sorted_index::select_sorted_count(state, identity, database_id, sql, txn_ctx).await, + ); } // RANGE as a sorted index function (check it's not a standard SQL RANGE). if (upper.starts_with("SELECT * FROM RANGE(") || upper.starts_with("SELECT * FROM RANGE (")) && !upper.contains(" BETWEEN ") { - return Some(kv_sorted_index::select_range(state, identity, database_id, sql).await); + return Some( + kv_sorted_index::select_range(state, identity, database_id, sql, txn_ctx).await, + ); } // KV_INCR / KV_DECR / KV_INCR_FLOAT / KV_CAS / KV_GETSET — atomic KV operations. diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index fc4906dcd..db5736066 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -35,6 +35,11 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, sqlstate::CHECK_VIOLATION, format!("pre-validation rejected: {reason}"), ), + ErrorCode::SyncRejected { violation, .. } => ( + "ERROR", + sqlstate::CHECK_VIOLATION, + format!("sync frame rejected: {violation}"), + ), // Nothing applied and the identical statement is expected to succeed // later, so drivers get the same class they already retry on rather // than a check violation they would surface as permanent. diff --git a/nodedb/src/control/server/shared/returning/inject.rs b/nodedb/src/control/server/shared/returning/inject.rs index dc6cbb826..9c00d0959 100644 --- a/nodedb/src/control/server/shared/returning/inject.rs +++ b/nodedb/src/control/server/shared/returning/inject.rs @@ -251,6 +251,7 @@ pub fn inject_returning_spec(plan: &mut PhysicalPlan, spec: ReturningSpec) { | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } | KvOp::MaterializeScan { .. } | KvOp::ResolveWrite(_) | KvOp::ResolvedWrite { .. }, diff --git a/nodedb/src/control/server/shared/session/commit/run.rs b/nodedb/src/control/server/shared/session/commit/run.rs index cf30af905..34873c1b3 100644 --- a/nodedb/src/control/server/shared/session/commit/run.rs +++ b/nodedb/src/control/server/shared/session/commit/run.rs @@ -82,6 +82,10 @@ pub async fn run_commit( dp: &impl TxnDataPlane, ) -> CommitOutcome { let read_set = sessions.take_read_set(session_id); + // The engine side effects this transaction's index DDL deferred. They run + // once its catalog entries landed, below. An abort before then drops them + // with the buffer. + let deferred_effects = super::super::ddl_buffer::drain_effects(); // Collections this transaction wrote itself. A read of a collection the // same transaction has written is a read-your-own-write, not a // serialization conflict — reading uncommitted own state (served from the @@ -374,6 +378,21 @@ pub async fn run_commit( return CommitOutcome::Aborted { reason }; } + // Index DDL's engine work follows its catalog entries: backfills, index + // teardown, analyzer bindings and sorted-index trees. A failure after the + // catalog landed reports as an abort, as a failed `flush_local` does. + if let Err(error) = + crate::control::server::shared::ddl::neutral::deferred_effects::run_deferred_effects( + state, + deferred_effects, + ) + .await + { + return CommitOutcome::Aborted { + reason: AbortReason::DdlPropose(error), + }; + } + // Record the schema fields this transaction's writes inferred, deferred // from statement time (`staging_gate`-buffered writes are planned against // the descriptor version a statement-time bump would invalidate). The diff --git a/nodedb/src/control/server/shared/session/ddl_buffer.rs b/nodedb/src/control/server/shared/session/ddl_buffer.rs index 970f930ea..ef48c925b 100644 --- a/nodedb/src/control/server/shared/session/ddl_buffer.rs +++ b/nodedb/src/control/server/shared/session/ddl_buffer.rs @@ -17,6 +17,7 @@ use std::cell::RefCell; use crate::control::catalog_entry::CatalogEntry; use super::audit_context::AuditCtx; +use super::ddl_effect::DeferredDdlEffect; /// One buffered DDL statement: the unstamped `CatalogEntry` /// plus the optional audit context captured from @@ -28,6 +29,9 @@ use super::audit_context::AuditCtx; pub struct BufferedDdl { pub entry: CatalogEntry, pub audit: Option, + /// Engine side effects the statement that buffered this entry owes at + /// COMMIT, in statement order. + pub effects: Vec, } /// Unstamped DDL entries buffered during a transaction. @@ -61,6 +65,7 @@ pub fn try_buffer(entry: CatalogEntry) -> bool { buf.push(BufferedDdl { entry, audit: super::audit_context::current(), + effects: Vec::new(), }); true } else { @@ -69,6 +74,38 @@ pub fn try_buffer(entry: CatalogEntry) -> bool { }) } +/// Attach an engine side effect to the entry buffered last, to run at +/// COMMIT. Returns `false` when no buffer is active or it holds no entry: the +/// caller then runs the effect at once. +pub fn defer_effect(effect: DeferredDdlEffect) -> bool { + with_slot(false, |b| { + let mut guard = b.borrow_mut(); + match guard.as_mut().and_then(|buf| buf.last_mut()) { + Some(last) => { + last.effects.push(effect); + true + } + None => false, + } + }) +} + +/// Remove every deferred engine side effect from the active buffer, in +/// statement order, leaving its entries in place. COMMIT calls this once, +/// before it takes the entries. +pub fn drain_effects() -> Vec { + with_slot(Vec::new(), |b| { + b.borrow_mut() + .as_mut() + .map(|buf| { + buf.iter_mut() + .flat_map(|item| std::mem::take(&mut item.effects)) + .collect() + }) + .unwrap_or_default() + }) +} + /// Run `f` over the entries this connection's open transaction has buffered, /// in statement order. Returns `None` when no buffer is active — outside a /// transaction, and outside any connection scope, there is nothing to overlay. @@ -159,6 +196,38 @@ mod tests { .await; } + #[tokio::test] + async fn a_deferred_effect_rides_the_last_entry_and_a_truncate_drops_it() { + use super::super::ddl_effect::DeferredDdlEffect; + let drop = |name: &str| DeferredDdlEffect::SortedIndexDrop { + tenant_id: crate::types::TenantId::new(1), + database_id: crate::types::DatabaseId::DEFAULT, + collection: "scores".into(), + index_name: name.into(), + }; + conn_scope::scoped(async { + activate(); + assert!( + !defer_effect(drop("none")), + "an empty buffer takes no effect" + ); + try_buffer(sample_entry("one")); + assert!(defer_effect(drop("kept"))); + try_buffer(sample_entry("two")); + assert!(defer_effect(drop("dropped"))); + truncate(1); + let effects = drain_effects(); + assert_eq!(effects.len(), 1); + assert!(matches!( + &effects[0], + DeferredDdlEffect::SortedIndexDrop { index_name, .. } if index_name == "kept" + )); + assert!(drain_effects().is_empty(), "draining removes them"); + assert_eq!(buffer_len(), 1, "the entries stay"); + }) + .await; + } + #[tokio::test] async fn discard_clears_buffer() { conn_scope::scoped(async { diff --git a/nodedb/src/control/server/shared/session/ddl_effect.rs b/nodedb/src/control/server/shared/session/ddl_effect.rs new file mode 100644 index 000000000..6c176202d --- /dev/null +++ b/nodedb/src/control/server/shared/session/ddl_effect.rs @@ -0,0 +1,79 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Engine side effects of transactional index DDL, held until COMMIT. +//! +//! Index DDL inside an explicit transaction buffers its catalog entries (see +//! [`super::ddl_buffer`]). Some index kinds also change Data Plane state: a +//! secondary index is backfilled from existing rows, a dropped one is purged, +//! a full-text index binds or resets its analyzer, a sorted index builds +//! or drops its order-statistic tree, and a key-value collection's index is +//! built or dropped in the KV engine. That engine +//! work must follow the catalog: it runs at COMMIT, once the buffered entries +//! landed, and never after a ROLLBACK. Each effect rides on the buffered entry +//! its statement wrote last, so a savepoint rollback drops it with the entry. + +use crate::types::{DatabaseId, TenantId}; +use nodedb_physical::physical_plan::PhysicalPlan; + +/// One engine side effect a buffered index statement owes at COMMIT. +#[derive(Debug, Clone)] +pub enum DeferredDdlEffect { + /// Backfill a secondary index the transaction created, then mark it + /// `Ready`. + SecondaryIndexBuild(SecondaryIndexBuild), + /// Apply an index's engine configuration, such as a full-text analyzer + /// binding. A refusal fails the COMMIT with `sqlstate`. + EngineApply { + tenant_id: TenantId, + database_id: DatabaseId, + collection: String, + plan: PhysicalPlan, + sqlstate: String, + context: String, + }, + /// Remove a dropped index's engine state: a secondary index's entries, + /// or a full-text collection's analyzer binding. + IndexTeardown { + tenant_id: TenantId, + database_id: DatabaseId, + collection: String, + plan: PhysicalPlan, + }, + /// Build a sorted index's tree on the core that owns its collection. + SortedIndexRegister { + tenant_id: TenantId, + database_id: DatabaseId, + collection: String, + plan: PhysicalPlan, + }, + /// Drop a sorted index's tree. + SortedIndexDrop { + tenant_id: TenantId, + database_id: DatabaseId, + collection: String, + index_name: String, + }, + /// Drop a key-value collection's secondary index on `field` from the KV + /// engine. + KvIndexDrop { + tenant_id: TenantId, + database_id: DatabaseId, + collection: String, + field: String, + }, +} + +/// A secondary index a transaction created in the `Building` state. +#[derive(Debug, Clone)] +pub struct SecondaryIndexBuild { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub collection: String, + pub index_name: String, + /// The path the index extracts, without an array suffix. + pub extraction_path: String, + pub is_array: bool, + pub unique: bool, + pub case_insensitive: bool, + pub predicate: Option, +} diff --git a/nodedb/src/control/server/shared/session/ddl_rollback.rs b/nodedb/src/control/server/shared/session/ddl_rollback.rs index 288e1fe50..1c2e785d6 100644 --- a/nodedb/src/control/server/shared/session/ddl_rollback.rs +++ b/nodedb/src/control/server/shared/session/ddl_rollback.rs @@ -110,7 +110,11 @@ mod tests { const TENANT: u64 = 3; fn buffered(entry: CatalogEntry) -> BufferedDdl { - BufferedDdl { entry, audit: None } + BufferedDdl { + entry, + audit: None, + effects: Vec::new(), + } } fn put(name: &str) -> BufferedDdl { diff --git a/nodedb/src/control/server/shared/session/mod.rs b/nodedb/src/control/server/shared/session/mod.rs index 0306c49f2..7f59ca0ed 100644 --- a/nodedb/src/control/server/shared/session/mod.rs +++ b/nodedb/src/control/server/shared/session/mod.rs @@ -11,6 +11,7 @@ pub mod cross_shard_mode; mod cursor; pub mod cursor_spill; pub mod ddl_buffer; +pub mod ddl_effect; mod ddl_flush; pub mod ddl_rollback; pub mod deadline; diff --git a/nodedb/src/control/server/shared/sql/staging_predicates.rs b/nodedb/src/control/server/shared/sql/staging_predicates.rs index b2b763379..abd0a0a42 100644 --- a/nodedb/src/control/server/shared/sql/staging_predicates.rs +++ b/nodedb/src/control/server/shared/sql/staging_predicates.rs @@ -205,6 +205,7 @@ fn kv_write_shape(op: &KvOp) -> Option { | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } | KvOp::MaterializeScan { .. } // Autocommit-only: transaction resolve rejects both. | KvOp::ResolveWrite(_) diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 4cbf917ca..891b8278c 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -62,7 +62,8 @@ pub fn plan_requires_txn_buffering(plan: &PhysicalPlan) -> bool { ) => false, // `Merge`/`UpdateFromJoin` never reach this — staged as point ops instead. - // `BatchInsert` replays via `exec_tx_passthrough` (oracle divergence, see module doc). + // The session expander reshapes a `BatchInsert` page into point inserts, + // which stage like any other point write. PhysicalPlan::Document( DocumentOp::BatchInsert { .. } | DocumentOp::Merge { .. } @@ -77,11 +78,11 @@ pub fn plan_requires_txn_buffering(plan: &PhysicalPlan) -> bool { PhysicalPlan::Vector( VectorOp::Insert { .. } | VectorOp::BatchInsert { .. } | VectorOp::Delete { .. }, ) => true, - // `SetParams` is `Permission::Alter`, but `to_replicated_entry` encodes it — - // classified as a write despite the DDL-like permission tier. - PhysicalPlan::Vector(VectorOp::SetParams { .. }) => true, - // `DropIndex` replicates the same way as `SetParams`. - PhysicalPlan::Vector(VectorOp::DropIndex { .. }) => true, + // ---- Vector: index DDL — encoded, but autocommit-only ---- + // Like the KV and document index DDL: each rides its own autocommit + // `VectorParams` / `VectorIndexDrop` record, and a transaction's redo + // record carries no index DDL, so it executes at the statement. + PhysicalPlan::Vector(VectorOp::SetParams { .. } | VectorOp::DropIndex { .. }) => false, // ---- Vector: reads, not encoded ---- PhysicalPlan::Vector( @@ -144,8 +145,9 @@ pub fn plan_requires_txn_buffering(plan: &PhysicalPlan) -> bool { | CrdtOp::ExportDelta { .. }, ) => false, - // No encoder arm here (see module doc); reaches `exec_tx_passthrough` at COMMIT - // with no reject arm, at the cost of RYOW loss + the no-undo gap. + // Constraint installs arrive from the committed Raft applier, and a + // restore only previews a delta. Resolve emits no redo sub-record for + // either: neither changes a row. PhysicalPlan::Crdt( CrdtOp::SetConstraints { .. } | CrdtOp::DropConstraints { .. } @@ -204,8 +206,8 @@ pub fn plan_requires_txn_buffering(plan: &PhysicalPlan) -> bool { | KvOp::Transfer { .. } | KvOp::TransferItem { .. }, ) => true, - // `Expire`/`Persist` are encoded (buffered), but `execute_tx_kv` rejects - // them inside a `TransactionBatch` — a COMMIT replay fails there regardless. + // `Expire`/`Persist` are buffered: staging records the TTL change in + // the overlay and COMMIT resolves it into the redo record. PhysicalPlan::Kv(KvOp::Expire { .. } | KvOp::Persist { .. }) => true, // ---- Kv: reads, not encoded ---- @@ -220,6 +222,7 @@ pub fn plan_requires_txn_buffering(plan: &PhysicalPlan) -> bool { | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } | KvOp::MaterializeScan { .. } // Read-only: reports what a governed write would apply; encodes nothing. | KvOp::ResolveWrite(_), @@ -782,18 +785,6 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), vector_id: 0, }), - PhysicalPlan::Vector(VectorOp::SetParams { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), - field_name: String::new(), - dim: 0, - m: 0, - ef_construction: 0, - metric: String::new(), - index_type: String::new(), - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - }), PhysicalPlan::Vector(VectorOp::QueryStats { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), field_name: String::new(), @@ -1929,12 +1920,13 @@ mod tests { PhysicalPlan::Meta(MetaOp::RecordCalvinWriteVersions { tenant_id: tenant(), plans: Vec::new(), - epoch: 0, - position: 0, }), PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch: 0, position: 0, + redo: Vec::new(), + collections: Vec::new(), + sum_targets: Vec::new(), }), PhysicalPlan::Meta(MetaOp::CalvinDrop { epoch: 0, @@ -2165,6 +2157,22 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), field: "f".into(), }), + PhysicalPlan::Vector(VectorOp::SetParams { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + field_name: String::new(), + dim: 0, + m: 0, + ef_construction: 0, + metric: String::new(), + index_type: String::new(), + pq_m: 0, + ivf_cells: 0, + ivf_nprobe: 0, + }), + PhysicalPlan::Vector(VectorOp::DropIndex { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + field_name: String::new(), + }), ]; for p in &plans { assert_encoded_but_not_buffered(p); diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/mod.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/mod.rs index 52207191e..2ae4bcc9e 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/mod.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/mod.rs @@ -52,8 +52,8 @@ //! / `crdt_variants_match_oracle` below. Without buffering, //! each of these executed immediately against base state inside an explicit //! transaction, was visible before COMMIT, and survived ROLLBACK: a -//! correctness bug, not a classification nuance. Closing it costs two -//! deliberate, documented trade-offs: +//! correctness bug, not a classification nuance. Closing it has two +//! documented consequences: //! //! 1. RYOW LOSS: a `Buffered` plan does not stage into the per-transaction //! overlay, so a read later in the SAME transaction does not observe the @@ -62,13 +62,10 @@ //! `ArrayOp::{Put, Delete}` is exempt: `is_stageable_write` routes it //! through `MetaOp::StageWrite` into `ArrayTxnOverlay`, so same-transaction //! array reads see it. The `ClusterArrayOp` wrapper is still `Buffered`. -//! 2. NO-UNDO GAP (pre-existing, not fixed here): every flipped variant -//! reaches `exec_tx_passthrough` -//! (`data/executor/handlers/transaction/sub_plan_write.rs`) at COMMIT, -//! which pushes no `UndoEntry`. If a sibling sub-plan fails later in the -//! same COMMIT batch, these writes cannot be reversed by -//! `rollback_undo_log`. Spatial, Text, and bulk Document writes already -//! ride this exact path. +//! 2. ONE INSTALL: COMMIT resolves every buffered plan into the +//! transaction's redo record, and the record installs with undo. A +//! sub-record that fails while it installs rolls back every write of the +//! record before it. //! //! `ClusterArrayOp::{Put, Delete}` also classifies `true` while //! `to_replicated_entry` has no encoder arm for either: they are Control-Plane @@ -90,7 +87,10 @@ //! normally. Pinned by `truncate_is_buffered_and_index_variants_are_not` //! below via `assert_encoded_but_not_buffered` — the inverse of //! `assert_buffered_but_unencoded` — and correspondingly excluded from -//! `kv_variants_match_oracle`. +//! `kv_variants_match_oracle`. `VectorOp::{SetParams, DropIndex}` take the +//! same inverse divergence for the same reason: each rides its own +//! autocommit `VectorParams` / `VectorIndexDrop` record, and a transaction's +//! redo record carries no index DDL. //! //! `DocumentOp::Truncate`, `KvOp::Truncate`, `VectorOp::DirectTruncate`, //! `ColumnarOp::Truncate`, and `TimeseriesOp::Truncate` classify `true`: in a diff --git a/nodedb/src/control/server/sync/async_dispatch/delta/outcome.rs b/nodedb/src/control/server/sync/async_dispatch/delta/outcome.rs index be7e6198a..aa5b50871 100644 --- a/nodedb/src/control/server/sync/async_dispatch/delta/outcome.rs +++ b/nodedb/src/control/server/sync/async_dispatch/delta/outcome.rs @@ -144,6 +144,15 @@ fn reject_frame(delta_msg: &DeltaPushMsg, violation: &ViolationType) -> Option Option { + // The validator's terminal verdict on the frame keeps its structured + // violation and compensation hint. + if let crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::SyncRejected { + violation, + .. + }) = error + { + return reject_frame(delta_msg, violation); + } if let Some(reason) = retryable_refusal_reason(error) { warn!( collection = %delta_msg.collection, @@ -316,6 +325,35 @@ mod tests { ); } + /// A frame the gate refused for good arrives on the error channel, since + /// its record is cancelled. It keeps the validator's structured hint. + #[test] + fn a_terminal_refusal_on_the_error_channel_reaches_the_client_as_a_rejection() { + let error = crate::Error::DataPlane(ErrorCode::SyncRejected { + violation: ViolationType::UniqueViolation { + field: "email".into(), + value: "a@b.com".into(), + }, + applied_seq: 5, + provenance: nodedb_types::sync::wire::SyncProvenance { + producer_id: 1, + epoch: 1, + stream_id: 1, + seq: 5, + }, + }); + let frame = frame_for_dispatch(&delta(), &provisional(), Err(error)).expect("frame"); + assert_eq!(frame.msg_type, SyncMessageType::DeltaReject); + let reject: DeltaRejectMsg = frame.decode_body().expect("reject decodes"); + assert_eq!( + reject.compensation, + Some(CompensationHint::UniqueViolation { + field: "email".into(), + conflicting_value: "a@b.com".into(), + }) + ); + } + #[test] fn an_unreadable_outcome_is_refused_retryably_not_acked_or_rejected() { // Neither "it applied" nor "it never will" is knowable here. Only the diff --git a/nodedb/src/control/server/wal_dispatch/crdt.rs b/nodedb/src/control/server/wal_dispatch/crdt.rs index a2ad505a0..0bd5b3572 100644 --- a/nodedb/src/control/server/wal_dispatch/crdt.rs +++ b/nodedb/src/control/server/wal_dispatch/crdt.rs @@ -80,6 +80,7 @@ pub(crate) fn encode_crdt_op_record( surrogate, provenance, expected_frontier_digest, + peer_id, .. } => { // Versioned payload preserves the admission fence for deterministic @@ -91,7 +92,8 @@ pub(crate) fn encode_crdt_op_record( *expected_frontier_digest, Some(document_id.clone()), Some(surrogate.as_u32()), - ); + ) + .with_peer_id(*peer_id); let crdt_payload = payload.encode().map_err(|e| crate::Error::Serialization { format: "msgpack".into(), detail: format!("wal crdt delta: {e}"), @@ -110,6 +112,7 @@ pub(crate) fn encode_crdt_op_record( auth_seq_no, delta_signature, signing_required, + peer_id, .. } => { let payload = crate::wal::CrdtDeltaWalPayload::new( @@ -126,7 +129,8 @@ pub(crate) fn encode_crdt_op_record( auth_seq_no: *auth_seq_no, delta_signature: *delta_signature, required: *signing_required, - }); + }) + .with_peer_id(*peer_id); let crdt_payload = payload.encode().map_err(|e| crate::Error::Serialization { format: "msgpack".into(), detail: format!("wal authenticated crdt delta: {e}"), diff --git a/nodedb/src/control/server/wal_dispatch/mod.rs b/nodedb/src/control/server/wal_dispatch/mod.rs index 903cee090..ec9f119bd 100644 --- a/nodedb/src/control/server/wal_dispatch/mod.rs +++ b/nodedb/src/control/server/wal_dispatch/mod.rs @@ -43,15 +43,12 @@ pub(crate) use timeseries::{ encode_columnar_truncate_payload, encode_timeseries_batch_payload_with_format, }; pub(crate) use vector::{ - VectorDirectDeleteRecord, VectorDirectTruncateRecord, VectorDirectUpdatePayload, - VectorDirectUpdateRecord, VectorDirectUpsertPayload, VectorDirectUpsertRecord, - VectorResolvedDirectWritePayload, VectorResolvedDirectWriteRecord, + VectorDirectDeleteRecord, VectorDirectTruncateRecord, VectorDirectUpdateRecord, + VectorDirectUpsertRecord, VectorResolvedDirectWritePayload, VectorResolvedDirectWriteRecord, encode_multi_vector_delete_payload, encode_multi_vector_put_payload, encode_sparse_vector_delete_payload, encode_sparse_vector_put_payload, encode_vector_batch_put_payload, encode_vector_delete_by_surrogate_payload, - encode_vector_delete_payload, encode_vector_direct_delete_payload, - encode_vector_direct_truncate_payload, encode_vector_direct_update_payload, - encode_vector_direct_upsert_payload, encode_vector_put_payload, + encode_vector_delete_payload, encode_vector_direct_truncate_payload, encode_vector_put_payload, encode_vector_resolved_direct_write_payload, }; diff --git a/nodedb/src/control/server/wal_dispatch/vector/mod.rs b/nodedb/src/control/server/wal_dispatch/vector/mod.rs index 3bbd8a89f..91eb743a3 100644 --- a/nodedb/src/control/server/wal_dispatch/vector/mod.rs +++ b/nodedb/src/control/server/wal_dispatch/vector/mod.rs @@ -12,14 +12,11 @@ pub use append::{ wal_append_vector_put, }; pub(crate) use encode::{ - VectorDirectDeleteRecord, VectorDirectTruncateRecord, VectorDirectUpdatePayload, - VectorDirectUpdateRecord, VectorDirectUpsertPayload, VectorDirectUpsertRecord, - encode_multi_vector_delete_payload, encode_multi_vector_put_payload, + VectorDirectDeleteRecord, VectorDirectTruncateRecord, VectorDirectUpdateRecord, + VectorDirectUpsertRecord, encode_multi_vector_delete_payload, encode_multi_vector_put_payload, encode_sparse_vector_delete_payload, encode_sparse_vector_put_payload, encode_vector_batch_put_payload, encode_vector_delete_by_surrogate_payload, - encode_vector_delete_payload, encode_vector_direct_delete_payload, - encode_vector_direct_truncate_payload, encode_vector_direct_update_payload, - encode_vector_direct_upsert_payload, encode_vector_put_payload, + encode_vector_delete_payload, encode_vector_direct_truncate_payload, encode_vector_put_payload, }; pub(crate) use encode::{ VectorResolvedDirectWritePayload, VectorResolvedDirectWriteRecord, diff --git a/nodedb/src/control/server/wal_dispatch_kv/append.rs b/nodedb/src/control/server/wal_dispatch_kv/append.rs index 80e1090f0..491c7545d 100644 --- a/nodedb/src/control/server/wal_dispatch_kv/append.rs +++ b/nodedb/src/control/server/wal_dispatch_kv/append.rs @@ -372,6 +372,7 @@ pub fn wal_append_kv_op( | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } | KvOp::SortedIndexTopK { .. } + | KvOp::SortedIndexTxnRead { .. } | KvOp::MaterializeScan { .. } => None, }; Ok(KvAppendOutcome { diff --git a/nodedb/src/control/surrogate/assign/bind_plan/kv.rs b/nodedb/src/control/surrogate/assign/bind_plan/kv.rs index db15da20f..204dd2436 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/kv.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/kv.rs @@ -145,6 +145,7 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut KvOp) -> crate::Result< | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } | KvOp::MaterializeScan { .. } => Ok(()), } } diff --git a/nodedb/src/control/system_txn/data_plane.rs b/nodedb/src/control/system_txn/data_plane.rs index 7608ae0b0..d76054151 100644 --- a/nodedb/src/control/system_txn/data_plane.rs +++ b/nodedb/src/control/system_txn/data_plane.rs @@ -16,9 +16,9 @@ use nodedb_physical::physical_task::PhysicalTask; /// that owns their vShard. /// /// The gateway must NOT be used here: commit-time tasks carry `MetaOp` plans -/// (`ResolveTxn`, `TransactionBatch`) with no named collection, so the +/// (`ResolveTxn`, `ApplyTransactionRedo`) with no named collection, so the /// gateway's router cannot derive a route for them and falls back to vShard 0, -/// durably applying the commit batch on the wrong core. +/// durably applying the commit on the wrong core. pub(super) struct SystemTxnDataPlane<'a> { pub(super) state: &'a SharedState, /// Provenance stamped on the writes this transaction applies. diff --git a/nodedb/src/control/system_txn/mod.rs b/nodedb/src/control/system_txn/mod.rs index d8b34545b..095f2a927 100644 --- a/nodedb/src/control/system_txn/mod.rs +++ b/nodedb/src/control/system_txn/mod.rs @@ -12,6 +12,10 @@ mod data_plane; mod run; mod scope; +#[cfg(test)] +mod tests; -pub use self::run::{SystemTxnError, run_tasks_atomically}; +pub use self::run::{ + SystemTxnError, SystemTxnStatement, run_statements_atomically, run_tasks_atomically, +}; pub use self::scope::SystemTxnScope; diff --git a/nodedb/src/control/system_txn/run.rs b/nodedb/src/control/system_txn/run.rs index 72d48a45e..188cba6a1 100644 --- a/nodedb/src/control/system_txn/run.rs +++ b/nodedb/src/control/system_txn/run.rs @@ -6,10 +6,12 @@ use std::sync::Arc; use crate::control::lease::QueryLeaseScope; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::security::identity::{Permission, required_permission}; use crate::control::server::dispatch_utils; use crate::control::server::shared::session::{ - AbortReason, CommitOutcome, StagingGateError, commit, lifecycle, route_in_tx_write, + AbortReason, CommitOutcome, InTxnRoute, StagingGateError, commit, lifecycle, route_in_tx_write, }; +use crate::control::server::shared::write_admission::plan_requires_txn_buffering; use crate::control::state::SharedState; use crate::event::EventSource; use crate::types::TraceId; @@ -38,9 +40,32 @@ pub enum SystemTxnError { source: crate::Error, }, - /// COMMIT itself aborted. The transaction applied nothing. + /// COMMIT itself aborted. The transaction applied nothing. `code` is the + /// Data Plane's verdict when one decided the abort. #[error("system transaction aborted at commit: {detail}")] - Commit { detail: String }, + Commit { + detail: String, + code: Option>, + }, +} + +impl From for crate::Error { + fn from(error: SystemTxnError) -> Self { + match error { + SystemTxnError::Begin { source } | SystemTxnError::Statement { source, .. } => source, + SystemTxnError::Commit { + code: Some(code), .. + } => crate::Error::DataPlane(*code), + SystemTxnError::Commit { detail, code: None } => crate::Error::Internal { detail }, + } + } +} + +/// One planned statement of a system transaction, with the descriptor +/// leases its plan was built against. +pub struct SystemTxnStatement { + pub tasks: Vec, + pub lease_scope: Arc, } /// Run every task as one transaction: all of them apply, or none do. @@ -59,14 +84,65 @@ pub async fn run_tasks_atomically( lease_scope: Arc, event_source: EventSource, ) -> Result<(), SystemTxnError> { + run_statements_atomically( + state, + identity, + vec![SystemTxnStatement { tasks, lease_scope }], + event_source, + ) + .await +} + +/// Run every statement's tasks as one transaction: all of them apply, or +/// none do. Each statement's leases stay on the tasks it buffered, so COMMIT +/// re-checks every version any statement was planned against. +/// +/// A task that is neither buffered nor a read, such as a data write the +/// transaction cannot buffer or index DDL, applies at once and survives a +/// rollback, so a statement carrying one is refused before BEGIN with +/// [`crate::Error::NotInTransactionBlock`]. A read runs at once against the +/// transaction's overlay, and its error fails the transaction. +pub async fn run_statements_atomically( + state: &SharedState, + identity: &AuthenticatedIdentity, + statements: Vec, + event_source: EventSource, +) -> Result<(), SystemTxnError> { + let total: usize = statements + .iter() + .map(|statement| statement.tasks.len()) + .sum(); + if let Some((index, task)) = statements + .iter() + .flat_map(|statement| statement.tasks.iter()) + .enumerate() + .find(|(_, task)| !runs_in_a_system_transaction(&task.plan)) + { + return Err(SystemTxnError::Statement { + index, + total, + source: crate::Error::NotInTransactionBlock { + statement: match task.plan.collection() { + Some(collection) => format!("a non-transactional write to '{collection}'"), + None => "a non-transactional write".to_owned(), + }, + }, + }); + } let scope = SystemTxnScope::begin(state).map_err(|source| SystemTxnError::Begin { source })?; - let total = tasks.len(); let dp = SystemTxnDataPlane { state, event_source, }; + let tasks = statements.into_iter().flat_map(|statement| { + let lease_scope = statement.lease_scope; + statement + .tasks + .into_iter() + .map(move |task| (task, Arc::clone(&lease_scope))) + }); - for (index, task) in tasks.into_iter().enumerate() { + for (index, (task, lease_scope)) in tasks.enumerate() { let buffered_before = scope.sessions().buffered_task_count(scope.session_id()); let routed = route_in_tx_write( state, @@ -77,8 +153,12 @@ pub async fn run_tasks_atomically( ) .await; - if let Err(error) = routed { - let source = staging_error(error); + let read = match routed { + Ok(InTxnRoute::Read(task)) => dispatch_read(state, *task).await.err(), + Ok(InTxnRoute::Buffered | InTxnRoute::Staged(_)) => None, + Err(error) => Some(staging_error(error)), + }; + if let Some(source) = read { lifecycle::run_rollback(scope.sessions(), scope.session_id(), identity, state, &dp) .await; return Err(SystemTxnError::Statement { @@ -115,6 +195,7 @@ pub async fn run_tasks_atomically( CommitOutcome::Committed => Ok(()), CommitOutcome::Aborted { reason } => Err(SystemTxnError::Commit { detail: describe(&reason), + code: abort_code(&reason).map(Box::new), }), } } @@ -143,6 +224,33 @@ async fn dispatch_staged( .await } +/// Whether a task can run inside a system transaction: it is buffered for +/// COMMIT, or it only reads. +fn runs_in_a_system_transaction(plan: &nodedb_physical::physical_plan::PhysicalPlan) -> bool { + plan_requires_txn_buffering(plan) + || matches!( + required_permission(plan), + Permission::Read | Permission::Monitor | Permission::Execute + ) +} + +/// Run one read of the transaction against its overlay. The rows are not +/// kept: a system transaction answers no rows. Its error fails the +/// transaction. +async fn dispatch_read(state: &SharedState, task: PhysicalTask) -> crate::Result<()> { + let response = dispatch_utils::dispatch_to_data_plane_with_txn( + state, + task.tenant_id, + task.database_id, + task.vshard_id, + task.plan, + TraceId::ZERO, + task.txn_id, + ) + .await?; + dispatch_utils::reject_data_plane_error(&response) +} + /// Flatten a staging-gate refusal into the crate error type. fn staging_error(error: StagingGateError) -> crate::Error { match error { @@ -156,6 +264,22 @@ fn staging_error(error: StagingGateError) -> crate::Error { } } +/// The Data-Plane verdict a commit abort carries, so a caller surfaces the +/// same class a client COMMIT would. +fn abort_code(reason: &AbortReason) -> Option { + match reason { + AbortReason::BatchRejected { code } => code.clone(), + AbortReason::Serialization | AbortReason::SchemaChanged { .. } => { + Some(crate::bridge::envelope::ErrorCode::ConflictRetry) + } + AbortReason::NoTransaction + | AbortReason::CalvinCancelled + | AbortReason::CalvinTimeout + | AbortReason::Dispatch(_) + | AbortReason::DdlPropose(_) => None, + } +} + /// Render a commit abort for the caller's log and retry record. fn describe(reason: &AbortReason) -> String { match reason { diff --git a/nodedb/src/control/system_txn/tests.rs b/nodedb/src/control/system_txn/tests.rs new file mode 100644 index 000000000..4586a6846 --- /dev/null +++ b/nodedb/src/control/system_txn/tests.rs @@ -0,0 +1,157 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A system transaction refuses a write it cannot buffer before BEGIN, and +//! runs a read, whose error fails the transaction. + +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::time::{Duration, Instant}; + +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan, VectorOp}; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; +use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate, TenantId}; + +use super::{SystemTxnError, SystemTxnStatement, run_statements_atomically}; +use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; +use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; +use crate::control::lease::QueryLeaseScope; +use crate::control::security::identity::{AuthenticatedIdentity, DatabaseSet, Role}; +use crate::control::state::SharedState; +use crate::event::EventSource; +use crate::types::{Lsn, VShardId}; +use crate::wal::WalManager; + +fn fixture() -> (Arc, CoreChannelDataSide, tempfile::TempDir) { + let directory = tempfile::tempdir().expect("temporary WAL directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("system_txn.wal")).expect("test WAL"), + ); + let (dispatcher, mut sides) = Dispatcher::new(1, 64); + let side = sides.pop().expect("one data side"); + let state = SharedState::new(dispatcher, wal).expect("shared state"); + (state, side, directory) +} + +fn identity() -> AuthenticatedIdentity { + AuthenticatedIdentity::new_internal_service( + 0, + "_system_txn_test", + TenantId::new(1), + vec![Role::Superuser], + true, + None, + DatabaseSet::All, + ) +} + +fn statement(plan: PhysicalPlan) -> Vec { + vec![SystemTxnStatement { + tasks: vec![PhysicalTask { + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + database_id: DatabaseId::DEFAULT, + plan, + post_set_op: PostSetOp::None, + txn_id: None, + }], + lease_scope: Arc::new(QueryLeaseScope::empty()), + }] +} + +#[tokio::test] +async fn a_write_it_cannot_buffer_is_refused_before_begin() { + let (state, mut side, _directory) = fixture(); + let plan = PhysicalPlan::Vector(VectorOp::DropIndex { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), + field_name: String::new(), + }); + + let result = + run_statements_atomically(&state, &identity(), statement(plan), EventSource::Trigger).await; + + assert!( + matches!( + result, + Err(SystemTxnError::Statement { + index: 0, + source: crate::Error::NotInTransactionBlock { .. }, + .. + }) + ), + "{result:?}" + ); + assert!( + side.request_rx.try_pop().is_err(), + "nothing reaches the data plane" + ); +} + +/// Answer every data-plane request until `stop` is set: the read with an +/// error, everything else with `Ok`. +async fn answer_all(state: Arc, mut side: CoreChannelDataSide, stop: Arc) { + let deadline = Instant::now() + Duration::from_secs(5); + while !stop.load(Ordering::Relaxed) && Instant::now() < deadline { + if let Ok(request) = side.request_rx.try_pop() { + let request = request.inner; + let (status, error_code) = match request.plan { + PhysicalPlan::Document(DocumentOp::PointGet { .. }) => ( + Status::Error, + Some(Box::new(ErrorCode::Internal { + detail: "read failed".into(), + })), + ), + _ => (Status::Ok, None), + }; + let response = Response { + request_id: request.request_id, + status, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }; + side.response_tx + .try_push(BridgeResponse { inner: response }) + .expect("fake data-plane response queue has capacity"); + } + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } +} + +#[tokio::test] +async fn a_read_runs_and_its_error_fails_the_transaction() { + let (state, side, _directory) = fixture(); + let stop = Arc::new(AtomicBool::new(false)); + let responder = tokio::spawn(answer_all(Arc::clone(&state), side, Arc::clone(&stop))); + let plan = PhysicalPlan::Document(DocumentOp::PointGet { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), + document_id: "d1".into(), + surrogate: Surrogate::new(1), + pk_bytes: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }); + + let result = + run_statements_atomically(&state, &identity(), statement(plan), EventSource::Trigger).await; + stop.store(true, Ordering::Relaxed); + responder.await.expect("responder completes"); + + assert!( + matches!( + result, + Err(SystemTxnError::Statement { + index: 0, + source: crate::Error::DataPlane(ErrorCode::Internal { .. }), + .. + }) + ), + "the read ran and its error failed the transaction: {result:?}" + ); +} diff --git a/nodedb/src/control/wal_catchup.rs b/nodedb/src/control/wal_catchup.rs index 74897fdcc..107e915b0 100644 --- a/nodedb/src/control/wal_catchup.rs +++ b/nodedb/src/control/wal_catchup.rs @@ -79,6 +79,32 @@ enum CatchupResult { Idle, } +/// What catch-up does with one timeseries record. +enum Resend { + /// Send the record again under this window. + Send(crate::control::server::dispatch_utils::MintedRecords), + /// The record's outcome is final: step past it. + Skip, + /// A live or held window owns the record and carries it to its outcome. + /// The cursor must not pass it, or the record is never reached again. + Wait, +} + +/// Decide whether the record at `lsn` is sent again. +/// +/// A record at or below the outcome floor, or one whose last owner closed, +/// has a final outcome: sending it again would apply it twice or below a +/// published watermark. A record a window still owns is left for that +/// window. +fn plan_resend(floor: &Arc, lsn: Lsn) -> Resend { + use crate::bridge::dispatch::ResendRefusal; + match crate::control::server::dispatch_utils::MintedRecords::resend(floor, lsn) { + Ok(minted) => Resend::Send(minted), + Err(ResendRefusal::BelowFloor | ResendRefusal::Closed) => Resend::Skip, + Err(ResendRefusal::Owned) => Resend::Wait, + } +} + /// Run one catch-up cycle: read new WAL records, dispatch timeseries batches. /// /// Uses paginated mmap replay to bound memory. Passes WAL LSNs to the @@ -176,15 +202,15 @@ async fn run_catchup_cycle(shared: &SharedState) -> CatchupResult { continue; } - // A record at or below the outcome floor has a final outcome. Sending - // it again would apply it below the floor, where a published - // watermark already claims its outcome. - let Some(minted) = crate::control::server::dispatch_utils::MintedRecords::resend( - &shared.outcome_floor, - Lsn::new(record.header.lsn), - ) else { - max_lsn = max_lsn.max(record.header.lsn); - continue; + let minted = match plan_resend(&shared.outcome_floor, Lsn::new(record.header.lsn)) { + Resend::Send(minted) => minted, + Resend::Skip => { + max_lsn = max_lsn.max(record.header.lsn); + continue; + } + // The cursor stays below this record, so a later cycle reaches + // it again once its window settles. + Resend::Wait => break, }; let tenant_id = TenantId::new(record.header.tenant_id); @@ -258,3 +284,25 @@ async fn run_catchup_cycle(shared: &SharedState) -> CatchupResult { CatchupResult::Idle } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::dispatch::OutcomeFloor; + + /// Catch-up never steps past a record a live or held window owns, skips + /// one whose outcome is final, and sends a free one. + #[test] + fn catchup_waits_on_an_owned_record_and_skips_a_closed_one() { + let floor = OutcomeFloor::new(); + floor.open_dispatched(Lsn::new(5)).hold(); + floor.open_dispatched(Lsn::new(6)).settle(); + + assert!(matches!(plan_resend(&floor, Lsn::new(5)), Resend::Wait)); + assert!(matches!(plan_resend(&floor, Lsn::new(6)), Resend::Skip)); + match plan_resend(&floor, Lsn::new(7)) { + Resend::Send(minted) => minted.settle(), + Resend::Skip | Resend::Wait => panic!("a free record above the floor is sent"), + } + } +} diff --git a/nodedb/src/control/wal_replication/decode/transaction_redo.rs b/nodedb/src/control/wal_replication/decode/transaction_redo.rs index 0dfc4ff18..558009313 100644 --- a/nodedb/src/control/wal_replication/decode/transaction_redo.rs +++ b/nodedb/src/control/wal_replication/decode/transaction_redo.rs @@ -66,6 +66,8 @@ mod tests { epoch: 11, position: 2, vshard_id: 7, + collections: Vec::new(), + sum_targets: Vec::new(), }), }, collections: vec!["accounts".into(), "entries".into()], diff --git a/nodedb/src/control/wal_replication/encode/entry_kv.rs b/nodedb/src/control/wal_replication/encode/entry_kv.rs index a43398d8f..59bf3c677 100644 --- a/nodedb/src/control/wal_replication/encode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/encode/entry_kv.rs @@ -350,7 +350,8 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { | KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } - | KvOp::SortedIndexScore { .. } => return Ok(None), + | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } => return Ok(None), })) } diff --git a/nodedb/src/control/write_resolve/kv.rs b/nodedb/src/control/write_resolve/kv.rs index 50f781610..eff276603 100644 --- a/nodedb/src/control/write_resolve/kv.rs +++ b/nodedb/src/control/write_resolve/kv.rs @@ -130,6 +130,7 @@ pub(super) fn resolver_for_kv_op(op: &KvOp) -> Option return None, diff --git a/nodedb/src/data/executor/core_loop/accessors.rs b/nodedb/src/data/executor/core_loop/accessors.rs index 7907b3f58..0eb82331b 100644 --- a/nodedb/src/data/executor/core_loop/accessors.rs +++ b/nodedb/src/data/executor/core_loop/accessors.rs @@ -238,6 +238,16 @@ impl CoreLoop { .unwrap_or_else(crate::engine::kv::current_ms) } + /// The instant a KV liveness read evaluates expiry at. A Calvin + /// transaction reads at its epoch instant while it stages, resolves or + /// renders its reply, so every replica sees the same live rows. Every + /// other read uses the wall clock. + pub(in crate::data::executor) fn kv_read_now_ms(&self) -> u64 { + self.epoch_system_ms + .map(|ms| ms as u64) + .unwrap_or_else(crate::engine::kv::current_ms) + } + /// Write a raw segment blob directly into the FTS LSM segment store for /// a given `(tenant, collection)`. /// diff --git a/nodedb/src/data/executor/core_loop/calvin_fence.rs b/nodedb/src/data/executor/core_loop/calvin_fence.rs new file mode 100644 index 000000000..0c04d2558 --- /dev/null +++ b/nodedb/src/data/executor/core_loop/calvin_fence.rs @@ -0,0 +1,340 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Calvin key ownership on a core. +//! +//! A staged Calvin transaction owns the rows it staged from its stage until +//! its flush or drop. Its flush installs images the stage computed, so a +//! write that lands on one of those rows in between would be overwritten. +//! A write to an owned row therefore waits here, in arrival order. +//! +//! - A document or KV point write waits only when a staged transaction staged +//! its row, or truncated its collection. +//! - Every other write, and a committed session transaction's redo, waits +//! when a staged transaction writes its collection. +//! +//! When the owner resolves, each waiting write that no other owner fences +//! runs, in arrival order. A write whose WAL record sits below the redo record +//! of a flush it waited on is refused with `RetryableRefusal`: restart replay +//! applies records in LSN order, so it would install that write before the +//! flush. The refusal cancels its record, and the client retries. A write +//! still waiting at its request deadline is refused the same way. + +use std::collections::VecDeque; +use std::time::Instant; + +use nodedb_physical::physical_plan::{DocumentOp, KvOp, MetaOp, PhysicalPlan}; +use nodedb_types::{RowIdentity, TenantId}; + +use super::CoreLoop; +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::control::server::shared::write_admission::plan_is_write; +use crate::control::wal_replication::transaction_redo::collections::written_collections; +use crate::data::executor::handlers::control::calvin_synthetic_txn_id; +use crate::data::executor::handlers::transaction::stage_write::kv_row_identity; +use crate::data::executor::task::ExecutionTask; +use crate::types::Lsn; + +/// `(epoch, position, vshard)` of a staged Calvin transaction. +type Owner = (u64, u32, u32); + +/// A write waiting for the Calvin transaction that owns its rows. +pub(in crate::data::executor) struct ParkedWrite { + task: ExecutionTask, + owner: Owner, + /// The highest redo LSN among the flushes this write waited on. + passed_flush_lsn: Option, +} + +/// The writes waiting on this core, in arrival order, and the owners that +/// resolved since the core last released writes. +#[derive(Default)] +pub(in crate::data::executor) struct CalvinFence { + parked: VecDeque, + /// Each owner that flushed, with its redo LSN, or dropped, with `None`. + resolved: Vec<(Owner, Option)>, +} + +#[cfg(test)] +impl CalvinFence { + /// Number of waiting writes. + pub(in crate::data::executor) fn len(&self) -> usize { + self.parked.len() + } +} + +impl CalvinFence { + /// Record that the staged transaction `owner` flushed at `flush_lsn`, or + /// dropped when `flush_lsn` is `None`. The core releases its waiting + /// writes once the resolving request answered. + pub(in crate::data::executor) fn note_resolved( + &mut self, + owner: (u64, u32, u32), + flush_lsn: Option, + ) { + if !self.parked.is_empty() { + self.resolved.push((owner, flush_lsn)); + } + } +} + +/// What a write targets, as the fence compares it with staged state. +enum FenceTarget { + /// Every row of a collection. + Collection(String), + /// One document row, by surrogate. + Surrogate(String, u32), + /// One KV row, by its overlay identity. + Row(String, RowIdentity), +} + +impl FenceTarget { + fn collection(&self) -> &str { + match self { + Self::Collection(collection) + | Self::Surrogate(collection, _) + | Self::Row(collection, _) => collection, + } + } +} + +/// The targets of `plan`, or `None` when it writes no base state. +fn fence_targets(plan: &PhysicalPlan) -> Option> { + if let PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { collections, .. }) = plan { + return Some( + collections + .iter() + .cloned() + .map(FenceTarget::Collection) + .collect(), + ); + } + if matches!( + plan, + PhysicalPlan::Meta( + MetaOp::CalvinExecuteStatic { .. } + | MetaOp::CalvinExecuteActive { .. } + | MetaOp::CalvinFlush { .. } + | MetaOp::CalvinDrop { .. } + | MetaOp::CalvinResolve { .. } + ) + ) || !plan_is_write(plan) + { + return None; + } + let point = match plan { + PhysicalPlan::Document( + DocumentOp::PointPut { + collection, + surrogate, + .. + } + | DocumentOp::PointInsert { + collection, + surrogate, + .. + } + | DocumentOp::PointUpdate { + collection, + surrogate, + .. + } + | DocumentOp::PointDelete { + collection, + surrogate, + .. + }, + ) => vec![FenceTarget::Surrogate( + collection.as_str().to_string(), + surrogate.as_u32(), + )], + PhysicalPlan::Kv( + KvOp::Put { + collection, key, .. + } + | KvOp::Insert { + collection, key, .. + } + | KvOp::InsertIfAbsent { + collection, key, .. + } + | KvOp::Expire { + collection, key, .. + } + | KvOp::Persist { + collection, key, .. + }, + ) => vec![FenceTarget::Row( + collection.as_str().to_string(), + kv_row_identity(key), + )], + PhysicalPlan::Kv(KvOp::Delete { + collection, keys, .. + }) => keys + .iter() + .map(|key| FenceTarget::Row(collection.as_str().to_string(), kv_row_identity(key))) + .collect(), + _ => written_collections(std::slice::from_ref(plan)) + .into_iter() + .map(FenceTarget::Collection) + .collect(), + }; + Some(point) +} + +impl CoreLoop { + /// Hold `task` when a staged Calvin transaction owns a row it writes. + /// Returns the task when nothing owns its rows. + pub(in crate::data::executor) fn park_if_calvin_owned( + &mut self, + task: ExecutionTask, + ) -> Option { + match self.calvin_owner_of(&task) { + Some(owner) => { + self.calvin.fence.parked.push_back(ParkedWrite { + task, + owner, + passed_flush_lsn: None, + }); + None + } + None => Some(task), + } + } + + /// The earliest staged Calvin transaction that owns a row `task` writes. + fn calvin_owner_of(&self, task: &ExecutionTask) -> Option { + if self.calvin.commit_pending.is_empty() { + return None; + } + let targets = fence_targets(task.plan())?; + let database_id = task.request.database_id; + let tenant_id = task.request.tenant_id; + let mut owners: Vec = self.calvin.commit_pending.keys().copied().collect(); + owners.sort_unstable(); + owners + .into_iter() + .find(|owner| self.owns_any(*owner, database_id, tenant_id, &targets)) + } + + /// Whether the staged transaction `owner` owns any of `targets`. + fn owns_any( + &self, + owner: Owner, + database_id: crate::types::DatabaseId, + tenant_id: TenantId, + targets: &[FenceTarget], + ) -> bool { + let Some(pending) = self.calvin.commit_pending.get(&owner) else { + return false; + }; + if pending.tenant_id != tenant_id { + return false; + } + let written = written_collections(&pending.plans); + let overlay = calvin_synthetic_txn_id(owner.0, owner.1, owner.2) + .ok() + .and_then(|txn_id| self.txn_overlays.get(&txn_id)); + targets.iter().any(|target| { + if !written.iter().any(|c| c == target.collection()) { + return false; + } + let coll_key = (database_id, tenant_id, target.collection().to_string()); + match target { + FenceTarget::Collection(_) => true, + FenceTarget::Surrogate(_, surrogate) => overlay.is_none_or(|overlay| { + overlay.is_truncated(&coll_key) + || overlay.get(&coll_key, *surrogate).is_some() + || overlay.get_ttl(&coll_key, *surrogate).is_some() + }), + FenceTarget::Row(_, identity) => overlay.is_none_or(|overlay| { + overlay.is_truncated(&coll_key) + || overlay.surrogate_for_doc_id(&coll_key, identity).is_some() + }), + } + }) + } + + /// Run every waiting write no owner fences any more, once an owner + /// resolved. + pub(in crate::data::executor) fn release_resolved_calvin_owners(&mut self) { + let resolved = std::mem::take(&mut self.calvin.fence.resolved); + if resolved.is_empty() { + return; + } + for (owner, flush_lsn) in resolved { + let Some(lsn) = flush_lsn else { + continue; + }; + for parked in &mut self.calvin.fence.parked { + if parked.owner == owner { + parked.passed_flush_lsn = parked.passed_flush_lsn.max(Some(lsn)); + } + } + } + let parked = std::mem::take(&mut self.calvin.fence.parked); + for mut write in parked { + if let Some(next) = self.calvin_owner_of(&write.task) { + write.owner = next; + self.calvin.fence.parked.push_back(write); + continue; + } + let below_flush = write + .passed_flush_lsn + .is_some_and(|flush| write.task.wal_lsn().is_some_and(|lsn| lsn < flush)); + if below_flush { + let response = self.response_error( + &write.task, + ErrorCode::RetryableRefusal { + reason: "a Calvin transaction installed this row at a later log \ + position while the write waited" + .into(), + }, + ); + self.send_parked_response(response); + } else { + self.run_task(write.task); + } + } + } + + /// Refuse every waiting write whose request deadline passed. + pub(in crate::data::executor) fn expire_calvin_parked(&mut self) { + if self.calvin.fence.parked.is_empty() { + return; + } + // no-determinism: the request deadline bounds a wait; the refusal + // cancels the write's record, so no replica applies it. + let now = Instant::now(); + let parked = std::mem::take(&mut self.calvin.fence.parked); + for write in parked { + if now > write.task.request.deadline { + let response = self.response_error( + &write.task, + ErrorCode::RetryableRefusal { + reason: format!( + "the write waited past its deadline for Calvin transaction \ + {}/{} that owns its row", + write.owner.0, write.owner.1 + ), + }, + ); + self.send_parked_response(response); + } else { + self.calvin.fence.parked.push_back(write); + } + } + } + + fn send_parked_response(&mut self, response: Response) { + if let Err(e) = self + .response_tx + .try_push(crate::bridge::dispatch::BridgeResponse { inner: response }) + { + tracing::warn!( + core = self.core_id, + error = %e, + "failed to send a parked write's refusal: response queue full" + ); + } + } +} diff --git a/nodedb/src/data/executor/core_loop/calvin_state.rs b/nodedb/src/data/executor/core_loop/calvin_state.rs new file mode 100644 index 000000000..212f96085 --- /dev/null +++ b/nodedb/src/data/executor/core_loop/calvin_state.rs @@ -0,0 +1,53 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The Calvin transaction state a core holds. + +use std::collections::HashMap; + +use super::calvin_fence::CalvinFence; +use super::commit_pending::PendingCommit; + +/// Staged Calvin transactions, the writes that wait on them, and the leader +/// flag of the transaction executing now. +pub(in crate::data::executor) struct CalvinCoreState { + /// Staged Calvin transactions awaiting the global verdict, keyed by + /// `(epoch, position, vshard)`. `CalvinExecuteStatic` validates a + /// transaction, stages its plans into the synthetic overlay, and inserts + /// its entry here without mutating base. The verdict-driven `CalvinFlush` + /// installs the transaction's redo record, and `CalvinDrop` discards the + /// staged state. Nothing staged here is observable in the base engines + /// until a flush. + /// + /// The vShard is part of the key because vShards round-robin onto cores. + /// Several vShard slices of one multi-participant transaction share the + /// same `(epoch, position)` and can land on the same core. Each slice + /// stages and flushes on its own. + pub(in crate::data::executor) commit_pending: HashMap<(u64, u32, u32), PendingCommit>, + + /// Whether this node leads the data group of the Calvin transaction + /// executing now. + /// + /// `execute_calvin_execute_active` sets it from the scheduler-stamped, + /// per-node `is_group_leader` around its OLLP verification, then restores + /// the resting value `true`. A shard that runs a bulk DML directly has no + /// replication followers, so it must run OLLP drift verification. + /// + /// OLLP determinism: the verification emits `OllpRetryRequired` only when + /// this is `true`. Every replica stages the carried + /// `ollp_predicted_surrogates` set verbatim, so all replicas write the same + /// surrogate set whatever their local scans read. + pub(in crate::data::executor) ollp_is_group_leader: bool, + + /// Writes that wait for the staged transaction owning their rows. + pub(in crate::data::executor) fence: CalvinFence, +} + +impl CalvinCoreState { + pub(in crate::data::executor) fn new() -> Self { + Self { + commit_pending: HashMap::new(), + ollp_is_group_leader: true, + fence: CalvinFence::default(), + } + } +} diff --git a/nodedb/src/data/executor/core_loop/commit_pending.rs b/nodedb/src/data/executor/core_loop/commit_pending.rs index ea3704161..59f46f60b 100644 --- a/nodedb/src/data/executor/core_loop/commit_pending.rs +++ b/nodedb/src/data/executor/core_loop/commit_pending.rs @@ -2,28 +2,30 @@ //! Staged Calvin write plans awaiting the local commit verdict. +use crate::data::executor::handlers::control::calvin_reply::CalvinReply; use crate::types::TenantId; use nodedb_physical::physical_plan::PhysicalPlan; -/// A Calvin transaction's write plans staged for commit, held between the +/// A Calvin transaction staged for commit, held between the /// validate-and-stage step and the verdict-driven flush-or-drop. /// -/// `CalvinExecuteStatic` inserts this into the core's commit-pending buffer -/// WITHOUT mutating base or firing side effects, and returns the local commit -/// verdict to the scheduler. A verdict-driven `CalvinFlush` replays `plans` -/// through the durable apply funnel (setting the epoch time anchor and -/// leadership scope captured here); a `CalvinDrop` discards it. Nothing here is -/// observable in the base engines until a flush. +/// `CalvinExecuteStatic` and `CalvinExecuteActive` stage the plans into the +/// synthetic overlay and insert this entry WITHOUT mutating base or firing +/// side effects. `CalvinResolve` resolves the overlay into the transaction's +/// redo record. A verdict-driven `CalvinFlush` installs that record and +/// answers with `reply`. A `CalvinDrop` discards the entry. +/// Nothing here is observable in the base engines until a flush. pub(in crate::data::executor) struct PendingCommit { - /// Physical write plans replayed through `execute_transaction_batch` on flush. + /// The staged write plans `CalvinResolve` resolves against the overlay. pub plans: Vec, - /// Tenant scope for `plans`, applied when the flush replays them. + /// Tenant scope for `plans`. pub tenant_id: TenantId, - /// Deterministic epoch timestamp anchor, restored on flush so time-dependent - /// writes (bitemporal sys_from, KV TTL, timeseries system_ms) stay identical - /// across replicas. + /// Deterministic epoch timestamp anchor. Resolve restores it so the + /// stamps it assigns are identical across replicas, and the flush + /// advances the core clock to it. pub epoch_system_ms: i64, - /// Whether this node led the data-group at stage time; restored on flush so - /// the leader-only OLLP verification runs on the same participant. - pub is_group_leader: bool, + /// The reply the transaction's plans decided when they staged: the last + /// `RETURNING` plan's rows, else the last plan's affected count. The + /// flush answers with it. + pub reply: CalvinReply, } diff --git a/nodedb/src/data/executor/core_loop/deferred.rs b/nodedb/src/data/executor/core_loop/deferred.rs index e1bf27b22..2e0a21aeb 100644 --- a/nodedb/src/data/executor/core_loop/deferred.rs +++ b/nodedb/src/data/executor/core_loop/deferred.rs @@ -1,9 +1,9 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Deferred trigger event collection during transaction batches. +//! Deferred trigger events of a committed transaction. //! -//! Accumulates write metadata during `execute_transaction_batch()`. -//! After successful commit, emits these as WriteEvents with +//! The redo install collects the document writes of a committed record and, +//! once the record settled, emits them as WriteEvents with //! `EventSource::Deferred` so the Event Plane fires DEFERRED-mode triggers. use std::sync::Arc; @@ -22,9 +22,9 @@ pub(in crate::data::executor) struct DeferredWrite { } impl CoreLoop { - /// Emit deferred trigger events for a completed transaction batch. + /// Emit deferred trigger events for a committed transaction. /// - /// Called after `execute_transaction_batch()` commits successfully. + /// Called after a committed redo record installed and settled. /// Each write in the transaction is emitted as a WriteEvent with /// `EventSource::Deferred`, which the Event Plane consumer routes /// to DEFERRED-mode triggers. diff --git a/nodedb/src/data/executor/core_loop/event_emit.rs b/nodedb/src/data/executor/core_loop/event_emit.rs index 16ca433d6..3f60ca90f 100644 --- a/nodedb/src/data/executor/core_loop/event_emit.rs +++ b/nodedb/src/data/executor/core_loop/event_emit.rs @@ -182,7 +182,9 @@ impl CoreLoop { labels: &[String], op: crate::event::WriteOp, ) { - if let Some(lsn) = task.wal_lsn() + // A committed-redo apply moves the watermark once the record settled. + if self.redo_apply.scope.is_none() + && let Some(lsn) = task.wal_lsn() && lsn > self.watermark { self.watermark = lsn; diff --git a/nodedb/src/data/executor/core_loop/maintenance.rs b/nodedb/src/data/executor/core_loop/maintenance.rs index 39060c111..29804b0a6 100644 --- a/nodedb/src/data/executor/core_loop/maintenance.rs +++ b/nodedb/src/data/executor/core_loop/maintenance.rs @@ -27,8 +27,8 @@ impl CoreLoop { interval: std::time::Duration, tombstone_threshold: f64, ) { - self.compaction_interval = interval; - self.compaction_tombstone_threshold = tombstone_threshold; + self.maintenance.compaction_interval = interval; + self.maintenance.compaction_tombstone_threshold = tombstone_threshold; } /// Set shared system metrics reference (called after open, before event loop). @@ -55,7 +55,7 @@ impl CoreLoop { &mut self, tracker: Arc, ) { - self.maintenance_budget = Some(tracker); + self.maintenance.maintenance_budget = Some(tracker); } /// Set checkpoint coordinator config (called after open, before event loop). @@ -69,7 +69,7 @@ impl CoreLoop { &mut self, config: crate::storage::compaction::CompactionConfig, ) { - self.segment_compaction_config = config; + self.maintenance.segment_compaction_config = config; } /// Set the number of Data Plane cores on this node. The committed-redo diff --git a/nodedb/src/data/executor/core_loop/maintenance_state.rs b/nodedb/src/data/executor/core_loop/maintenance_state.rs new file mode 100644 index 000000000..c5c151411 --- /dev/null +++ b/nodedb/src/data/executor/core_loop/maintenance_state.rs @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The background maintenance state a core holds. + +use std::sync::Arc; + +use crate::data::executor::handlers::control::reindex::PendingReindex; + +/// Compaction pacing, the maintenance CPU budget, and in-flight index +/// rebuilds. +pub(in crate::data::executor) struct MaintenanceState { + /// Last time periodic maintenance (compaction, edge sweep) ran. + pub(in crate::data::executor) last_maintenance: Option, + + /// How often `maybe_run_maintenance` triggers. + pub(in crate::data::executor) compaction_interval: std::time::Duration, + + /// Tombstone ratio threshold for auto-compaction (0.0–1.0). + pub(in crate::data::executor) compaction_tombstone_threshold: f64, + + /// L1 segment compaction config for the storage layer. + pub(in crate::data::executor) segment_compaction_config: + crate::storage::compaction::CompactionConfig, + + /// Shared per-database maintenance CPU budget tracker. Every maintenance + /// site gates per-database background work against the quota's + /// `maintenance_cpu_pct`. `set_maintenance_budget` sets it after spawn. + pub(in crate::data::executor) maintenance_budget: + Option>, + + /// In-flight concurrent index rebuilds, polled each tick. + /// + /// Each entry yields a `RebuildResult` once its background OS thread + /// finishes the shadow build. Only one rebuild per collection runs at a + /// time. `execute_rebuild_index` returns `ErrorCode::Conflict` for a + /// second one. + pub(in crate::data::executor) pending_reindex: Vec, +} + +impl MaintenanceState { + pub(in crate::data::executor) fn new() -> Self { + Self { + last_maintenance: None, + compaction_interval: std::time::Duration::from_secs(600), + compaction_tombstone_threshold: 0.2, + segment_compaction_config: crate::storage::compaction::CompactionConfig::default(), + maintenance_budget: None, + pending_reindex: Vec::new(), + } + } +} diff --git a/nodedb/src/data/executor/core_loop/mod.rs b/nodedb/src/data/executor/core_loop/mod.rs index 872e75ad6..6da403379 100644 --- a/nodedb/src/data/executor/core_loop/mod.rs +++ b/nodedb/src/data/executor/core_loop/mod.rs @@ -2,6 +2,8 @@ mod accessors; mod bitemporal_time; +pub(in crate::data::executor) mod calvin_fence; +pub(in crate::data::executor) mod calvin_state; pub(in crate::data::executor) mod checkpoint_floors; mod columnar_schema_seed; pub(in crate::data::executor) mod commit_pending; @@ -15,6 +17,7 @@ pub(in crate::data::executor) mod filter_match; mod graph_partition; pub(in crate::data::executor) mod index_value_versions; pub(in crate::data::executor) mod maintenance; +pub(in crate::data::executor) mod maintenance_state; mod open; pub mod pressure; pub(in crate::data::executor) mod priority_queues; diff --git a/nodedb/src/data/executor/core_loop/open.rs b/nodedb/src/data/executor/core_loop/open.rs index 45b6366c3..c816ceaee 100644 --- a/nodedb/src/data/executor/core_loop/open.rs +++ b/nodedb/src/data/executor/core_loop/open.rs @@ -128,9 +128,7 @@ impl CoreLoop { producer_epoch_floor: HashMap::new(), stats_store, aggregate_cache: HashMap::new(), - last_maintenance: None, - compaction_interval: std::time::Duration::from_secs(600), - compaction_tombstone_threshold: 0.2, + maintenance: super::maintenance_state::MaintenanceState::new(), index_configs: HashMap::new(), ivf_indexes: HashMap::new(), sparse_vector_indexes: HashMap::new(), @@ -154,7 +152,6 @@ impl CoreLoop { checkpoint_coordinator: crate::storage::checkpoint::CheckpointCoordinator::new( crate::storage::checkpoint::CheckpointConfig::default(), ), - segment_compaction_config: crate::storage::compaction::CompactionConfig::default(), spatial_indexes: std::collections::HashMap::new(), spatial_doc_map: std::collections::HashMap::new(), vector_doc_map: std::collections::HashMap::new(), @@ -182,7 +179,6 @@ impl CoreLoop { ts_segment_kek: None, }, governor, - maintenance_budget: None, throttle: super::pressure::SpscThrottle::new(), collection_arena_registry: None, metrics: None, @@ -190,26 +186,15 @@ impl CoreLoop { event_sequence: 0, quiesce: None, quarantine_registry: None, - pending_reindex: Vec::new(), epoch_system_ms: None, - // Resting state is authoritative: a shard executing a bulk DML - // directly (single-shard / non-Calvin dispatch) has no replication - // followers, so it MUST run OLLP drift verification. The Calvin - // replicated path scopes this to actual group leadership for the - // duration of a batch and restores it afterward — followers in a - // group skip verification and apply the leader's predicted set. - ollp_is_group_leader: true, txn_overlays: HashMap::new(), graph_txn_overlays: HashMap::new(), array_txn_overlays: HashMap::new(), txn_created_columnar_engines: HashMap::new(), write_index: super::write_index::WriteVersionIndex::new(), - commit_pending: HashMap::new(), - calvin_flush_key: None, - calvin_flush_index_tuples: HashMap::new(), + calvin: super::calvin_state::CalvinCoreState::new(), active_bitemporal_stamps: HashMap::new(), active_graph_system_from: None, - balanced_txn_entries: None, redo_apply: crate::data::executor::handlers::transaction::redo_apply::RedoApplyState::new(), fail_stop: super::fail_stop::CoreFailStop::default(), diff --git a/nodedb/src/data/executor/core_loop/state.rs b/nodedb/src/data/executor/core_loop/state.rs index f8713a2a1..950151a83 100644 --- a/nodedb/src/data/executor/core_loop/state.rs +++ b/nodedb/src/data/executor/core_loop/state.rs @@ -209,9 +209,6 @@ pub struct CoreLoop { super::super::handlers::aggregate::AggregateCacheEntry, >, - /// Last time periodic maintenance (compaction, edge sweep) was run. - pub(in crate::data::executor) last_maintenance: Option, - /// Per-collection full index config (includes index_type, PQ params, IVF params). /// Stored alongside vector_params for collections that use non-default index types. /// Key: `(DatabaseId, TenantId, collection_key)` — same shape as `vector_collections`. @@ -229,12 +226,6 @@ pub struct CoreLoop { pub(in crate::data::executor) sparse_vector_indexes: HashMap<(DatabaseId, TenantId, String, String), SparseInvertedIndex>, - /// Compaction interval (how often `maybe_run_maintenance` triggers). - pub(in crate::data::executor) compaction_interval: std::time::Duration, - - /// Tombstone ratio threshold for auto-compaction (0.0–1.0). - pub(in crate::data::executor) compaction_tombstone_threshold: f64, - /// Per-core LRU document cache for O(1) hot-key point lookups. /// Invalidated write-through on PointPut/Delete/Update. pub(in crate::data::executor) doc_cache: DocCache, @@ -338,10 +329,6 @@ pub struct CoreLoop { pub(in crate::data::executor) checkpoint_coordinator: crate::storage::checkpoint::CheckpointCoordinator, - /// L1 segment compaction config for the storage layer. - pub(in crate::data::executor) segment_compaction_config: - crate::storage::compaction::CompactionConfig, - /// Per-collection document index configurations. /// Maps (DatabaseId, TenantId, collection) → CollectionConfig. /// Populated via RegisterDocumentCollection plans. @@ -391,14 +378,6 @@ pub struct CoreLoop { /// Memory governor for per-engine budget enforcement. pub(in crate::data::executor) governor: Arc, - /// Shared per-database maintenance CPU budget tracker. - /// - /// Used by all maintenance sites (`run_compaction` and friends) to gate - /// per-database background work against the quota's `maintenance_cpu_pct`. - /// Set by `set_maintenance_budget` after core spawn. - pub(in crate::data::executor) maintenance_budget: - Option>, - /// Request intake level for this tick, folded from engine memory /// pressure and response-ring utilization. Drain depth and the suspend /// decision both read off it. @@ -445,52 +424,23 @@ pub struct CoreLoop { pub(in crate::data::executor) quarantine_registry: Option>, - /// In-flight concurrent index rebuilds, polled each tick. - /// - /// Each entry is a `(collection_key, receiver)` pair. The receiver - /// yields a `RebuildResult` once the background OS thread finishes - /// the shadow build. Only one rebuild per collection may be in - /// progress at a time; `execute_rebuild_index` returns - /// `ErrorCode::Conflict` when a second is attempted. - pub(in crate::data::executor) pending_reindex: - Vec, + /// Compaction pacing, the maintenance CPU budget, and index rebuilds. + pub(in crate::data::executor) maintenance: super::maintenance_state::MaintenanceState, /// Ambient deterministic timestamp for the current Calvin epoch. /// - /// Set to `Some(ms)` by `execute_calvin_execute_static` and - /// `execute_calvin_execute_active` before dispatching the inner - /// transaction batch, then reset to `None` immediately after. - /// Engine handlers that need "current time" (bitemporal sys_from, KV TTL - /// expire_at, timeseries system_ms) call + /// Set to `Some(ms)` by `execute_calvin_execute_static`, + /// `execute_calvin_execute_active` and `execute_calvin_resolve` while they + /// stage or resolve a transaction's plans, and by `execute_calvin_flush` + /// while it renders the reply, then restored immediately after. Engine handlers that need "current time" (bitemporal sys_from, + /// KV TTL expire_at, timeseries system_ms) call /// `self.epoch_system_ms.unwrap_or_else()` so that - /// single-shard (non-Calvin) paths continue working without change. + /// single-shard (non-Calvin) paths read the wall clock. /// /// Safety: this is safe because `CoreLoop` is `!Send` and single-threaded - /// per core. Sub-plans inside `execute_transaction_batch` do not recurse - /// back into a Calvin execute variant, so the reset after the batch is not - /// premature. + /// per core, and staging never recurses into another Calvin execute. pub(in crate::data::executor) epoch_system_ms: Option, - /// Whether THIS node is the leader of the data-group owning the vshard for - /// the currently-executing Calvin transaction. - /// - /// Set to `true`/`false` by `execute_calvin_execute_static` and - /// `execute_calvin_execute_active` (from the scheduler-stamped, per-node, - /// non-replicated `MetaOp::is_group_leader`) before dispatching the inner - /// transaction batch, then reset to `false` immediately after. - /// - /// OLLP determinism: the bulk-DML handlers run the optimistic-lock - /// verification (`actual != predicted`) and emit `OllpRetryRequired` ONLY - /// when this is `true`. Every replica — leader and follower alike — applies - /// the carried `ollp_predicted_surrogates` set verbatim, so all replicas - /// mutate the identical surrogate set regardless of any per-replica local - /// scan drift. - /// - /// Defaults to `false` outside a Calvin execute (single-shard / non-Calvin - /// paths never set predicted surrogates, so the flag is never read there). - /// Safe for the same single-threaded `!Send` reasons as `epoch_system_ms`. - pub(in crate::data::executor) ollp_is_group_leader: bool, - /// Per-transaction staging overlay: not-yet-durable writes for each /// in-flight transaction on this core, keyed by `TxnId`. Populated by /// `MetaOp::StageWrite`, released by `MetaOp::DropTxnOverlay`. @@ -536,46 +486,9 @@ pub struct CoreLoop { /// mutation in the current live apply/replay scope. pub(in crate::data::executor) active_graph_system_from: Option, - /// Signed BALANCED entries accumulated for the transaction batch currently - /// applying on this core, as `(collection, entry)`. - /// - /// `Some` ONLY between the start and the end of - /// `execute_transaction_batch`, which is what makes the same write handler - /// serve both scopes: with a batch open, a handler's entries accumulate - /// here and the batch checks them once at its commit boundary; with none - /// open the handler is its own boundary and checks its entries itself. An - /// explicit transaction may legitimately write one leg of a journal per - /// statement, so a per-statement check inside a batch would refuse writes - /// the constraint permits. - pub(in crate::data::executor) balanced_txn_entries: Option< - Vec<( - String, - crate::data::executor::enforcement::balanced::BalancedEntry, - )>, - >, - - /// Staged Calvin write plans awaiting the local commit verdict, keyed by - /// `(epoch, position, vshard)`. `CalvinExecuteStatic` validates a - /// transaction and inserts its plans here WITHOUT mutating base; the - /// verdict-driven `CalvinFlush` replays them through - /// `execute_transaction_batch` to make them durable, and `CalvinDrop` - /// discards them. Nothing staged here is observable in the base engines - /// until a flush. - /// - /// The vShard is part of the key because vShards round-robin onto cores: - /// several vShard slices of one multi-participant transaction — which share - /// the same `(epoch, position)` — can land on the same core, and each must - /// stage and flush its own slice independently rather than clobber a peer's. - pub(in crate::data::executor) commit_pending: - HashMap<(u64, u32, u32), super::commit_pending::PendingCommit>, - - /// `Some((epoch, position, vshard))` only during a distributed Calvin flush apply (staging gate). - pub(in crate::data::executor) calvin_flush_key: Option<(u64, u32, u32)>, - - /// Per-`(epoch, position, vshard)` index-value tuples drained from a Calvin - /// flush's undo log, awaiting the post-apply `RecordCalvinWriteVersions` op. - pub(in crate::data::executor) calvin_flush_index_tuples: - crate::data::executor::handlers::transaction::index_write_values::StagedCalvinIndexTuples, + /// Staged Calvin transactions, the writes waiting on them, and the + /// executing transaction's leader flag. + pub(in crate::data::executor) calvin: super::calvin_state::CalvinCoreState, /// Core count and per-record scratch of the committed-redo apply. pub(in crate::data::executor) redo_apply: diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index af690811c..229f4474e 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -57,8 +57,17 @@ impl CoreLoop { }; self.io_metrics.record_wait(tier, wait_ns); - let mut task = qt.task; + // A write to a row a staged Calvin transaction owns waits for it. + if let Some(task) = self.park_if_calvin_owned(qt.task) { + self.run_task(task); + } + self.release_resolved_calvin_owners(); + true + } + /// Execute `task` and send its response: an idempotent replay answers + /// from the cache, and an expired task answers that it never started. + pub(in crate::data::executor) fn run_task(&mut self, mut task: ExecutionTask) { if let Some(key) = task.request.idempotency_key && let Some(&succeeded) = self.idempotency_cache.get(&key) { @@ -73,7 +82,7 @@ impl CoreLoop { { warn!(core = self.core_id, error = %e, "failed to send idempotent response"); } - return true; + return; } let response = if task.is_expired() { @@ -127,8 +136,6 @@ impl CoreLoop { { warn!(core = self.core_id, error = %e, "failed to send response — response queue full"); } - - true } /// Run one iteration of the event loop: drain requests, process tasks. @@ -141,6 +148,7 @@ impl CoreLoop { // Adjust SPSC read depth based on current memory pressure. self.apply_spsc_pressure(); self.drain_requests(); + self.expire_calvin_parked(); let mut processed = 0; while !self.task_queue.is_empty() { // A fail-stopped core serves nothing, including the rest of the diff --git a/nodedb/src/data/executor/core_loop/write_index.rs b/nodedb/src/data/executor/core_loop/write_index.rs index 558925449..98da25d60 100644 --- a/nodedb/src/data/executor/core_loop/write_index.rs +++ b/nodedb/src/data/executor/core_loop/write_index.rs @@ -246,6 +246,10 @@ impl CoreLoop { /// the write task. `key` is `None` for engines whose per-key identity is /// internal (columnar / timeseries / array / spatial / FTS) — those record /// only the collection floor. + /// + /// In the install pass of a committed-redo apply the version waits in the + /// apply scope. The apply publishes it once the record settled, so a + /// rolled-back install leaves no version and no watermark behind. pub(in crate::data::executor) fn note_write_lsn( &mut self, db: DatabaseId, @@ -253,6 +257,22 @@ impl CoreLoop { collection: &str, key: Option, lsn: Lsn, + ) { + if let Some(scope) = self.redo_apply.scope.as_mut() { + scope.defer_write_version(db, tenant, collection, key, lsn); + return; + } + self.publish_write_version(db, tenant, collection, key, lsn); + } + + /// Record a write version and advance the core watermark monotonically. + pub(in crate::data::executor) fn publish_write_version( + &mut self, + db: DatabaseId, + tenant: TenantId, + collection: &str, + key: Option, + lsn: Lsn, ) { self.write_index .note_write_lsn(db, tenant, collection, key, lsn); @@ -949,7 +969,7 @@ pub(crate) mod tests { } #[test] - fn transaction_batch_records_sub_plan_versions() { + fn a_committed_transaction_records_its_write_versions() { let (mut core, _, _, _dir) = make_core(); let task = wal_task(60); let plans = vec![PhysicalPlan::Document(DocumentOp::PointPut { @@ -962,8 +982,8 @@ pub(crate) mod tests { rls_filters: Vec::new(), resolved_sum_targets: Vec::new(), })]; - let resp = core.execute_transaction_batch(&task, 1, &plans, &[], None); - assert_eq!(resp.status, Status::Ok); + let resp = core.commit_plans_for_test(&task, 1, &plans, 60); + assert_eq!(resp.status, Status::Ok, "{:?}", resp.error_code); assert_eq!( core.write_index.key_write_lsn(&surrogate_key("batch", 11)), @@ -1202,12 +1222,11 @@ pub(crate) mod tests { } #[test] - fn conflicting_read_set_is_flagged_invalid_but_batch_still_applies() { + fn a_read_that_predates_a_committed_write_is_no_longer_current() { let (mut core, _, _, _dir) = make_core(); let vshard = local_vshard("orders"); - // First batch: a write to key 7 in "orders", recording its version at - // LSN 10 (this is the same chokepoint a Calvin apply funnels through). + // A committed write to key 7 in "orders" records its version at LSN 10. let write_task = wal_task_with_vshard(10, vshard); let write_plans = vec![PhysicalPlan::Document(DocumentOp::PointPut { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), @@ -1219,50 +1238,13 @@ pub(crate) mod tests { rls_filters: Vec::new(), resolved_sum_targets: Vec::new(), })]; - let write_resp = core.execute_transaction_batch(&write_task, 1, &write_plans, &[], None); - assert_eq!(write_resp.status, Status::Ok); - assert_eq!( - write_resp.read_set_valid, - Some(true), - "empty read-set is vacuously current" - ); - - // Second batch carries a synthetic read-set observing key 7 BEFORE the - // write above (read_lsn = 5 < the recorded write's LSN 10), alongside its - // own unrelated write. Proves: (a) the first batch's write really was - // recorded into the version index (without it this would false-report - // valid), and (b) an invalid read-set does not block the batch's own - // apply (non-enforcing). - let second_task = wal_task_with_vshard(20, vshard); - let second_plans = vec![PhysicalPlan::Document(DocumentOp::PointPut { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), - document_id: "o8".into(), - value: doc_value("a", "2"), - surrogate: Surrogate::new(8), - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - })]; - let stale_reads = vec![point_entry("orders", 7, 5)]; - let second_resp = - core.execute_transaction_batch(&second_task, 1, &second_plans, &stale_reads, None); - - assert_eq!( - second_resp.status, - Status::Ok, - "apply proceeds regardless of the read-set validation outcome" - ); - assert_eq!( - second_resp.read_set_valid, - Some(false), - "stale read against the recorded write must be detected as no longer current" - ); - - // The second batch's own write still landed despite the invalid read-set. - assert_eq!( - core.write_index.key_write_lsn(&surrogate_key("orders", 8)), - Some(Lsn::new(20)) - ); + let write_resp = core.commit_plans_for_test(&write_task, 1, &write_plans, 10); + assert_eq!(write_resp.status, Status::Ok, "{:?}", write_resp.error_code); + + // A read of key 7 observed at LSN 5 predates the write; one observed + // at LSN 10 does not. + let task = task_with_vshard(vshard); + assert!(!core.read_set_still_current(&task, 1, &[point_entry("orders", 7, 5)])); + assert!(core.read_set_still_current(&task, 1, &[point_entry("orders", 7, 10)])); } } diff --git a/nodedb/src/data/executor/dispatch/meta.rs b/nodedb/src/data/executor/dispatch/meta.rs index e47b32d62..fc7151019 100644 --- a/nodedb/src/data/executor/dispatch/meta.rs +++ b/nodedb/src/data/executor/dispatch/meta.rs @@ -2,7 +2,7 @@ //! Dispatch for MetaOp variants (WAL, snapshots, retention, continuous aggregates). -use crate::bridge::envelope::Response; +use crate::bridge::envelope::{ErrorCode, Response}; use nodedb_physical::physical_plan::{MetaOp, SAVEPOINT_MARKER_BYTES}; use crate::data::executor::core_loop::CoreLoop; @@ -22,9 +22,16 @@ impl CoreLoop { MetaOp::Cancel { target_request_id } => self.execute_cancel(task, *target_request_id), - MetaOp::TransactionBatch { plans, txn_id } => { - self.execute_transaction_batch(task, tid, plans, &[], *txn_id) - } + // A committed transaction installs only through its redo record + // (`ApplyTransactionRedo`, `CalvinFlush`). A plan batch carries no + // record restart replay reads, so an Origin core refuses it. + MetaOp::TransactionBatch { .. } => self.response_error( + task, + ErrorCode::Unsupported { + detail: "a transaction batch installs only through a committed redo record" + .into(), + }, + ), MetaOp::CreateSnapshot => self.execute_create_snapshot(task), MetaOp::Compact => self.execute_compact(task), @@ -249,36 +256,32 @@ impl CoreLoop { }, ), - MetaOp::RecordCalvinWriteVersions { - tenant_id, - plans, - epoch, - position, - } => { + MetaOp::RecordCalvinWriteVersions { tenant_id, plans } => { // The Calvin apply already committed; this records the write - // version of every key it wrote at the CalvinApplied WAL LSN the - // scheduler threaded onto the request envelope, reusing the same - // recorder the single-shard fast-path commit funnels through. A - // no-op when the envelope carries no LSN. + // version of every key it wrote at the applied WAL LSN the + // scheduler threaded onto the request envelope. A no-op when + // the envelope carries no LSN. The install recorded the + // index-value versions of its document rows itself. self.record_batch_write_versions(task, tenant_id.as_u64(), plans); - // Drain the per-index value tuples the distributed flush staged - // for this batch and record them at the same applied LSN. - if let Some(lsn) = task.wal_lsn() { - self.record_staged_calvin_index_values( - task.request.database_id, - *tenant_id, - *epoch, - *position, - task.request.vshard_id.as_u32(), - lsn, - ); - } self.response_ok(task) } - MetaOp::CalvinFlush { epoch, position } => { - self.execute_calvin_flush(task, *epoch, *position) - } + MetaOp::CalvinFlush { + epoch, + position, + redo, + collections, + sum_targets, + } => self.execute_calvin_flush( + task, + crate::data::executor::handlers::control::calvin::CalvinFlushRedo { + epoch: *epoch, + position: *position, + redo, + collections, + sum_targets, + }, + ), MetaOp::CalvinDrop { epoch, position } => { self.execute_calvin_drop(task, *epoch, *position) @@ -328,10 +331,9 @@ impl CoreLoop { // dropped here too, but ONLY if still empty. On ROLLBACK the // staged rows never left the overlay, so the engine's memtable is // still empty and gets dropped -- no phantom empty engine survives - // the rollback. On COMMIT, `TransactionBatch` has already replayed - // the insert through `execute_columnar_insert` (populating the - // memtable) before this dispatches, so the empty-check fails and - // the engine correctly stays registered with its committed rows. + // the rollback. On COMMIT, the redo install has already written + // the committed rows into the memtable before this dispatches, so + // the empty-check fails and the engine stays registered with them. MetaOp::DropTxnOverlay { txn_id } => { // Behaviour-preserving delegation to the shared teardown, which // the lease reaper also calls (see `CoreLoop::drop_overlay_entry`). @@ -402,7 +404,7 @@ mod txn_created_columnar_engine_tests { //! per-txn overlay for the engine's memtable) — but ONLY while that engine //! is still empty. On ROLLBACK the memtable is empty, so the phantom engine //! is dropped; on COMMIT the memtable has already been populated by the - //! `TransactionBatch` replay, so the engine (and its rows) survive. + //! redo install, so the engine (and its rows) survive. //! //! Observed directly on `CoreLoop::columnar_engines` membership — the field //! the fix mutates — because a leaked empty engine is invisible to ordinary @@ -579,8 +581,8 @@ mod txn_created_columnar_engine_tests { let key = stage_new_collection(&mut core, &task, txn_id, "committed"); assert!(core.columnar_engines.contains_key(&key)); - // Mimic COMMIT: the `TransactionBatch` replay applies the buffered - // insert to the engine's memtable BEFORE DropTxnOverlay dispatches. + // Mimic COMMIT: the redo install writes the committed row into the + // engine's memtable BEFORE DropTxnOverlay dispatches. core.columnar_engines .get_mut(&key) .expect("engine present") diff --git a/nodedb/src/data/executor/enforcement/hash_chain.rs b/nodedb/src/data/executor/enforcement/hash_chain.rs index c6889a40a..3c2dc702f 100644 --- a/nodedb/src/data/executor/enforcement/hash_chain.rs +++ b/nodedb/src/data/executor/enforcement/hash_chain.rs @@ -285,12 +285,11 @@ mod tests { core.doc_configs.insert(config_key.clone(), config); for (surrogate, doc_id, value) in rows.iter().take(2) { - let resp = core.execute_transaction_batch( + let resp = core.commit_plans_for_test( &task, TID, &[put_plan(*surrogate, doc_id, value)], - &[], - None, + 10 + u64::from(*surrogate), ); assert_eq!(resp.status, Status::Ok, "pre-restart insert must succeed"); } @@ -318,12 +317,11 @@ mod tests { core.doc_configs.insert(config_key.clone(), config); let (surrogate, doc_id, value) = &rows[2]; - let resp = core.execute_transaction_batch( + let resp = core.commit_plans_for_test( &task, TID, &[put_plan(*surrogate, doc_id, value)], - &[], - None, + 10 + u64::from(*surrogate), ); assert_eq!(resp.status, Status::Ok, "post-restart insert must succeed"); diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs index 77d8b3f83..f065404ee 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs @@ -1529,12 +1529,10 @@ mod tests { /// The CALVIN apply path honours a deferral, and it is the only path that ever /// sees one. /// - /// `execute_calvin_flush` replays every staged plan through - /// `execute_transaction_batch`, which intercepts `PointInsert` for undo - /// tracking instead of re-dispatching it. That interception must forward both - /// `resolved_sum_targets` and `deferred_sum_targets`; dropping the latter lets - /// the source core fold a balance the Control Plane already shipped on its own - /// `ApplyBalanceDelta` task, moving the total twice. + /// A committed Calvin transaction installs its redo record with the + /// transaction's resolved and deferred sum targets. Dropping the deferral + /// lets the source core fold a balance the Control Plane already shipped on + /// its own `ApplyBalanceDelta` task, moving the total twice. /// /// A dropped marker is invisible under two compounding conditions: a deferral /// is only ever set on a CROSS-SHARD statement, and a cross-shard statement @@ -1542,11 +1540,8 @@ mod tests { /// carries the marker, and never on a write that could notice. The /// direct-dispatch path forwards it correctly, which is why a regression here /// leaves every single-shard test green. - /// - /// Asserted through `execute_transaction_batch` rather than through the funnel: - /// the funnel was never where the field was lost. #[test] - fn the_transaction_batch_path_honours_a_deferred_binding() { + fn the_calvin_path_honours_a_deferred_binding() { let dir = tempfile::tempdir().expect("tempdir"); let (mut core, _req, _resp) = make_core_with_dir(dir.path()); register_collections_onto(&mut core, REMOTE_TARGET); @@ -1576,7 +1571,7 @@ mod tests { )]; let task = make_default_task(); - let response = core.execute_transaction_batch(&task, TID, &plans, &[], None); + let response = core.calvin_commit_for_test(&task, TID, &plans, 1, 100); assert_eq!( response.status, Status::Ok, @@ -1606,12 +1601,12 @@ mod tests { /// The same path, with the deferral ABSENT, must still apply the balance. /// - /// Without this the fix above could be "never fold on the transaction batch - /// path", which would silently drop every CO-RESIDENT balance committed through + /// Without this the fix above could be "never fold on the Calvin path", + /// which would silently drop every CO-RESIDENT balance committed through /// Calvin — a wrong total in the other direction, and one no cross-shard test /// would catch. #[test] - fn the_transaction_batch_path_still_folds_an_undeferred_binding() { + fn the_calvin_path_still_folds_an_undeferred_binding() { let dir = tempfile::tempdir().expect("tempdir"); let (mut core, _req, _resp) = make_core_with_dir(dir.path()); register_collections(&mut core); @@ -1634,7 +1629,7 @@ mod tests { )]; let task = make_default_task(); - let response = core.execute_transaction_batch(&task, TID, &plans, &[], None); + let response = core.calvin_commit_for_test(&task, TID, &plans, 2, 101); assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); assert_eq!( balance_of(&core, SURROGATE_A), @@ -1643,25 +1638,19 @@ mod tests { ); } - /// A source write replayed on the transaction-batch path reports its - /// affected-row count. + /// A source write committed through Calvin reports its affected-row count. /// - /// `execute_calvin_flush` replays a participant's staged plans through - /// `execute_transaction_batch` and returns the LAST sub-plan's payload as that - /// participant's applied response. The scheduler deposits it, and the - /// coordinator shapes the statement's `INSERT ` tag from it — so a bare `Ok` - /// from the sub-plan leaves an autocommit CROSS-SHARD insert with no count at - /// all and the statement fails with "write response carried no affected-row - /// count". + /// The flush answers with the reply the transaction's last plan computed + /// when it staged. The scheduler deposits it, and the coordinator shapes + /// the statement's `INSERT ` tag from it — so a bare `Ok` leaves an + /// autocommit CROSS-SHARD insert with no count at all and the statement + /// fails with "write response carried no affected-row count". /// - /// It stayed hidden because the other consumer of this payload is the - /// single-shard COMMIT flush, whose tag is `COMMIT` and which discards the - /// count entirely. The cross-shard materialized-sum pair is the first shape - /// that makes a ONE-STATEMENT autocommit write commit through Calvin, and it - /// only appeared to work while the sibling balance participant — whose handler - /// does report a count — happened to win the race to deposit first. + /// The single-shard COMMIT answers with the `COMMIT` tag and discards the + /// count. The cross-shard materialized-sum pair is the shape that makes a + /// ONE-STATEMENT autocommit write commit through Calvin. #[test] - fn the_transaction_batch_path_reports_its_affected_count() { + fn the_calvin_path_reports_its_affected_count() { use crate::control::server::shared::sql::staging_predicates::require_affected_count; let dir = tempfile::tempdir().expect("tempdir"); @@ -1686,13 +1675,13 @@ mod tests { )]; let task = make_default_task(); - let response = core.execute_transaction_batch(&task, TID, &plans, &[], None); + let response = core.calvin_commit_for_test(&task, TID, &plans, 3, 102); assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); assert_eq!( require_affected_count(response.payload.as_bytes()) - .expect("a batch whose last sub-plan renders an INSERT tag must carry its count"), + .expect("a flush whose last plan renders an INSERT tag must carry its count"), 1, - "one row was inserted, so the batch's applied response reports one" + "one row was inserted, so the flush reports one" ); } } diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs b/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs index e8f8136e2..9115eb9dd 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs @@ -78,7 +78,7 @@ impl CoreLoop { check: &SumTargetCheck<'_>, rows: &[serde_json::Value], ) -> bool { - if !self.ollp_is_group_leader { + if !self.calvin.ollp_is_group_leader { return false; } let key = ( @@ -147,7 +147,7 @@ impl CoreLoop { check: &SumTargetCheck<'_>, doc_ids: &[nodedb_types::StorageKey], ) -> bool { - if !self.ollp_is_group_leader || !self.declares_materialized_sums(check) { + if !self.calvin.ollp_is_group_leader || !self.declares_materialized_sums(check) { return false; } let mut rows: Vec = Vec::with_capacity(doc_ids.len()); diff --git a/nodedb/src/data/executor/enforcement/statement.rs b/nodedb/src/data/executor/enforcement/statement.rs index 1b1858f20..763fb5cd5 100644 --- a/nodedb/src/data/executor/enforcement/statement.rs +++ b/nodedb/src/data/executor/enforcement/statement.rs @@ -10,11 +10,10 @@ //! definition and is refused. That is the point of the constraint: a balanced //! ledger cannot be populated one leg at a time. //! -//! Which boundary a write belongs to is decided by -//! [`CoreLoop::balanced_txn_entries`]: with a transaction batch open the -//! entries accumulate onto it and the batch checks them once at commit; with -//! none open the statement is its own boundary and checks immediately, before -//! it commits, so a refusal writes nothing. +//! An autocommit statement checks its entries immediately, before it commits, +//! so a refusal writes nothing. A committed transaction's redo install checks +//! the entries of every write in the record at its own commit boundary +//! (`redo_apply::validate`). //! //! # Why some callers pre-compute their entries here instead of collecting the //! # funnel's @@ -54,25 +53,15 @@ impl CoreLoop { /// Account for the entries one write boundary produced. /// - /// Inside a transaction batch the entries are accumulated onto the batch, - /// which checks them once at its own commit boundary. Outside one, the - /// caller IS the boundary and the check runs here — the caller must not - /// have committed yet, so an `Err` leaves nothing written. + /// The caller IS the boundary and the check runs here — the caller must + /// not have committed yet, so an `Err` leaves nothing written. pub(in crate::data::executor) fn settle_balanced_entries( - &mut self, + &self, database_id: u64, tid: u64, collection: &str, entries: Vec, ) -> crate::Result<()> { - if let Some(pending) = self.balanced_txn_entries.as_mut() { - pending.extend( - entries - .into_iter() - .map(|entry| (collection.to_string(), entry)), - ); - return Ok(()); - } if entries.is_empty() { return Ok(()); } diff --git a/nodedb/src/data/executor/handlers/bulk_dml/admission.rs b/nodedb/src/data/executor/handlers/bulk_dml/admission.rs index 3980f0cb2..9e5a36920 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/admission.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/admission.rs @@ -59,7 +59,7 @@ impl CoreLoop { let apply_ids: Vec = match admission.predicted_surrogates { Some(predicted) => { // The set comparison is deterministic: both sides are sorted. - if self.ollp_is_group_leader + if self.calvin.ollp_is_group_leader && !super::scan::ollp_surrogates_match(&matching_ids, predicted) { return Err(ErrorCode::OllpRetryRequired); @@ -78,7 +78,7 @@ impl CoreLoop { // unchanged. This runs BEFORE any write, so `sparse.get` still returns // pre-mutation content. if let Some(predicted) = admission.predicted_edges - && self.ollp_is_group_leader + && self.calvin.ollp_is_group_leader { let actual = self.ollp_actual_edges(database_id, tid, admission.collection, &apply_ids); if !super::scan::ollp_edges_match(actual, predicted) { diff --git a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs index 65ae806d4..25ccf2720 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/delete_cascade.rs @@ -103,14 +103,14 @@ impl CoreLoop { crate::diag::orphaned_index_entry_after_delete(&e, collection, "secondary"); warn!(core = self.core_id, %collection, %doc_id, error = %e, "bulk delete: secondary index cascade failed"); } - // Cascade: graph edges. + // Cascade: graph edges. The graph keys a row's node by its client key. // On an error neither edge store changed: the edges stay in both, // and the dangling-edge sweep retries them. - if let Err(e) = self.cascade_node_edges(database_id, tid, doc_id) { + if let Err(e) = self.cascade_node_edges(database_id, tid, row_identity.as_str()) { crate::diag::orphaned_index_entry_after_delete(&e, collection, "graph_edge"); warn!(core = self.core_id, %doc_id, error = %e, "bulk delete: edge cascade failed"); } - self.mark_node_deleted(database_id, tid, doc_id); + self.mark_node_deleted(database_id, tid, row_identity.as_str()); // Cascade: secondary HNSW vector index. The put path indexed // this row's vectors under its surrogate; the delete must // soft-delete those nodes and drop the reverse-map entry, or the diff --git a/nodedb/src/data/executor/handlers/compact/budget.rs b/nodedb/src/data/executor/handlers/compact/budget.rs index e30813820..76fdb64d2 100644 --- a/nodedb/src/data/executor/handlers/compact/budget.rs +++ b/nodedb/src/data/executor/handlers/compact/budget.rs @@ -40,7 +40,7 @@ impl CoreLoop { if force { return BudgetGate::Granted(None); } - match self.maintenance_budget.as_ref() { + match self.maintenance.maintenance_budget.as_ref() { None => BudgetGate::Granted(None), Some(tracker) => match tracker.try_acquire(db, 0.0) { Some(lease) => BudgetGate::Granted(Some(lease)), diff --git a/nodedb/src/data/executor/handlers/compact/maintenance.rs b/nodedb/src/data/executor/handlers/compact/maintenance.rs index b111adfca..bcebf168f 100644 --- a/nodedb/src/data/executor/handlers/compact/maintenance.rs +++ b/nodedb/src/data/executor/handlers/compact/maintenance.rs @@ -200,12 +200,12 @@ impl CoreLoop { // Compaction: periodic tombstone removal + segment merge. let now = std::time::Instant::now(); - if let Some(last) = self.last_maintenance - && now.duration_since(last) < self.compaction_interval + if let Some(last) = self.maintenance.last_maintenance + && now.duration_since(last) < self.maintenance.compaction_interval { return !flush_plan.is_empty(); } - self.last_maintenance = Some(now); + self.maintenance.last_maintenance = Some(now); // Horizon-GC the per-core last-write-LSN version index: evict entries // far below the watermark and enforce the entry-count backstop. Rides // the compaction interval — no dedicated timer. diff --git a/nodedb/src/data/executor/handlers/compact/runner.rs b/nodedb/src/data/executor/handlers/compact/runner.rs index b382243b2..0cf9f40cc 100644 --- a/nodedb/src/data/executor/handlers/compact/runner.rs +++ b/nodedb/src/data/executor/handlers/compact/runner.rs @@ -99,7 +99,7 @@ impl CoreLoop { 0.0 }; - if !force && ratio < self.compaction_tombstone_threshold { + if !force && ratio < self.maintenance.compaction_tombstone_threshold { continue; } diff --git a/nodedb/src/data/executor/handlers/compact/segments.rs b/nodedb/src/data/executor/handlers/compact/segments.rs index 96f08b86d..58fb4305a 100644 --- a/nodedb/src/data/executor/handlers/compact/segments.rs +++ b/nodedb/src/data/executor/handlers/compact/segments.rs @@ -28,7 +28,10 @@ impl CoreLoop { .map(|d| d.as_millis() as i64) .unwrap_or(0); - let max_per_pass = self.segment_compaction_config.max_segments_per_pass; + let max_per_pass = self + .maintenance + .segment_compaction_config + .max_segments_per_pass; let mut total_merged = 0usize; let mut total_deferred = 0usize; @@ -46,7 +49,7 @@ impl CoreLoop { }) .collect(); - let budget = self.maintenance_budget.clone(); + let budget = self.maintenance.maintenance_budget.clone(); let core_id = self.core_id; for ((db, tid, collection), registry) in &mut self.ts_registries { diff --git a/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs b/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs index f08eb869a..c982cd7a0 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/active_passive.rs @@ -3,6 +3,7 @@ //! Passive and active participants for a dependent-read Calvin transaction. use std::collections::BTreeMap; +use std::panic::{AssertUnwindSafe, catch_unwind}; use tracing::{debug, info_span}; @@ -12,6 +13,7 @@ use nodedb_types::Value; use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::commit_pending::PendingCommit; +use crate::data::executor::handlers::control::calvin_reply::CalvinReply; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; use crate::types::TenantId; @@ -19,6 +21,7 @@ use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_plan::meta::PassiveReadKeyId; use crate::data::executor::handlers::control::calvin_txn_id::calvin_synthetic_txn_id; +use crate::data::panic_payload::panic_payload_to_string; use super::shared::CalvinExecCtx; @@ -74,27 +77,22 @@ impl CoreLoop { /// Stage an active-participant dependent-read Calvin txn for commit. /// /// Mirrors [`CoreLoop::execute_calvin_execute_static`]: it performs NO base - /// mutation and fires NO side effects — it buffers the write plans in - /// `commit_pending` and stages each into `txn_overlays` under the synthetic - /// `TxnId`, so a subsequent `CalvinResolve` reconstitutes them as one - /// replayable `RedoRecord` and [`CoreLoop::execute_calvin_flush`] applies - /// them. This restores WAL-only-restart durability for the dependent-read - /// path, which previously applied directly with `wal_lsn: None` (only a - /// non-replayable `CalvinApplied` marker survived). + /// mutation and fires NO side effects. It stages each write plan into + /// `txn_overlays` under the synthetic `TxnId` and buffers the plans in + /// `commit_pending`, so a subsequent `CalvinResolve` builds one redo record + /// and [`CoreLoop::execute_calvin_flush`] installs it. /// /// The one divergence from the static path: OLLP predicate verification /// (leader-only) runs HERE, before staging, via /// [`CoreLoop::verify_calvin_active_ollp`]. The dependent-read path has no /// LSN-versioned read-set to vote on; its conflict detector is the OLLP - /// `actual != predicted` re-check. Running it at stage time (not flush) - /// ensures a mismatch returns `OllpRetryRequired` and stages nothing — - /// otherwise a stale redo would be WAL-appended before the flush-time check - /// (whose retry signal is swallowed as a degraded shard). The Control Plane - /// scheduler releases locks and re-recons on `OllpRetryRequired`. + /// `actual != predicted` re-check. A mismatch returns `OllpRetryRequired` + /// and stages nothing, so no redo record is appended for it. The Control + /// Plane scheduler releases locks and re-recons on `OllpRetryRequired`. /// /// `injected_reads` is retained on the wire for future plan variants that - /// reference resolved read values by `PassiveReadKeyId`; in v1 the - /// coordinator baked the read values into concrete point ops / the predicted + /// reference resolved read values by `PassiveReadKeyId`. The coordinator + /// bakes the read values into concrete point ops and the predicted /// surrogate set at recon, so the plans are self-contained and stage /// byte-identically to the static path. pub(in crate::data::executor) fn execute_calvin_execute_active( @@ -137,48 +135,65 @@ impl CoreLoop { // mismatch surfaces on THIS stage response (where the scheduler releases // locks and re-recons) and nothing is staged, resolved, or WAL-appended. // Scoped to this replica's staged leadership for the check, then the - // resting (authoritative) state is restored. A read-only scan needs no - // time anchor, so `epoch_system_ms`/`hlc` stay unset until flush - // (mirroring the static path, where they ride `PendingCommit`). - let prev_group_leader = self.ollp_is_group_leader; - self.ollp_is_group_leader = is_group_leader; + // resting (authoritative) state is restored. + let prev_group_leader = self.calvin.ollp_is_group_leader; + self.calvin.ollp_is_group_leader = is_group_leader; let verified = self.verify_calvin_active_ollp(task, tenant_id.as_u64(), plans); - self.ollp_is_group_leader = prev_group_leader; + self.calvin.ollp_is_group_leader = prev_group_leader; match verified { Ok(true) => {} Ok(false) => return self.response_error(task, ErrorCode::OllpRetryRequired), Err(e) => return self.response_error(task, e), } - // Stage exactly like `execute_calvin_execute_static`: buffer the plans in - // `commit_pending` (the sole durable apply the flush replays) and stage - // each write into `txn_overlays` under the synthetic `TxnId` (producer - // side for `CalvinResolve`). No base mutation, no side effects; the time - // anchor + leadership scope captured here are restored at flush time. - self.commit_pending.insert( - (epoch, position, vshard_id), - PendingCommit { - plans: plans.to_vec(), - tenant_id: *tenant_id, - epoch_system_ms, - is_group_leader, - }, - ); let synthetic_txn_id = match calvin_synthetic_txn_id(epoch, position, vshard_id) { Ok(id) => id, - Err(e) => return self.response_error(task, e), + Err(e) => return self.calvin_stage_failure(task, epoch, position, vshard_id, e), }; // Staging reads the epoch's time anchor, so every replica stages the // same images. let prev_epoch_ms = self.epoch_system_ms; self.epoch_system_ms = Some(epoch_system_ms); - let staged = plans.iter().try_for_each(|plan| { - self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan) - }); + // A panic while staging drops the staged state like any refusal, + // instead of unwinding past a half-staged overlay. + let staged = catch_unwind(AssertUnwindSafe(|| { + let mut reply = CalvinReply::default(); + for plan in plans { + self.stage_calvin_plan(task, synthetic_txn_id, *tenant_id, plan, &mut reply)?; + } + Ok::(reply) + })); self.epoch_system_ms = prev_epoch_ms; - if let Err(e) = staged { - return self.response_error(task, e); - } + let reply = match staged { + Ok(Ok(reply)) => reply, + Ok(Err(e)) => return self.calvin_stage_failure(task, epoch, position, vshard_id, e), + Err(payload) => { + return self.calvin_stage_failure( + task, + epoch, + position, + vshard_id, + ErrorCode::Internal { + detail: format!( + "panic while staging active Calvin transaction: {}", + panic_payload_to_string(payload.as_ref()) + ), + }, + ); + } + }; + + // Publish only a fully staged transaction. Resolve reads these plans + // against the overlay; a global abort drops both. + self.calvin.commit_pending.insert( + (epoch, position, vshard_id), + PendingCommit { + plans: plans.to_vec(), + tenant_id: *tenant_id, + epoch_system_ms, + reply, + }, + ); Response { request_id: task.request_id(), @@ -212,12 +227,10 @@ mod tests { }; /// The dependent-read ACTIVE path STAGES its writes (into `commit_pending` + - /// the synthetic overlay) instead of applying them to base directly. This is - /// the direct regression guard for U-CAL5: before it, this handler called - /// `execute_transaction_batch` inline (`wal_lsn: None`), so a Calvin-committed - /// dependent-read write left only a non-replayable `CalvinApplied` marker and - /// was lost on a WAL-only restart. Staging routes it through the same - /// resolve → redo → flush the static path uses. + /// the synthetic overlay) instead of applying them to base directly, so a + /// Calvin-committed dependent-read write reaches the WAL as a redo record. + /// Staging routes it through the same resolve → redo → flush the static + /// path uses. #[test] fn calvin_execute_active_stages_point_insert_into_overlay() { let dir = tempfile::tempdir().unwrap(); @@ -244,7 +257,7 @@ mod tests { // STAGED, not applied: the plans are buffered for the flush replay. assert!( - core.commit_pending.contains_key(&(1, 0, vshard_id)), + core.calvin.commit_pending.contains_key(&(1, 0, vshard_id)), "active-path write must be STAGED into commit_pending, not applied directly" ); @@ -307,7 +320,7 @@ mod tests { // Drift stages NOTHING — neither the raw buffer nor the overlay. let vshard_id = task.request.vshard_id.as_u32(); assert!( - !core.commit_pending.contains_key(&(1, 0, vshard_id)), + !core.calvin.commit_pending.contains_key(&(1, 0, vshard_id)), "an OLLP-drift retry must not leave a staged commit buffer" ); let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); diff --git a/nodedb/src/data/executor/handlers/control/calvin/discard.rs b/nodedb/src/data/executor/handlers/control/calvin/discard.rs index 420aaded3..e3df9fe9d 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/discard.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/discard.rs @@ -23,12 +23,17 @@ impl CoreLoop { ) -> Response { let vshard_id = task.request.vshard_id.as_u32(); let existed = self + .calvin .commit_pending .remove(&(epoch, position, vshard_id)) .is_some(); // Discard the synthetic overlay entry alongside the raw plan buffer; // idempotent no-op if it was never staged or already removed. self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); + // Writes waiting on the rows this transaction owned run next. + self.calvin + .fence + .note_resolved((epoch, position, vshard_id), None); debug!( core = self.core_id, epoch, position, vshard_id, existed, "calvin drop: discarding staged commit" diff --git a/nodedb/src/data/executor/handlers/control/calvin/flush.rs b/nodedb/src/data/executor/handlers/control/calvin/flush.rs index 84d154a15..5987e67d6 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/flush.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/flush.rs @@ -1,53 +1,67 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Flush a staged Calvin transaction to base storage. +//! Flush a staged Calvin transaction: install its committed redo record. +use nodedb_physical::physical_plan::RedoSumTargets; use tracing::{debug, info_span}; -use crate::bridge::envelope::Response; +use crate::bridge::envelope::{Payload, Response, Status}; use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::handlers::transaction::overlay::BitemporalStamp; +use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; use crate::data::executor::task::ExecutionTask; -use crate::data::executor::handlers::control::calvin_txn_id::calvin_synthetic_txn_id; +/// The `MetaOp::CalvinFlush` fields the flush reads. +pub(in crate::data::executor) struct CalvinFlushRedo<'a> { + pub epoch: u64, + pub position: u32, + /// The encoded `RedoRecord` the scheduler appended at the request's LSN. + /// Empty when the transaction wrote nothing on this vShard. + pub redo: &'a [u8], + /// Every collection the local slice of the transaction writes. + pub collections: &'a [String], + /// The materialized-sum targets the local slice folds into. + pub sum_targets: &'a [RedoSumTargets], +} impl CoreLoop { - /// Flush a staged Calvin transaction to base storage. + /// Flush the Calvin transaction staged under `(epoch, position)`. + /// + /// The flush installs the transaction's committed redo record through + /// [`CoreLoop::install_committed_redo`], the install every committed + /// transaction and restart replay run: validate, install with undo, then + /// settle and cover. An absent staged entry means a duplicate dispatch, + /// which answers `Ok` and installs nothing. /// - /// Pops the plans staged by [`CoreLoop::execute_calvin_execute_static`] - /// under `(epoch, position)` and replays them through the durable apply - /// funnel (`execute_transaction_batch`) — the same funnel the single-shard - /// commit and recovery use — so base mutation, side effects, and - /// version recording all run exactly once here. The deterministic epoch - /// time anchor and leadership scope captured at stage time are restored - /// around the apply so time-dependent writes stay identical across - /// replicas. An absent key (already flushed or dropped, e.g. a duplicate - /// dispatch) is an idempotent no-op returning `Ok`. + /// The staged entry and its overlay stay until the install succeeds. A + /// refused install rolled every write back, so the scheduler can send the + /// same flush again. + /// + /// The reply reads at the epoch instant, so every replica sees the same + /// live rows. The response carries the reply the plans decided when they + /// staged. A document or CRDT `RETURNING` row is read from base after the + /// install, so it reports the row the install stored. A reply that fails + /// to render leaves the flush `Ok`, since the record installed. Its error + /// travels in `error_code` for the statement to report. pub(in crate::data::executor) fn execute_calvin_flush( &mut self, task: &ExecutionTask, - epoch: u64, - position: u32, + flush: CalvinFlushRedo<'_>, ) -> Response { + let CalvinFlushRedo { + epoch, + position, + redo, + collections, + sum_targets, + } = flush; let vshard_id = task.request.vshard_id.as_u32(); - // Capture the resolve-time bitemporal stamps (if `CalvinResolve` staged - // them into the synthetic overlay) BEFORE the overlay is dropped, so the - // base install below reuses the exact stamp the redo carries rather than - // minting a fresh one. Empty when this transaction wrote no bitemporal - // document rows or resolve never ran. - let synthetic_txn_id = calvin_synthetic_txn_id(epoch, position, vshard_id).ok(); - let bitemporal_stamps: Vec<(u32, BitemporalStamp)> = synthetic_txn_id - .and_then(|synthetic| self.txn_overlays.get(&synthetic)) - .map(|overlay| overlay.all_bitemporal_stamps().collect()) - .unwrap_or_default(); - let graph_system_from = synthetic_txn_id - .and_then(|synthetic| self.graph_txn_overlays.get(&synthetic)) - .and_then(|overlay| overlay.resolved_system_from()); - // Drop the synthetic overlay entry staged by - // `execute_calvin_execute_static` unconditionally, before the apply - // below: idempotent no-op on a duplicate dispatch. - self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); - let Some(pending) = self.commit_pending.remove(&(epoch, position, vshard_id)) else { + let key = (epoch, position, vshard_id); + let Some((tenant_id, epoch_system_ms)) = self + .calvin + .commit_pending + .get(&key) + .map(|pending| (pending.tenant_id, pending.epoch_system_ms)) + else { debug!( core = self.core_id, epoch, position, vshard_id, "calvin flush: no staged commit (already resolved)" @@ -59,117 +73,394 @@ impl CoreLoop { epoch, position, vshard = vshard_id, - tenant_id = pending.tenant_id.as_u64(), + tenant_id = tenant_id.as_u64(), trace_id = ?task.request.trace_id, ) .entered(); const NANOS_PER_MS: i64 = 1_000_000; self.hlc - .update_from_remote(pending.epoch_system_ms.saturating_mul(NANOS_PER_MS)); - self.epoch_system_ms = Some(pending.epoch_system_ms); - // Scope OLLP verification to this participant's staged leadership for the - // batch, then restore the resting (authoritative) state. - let prev_group_leader = self.ollp_is_group_leader; - self.ollp_is_group_leader = pending.is_group_leader; - // Install the captured resolve-time stamps into apply scratch; the - // batch consumes them for its bitemporal document puts and clears the - // scratch when it returns. `txn_id = None`: the synthetic overlay was - // already dropped above, so the stamps are threaded in directly here. - for (surrogate, stamp) in bitemporal_stamps { - self.active_bitemporal_stamps.insert(surrogate, stamp); + .update_from_remote(epoch_system_ms.saturating_mul(NANOS_PER_MS)); + + let mut response = if redo.is_empty() { + self.response_ok(task) + } else { + self.install_committed_redo( + task, + tenant_id.as_u64(), + CommittedRedo { + redo, + collections, + sum_targets, + }, + ) + }; + if response.status != Status::Ok { + return response; } - self.active_graph_system_from = graph_system_from; - // The read-set was already validated at stage time and drives the - // flush/drop decision; the replay itself carries no read-set to re-check. - // Scope the flush key so `record_batch_index_write_values` stages this - // batch's index tuples (the apply carries `wal_lsn: None`); the post-apply - // `RecordCalvinWriteVersions` op drains them at the replicated applied LSN. - self.calvin_flush_key = Some((epoch, position, vshard_id)); - let result = self.execute_transaction_batch( - task, - pending.tenant_id.as_u64(), - &pending.plans, - &[], - None, - ); - self.calvin_flush_key = None; - self.ollp_is_group_leader = prev_group_leader; - self.epoch_system_ms = None; - result + // The record installed: the staged entry and its overlay are spent. + let reply = self + .calvin + .commit_pending + .remove(&key) + .map(|pending| pending.reply) + .unwrap_or_default(); + self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); + // Writes waiting on the rows this transaction owned run next. + self.calvin.fence.note_resolved(key, task.wal_lsn()); + let prev_epoch_ms = self.epoch_system_ms; + self.epoch_system_ms = Some(epoch_system_ms); + let rendered = self.calvin_reply_payload(task, tenant_id.as_u64(), reply); + self.epoch_system_ms = prev_epoch_ms; + match rendered { + Ok(payload) => response.payload = Payload::from_vec(payload), + Err(error) => response.error_code = Some(Box::new(error)), + } + response } } #[cfg(test)] mod tests { use super::*; - use crate::bridge::envelope::Status; + use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::tests::make_core_with_dir; - use crate::types::TenantId; + use crate::data::executor::handlers::control::calvin_txn_id::calvin_synthetic_txn_id; + use crate::types::{DatabaseId, Lsn, TenantId}; + use nodedb_physical::physical_plan::{ + DocumentOp, PhysicalPlan, ReturningColumns, ReturningSpec, + }; + use nodedb_types::{QualifiedCollection, StorageKey, Surrogate}; + + use crate::engine::document::store::{CollectionConfig, IndexPath}; use super::super::shared::CalvinExecCtx; use super::super::shared::test_support::{ - bulk_delete_plan, make_task, point_insert_plan, seed_row, + bulk_delete_plan, doc_value, make_task, point_insert_plan, seed_row, }; + const CTX: CalvinExecCtx = CalvinExecCtx { + epoch: 1, + position: 0, + epoch_system_ms: 0, + is_group_leader: true, + }; + + /// Stage `plans` at `(1, 0)`, resolve them, and return the redo bytes. + fn stage_and_resolve(core: &mut CoreLoop, plans: &[PhysicalPlan]) -> Vec { + let task = make_task(); + let staged = core.execute_calvin_execute_static(&task, CTX, &TenantId::new(1), plans, &[]); + assert_eq!(staged.status, Status::Ok, "{:?}", staged.error_code); + let resolved = core.execute_calvin_resolve(&task, 1, 0); + assert_eq!(resolved.status, Status::Ok, "{:?}", resolved.error_code); + resolved.payload.as_bytes().to_vec() + } + + fn flush_at(core: &mut CoreLoop, lsn: u64, redo: &[u8], collections: &[String]) -> Response { + let mut task = make_task(); + task.wal_lsn = Some(Lsn::new(lsn)); + core.execute_calvin_flush( + &task, + CalvinFlushRedo { + epoch: 1, + position: 0, + redo, + collections, + sum_targets: &[], + }, + ) + } + + fn base_row(core: &CoreLoop, collection: &str, surrogate: u32) -> Option> { + core.sparse + .get( + DatabaseId::DEFAULT.as_u64(), + 1, + collection, + &StorageKey::for_surrogate(Surrogate::new(surrogate)), + ) + .expect("read base row") + } + #[test] fn calvin_flush_drops_synthetic_overlay() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let redo = stage_and_resolve(&mut core, &[point_insert_plan("orders", "o1", 7)]); - let task = make_task(); - let tenant_id = TenantId::new(1); - let plans = vec![point_insert_plan("orders", "o1", 7)]; - let ctx = CalvinExecCtx { - epoch: 1, - position: 0, - epoch_system_ms: 0, - is_group_leader: true, - }; - let resp = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); - assert_eq!(resp.status, Status::Ok); - - let vshard_id = task.request.vshard_id.as_u32(); + let vshard_id = make_task().request.vshard_id.as_u32(); let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); assert!(core.txn_overlays.contains_key(&synthetic)); - let flush_resp = core.execute_calvin_flush(&task, 1, 0); - assert_eq!(flush_resp.status, Status::Ok); + let flushed = flush_at(&mut core, 40, &redo, &["orders".to_string()]); + assert_eq!(flushed.status, Status::Ok, "{:?}", flushed.error_code); assert!( !core.txn_overlays.contains_key(&synthetic), "flush must drop the synthetic overlay entry alongside commit_pending" ); + assert!( + base_row(&core, "orders", 7).is_some(), + "the flush installs the staged row from the redo record" + ); } - /// A staged plan keeps its OLLP prediction until the flush replays it. - /// A row that joins the predicate between stage and flush makes the - /// leader's flush answer `OllpRetryRequired`, although the verdict - /// already committed the transaction. + /// The flush installs the redo record, not the staged plans. A row that + /// joins the bulk delete's predicate after staging stays, and the + /// predicted row goes. #[test] - fn a_staged_prediction_that_drifts_before_the_flush_answers_ollp_retry() { + fn a_flush_installs_the_resolved_record_not_a_replay_of_the_plans() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); seed_row(&mut core, "orders", 1); - - let task = make_task(); - let tenant_id = TenantId::new(1); - let plans = vec![bulk_delete_plan("orders", Some(vec![1]))]; - let ctx = CalvinExecCtx { - epoch: 1, - position: 0, - epoch_system_ms: 0, - is_group_leader: true, - }; - let staged = core.execute_calvin_execute_static(&task, ctx, &tenant_id, &plans, &[]); - assert_eq!(staged.status, Status::Ok, "{:?}", staged.error_code); + let redo = stage_and_resolve(&mut core, &[bulk_delete_plan("orders", Some(vec![1]))]); seed_row(&mut core, "orders", 2); - let flushed = core.execute_calvin_flush(&task, 1, 0); + let flushed = flush_at(&mut core, 41, &redo, &["orders".to_string()]); - assert_eq!(flushed.status, Status::Error); + assert_eq!(flushed.status, Status::Ok, "{:?}", flushed.error_code); + assert!(base_row(&core, "orders", 1).is_none()); + assert!(base_row(&core, "orders", 2).is_some()); + } + + /// The flush answers with the reply the staged statement computed: + /// RETURNING rows for a point insert that asked for them. + #[test] + fn a_calvin_flush_answers_the_returning_rows_its_statement_staged() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let plan = PhysicalPlan::Document(DocumentOp::PointInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), + document_id: "o9".to_string(), + value: doc_value("a", "returned"), + if_absent: false, + surrogate: Surrogate::new(9), + returning: Some(ReturningSpec { + columns: ReturningColumns::Star, + }), + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + let redo = stage_and_resolve(&mut core, &[plan]); + + let flushed = flush_at(&mut core, 42, &redo, &["orders".to_string()]); + + assert_eq!(flushed.status, Status::Ok, "{:?}", flushed.error_code); + let payload = flushed.payload.as_bytes(); + assert!( + payload.windows(b"returned".len()).any(|w| w == b"returned"), + "the flush reply carries the inserted row" + ); + assert!(base_row(&core, "orders", 9).is_some()); + } + + /// The install notes the index values of every document row at the + /// flush's LSN, so no flush-time staging carries them to a later op. + #[test] + fn a_calvin_flush_notes_index_values_at_the_redo_lsn() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let mut config = CollectionConfig::new("orders"); + config.index_paths.push(IndexPath::new("a")); + core.doc_configs.insert( + (DatabaseId::DEFAULT, TenantId::new(1), "orders".to_string()), + config, + ); + let redo = stage_and_resolve(&mut core, &[point_insert_plan("orders", "o1", 7)]); + + let flushed = flush_at(&mut core, 43, &redo, &["orders".to_string()]); + + assert_eq!(flushed.status, Status::Ok, "{:?}", flushed.error_code); assert_eq!( + core.write_index.index_values.value_lsn( + DatabaseId::DEFAULT, + TenantId::new(1), + "orders", + "a", + "1" + ), + Some(Lsn::new(43)), + "the install records the row's index value at the redo LSN" + ); + } + + /// A flush whose record fails to install answers the install's error and + /// leaves nothing of the record in base. + #[test] + fn a_calvin_flush_whose_record_cannot_install_answers_an_error() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + stage_and_resolve(&mut core, &[point_insert_plan("orders", "o1", 7)]); + + let flushed = flush_at(&mut core, 44, &[0xff, 0x00], &["orders".to_string()]); + + assert_eq!(flushed.status, Status::Error); + assert!(!matches!( flushed.error_code.as_deref(), - Some(&crate::bridge::envelope::ErrorCode::OllpRetryRequired) + Some(ErrorCode::OllpRetryRequired) + )); + assert!(base_row(&core, "orders", 7).is_none()); + } + + /// A refused install keeps the staged entry and its overlay, so the + /// scheduler's resent flush installs the record. + #[test] + fn a_refused_install_keeps_the_staged_commit_for_the_resent_flush() { + use crate::data::executor::handlers::transaction::redo_apply::test_commit::failing_install_sub_record; + use crate::wal::RedoRecord; + + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let redo = stage_and_resolve(&mut core, &[point_insert_plan("orders", "o1", 7)]); + let mut failing = RedoRecord::from_bytes(&redo).expect("decode redo"); + failing.ops.push(failing_install_sub_record()); + let failing = failing.to_bytes().expect("encode redo"); + let vshard_id = make_task().request.vshard_id.as_u32(); + let synthetic = calvin_synthetic_txn_id(1, 0, vshard_id).unwrap(); + + let refused = flush_at(&mut core, 45, &failing, &["orders".to_string()]); + + assert!( + matches!( + refused.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ), + "{:?}", + refused.error_code + ); + assert!(core.calvin.commit_pending.contains_key(&(1, 0, vshard_id))); + assert!(core.txn_overlays.contains_key(&synthetic)); + assert!(base_row(&core, "orders", 7).is_none()); + + let flushed = flush_at(&mut core, 46, &redo, &["orders".to_string()]); + + assert_eq!(flushed.status, Status::Ok, "{:?}", flushed.error_code); + assert!(base_row(&core, "orders", 7).is_some()); + assert!(!core.calvin.commit_pending.contains_key(&(1, 0, vshard_id))); + assert!(!core.txn_overlays.contains_key(&synthetic)); + } + + /// A reply that fails to render after a successful install leaves the + /// flush `Ok` and carries the render error for the statement. + #[test] + fn a_reply_render_error_leaves_the_installed_flush_ok() { + use crate::data::executor::handlers::control::calvin_reply::CalvinReply; + + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let redo = stage_and_resolve(&mut core, &[point_insert_plan("orders", "o1", 7)]); + let vshard_id = make_task().request.vshard_id.as_u32(); + core.calvin + .commit_pending + .get_mut(&(1, 0, vshard_id)) + .expect("staged commit") + .reply = CalvinReply::unrenderable_for_test("orders"); + + let flushed = flush_at(&mut core, 47, &redo, &["orders".to_string()]); + + assert_eq!(flushed.status, Status::Ok); + assert!( + matches!( + flushed.error_code.as_deref(), + Some(ErrorCode::Internal { .. }) + ), + "{:?}", + flushed.error_code + ); + assert!(base_row(&core, "orders", 7).is_some()); + assert!(!core.calvin.commit_pending.contains_key(&(1, 0, vshard_id))); + } + /// An autocommit put of row `surrogate` in "orders" at `wal_lsn`. + fn autocommit_put(surrogate: u32, value: &str, wal_lsn: u64) -> ExecutionTask { + let mut task = make_task(); + task.request.plan = PhysicalPlan::Document(DocumentOp::PointPut { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), + document_id: "o1".to_string(), + value: doc_value("a", value), + surrogate: Surrogate::new(surrogate), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }); + task.wal_lsn = Some(Lsn::new(wal_lsn)); + task + } + + /// The flush of the transaction staged at `(1, 0)`, as a queued task. + fn flush_task(redo: Vec, lsn: u64) -> ExecutionTask { + let mut task = make_task(); + task.request.plan = + PhysicalPlan::Meta(nodedb_physical::physical_plan::MetaOp::CalvinFlush { + epoch: 1, + position: 0, + redo, + collections: vec!["orders".to_string()], + sum_targets: Vec::new(), + }); + task.wal_lsn = Some(Lsn::new(lsn)); + task + } + + fn holds(core: &CoreLoop, needle: &[u8]) -> bool { + base_row(core, "orders", 7) + .is_some_and(|row| row.windows(needle.len()).any(|w| w == needle)) + } + + /// An autocommit write to a row a staged Calvin transaction owns waits + /// for the flush, then applies on top of it. + #[test] + fn a_write_to_an_owned_row_applies_after_the_flush() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let redo = stage_and_resolve(&mut core, &[point_insert_plan("orders", "o1", 7)]); + + core.task_queue.push(autocommit_put(7, "autocommit", 60)); + core.tick(); + assert_eq!(core.calvin.fence.len(), 1, "the write waits for the owner"); + assert!(base_row(&core, "orders", 7).is_none()); + + core.task_queue.push(flush_task(redo, 50)); + core.tick(); + + assert_eq!(core.calvin.fence.len(), 0); + assert!( + holds(&core, b"autocommit"), + "the waiting write applies after the flush and is not lost" ); } + + /// A waiting write whose record sits below the flush's redo record is + /// refused: restart replay would apply it before the flush. + #[test] + fn a_waiting_write_below_the_flush_lsn_is_refused() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let redo = stage_and_resolve(&mut core, &[point_insert_plan("orders", "o1", 7)]); + + core.task_queue.push(autocommit_put(7, "autocommit", 40)); + core.task_queue.push(flush_task(redo, 50)); + core.tick(); + + assert_eq!(core.calvin.fence.len(), 0); + assert!(base_row(&core, "orders", 7).is_some()); + assert!( + !holds(&core, b"autocommit"), + "the refused write never applies" + ); + } + + /// A write to a row the staged transaction does not own runs at once. + #[test] + fn a_write_to_an_unowned_row_does_not_wait() { + let dir = tempfile::tempdir().unwrap(); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + stage_and_resolve(&mut core, &[point_insert_plan("orders", "o1", 7)]); + + core.task_queue.push(autocommit_put(8, "autocommit", 60)); + core.tick(); + + assert_eq!(core.calvin.fence.len(), 0); + assert!(base_row(&core, "orders", 8).is_some()); + } } diff --git a/nodedb/src/data/executor/handlers/control/calvin/mod.rs b/nodedb/src/data/executor/handlers/control/calvin/mod.rs index ead7dde45..33fb2b19c 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/mod.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/mod.rs @@ -6,11 +6,11 @@ //! //! - `CoreLoop::execute_calvin_execute_static`: static-set multi-shard txn //! (the common case). It VALIDATES the read-set to compute the local commit -//! vote and STAGES the transaction's plans into the commit-pending buffer -//! WITHOUT mutating base or firing side effects, then returns the vote. -//! `CoreLoop::execute_calvin_flush` later replays the staged plans through -//! the durable apply funnel, or `CoreLoop::execute_calvin_drop` discards -//! them. +//! vote and STAGES the transaction's plans into the synthetic overlay and +//! the commit-pending buffer WITHOUT mutating base or firing side effects, +//! then returns the vote. `CoreLoop::execute_calvin_flush` later installs +//! the transaction's committed redo record, or +//! `CoreLoop::execute_calvin_drop` discards the staged state. //! //! - `CoreLoop::execute_calvin_execute_passive`: passive participant for a //! dependent-read txn. Reads each declared key from the local engine and @@ -19,20 +19,24 @@ //! receiving this response. //! //! - `CoreLoop::execute_calvin_execute_active`: active participant for a -//! dependent-read txn. Executes the physical plans with the injected read -//! values already resolved. Performs an OLLP verification hook: if the -//! active participant detects that the declared predicate no longer matches -//! the current engine state, it returns `OllpRetryRequired` WITHOUT writing. -//! The OLLP orchestrator on the Control Plane retries via `Inbox::submit`. +//! dependent-read txn. Stages the physical plans, with the injected read +//! values already resolved, the way the static path does. Performs an OLLP +//! verification hook first: if the active participant detects that the +//! declared predicate no longer matches the current engine state, it +//! returns `OllpRetryRequired` and stages nothing. The OLLP orchestrator on +//! the Control Plane retries via `Inbox::submit`. //! -//! The `CalvinApplied` WAL record is written on the Control Plane side (in the -//! scheduler's response path) after a successful response is received through -//! the SPSC bridge; not here in the Data Plane. +//! The scheduler appends the transaction's `TransactionRedo` record, or a +//! `CalvinApplied` marker when the transaction wrote nothing here, on the +//! Control Plane. The Data Plane writes no WAL record. mod active_passive; mod discard; mod flush; mod shared; mod static_stage; +#[cfg(test)] +mod test_commit; +pub(in crate::data::executor) use flush::CalvinFlushRedo; pub(in crate::data::executor) use shared::CalvinExecCtx; diff --git a/nodedb/src/data/executor/handlers/control/calvin/shared.rs b/nodedb/src/data/executor/handlers/control/calvin/shared.rs index 8d83cd6a9..c941c09d3 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/shared.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/shared.rs @@ -33,8 +33,13 @@ impl CoreLoop { where E: Into, { - self.commit_pending.remove(&(epoch, position, vshard_id)); + self.calvin + .commit_pending + .remove(&(epoch, position, vshard_id)); self.drop_calvin_synthetic_overlay(epoch, position, vshard_id); + self.calvin + .fence + .note_resolved((epoch, position, vshard_id), None); let mut response = self.response_error(task, error.into()); // Scheduler treats this as a durable local abort vote and still waits // for the authoritative global verdict before issuing any drop. diff --git a/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs b/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs index bb66435b2..9c8800e6a 100644 --- a/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs +++ b/nodedb/src/data/executor/handlers/control/calvin/static_stage.rs @@ -11,6 +11,7 @@ use nodedb_types::calvin::VersionedReadEntry; use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::commit_pending::PendingCommit; +use crate::data::executor::handlers::control::calvin_reply::CalvinReply; use crate::data::executor::task::ExecutionTask; use crate::types::TenantId; use nodedb_physical::physical_plan::PhysicalPlan; @@ -26,13 +27,14 @@ impl CoreLoop { /// Computes the local commit vote by checking whether this participant's /// slice of the transaction's LSN-versioned read-set is still current /// against the per-core write versions, then STAGES the write plans into - /// the commit-pending buffer keyed by `(epoch, position)`. It performs NO - /// base mutation and fires NO side effects — nothing is observable until a - /// subsequent [`CoreLoop::execute_calvin_flush`] replays the staged plans - /// (or [`CoreLoop::execute_calvin_drop`] discards them). The response - /// carries the vote on `read_set_valid`; the deterministic time anchor and - /// leadership scope are captured with the staged plans and restored at - /// flush time (when the actual apply — and any time-dependent writes — run). + /// the synthetic overlay and the commit-pending buffer keyed by + /// `(epoch, position)`. It performs NO base mutation and fires NO side + /// effects — nothing is observable until a subsequent + /// [`CoreLoop::execute_calvin_flush`] installs the transaction's redo + /// record (or [`CoreLoop::execute_calvin_drop`] discards the staged + /// state). The response carries the vote on `read_set_valid`. Staging runs + /// under the epoch's deterministic time anchor, which the pending entry + /// keeps for resolve. pub(in crate::data::executor) fn execute_calvin_execute_static( &mut self, task: &ExecutionTask, @@ -85,16 +87,17 @@ impl CoreLoop { let prev_epoch_ms = self.epoch_system_ms; self.epoch_system_ms = Some(epoch_system_ms); let stage_result = catch_unwind(AssertUnwindSafe(|| { + let mut reply = CalvinReply::default(); for plan in plans { - self.stage_calvin_overlay(task, synthetic_txn_id, *tenant_id, plan)?; + self.stage_calvin_plan(task, synthetic_txn_id, *tenant_id, plan, &mut reply)?; // Test-only fault boundary after a potentially-mutating stage. crate::fail_point!("calvin_static::during_overlay_stage"); } - Ok::<(), ErrorCode>(()) + Ok::(reply) })); self.epoch_system_ms = prev_epoch_ms; - match stage_result { - Ok(Ok(())) => {} + let reply = match stage_result { + Ok(Ok(reply)) => reply, Ok(Err(error)) => { return self.calvin_stage_failure(task, epoch, position, vshard_id, error); } @@ -112,7 +115,7 @@ impl CoreLoop { }, ); } - } + }; // Local commit vote: is this participant's slice of the read-set still // current against the local write versions? Empty read-set is vacuously @@ -120,15 +123,15 @@ impl CoreLoop { // retains the fully staged state until the durable global verdict. let vote = self.read_set_still_current(task, tenant_id.as_u64(), versioned_reads); - // Publish only a fully staged transaction. The verdict-driven flush - // replays this raw plan buffer; a global abort drops it and its overlay. - self.commit_pending.insert( + // Publish only a fully staged transaction. Resolve reads these plans + // against the overlay; a global abort drops both. + self.calvin.commit_pending.insert( (epoch, position, vshard_id), PendingCommit { plans: plans.to_vec(), tenant_id: *tenant_id, epoch_system_ms, - is_group_leader, + reply, }, ); @@ -182,11 +185,10 @@ mod tests { let vshard_id = task.request.vshard_id.as_u32(); - // `commit_pending` is unchanged -- it still holds the raw plans that - // drive the base install at flush time. + // `commit_pending` holds the staged plans resolve reads. assert!( - core.commit_pending.contains_key(&(1, 0, vshard_id)), - "commit_pending must still be populated exactly as before this unit" + core.calvin.commit_pending.contains_key(&(1, 0, vshard_id)), + "commit_pending must hold the staged transaction" ); // The synthetic overlay entry additionally holds the resolved @@ -226,7 +228,7 @@ mod tests { assert_eq!(response.status, Status::Error); assert_eq!(response.read_set_valid, Some(false)); - assert!(core.commit_pending.is_empty()); + assert!(core.calvin.commit_pending.is_empty()); assert!(core.txn_overlays.is_empty()); assert!(core.graph_txn_overlays.is_empty()); } @@ -270,7 +272,7 @@ mod tests { assert_eq!(response.status, Status::Error); assert_eq!(response.read_set_valid, Some(false)); - assert!(!core.commit_pending.contains_key(&(9, 3, vshard))); + assert!(!core.calvin.commit_pending.contains_key(&(9, 3, vshard))); assert!(!core.txn_overlays.contains_key(&synthetic)); assert!(!core.graph_txn_overlays.contains_key(&synthetic)); assert_eq!( @@ -353,7 +355,7 @@ mod tests { assert_eq!(response.status, Status::Error); assert_eq!(response.read_set_valid, Some(false)); - assert!(!core.commit_pending.contains_key(&(9, 4, vshard))); + assert!(!core.calvin.commit_pending.contains_key(&(9, 4, vshard))); assert!(!core.txn_overlays.contains_key(&synthetic)); assert!(!core.graph_txn_overlays.contains_key(&synthetic)); assert_eq!( @@ -400,7 +402,7 @@ mod tests { failed.error_code.as_deref(), Some(ErrorCode::RejectedPrevalidation { .. }) )); - assert!(!core.commit_pending.contains_key(&(21, 1, vshard))); + assert!(!core.calvin.commit_pending.contains_key(&(21, 1, vshard))); assert!(!core.txn_overlays.contains_key(&synthetic)); let valid = canonical_ilp_plan("cpu", vec!["cpu value=1i", "cpu value=2i"], vec![1, 2]); @@ -439,7 +441,7 @@ mod tests { ); let mismatch_id = calvin_synthetic_txn_id(21, 2, vshard).expect("synthetic transaction id"); assert_eq!(failed.read_set_valid, Some(false)); - assert!(!core.commit_pending.contains_key(&(21, 2, vshard))); + assert!(!core.calvin.commit_pending.contains_key(&(21, 2, vshard))); assert!(!core.txn_overlays.contains_key(&mismatch_id)); core.ts_tuning.max_tag_cardinality = 1; @@ -468,7 +470,7 @@ mod tests { Some(ErrorCode::RejectedPrevalidation { .. }) )); assert!( - !core.commit_pending.contains_key(&(21, 3, vshard)), + !core.calvin.commit_pending.contains_key(&(21, 3, vshard)), "a rejected stage cannot reach the TransactionRedo-producing flush path" ); assert!(!core.txn_overlays.contains_key(&overflow_id)); diff --git a/nodedb/src/data/executor/handlers/control/calvin/test_commit.rs b/nodedb/src/data/executor/handlers/control/calvin/test_commit.rs new file mode 100644 index 000000000..82e6d9aa3 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin/test_commit.rs @@ -0,0 +1,77 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Test driver for a committed Calvin transaction on one core: stage the +//! plans, resolve them into the redo record, and flush it. + +use nodedb_physical::physical_plan::PhysicalPlan; + +use crate::bridge::envelope::{Response, Status}; +use crate::control::wal_replication::transaction_redo::collections::written_collections; +use crate::control::wal_replication::transaction_redo::sum_targets::redo_sum_targets; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; +use crate::types::{Lsn, TenantId}; + +use super::flush::CalvinFlushRedo; +use super::shared::CalvinExecCtx; + +impl CoreLoop { + /// Commit `plans` as the Calvin transaction at `(epoch, 0)` and flush its + /// redo record at `lsn`, the path a committed Calvin transaction takes on + /// its participant core. + /// + /// Returns the first refusal: the stage, the resolve, or the flush. + pub(in crate::data::executor) fn calvin_commit_for_test( + &mut self, + task: &ExecutionTask, + tid: u64, + plans: &[PhysicalPlan], + epoch: u64, + lsn: u64, + ) -> Response { + self.calvin_commit_at_for_test(task, tid, plans, epoch, 0, lsn) + } + + /// Commit `plans` like [`Self::calvin_commit_for_test`], with the epoch + /// instant `epoch_system_ms`. + pub(in crate::data::executor) fn calvin_commit_at_for_test( + &mut self, + task: &ExecutionTask, + tid: u64, + plans: &[PhysicalPlan], + epoch: u64, + epoch_system_ms: i64, + lsn: u64, + ) -> Response { + let ctx = CalvinExecCtx { + epoch, + position: 0, + epoch_system_ms, + is_group_leader: true, + }; + let staged = self.execute_calvin_execute_static(task, ctx, &TenantId::new(tid), plans, &[]); + if staged.status != Status::Ok { + return staged; + } + let resolved = self.execute_calvin_resolve(task, epoch, 0); + if resolved.status != Status::Ok { + return resolved; + } + let redo = resolved.payload.as_bytes().to_vec(); + let mut request = task.request.clone(); + request.wal_lsn = Some(Lsn::new(lsn)); + let flush_task = ExecutionTask::new(request); + let collections = written_collections(plans); + let sum_targets = redo_sum_targets(plans); + self.execute_calvin_flush( + &flush_task, + CalvinFlushRedo { + epoch, + position: 0, + redo: &redo, + collections: &collections, + sum_targets: &sum_targets, + }, + ) + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin_active_verify.rs b/nodedb/src/data/executor/handlers/control/calvin_active_verify.rs index ab3f986fb..189c9e2af 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_active_verify.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_active_verify.rs @@ -25,7 +25,7 @@ impl CoreLoop { tid: u64, plans: &[PhysicalPlan], ) -> crate::Result { - if !self.ollp_is_group_leader { + if !self.calvin.ollp_is_group_leader { return Ok(true); } let database_id = task.request.database_id.as_u64(); diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs index cdc191c35..a45afab31 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage.rs @@ -1,26 +1,25 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Stage a Calvin static-execute write plan into the shared per-core -//! `txn_overlays` (and `graph_txn_overlays` / `array_txn_overlays`), keyed by -//! a synthetic `TxnId` +//! Stage a Calvin write plan into the shared per-core `txn_overlays` (and +//! `graph_txn_overlays` / `array_txn_overlays`), keyed by a synthetic `TxnId` //! (see `calvin_txn_id.rs`). //! -//! This is purely additive to -//! [`CoreLoop::execute_calvin_execute_static`]'s existing `commit_pending` -//! raw-plan buffering, which remains untouched and still drives the base -//! install at flush time. Staging here is the producer side for a later -//! `CalvinResolve` op that reads the overlay the same way -//! `MetaOp::ResolveTxn` already does for session transactions -//! (`resolve/entry.rs`). +//! Staging is the producer side for `CalvinResolve`, which reads the overlay +//! the same way `MetaOp::ResolveTxn` does for session transactions +//! (`resolve/entry.rs`). The redo record it builds is what the flush +//! installs. -use nodedb_physical::physical_plan::{ColumnarOp, DocumentOp, GraphOp, PhysicalPlan, TimeseriesOp}; +use nodedb_physical::physical_plan::{ + ArrayOp, ColumnarOp, CrdtOp, DocumentOp, GraphOp, PhysicalPlan, TimeseriesOp, VectorOp, +}; use nodedb_types::RowIdentity; use crate::bridge::envelope::{ErrorCode, Response, Status}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::stage_write::{ - StageCtx, StageTimeseriesInsertParams, + StageBalanceDeltaParams, StageBatchInsertParams, StageCtx, StageTimeseriesInsertParams, }; +use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; use crate::types::{TenantId, TxnId}; @@ -28,35 +27,36 @@ use super::calvin_txn_id::calvin_synthetic_txn_id; impl CoreLoop { /// Stage one Calvin write plan into the transaction overlay under the - /// synthetic `txn_id`, reusing the exact same statement-time staging - /// handlers a session `BEGIN..COMMIT` point write uses. + /// synthetic `txn_id`, reusing the statement-time staging handlers a + /// session `BEGIN..COMMIT` write uses, and return the handler's reply. + /// + /// Staged here: the Document point family (`PointInsert` / `PointPut` / + /// `PointDelete` / `PointUpdate` / `Upsert`), `BatchInsert`, + /// `ApplyBalanceDelta`, `Truncate`, KV ops, GRAPH edge/label ops, + /// `TimeseriesOp::Ingest`, every columnar write, the vector-primary direct + /// writes, CRDT row writes, and ARRAY cell writes. `MERGE` / + /// `UPDATE ... FROM` / `INSERT ... SELECT` are resolved into point writes + /// on the Control Plane before dispatch, so one reaching here is refused. /// - /// Concrete, surrogate-carrying point ops are staged here: the Document - /// point family (`PointInsert` / `PointPut` / `PointDelete` / - /// `PointUpdate` / `Upsert`), KV point ops, and GRAPH edge/label ops. - /// These are also the only op shapes that reach Calvin buffering for the - /// Document family in the first place — `MERGE` / `UPDATE ... FROM` / - /// `INSERT ... SELECT` are already expanded to concrete point ops at - /// statement time before Calvin buffering (`commit.rs`). + /// The remaining write families stage nothing because resolve serializes + /// them from the plan node itself: the other vector writes, CRDT deltas, + /// FTS and spatial writes, timeseries truncates and array flushes. /// - /// `DocumentOp::BulkUpdate` / `BulkDelete` (predicate DML) are also - /// staged, but via the predicted-surrogate-set primitives in + /// `DocumentOp::BulkUpdate` / `BulkDelete` (predicate DML) are staged via + /// the predicted-surrogate-set primitives in /// [`calvin_overlay_stage_bulk`][super::calvin_overlay_stage_bulk] rather - /// than a live predicate rescan — see that module's docs for the - /// determinism rationale. `TimeseriesOp::Ingest` is staged through the - /// same canonical row decoder as session writes; its per-row tokens are - /// overlay-local and never become base-storage identities. Every columnar - /// write is staged through the session handlers, so COMMIT resolve reads - /// its post-images from the overlay. Staging runs under the epoch's time - /// anchor, so every replica stages the same images. Spatial writes carry - /// their absolute post-image on the plan node and stay unstaged. + /// than a live predicate rescan. See that module's docs for the + /// determinism rationale. Staging runs under the epoch's time anchor, so + /// every replica stages the same images. Spatial and text writes carry + /// their absolute post-image on the plan node, stay unstaged, and answer + /// an empty reply. pub(in crate::data::executor) fn stage_calvin_overlay( &mut self, task: &ExecutionTask, txn_id: TxnId, tenant_id: TenantId, plan: &PhysicalPlan, - ) -> Result<(), ErrorCode> { + ) -> Result, ErrorCode> { let tid = tenant_id.as_u64(); match plan { PhysicalPlan::Document(DocumentOp::PointInsert { @@ -176,6 +176,7 @@ impl CoreLoop { rls_write_check, declared_primary_key: declared_primary_key.as_deref(), }) + .and_then(affected_reply) .map_err(ErrorCode::from), PhysicalPlan::Document(DocumentOp::BulkUpdate { collection, @@ -195,7 +196,66 @@ impl CoreLoop { rls_write_check, declared_primary_key: declared_primary_key.as_deref(), }) + .and_then(affected_reply) .map_err(ErrorCode::from), + PhysicalPlan::Document(DocumentOp::BatchInsert { + collection, + documents, + surrogates, + .. + }) => { + let resp = self.stage_document_batch_insert(StageBatchInsertParams { + task, + tid, + txn_id, + collection: collection.as_str(), + documents, + surrogates, + }); + Self::stage_result(&resp) + } + PhysicalPlan::Document(DocumentOp::ApplyBalanceDelta { + collection, + document_id, + surrogate, + column, + delta, + join_column, + join_value, + declared_primary_key, + }) => { + let resp = self.stage_apply_balance_delta(StageBalanceDeltaParams { + task, + tid, + txn_id, + collection: collection.as_str(), + document_id, + surrogate: *surrogate, + column, + delta, + join_column, + join_value, + declared_primary_key: declared_primary_key.as_deref(), + }); + Self::stage_result(&resp) + } + PhysicalPlan::Document(DocumentOp::Truncate { collection, .. }) => { + let resp = self.stage_collection_truncate(task, tid, txn_id, collection.as_str()); + Self::stage_result(&resp) + } + // The Control Plane resolves these into point writes, or proposes + // them through Raft, before any Calvin dispatch. Staging one would + // install nothing, so it is refused. + PhysicalPlan::Document( + DocumentOp::InsertSelect { .. } + | DocumentOp::UpdateFromJoin { .. } + | DocumentOp::Merge { .. } + | DocumentOp::ResolvedWrite { .. }, + ) => Err(ErrorCode::Internal { + detail: "a cross-collection or pre-resolved document write reached Calvin \ + staging; the Control Plane resolves it before dispatch" + .into(), + }), PhysicalPlan::Kv(op) => { let resp = self.execute_stage_kv(task, tid, txn_id, op); Self::stage_result(&resp) @@ -247,14 +307,60 @@ impl CoreLoop { let resp = self.execute_stage_graph(task, tid, txn_id, op); Self::stage_result(&resp) } - _ => Ok(()), + PhysicalPlan::Vector( + op @ (VectorOp::DirectInsert { .. } + | VectorOp::DirectInsertIfAbsent { .. } + | VectorOp::DirectUpsert { .. } + | VectorOp::DirectUpdate { .. } + | VectorOp::DirectDelete { .. } + | VectorOp::DirectTruncate { .. }), + ) => { + let resp = self.execute_stage_vector(task, tid, txn_id, op); + Self::stage_result(&resp) + } + PhysicalPlan::Crdt(op @ (CrdtOp::DocUpsert { .. } | CrdtOp::DocDelete { .. })) => { + let resp = self.execute_stage_crdt(task, tid, txn_id, op); + Self::stage_result(&resp) + } + PhysicalPlan::Array(op @ (ArrayOp::Put { .. } | ArrayOp::Delete { .. })) => { + let resp = self.execute_stage_array(task, txn_id, op); + Self::stage_result(&resp) + } + // Document reads and index DDL write no row. + PhysicalPlan::Document( + DocumentOp::ResolveWrite(_) + | DocumentOp::PointGet { .. } + | DocumentOp::Scan { .. } + | DocumentOp::RangeScan { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::IndexedFetch { .. } + | DocumentOp::EstimateCount { .. } + | DocumentOp::MaterializeScan { .. } + | DocumentOp::Register { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. }, + ) => Ok(Vec::new()), + // Resolve serializes these writes from the plan node, and the + // rest of these families write nothing. + PhysicalPlan::Timeseries(_) + | PhysicalPlan::Columnar(_) + | PhysicalPlan::Graph(_) + | PhysicalPlan::Vector(_) + | PhysicalPlan::Crdt(_) + | PhysicalPlan::Array(_) + | PhysicalPlan::Text(_) + | PhysicalPlan::Spatial(_) + | PhysicalPlan::Query(_) + | PhysicalPlan::Meta(_) + | PhysicalPlan::ClusterArray(_) + | PhysicalPlan::ClusterEvent(_) => Ok(Vec::new()), } } - /// Turn a staging handler's `Response` into a `Result`, so a staging + /// Turn a staging handler's `Response` into its reply, so a staging /// failure propagates loudly to the Calvin caller instead of being /// silently swallowed. - fn stage_result(resp: &Response) -> Result<(), ErrorCode> { + fn stage_result(resp: &Response) -> Result, ErrorCode> { if resp.status == Status::Error { return Err(resp.error_code.as_deref().cloned().unwrap_or_else(|| { ErrorCode::Internal { @@ -262,7 +368,7 @@ impl CoreLoop { } })); } - Ok(()) + Ok(resp.payload.as_bytes().to_vec()) } /// Discard the synthetic-`TxnId` overlay entries staged for @@ -292,3 +398,8 @@ impl CoreLoop { } } } + +/// The `{"affected": n}` reply of a Calvin bulk stage. +fn affected_reply(affected: usize) -> crate::Result> { + response_codec::encode_count("affected", affected) +} diff --git a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs index b3496654f..7d3f1cbdd 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_overlay_stage_bulk.rs @@ -6,20 +6,15 @@ //! //! # The determinism rule //! -//! The Calvin flush apply (`execute_bulk_delete` / `execute_bulk_update`) -//! mutates EXACTLY the CP-injected `ollp_predicted_surrogates` set, verbatim, -//! on every replica — see `super::super::bulk_dml::delete`'s `apply_ids` -//! derivation and `super::super::bulk_dml::scan::ollp_predicted_doc_ids`. A -//! live predicate rescan is NOT used as the apply set when a prediction is -//! present, because a follower's local snapshot can legitimately lag the -//! leader's verified prediction window; re-deriving the row set locally would -//! diverge across replicas. -//! -//! Staging here must therefore resolve rows from `ollp_predicted_surrogates` -//! — the SAME set the flush applies — via the SAME `ollp_predicted_doc_ids` -//! primitive, never via a fresh `scan_matching_documents` predicate scan (the +//! Staging resolves EXACTLY the CP-injected `ollp_predicted_surrogates` set, +//! verbatim, on every replica, via the `ollp_predicted_doc_ids` primitive. A +//! live predicate rescan is NOT used as the row set, because a follower's +//! local snapshot can legitimately lag the leader's verified prediction +//! window. Re-deriving the row set locally would diverge across replicas. +//! The flush installs the redo record `CalvinResolve` builds from this +//! staging, so the staged rows are the rows the flush writes. The //! `stage_bulk_delete` / `stage_bulk_update` session-transaction handlers do -//! exactly that live rescan and are NOT reused here for this reason). +//! a live rescan and are NOT reused here for this reason. //! //! Reading each predicted surrogate's CURRENT body (for `BulkUpdate`'s //! post-image and read-your-own-writes) is still safe to source from local @@ -29,11 +24,12 @@ //! replicas — unlike predicate *membership*, which is what the surrogate-set //! fixing above protects against. //! -//! `BulkUpdate`'s per-row transform reuses `CoreLoop::stage_apply_update` -//! verbatim — the exact same decode → apply-updates → recompute-generated → -//! re-encode pipeline `execute_bulk_update` and `stage_point_update` already -//! share — so the staged post-image is byte-identical to what the flush -//! apply would produce for the same input body. +//! `BulkUpdate`'s per-row transform reuses `CoreLoop::stage_apply_update`, +//! the decode → apply-updates → recompute-generated → re-encode pipeline +//! `execute_bulk_update` and `stage_point_update` share. +//! +//! Each stage returns the number of rows it staged, the affected count the +//! statement reports. use nodedb_physical::physical_plan::UpdateValue; use nodedb_types::Surrogate; @@ -95,11 +91,12 @@ impl CoreLoop { /// Stage a Calvin `BulkDelete` into the overlay: one tombstone per /// predicted surrogate, resolved to its doc-id via /// `ollp_predicted_doc_ids` — the identical primitive the flush apply - /// uses to derive `apply_ids`. NOT a live predicate rescan. + /// uses to derive `apply_ids`. NOT a live predicate rescan. Returns the + /// number of predicted rows that exist. pub(in crate::data::executor) fn stage_calvin_bulk_delete( &mut self, params: CalvinBulkDeleteStage<'_>, - ) -> crate::Result<()> { + ) -> crate::Result { let CalvinBulkDeleteStage { task, tid, @@ -120,16 +117,23 @@ impl CoreLoop { predicted_sorted.sort_unstable(); let doc_ids = ollp_predicted_doc_ids(predicted); - // Each row's identity is read once from its current body, by the rule - // INSERT minted it with. A row with no current body removes nothing; - // its tombstone is keyed by the decimal surrogate. + // Each row's identity is read once from its current body under + // BASE ∪ OVERLAY, by the rule INSERT minted it with, so an earlier + // plan of the same transaction is observed. A row with no current + // body removes nothing; its tombstone is keyed by the decimal + // surrogate. let strict_schema = self.resolve_strict_schema(database_id.as_u64(), tid, collection); + let bitemporal = self.is_bitemporal(database_id.as_u64(), tid, collection); let mut rows: Vec<(u32, nodedb_types::RowIdentity, Option>)> = Vec::with_capacity(doc_ids.len()); for (surrogate, doc_id) in predicted_sorted.into_iter().zip(doc_ids) { - let body = self - .sparse - .get(database_id.as_u64(), tid, collection, &doc_id)?; + let body = self.calvin_current_body(CalvinRowRead { + txn_id, + coll_key: &coll_key, + surrogate, + storage_key: &doc_id, + bitemporal, + })?; let identity = match &body { Some(body) => { stored_row_identity(body, strict_schema.as_ref(), declared_primary_key, doc_id) @@ -161,11 +165,12 @@ impl CoreLoop { } } + let removed = rows.iter().filter(|(_, _, body)| body.is_some()).count(); let overlay = self.txn_overlay_mut(txn_id); for (surrogate, identity, _body) in &rows { overlay.insert_tombstone(coll_key.clone(), *surrogate, identity); } - Ok(()) + Ok(removed) } /// Stage a Calvin `BulkUpdate` into the overlay: for each predicted @@ -181,11 +186,11 @@ impl CoreLoop { /// A predicted surrogate that resolves to no current body (already /// tombstoned in this transaction, or absent from BASE) is skipped — the /// identical `continue`-on-miss behavior `execute_bulk_update` exhibits - /// for its own `apply_ids` loop. + /// for its own `apply_ids` loop. Returns the number of rows staged. pub(in crate::data::executor) fn stage_calvin_bulk_update( &mut self, params: CalvinBulkUpdateStage<'_>, - ) -> crate::Result<()> { + ) -> crate::Result { let CalvinBulkUpdateStage { task, tid, @@ -208,37 +213,21 @@ impl CoreLoop { predicted_sorted.sort_unstable(); let strict_schema = self.resolve_strict_schema(database_id.as_u64(), tid, collection); + let mut updated = 0usize; for surrogate in predicted_sorted { let storage_key = nodedb_types::StorageKey::for_surrogate(Surrogate::new(surrogate)); // Current body: overlay wins over base (read-your-own-writes), // mirroring `stage_point_update`'s exact overlay-then-base read. - let overlay_cur = self - .txn_overlays - .get(&txn_id) - .and_then(|o| o.get(&coll_key, surrogate)) - .cloned(); - let current_bytes = match overlay_cur { - Some(Staged::Put(body)) => body, - Some(Staged::Tombstone) => continue, - None => { - let read = if bitemporal { - self.sparse.versioned_get_current( - database_id.as_u64(), - tid, - collection, - &storage_key, - ) - } else { - self.sparse - .get(database_id.as_u64(), tid, collection, &storage_key) - }; - match read { - Ok(Some(bytes)) => bytes, - Ok(None) => continue, - Err(e) => return Err(e), - } - } + let Some(current_bytes) = self.calvin_current_body(CalvinRowRead { + txn_id, + coll_key: &coll_key, + surrogate, + storage_key: &storage_key, + bitemporal, + })? + else { + continue; }; let new_body = self.stage_apply_update( @@ -266,7 +255,51 @@ impl CoreLoop { collection, )?; self.stage_bulk_put_capped(txn_id, &coll_key, surrogate, &identity, new_body)?; + updated += 1; + } + Ok(updated) + } +} + +/// One row a Calvin bulk stage reads under BASE ∪ OVERLAY. +struct CalvinRowRead<'a> { + txn_id: TxnId, + coll_key: &'a (DatabaseId, TenantId, String), + surrogate: u32, + storage_key: &'a nodedb_types::StorageKey, + bitemporal: bool, +} + +impl CoreLoop { + /// The row's current stored body inside the Calvin transaction. A staged + /// put wins over base. A staged tombstone or a staged TRUNCATE hides the + /// row. Otherwise the body is base storage: the current version on a + /// bitemporal collection. + fn calvin_current_body(&self, read: CalvinRowRead<'_>) -> crate::Result>> { + let overlay = self.txn_overlays.get(&read.txn_id); + match overlay.and_then(|o| o.get(read.coll_key, read.surrogate)) { + Some(Staged::Put(body)) => return Ok(Some(body.clone())), + Some(Staged::Tombstone) => return Ok(None), + None => {} + } + if overlay.is_some_and(|o| !o.base_visible(read.coll_key)) { + return Ok(None); + } + let (database_id, tenant, collection) = read.coll_key; + if read.bitemporal { + self.sparse.versioned_get_current( + database_id.as_u64(), + tenant.as_u64(), + collection, + read.storage_key, + ) + } else { + self.sparse.get( + database_id.as_u64(), + tenant.as_u64(), + collection, + read.storage_key, + ) } - Ok(()) } } diff --git a/nodedb/src/data/executor/handlers/control/calvin_reply/flush_read.rs b/nodedb/src/data/executor/handlers/control/calvin_reply/flush_read.rs new file mode 100644 index 000000000..35e4779f9 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin_reply/flush_read.rs @@ -0,0 +1,54 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The payload a flushed Calvin transaction answers with. + +use super::images::{RowLocation, StoredRow}; +use super::reply::CalvinReply; +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; + +impl CoreLoop { + /// The payload `reply` answers with once the flush has installed the + /// transaction. Post-images are read from base now, so each row is what + /// the install stored. + /// + /// A row the plan wrote that base does not hold after the install is an + /// install that dropped a write, and the reply refuses it. + pub(in crate::data::executor) fn calvin_reply_payload( + &self, + task: &ExecutionTask, + tid: u64, + reply: CalvinReply, + ) -> Result, ErrorCode> { + let images = match reply { + CalvinReply::Count(payload) | CalvinReply::Rows(payload) => return Ok(payload), + CalvinReply::PostImages(images) => images, + }; + let at = RowLocation { + engine: images.engine, + database_id: task.request.database_id.as_u64(), + tid, + collection: &images.collection, + }; + let mut rows = Vec::with_capacity(images.rows.len()); + for (identity, surrogate) in &images.rows { + let Some(bytes) = self.calvin_base_row(&at, identity, *surrogate)? else { + return Err(ErrorCode::Internal { + detail: format!( + "calvin RETURNING: row '{}' of '{}' is absent after the install that \ + wrote it", + identity.as_str(), + images.collection + ), + }); + }; + rows.push(StoredRow { + identity: identity.clone(), + surrogate: *surrogate, + bytes, + }); + } + self.calvin_render_rows(&at, &images.spec, &images.rls_filters, &rows) + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin_reply/images.rs b/nodedb/src/data/executor/handlers/control/calvin_reply/images.rs new file mode 100644 index 000000000..e43967bb8 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin_reply/images.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Stored images of the rows a Calvin plan touched, and the `RETURNING` +//! rows they render. +//! +//! Each engine renders through the converter its live write handler uses, +//! so a row reads the same whichever path wrote it. + +use nodedb_physical::physical_plan::ReturningSpec; +use nodedb_types::{RowIdentity, StorageKey, Surrogate}; + +use super::target::RowEngine; +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::returning_rows::{ + build_stored_rows_payload, kv_stored_rows_payload, vector_stored_rows_payload, +}; +use crate::data::executor::handlers::transaction::overlay::staged_vector_sidecar; +use crate::data::executor::handlers::transaction::stage_write::unhex_key; + +/// One row's identity and the stored bytes of the image `RETURNING` +/// reports. A vector-primary row's bytes are its payload sidecar. +pub(super) struct StoredRow { + pub identity: RowIdentity, + pub surrogate: Surrogate, + pub bytes: Vec, +} + +/// Where a row lives, for a base read. +pub(super) struct RowLocation<'a> { + pub engine: RowEngine, + pub database_id: u64, + pub tid: u64, + pub collection: &'a str, +} + +/// The stored bytes of a row staged as `body`. A staged vector-primary row +/// carries its vector beside its sidecar, and only the sidecar is rendered. +pub(super) fn staged_row_bytes(engine: RowEngine, body: &[u8]) -> Result, ErrorCode> { + match engine { + RowEngine::Vector => staged_vector_sidecar(body).map_err(ErrorCode::from), + RowEngine::Document + | RowEngine::Crdt + | RowEngine::Kv + | RowEngine::Columnar + | RowEngine::Timeseries => Ok(body.to_vec()), + } +} + +/// The raw key of a KV row, from its overlay identity. +fn kv_raw_key(identity: &RowIdentity) -> Result, ErrorCode> { + unhex_key(identity.as_str()).ok_or_else(|| ErrorCode::Internal { + detail: format!( + "calvin RETURNING: KV row identity '{}' is not a hex-encoded key", + identity.as_str() + ), + }) +} + +impl CoreLoop { + /// The row's body in base storage, or `None` when base holds no such row. + /// A bitemporal document row reads its current version. + pub(super) fn calvin_base_row( + &self, + at: &RowLocation<'_>, + identity: &RowIdentity, + surrogate: Surrogate, + ) -> Result>, ErrorCode> { + let RowLocation { + engine, + database_id, + tid, + collection, + } = *at; + match engine { + RowEngine::Document | RowEngine::Crdt => { + let key = StorageKey::for_surrogate(surrogate); + let read = if self.is_bitemporal(database_id, tid, collection) { + self.sparse + .versioned_get_current(database_id, tid, collection, &key) + } else { + self.sparse.get(database_id, tid, collection, &key) + }; + read.map_err(ErrorCode::from) + } + RowEngine::Kv => { + let key = kv_raw_key(identity)?; + Ok(self + .kv_engine + .get(database_id, tid, collection, &key, self.kv_read_now_ms())) + } + RowEngine::Vector => self.vector_sidecar_bytes(database_id, tid, collection, surrogate), + // Neither engine keys a base row by surrogate. Their `RETURNING` + // rows are always the images the plan staged. + RowEngine::Columnar | RowEngine::Timeseries => Err(ErrorCode::Internal { + detail: format!( + "calvin RETURNING: a {engine:?} row of '{collection}' has no keyed base read" + ), + }), + } + } + + /// Render `rows` as the `RETURNING` row set of `at.engine`. + pub(super) fn calvin_render_rows( + &self, + at: &RowLocation<'_>, + spec: &ReturningSpec, + rls_filters: &[u8], + rows: &[StoredRow], + ) -> Result, ErrorCode> { + match at.engine { + RowEngine::Document | RowEngine::Crdt => { + // A CRDT row is MessagePack in either storage mode. + let strict_schema = match at.engine { + RowEngine::Document => { + self.resolve_strict_schema(at.database_id, at.tid, at.collection) + } + RowEngine::Crdt + | RowEngine::Kv + | RowEngine::Vector + | RowEngine::Columnar + | RowEngine::Timeseries => None, + }; + let stored: Vec<(&RowIdentity, &[u8])> = rows + .iter() + .map(|row| (&row.identity, row.bytes.as_slice())) + .collect(); + build_stored_rows_payload(spec, rls_filters, strict_schema.as_ref(), &stored) + .map_err(ErrorCode::from) + } + RowEngine::Kv => { + let keys = rows + .iter() + .map(|row| kv_raw_key(&row.identity)) + .collect::, _>>()?; + let stored: Vec<(&[u8], &[u8])> = keys + .iter() + .zip(rows) + .map(|(key, row)| (key.as_slice(), row.bytes.as_slice())) + .collect(); + kv_stored_rows_payload(spec, rls_filters, &stored).map_err(ErrorCode::from) + } + RowEngine::Vector => { + let keys: Vec = rows + .iter() + .map(|row| StorageKey::for_surrogate(row.surrogate)) + .collect(); + let stored: Vec<(&StorageKey, &[u8])> = keys + .iter() + .zip(rows) + .map(|(key, row)| (key, row.bytes.as_slice())) + .collect(); + vector_stored_rows_payload(spec, rls_filters, &stored).map_err(ErrorCode::from) + } + RowEngine::Columnar | RowEngine::Timeseries => Err(ErrorCode::Internal { + detail: format!( + "calvin RETURNING: {:?} rows of '{}' render from their staged values", + at.engine, at.collection + ), + }), + } + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin_reply/mod.rs b/nodedb/src/data/executor/handlers/control/calvin_reply/mod.rs new file mode 100644 index 000000000..1ab795f8d --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin_reply/mod.rs @@ -0,0 +1,9 @@ +// SPDX-License-Identifier: BUSL-1.1 + +mod flush_read; +mod images; +mod reply; +mod stage; +mod target; + +pub(in crate::data::executor) use reply::CalvinReply; diff --git a/nodedb/src/data/executor/handlers/control/calvin_reply/reply.rs b/nodedb/src/data/executor/handlers/control/calvin_reply/reply.rs new file mode 100644 index 000000000..bba669ee2 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin_reply/reply.rs @@ -0,0 +1,66 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The reply a staged Calvin transaction answers with when its flush +//! installs it. + +use nodedb_physical::physical_plan::ReturningSpec; +use nodedb_types::{RowIdentity, Surrogate}; + +use super::target::RowEngine; + +/// The reply a Calvin transaction's flush answers with. +/// +/// The transaction answers with its last `RETURNING` plan's rows, or with +/// its last plan's affected count when no plan carries `RETURNING`. A plan +/// derived from the statement, such as a graph edge or a balance move, +/// therefore never replaces the rows the statement asked for. +#[derive(Debug)] +pub(in crate::data::executor) enum CalvinReply { + /// An affected count, decided when the plan staged. + Count(Vec), + /// `RETURNING` rows, decided when the plan staged. + Rows(Vec), + /// `RETURNING` rows the flush reads from base after the install, so each + /// row is exactly what a `SELECT` reads. + PostImages(PostImages), +} + +impl Default for CalvinReply { + fn default() -> Self { + Self::Count(Vec::new()) + } +} + +impl CalvinReply { + /// A reply the flush cannot render: post-images of a columnar row, which + /// base keys by no surrogate. + #[cfg(test)] + pub(in crate::data::executor) fn unrenderable_for_test(collection: &str) -> Self { + Self::PostImages(PostImages { + spec: ReturningSpec { + columns: nodedb_physical::physical_plan::ReturningColumns::Star, + }, + rls_filters: Vec::new(), + collection: collection.to_string(), + engine: RowEngine::Columnar, + rows: vec![(RowIdentity::from_user_key("r1"), Surrogate::new(1))], + }) + } + + /// Whether this reply carries `RETURNING` rows. + pub(super) fn has_rows(&self) -> bool { + !matches!(self, Self::Count(_)) + } +} + +/// The rows of a `RETURNING` plan the flush reads after the install. +#[derive(Debug)] +pub(in crate::data::executor) struct PostImages { + pub(super) spec: ReturningSpec, + pub(super) rls_filters: Vec, + pub(super) collection: String, + pub(super) engine: RowEngine, + /// Each row the plan wrote: its client identity and its surrogate, in + /// the order the plan wrote them. + pub(super) rows: Vec<(RowIdentity, Surrogate)>, +} diff --git a/nodedb/src/data/executor/handlers/control/calvin_reply/stage.rs b/nodedb/src/data/executor/handlers/control/calvin_reply/stage.rs new file mode 100644 index 000000000..260053674 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin_reply/stage.rs @@ -0,0 +1,760 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Stage one Calvin plan and fold its reply into the transaction's reply. +//! +//! A Calvin flush installs the transaction's redo record, and a redo record +//! carries no reply. Each plan therefore decides its reply when it stages. +//! +//! - A plan without `RETURNING` answers its staging handler's reply: the +//! affected count against BASE ∪ OVERLAY. +//! - A `RETURNING` plan answers the rows it touched. The overlay journal +//! names them: every slot the plan's staging mutated. Each row's image is +//! taken from BASE ∪ OVERLAY, so a row an earlier plan of the same +//! transaction wrote is reported as that plan left it. +//! - A delete reports each removed row's image from before the plan. +//! - A document or CRDT write reports each row as the flush's install +//! stored it, read back from base after the install. The install can +//! rewrite a body, as hash chaining does. +//! - A KV, vector-primary or columnar write reports the rows it staged, +//! which are the exact bytes the install stores. +//! - A timeseries ingest reports its stamped rows rendered through the +//! raw-scan row emitter, as the live ingest does. + +use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp}; +use nodedb_types::{Surrogate, Value}; + +use super::images::{RowLocation, StoredRow, staged_row_bytes}; +use super::reply::{CalvinReply, PostImages}; +use super::target::{ReplyImage, ReturningTarget, RowEngine, returning_target}; +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::columnar_write::row_values_to_object; +use crate::data::executor::handlers::returning_rows::build_rows_payload; +use crate::data::executor::handlers::timeseries::StampedIngest; +use crate::data::executor::handlers::transaction::overlay::{ + Staged, TouchedSlot, decode_staged_row, +}; +use crate::data::executor::task::ExecutionTask; +use crate::engine::timeseries::ilp; +use crate::types::{DatabaseId, TenantId, TxnId}; +use crate::util::rmpv_value::rmpv_to_value; + +type CollKey = (DatabaseId, TenantId, String); + +impl CoreLoop { + /// Stage one Calvin plan under `txn_id` and fold its reply into `reply`. + /// + /// A `RETURNING` plan's rows replace `reply`. A later plan without + /// `RETURNING` keeps them, and replaces only an affected count. + pub(in crate::data::executor) fn stage_calvin_plan( + &mut self, + task: &ExecutionTask, + txn_id: TxnId, + tenant_id: TenantId, + plan: &PhysicalPlan, + reply: &mut CalvinReply, + ) -> Result<(), ErrorCode> { + let tid = tenant_id.as_u64(); + // Every core that stages a plan holds the overlay, even when the plan + // stages no row. `CalvinResolve` refuses a missing one. + let marker = self.txn_overlay_mut(txn_id).journal_len(); + let staged = self.stage_calvin_overlay(task, txn_id, tenant_id, plan)?; + match returning_target(plan) { + Some(target) => { + *reply = self.calvin_returning_reply(task, txn_id, tid, plan, &target, marker)?; + } + None => { + self.settle_rows_before_later_write(task, txn_id, tid, reply, marker)?; + if !reply.has_rows() { + *reply = CalvinReply::Count(staged); + } + } + } + Ok(()) + } + + /// The reply of a `RETURNING` plan whose staging began at journal + /// position `marker`. + fn calvin_returning_reply( + &self, + task: &ExecutionTask, + txn_id: TxnId, + tid: u64, + plan: &PhysicalPlan, + target: &ReturningTarget<'_>, + marker: usize, + ) -> Result { + let database_id = task.request.database_id; + let coll_key: CollKey = ( + database_id, + TenantId::new(tid), + target.collection.to_string(), + ); + let overlay = self + .txn_overlays + .get(&txn_id) + .ok_or_else(|| ErrorCode::Internal { + detail: "calvin RETURNING: the transaction overlay is missing after staging".into(), + })?; + let slots = overlay.slots_touched_since(marker, &coll_key); + let at = RowLocation { + engine: target.engine, + database_id: database_id.as_u64(), + tid, + collection: target.collection, + }; + let written = slots.iter().filter_map(|slot| match slot.after { + Some(Staged::Put(body)) => Some((slot, body.as_slice())), + Some(Staged::Tombstone) | None => None, + }); + + match (target.image, target.engine) { + (ReplyImage::Before, _) => { + let rows = + self.calvin_removed_rows(&at, target, &slots, overlay.base_visible(&coll_key))?; + self.calvin_render_rows(&at, target.spec, target.rls_filters, &rows) + .map(CalvinReply::Rows) + } + (ReplyImage::After, RowEngine::Document | RowEngine::Crdt) => { + Ok(CalvinReply::PostImages(PostImages { + spec: target.spec.clone(), + rls_filters: target.rls_filters.to_vec(), + collection: target.collection.to_string(), + engine: target.engine, + rows: written + .map(|(slot, _)| (slot.doc_id.clone(), Surrogate::new(slot.surrogate))) + .collect(), + })) + } + (ReplyImage::After, RowEngine::Kv | RowEngine::Vector) => { + let rows = written + .map(|(slot, body)| { + Ok(StoredRow { + identity: slot.doc_id.clone(), + surrogate: Surrogate::new(slot.surrogate), + bytes: staged_row_bytes(target.engine, body)?, + }) + }) + .collect::, ErrorCode>>()?; + self.calvin_render_rows(&at, target.spec, target.rls_filters, &rows) + .map(CalvinReply::Rows) + } + (ReplyImage::After, RowEngine::Columnar) => { + let schema = self + .columnar_engines + .get(&coll_key) + .map(|engine| engine.schema().clone()) + .ok_or_else(|| ErrorCode::Internal { + detail: format!( + "calvin RETURNING: columnar collection '{}' has no engine after \ + staging", + target.collection + ), + })?; + let docs = written + .map(|(slot, body)| { + decode_staged_row(body) + .map(|row| row_values_to_object(&schema, &row)) + .ok_or_else(|| ErrorCode::Internal { + detail: format!( + "calvin RETURNING: staged columnar row {} of '{}' does not \ + decode", + slot.surrogate, target.collection + ), + }) + }) + .collect::, ErrorCode>>()?; + build_rows_payload(target.spec, target.rls_filters, &docs) + .map(CalvinReply::Rows) + .map_err(ErrorCode::from) + } + (ReplyImage::After, RowEngine::Timeseries) => self + .calvin_timeseries_rows(task, txn_id, tid, plan, target) + .map(CalvinReply::Rows), + } + } + + /// The prior image of every row a delete removed: the row as an earlier + /// plan of the transaction staged it, else its base row. + fn calvin_removed_rows( + &self, + at: &RowLocation<'_>, + target: &ReturningTarget<'_>, + slots: &[TouchedSlot<'_>], + base_visible: bool, + ) -> Result, ErrorCode> { + let mut rows = Vec::new(); + for slot in slots { + if !matches!(slot.after, Some(Staged::Tombstone)) { + continue; + } + let surrogate = Surrogate::new(slot.surrogate); + let prior = match slot.before { + Some(Staged::Put(body)) => Some(staged_row_bytes(at.engine, body)?), + Some(Staged::Tombstone) => None, + None if base_visible => { + self.calvin_removed_base_row(at, target, slot, surrogate)? + } + None => None, + }; + if let Some(bytes) = prior { + rows.push(StoredRow { + identity: slot.doc_id.clone(), + surrogate, + bytes, + }); + } + } + Ok(rows) + } + + /// The base image of a row a delete removed. A vector-primary row whose + /// node is bound but whose sidecar is absent has an empty image, as the + /// live delete reports it. + fn calvin_removed_base_row( + &self, + at: &RowLocation<'_>, + target: &ReturningTarget<'_>, + slot: &TouchedSlot<'_>, + surrogate: Surrogate, + ) -> Result>, ErrorCode> { + let base = self.calvin_base_row(at, slot.doc_id, surrogate)?; + match (base, target.vector_field) { + (None, Some(field)) => { + let index_key = + CoreLoop::vector_index_key(at.database_id, at.tid, at.collection, field); + Ok(self + .vector_direct_node(&index_key, surrogate) + .map(|_| Vec::new())) + } + (base, _) => Ok(base), + } + } + + /// The `RETURNING` rows of a timeseries ingest: the lines its resolve + /// stamps, rendered through the raw-scan row emitter. + fn calvin_timeseries_rows( + &self, + task: &ExecutionTask, + txn_id: TxnId, + tid: u64, + plan: &PhysicalPlan, + target: &ReturningTarget<'_>, + ) -> Result, ErrorCode> { + let PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection, + payload, + format, + surrogates, + .. + }) = plan + else { + return Err(ErrorCode::Internal { + detail: "calvin RETURNING: a timeseries reply target is not an ingest".into(), + }); + }; + let database_id = task.request.database_id; + let tenant = TenantId::new(tid); + let coll_key: CollKey = (database_id, tenant, collection.as_str().to_string()); + // The instant the staged batch read, which resolve stamps its untimed + // rows with. + let now_ms = surrogates + .first() + .and_then(|first| { + self.txn_overlays + .get(&txn_id)? + .ingest_now(&coll_key, first.as_u32()) + }) + .unwrap_or_else(|| self.ingest_now_ms()); + let lines = self.stamped_ingest_lines(StampedIngest { + database_id, + tid: tenant, + collection: collection.as_str(), + payload, + format, + now_ms, + })?; + let joined = lines.join("\n"); + let parsed = ilp::parse_batch(&joined).map_err(|e| ErrorCode::RejectedPrevalidation { + reason: format!("calvin RETURNING: unparsable line protocol: {e}"), + })?; + let rows = self.preview_ilp_ingest_rows( + task, + tenant, + collection.as_str(), + parsed.lines(), + now_ms, + )?; + let docs: Vec = rows.iter().map(rmpv_to_value).collect(); + build_rows_payload(target.spec, target.rls_filters, &docs).map_err(ErrorCode::from) + } + + /// Decide a document or CRDT post-image reply now when a later plan, + /// staged from journal position `marker`, rewrote one of its rows. A base + /// read after the install would report the later plan's row, so each row + /// takes the image the `RETURNING` plan staged instead. + fn settle_rows_before_later_write( + &self, + task: &ExecutionTask, + txn_id: TxnId, + tid: u64, + reply: &mut CalvinReply, + marker: usize, + ) -> Result<(), ErrorCode> { + let CalvinReply::PostImages(images) = &*reply else { + return Ok(()); + }; + let database_id = task.request.database_id; + let coll_key: CollKey = (database_id, TenantId::new(tid), images.collection.clone()); + let Some(overlay) = self.txn_overlays.get(&txn_id) else { + return Ok(()); + }; + let later = overlay.slots_touched_since(marker, &coll_key); + let rewritten = |surrogate: Surrogate| { + later + .iter() + .find(|slot| slot.surrogate == surrogate.as_u32()) + }; + if !images.rows.iter().any(|(_, s)| rewritten(*s).is_some()) { + return Ok(()); + } + let at = RowLocation { + engine: images.engine, + database_id: database_id.as_u64(), + tid, + collection: &images.collection, + }; + let mut rows = Vec::with_capacity(images.rows.len()); + for (identity, surrogate) in &images.rows { + // The row as the `RETURNING` plan left it: the later plan's prior + // value when that plan touched it, else the overlay's value now. + let staged = match rewritten(*surrogate) { + Some(slot) => slot.before, + None => overlay.get(&coll_key, surrogate.as_u32()), + }; + if let Some(Staged::Put(body)) = staged { + rows.push(StoredRow { + identity: identity.clone(), + surrogate: *surrogate, + bytes: staged_row_bytes(images.engine, body)?, + }); + } + } + let payload = self.calvin_render_rows(&at, &images.spec, &images.rls_filters, &rows)?; + *reply = CalvinReply::Rows(payload); + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use nodedb_physical::physical_plan::{ + ColumnarInsertIntent, ColumnarOp, CrdtOp, CrdtWriteVerb, DocumentOp, KvOp, PhysicalPlan, + ReturningColumns, ReturningItem, ReturningSpec, TimeseriesOp, UpdateValue, VectorOp, + }; + use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarSchema}; + use nodedb_types::{DatabaseId, QualifiedCollection, RlsWriteCheck, Surrogate, Value}; + + use crate::bridge::envelope::{Response, Status}; + use crate::data::executor::core_loop::CoreLoop; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::response_codec::RowsPayload; + + const TID: u64 = 1; + + fn returning(columns: &[&str]) -> Option { + Some(ReturningSpec { + columns: ReturningColumns::Named( + columns + .iter() + .map(|name| ReturningItem { + name: (*name).to_string(), + alias: None, + }) + .collect(), + ), + }) + } + + fn coll(name: &str) -> QualifiedCollection { + QualifiedCollection::new(DatabaseId::DEFAULT, name) + } + + /// A row body as the planner encodes one: a standard MessagePack map. + /// A KV value is stored and read back in exactly this form. + fn object(fields: &[(&str, Value)]) -> Vec { + let map: std::collections::HashMap = fields + .iter() + .map(|(k, v)| ((*k).to_string(), v.clone())) + .collect(); + nodedb_types::value_to_msgpack(&Value::Object(map)).expect("encode object") + } + + fn text(value: &str) -> Value { + Value::String(value.to_string()) + } + + /// Commit `plans` as one Calvin transaction on a fresh core prepared by + /// `setup`, and return the flush's reply. + fn commit(plans: &[PhysicalPlan], setup: impl FnOnce(&mut CoreLoop)) -> Response { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + setup(&mut core); + core.calvin_commit_for_test(&make_default_task(), TID, plans, 1, 100) + } + + /// The first returned column of every row in `response`. + fn returned(response: &Response) -> Vec { + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + let payload: RowsPayload = + zerompk::from_msgpack(response.payload.as_bytes()).expect("RETURNING rows"); + payload + .rows + .iter() + .map(|row| { + row.first() + .map(|cell| cell.0.clone()) + .unwrap_or(Value::Null) + }) + .collect() + } + + fn doc_insert(id: &str, surrogate: u32, a: &str, if_absent: bool, ret: bool) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointInsert { + collection: coll("orders"), + document_id: id.to_string(), + value: object(&[("a", text(a))]), + if_absent, + surrogate: Surrogate::new(surrogate), + returning: if ret { returning(&["a"]) } else { None }, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }) + } + + fn doc_put(id: &str, surrogate: u32, a: &str) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointPut { + collection: coll("orders"), + document_id: id.to_string(), + value: object(&[("a", text(a))]), + surrogate: Surrogate::new(surrogate), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }) + } + + fn doc_update(id: &str, surrogate: u32, a: &str) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointUpdate { + collection: coll("orders"), + document_id: id.to_string(), + surrogate: Surrogate::new(surrogate), + pk_bytes: Vec::new(), + updates: vec![( + "a".to_string(), + UpdateValue::Literal(nodedb_types::value_to_msgpack(&text(a)).expect("literal")), + )], + returning: returning(&["a"]), + rls_filters: Vec::new(), + rls_write_check: RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }) + } + + fn doc_delete(id: &str, surrogate: u32) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointDelete { + collection: coll("orders"), + document_id: id.to_string(), + surrogate: Surrogate::new(surrogate), + pk_bytes: Vec::new(), + returning: returning(&["a"]), + rls_filters: Vec::new(), + rls_write_check: RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + }) + } + + /// A point insert answers the row it stored. A later `ON CONFLICT DO + /// NOTHING` insert of the same key stages nothing and answers no rows. + #[test] + fn a_point_insert_answers_its_row_and_a_skipped_insert_answers_none() { + let inserted = commit(&[doc_insert("o1", 7, "kept", false, true)], |_| {}); + assert_eq!(returned(&inserted), vec![text("kept")]); + + let skipped = commit( + &[ + doc_insert("o1", 7, "kept", false, false), + doc_insert("o1", 7, "skipped", true, true), + ], + |_| {}, + ); + assert!(returned(&skipped).is_empty()); + } + + /// An update's RETURNING reports the row an earlier plan of the same + /// statement inserted, with the update applied. + #[test] + fn an_update_returns_the_post_image_of_the_statements_own_insert() { + let response = commit( + &[ + doc_insert("o1", 7, "first", false, false), + doc_update("o1", 7, "second"), + ], + |_| {}, + ); + assert_eq!(returned(&response), vec![text("second")]); + } + + /// A delete's RETURNING reports the row an earlier plan of the same + /// statement inserted. + #[test] + fn a_delete_returns_the_row_the_statement_inserted_before_it() { + let response = commit( + &[ + doc_insert("o1", 7, "first", false, false), + doc_delete("o1", 7), + ], + |_| {}, + ); + assert_eq!(returned(&response), vec![text("first")]); + } + + /// A bulk delete's RETURNING reports a predicted row as an earlier plan + /// of the same statement updated it. + #[test] + fn a_bulk_delete_returns_the_row_as_the_statement_updated_it() { + let mut update = doc_update("1", 1, "changed"); + if let PhysicalPlan::Document(DocumentOp::PointUpdate { returning, .. }) = &mut update { + *returning = None; + } + let bulk_delete = PhysicalPlan::Document(DocumentOp::BulkDelete { + collection: coll("orders"), + filters: Vec::new(), + returning: returning(&["a"]), + ollp_predicted_surrogates: Some(vec![1]), + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }); + let response = commit(&[update, bulk_delete], |core| { + let body = crate::data::executor::doc_format::canonicalize_document_for_storage( + &object(&[("a", text("seeded"))]), + ); + core.sparse + .put( + DatabaseId::DEFAULT.as_u64(), + TID, + "orders", + &nodedb_types::StorageKey::for_surrogate(Surrogate::new(1)), + &body, + ) + .expect("seed row"); + }); + assert_eq!(returned(&response), vec![text("changed")]); + } + + /// A later plan without RETURNING keeps the rows an earlier RETURNING + /// plan answered. + #[test] + fn a_later_count_only_plan_keeps_the_returning_rows() { + let response = commit( + &[ + doc_insert("o1", 7, "kept", false, true), + doc_insert("o2", 8, "other", false, false), + ], + |_| {}, + ); + assert_eq!(returned(&response), vec![text("kept")]); + } + + /// A later plan that rewrites a returned row leaves the reply at the row + /// the RETURNING plan wrote. + #[test] + fn a_later_rewrite_of_a_returned_row_keeps_the_returning_plans_image() { + let response = commit( + &[ + doc_insert("o1", 7, "first", false, true), + doc_put("o1", 7, "second"), + ], + |_| {}, + ); + assert_eq!(returned(&response), vec![text("first")]); + } + + /// A document batch insert answers every row it stored, in plan order. + #[test] + fn a_batch_insert_answers_every_row() { + let plan = PhysicalPlan::Document(DocumentOp::BatchInsert { + collection: coll("orders"), + documents: vec![ + ("o1".to_string(), object(&[("a", text("one"))])), + ("o2".to_string(), object(&[("a", text("two"))])), + ], + surrogates: vec![Surrogate::new(7), Surrogate::new(8)], + returning: returning(&["a"]), + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + let response = commit(&[plan], |_| {}); + assert_eq!(returned(&response), vec![text("one"), text("two")]); + } + + fn kv_put(value: &str) -> PhysicalPlan { + PhysicalPlan::Kv(KvOp::Put { + collection: coll("cache"), + key: b"k".to_vec(), + value: object(&[("v", text(value))]), + ttl_ms: 0, + surrogate: Surrogate::new(5), + returning: returning(&["v"]), + rls_filters: Vec::new(), + }) + } + + #[test] + fn a_kv_put_answers_the_value_it_stored() { + assert_eq!( + returned(&commit(&[kv_put("put")], |_| {})), + vec![text("put")] + ); + } + + /// A KV delete answers the row it removed, as an earlier plan of the + /// same statement left it. + #[test] + fn a_kv_delete_answers_the_row_the_statement_put_before_it() { + let delete = PhysicalPlan::Kv(KvOp::Delete { + collection: coll("cache"), + keys: vec![b"k".to_vec()], + rls_write_check: RlsWriteCheck::NoPolicyApplies, + returning: returning(&["v"]), + rls_filters: Vec::new(), + }); + let response = commit(&[kv_put("put"), delete], |_| {}); + assert_eq!(returned(&response), vec![text("put")]); + } + + /// A KV row live at the epoch instant and expired by the wall clock is + /// live for the stage and the reply. Every replica reads at the epoch + /// instant, whenever its core runs the transaction. + #[test] + fn a_kv_delete_reads_liveness_at_the_epoch_instant() { + let now = crate::engine::kv::current_ms(); + let delete = PhysicalPlan::Kv(KvOp::Delete { + collection: coll("cache"), + keys: vec![b"k".to_vec()], + rls_write_check: RlsWriteCheck::NoPolicyApplies, + returning: returning(&["v"]), + rls_filters: Vec::new(), + }); + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + // Expires 5 s before the wall clock, 3 s after the epoch instant. + let value = object(&[("v", text("old"))]); + core.kv_engine.put(crate::engine::kv::KvPutParams { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: TID, + collection: "cache", + key: b"k", + value: &value, + ttl_ms: 5_000, + now_ms: now - 10_000, + surrogate: Surrogate::new(5), + }); + + let epoch_ms = i64::try_from(now - 8_000).expect("epoch instant fits i64"); + let response = + core.calvin_commit_at_for_test(&make_default_task(), TID, &[delete], 1, epoch_ms, 100); + + assert_eq!(returned(&response), vec![text("old")]); + } + + #[test] + fn a_vector_primary_insert_answers_its_payload() { + let plan = PhysicalPlan::Vector(VectorOp::DirectInsert { + collection: coll("vp"), + field: "vec".into(), + surrogate: Surrogate::new(61), + pk_bytes: b"r".to_vec(), + vector: vec![1.0, 0.0], + payload: zerompk::to_msgpack_vec(&std::collections::HashMap::from([( + "label".to_string(), + text("tagged"), + )])) + .expect("encode payload"), + quantization: nodedb_types::VectorQuantization::None, + storage_dtype: nodedb_types::VectorStorageDtype::F32, + payload_indexes: Vec::new(), + returning: returning(&["label"]), + rls_filters: Vec::new(), + }); + assert_eq!(returned(&commit(&[plan], |_| {})), vec![text("tagged")]); + } + + #[test] + fn a_crdt_upsert_answers_the_row_the_install_materialized() { + let plan = PhysicalPlan::Crdt(CrdtOp::DocUpsert { + collection: coll("tasks"), + document_id: "t1".to_string(), + fields_json: r#"{"title":"kept"}"#.to_string(), + surrogate: Surrogate::new(71), + partial: false, + verb: CrdtWriteVerb::Insert, + returning: returning(&["title"]), + rls_filters: Vec::new(), + }); + assert_eq!(returned(&commit(&[plan], |_| {})), vec![text("kept")]); + } + + #[test] + fn a_columnar_insert_answers_its_row() { + let schema = ColumnarSchema::new(vec![ + ColumnDef::required("id", ColumnType::String).with_primary_key(), + ColumnDef::nullable("note", ColumnType::String), + ]) + .expect("valid columnar schema"); + let row = Value::Object(std::collections::HashMap::from([ + ("id".to_string(), text("c1")), + ("note".to_string(), text("noted")), + ])); + let plan = PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: coll("metrics_col"), + payload: nodedb_types::value_to_msgpack(&Value::Array(vec![row])) + .expect("encode columnar payload"), + format: "msgpack".to_string(), + intent: ColumnarInsertIntent::Insert, + on_conflict_updates: Vec::new(), + surrogates: vec![Surrogate::new(81)], + schema_bytes: zerompk::to_msgpack_vec(&schema).expect("encode schema"), + provenance: None, + wal_lsn: None, + rls_write_check: RlsWriteCheck::NoPolicyApplies, + returning: returning(&["note"]), + rls_filters: Vec::new(), + }); + assert_eq!(returned(&commit(&[plan], |_| {})), vec![text("noted")]); + } + + #[test] + fn a_timeseries_ingest_answers_its_rows() { + let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: coll("cpu"), + payload: zerompk::to_msgpack_vec(&vec!["cpu value=7i 1000000000"]) + .expect("canonical ILP payload"), + format: "ilp-msgpack".to_owned(), + wal_lsn: None, + surrogates: vec![Surrogate::new(91)], + provenance: None, + rls_write_check: RlsWriteCheck::NoPolicyApplies, + returning: returning(&["value"]), + rls_filters: Vec::new(), + }); + assert_eq!(returned(&commit(&[plan], |_| {})), vec![Value::Integer(7)]); + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin_reply/target.rs b/nodedb/src/data/executor/handlers/control/calvin_reply/target.rs new file mode 100644 index 000000000..d9a9bac95 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/calvin_reply/target.rs @@ -0,0 +1,320 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What a staged Calvin plan's `RETURNING` clause reports: which +//! collection, which engine's row shape, and whether each row is its image +//! before the plan or after it. + +use nodedb_physical::physical_plan::{ + ColumnarOp, CrdtOp, DocumentOp, KvOp, PhysicalPlan, ReturningSpec, TimeseriesOp, VectorOp, +}; + +/// The row shape an engine's `RETURNING` rows take. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::data::executor) enum RowEngine { + /// Stored document bodies in the sparse store. + Document, + /// CRDT rows: MessagePack bodies in the sparse store in either storage + /// mode. + Crdt, + /// KV values, keyed by their raw key. + Kv, + /// Vector-primary payload sidecars. + Vector, + /// Columnar rows in schema order. + Columnar, + /// Timeseries rows as the raw-scan row emitter renders them. + Timeseries, +} + +/// Which image of a written row `RETURNING` reports. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum ReplyImage { + /// The row as it was before the plan: a delete reports what it removed. + Before, + /// The row as the plan left it. + After, +} + +/// A `RETURNING` plan's reply target. +pub(super) struct ReturningTarget<'a> { + pub spec: &'a ReturningSpec, + pub rls_filters: &'a [u8], + pub collection: &'a str, + pub engine: RowEngine, + pub image: ReplyImage, + /// The vector field a vector-primary delete names. A removed row whose + /// node is bound but whose sidecar is absent reports an empty sidecar. + pub vector_field: Option<&'a str>, +} + +impl<'a> ReturningTarget<'a> { + fn new( + spec: &'a ReturningSpec, + rls_filters: &'a [u8], + collection: &'a str, + engine: RowEngine, + image: ReplyImage, + ) -> Self { + Self { + spec, + rls_filters, + collection, + engine, + image, + vector_field: None, + } + } +} + +/// The reply target of `plan`, or `None` when it carries no `RETURNING`. +/// +/// `UpdateFromJoin` and `Merge` are never Calvin plans: the Control Plane +/// resolves them into point writes, and staging refuses one that arrives. +pub(super) fn returning_target(plan: &PhysicalPlan) -> Option> { + use ReplyImage::{After, Before}; + match plan { + PhysicalPlan::Document(op) => match op { + DocumentOp::PointPut { + collection, + returning: Some(spec), + rls_filters, + .. + } + | DocumentOp::PointInsert { + collection, + returning: Some(spec), + rls_filters, + .. + } + | DocumentOp::PointUpdate { + collection, + returning: Some(spec), + rls_filters, + .. + } + | DocumentOp::BatchInsert { + collection, + returning: Some(spec), + rls_filters, + .. + } + | DocumentOp::Upsert { + collection, + returning: Some(spec), + rls_filters, + .. + } + | DocumentOp::BulkUpdate { + collection, + returning: Some(spec), + rls_filters, + .. + } => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Document, + After, + )), + DocumentOp::PointDelete { + collection, + returning: Some(spec), + rls_filters, + .. + } + | DocumentOp::BulkDelete { + collection, + returning: Some(spec), + rls_filters, + .. + } => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Document, + Before, + )), + _ => None, + }, + PhysicalPlan::Kv(op) => match op { + KvOp::Put { + collection, + returning: Some(spec), + rls_filters, + .. + } + | KvOp::Insert { + collection, + returning: Some(spec), + rls_filters, + .. + } + | KvOp::InsertIfAbsent { + collection, + returning: Some(spec), + rls_filters, + .. + } + | KvOp::InsertOnConflictUpdate { + collection, + returning: Some(spec), + rls_filters, + .. + } + | KvOp::BatchPut { + collection, + returning: Some(spec), + rls_filters, + .. + } + | KvOp::FieldSet { + collection, + returning: Some(spec), + rls_filters, + .. + } + | KvOp::PredicateUpdate { + collection, + returning: Some(spec), + rls_filters, + .. + } => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Kv, + After, + )), + KvOp::Delete { + collection, + returning: Some(spec), + rls_filters, + .. + } + | KvOp::PredicateDelete { + collection, + returning: Some(spec), + rls_filters, + .. + } => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Kv, + Before, + )), + _ => None, + }, + PhysicalPlan::Vector(op) => match op { + VectorOp::DirectUpsert { + collection, + returning: Some(spec), + rls_filters, + .. + } + | VectorOp::DirectInsert { + collection, + returning: Some(spec), + rls_filters, + .. + } + | VectorOp::DirectInsertIfAbsent { + collection, + returning: Some(spec), + rls_filters, + .. + } + | VectorOp::DirectUpdate { + collection, + returning: Some(spec), + rls_filters, + .. + } => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Vector, + After, + )), + VectorOp::DirectDelete { + collection, + field, + returning: Some(spec), + rls_filters, + .. + } => Some(ReturningTarget { + vector_field: Some(field.as_str()), + ..ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Vector, + Before, + ) + }), + _ => None, + }, + PhysicalPlan::Crdt(op) => match op { + CrdtOp::DocUpsert { + collection, + returning: Some(spec), + rls_filters, + .. + } => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Crdt, + After, + )), + CrdtOp::DocDelete { + collection, + returning: Some(spec), + rls_filters, + .. + } => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Crdt, + Before, + )), + _ => None, + }, + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection, + returning: Some(spec), + rls_filters, + .. + }) => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Columnar, + After, + )), + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection, + returning: Some(spec), + rls_filters, + .. + }) => Some(ReturningTarget::new( + spec, + rls_filters, + collection.as_str(), + RowEngine::Timeseries, + After, + )), + // No other plan of these engines carries `RETURNING`. + PhysicalPlan::Columnar(_) + | PhysicalPlan::Timeseries(_) + | PhysicalPlan::Graph(_) + | PhysicalPlan::Text(_) + | PhysicalPlan::Spatial(_) + | PhysicalPlan::Query(_) + | PhysicalPlan::Meta(_) + | PhysicalPlan::Array(_) + | PhysicalPlan::ClusterArray(_) + | PhysicalPlan::ClusterEvent(_) => None, + } +} diff --git a/nodedb/src/data/executor/handlers/control/calvin_resolve.rs b/nodedb/src/data/executor/handlers/control/calvin_resolve.rs index fc8a97bbe..1bfab99eb 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_resolve.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_resolve.rs @@ -19,7 +19,6 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; use super::calvin_txn_id::calvin_synthetic_txn_id; -use crate::data::executor::handlers::transaction::resolve::StagedWrites; impl CoreLoop { /// Resolve the Calvin transaction staged under `(epoch, position)` on @@ -53,37 +52,37 @@ impl CoreLoop { // `commit_pending` so the `&mut self` resolve call — which assigns // bitemporal stamps into the overlay — does not overlap the immutable // borrow of the pending buffer. - let (tid, plans, epoch_system_ms) = - match self.commit_pending.get(&(epoch, position, vshard_id)) { - Some(pending) => ( - pending.tenant_id.as_u64(), - pending.plans.clone(), - pending.epoch_system_ms, - ), - None => { - return self.response_error( - task, - crate::Error::Internal { - detail: format!( - "calvin resolve: no staged commit for epoch={epoch} \ + let (tid, plans, epoch_system_ms) = match self + .calvin + .commit_pending + .get(&(epoch, position, vshard_id)) + { + Some(pending) => ( + pending.tenant_id.as_u64(), + pending.plans.clone(), + pending.epoch_system_ms, + ), + None => { + return self.response_error( + task, + crate::Error::Internal { + detail: format!( + "calvin resolve: no staged commit for epoch={epoch} \ position={position} vshard={vshard_id} (must be staged via \ CalvinExecuteStatic before CalvinResolve)" - ), - }, - ); - } - }; + ), + }, + ); + } + }; // Restore the epoch's deterministic time anchor around resolve so the // bitemporal stamps `execute_resolve_txn` assigns are identical across - // replicas (mirrors `execute_calvin_flush`). `CalvinFlush` reads these - // stamps back from the overlay, so redo and base install agree. + // replicas. The redo record carries them, so every install writes the + // same version key. let prev_epoch_ms = self.epoch_system_ms; self.epoch_system_ms = Some(epoch_system_ms); - // Calvin stages no vector-primary direct write, so those resolve from - // their plan nodes. - let resp = - self.execute_resolve_staged(task, tid, synthetic_txn_id, &plans, StagedWrites::Calvin); + let resp = self.execute_resolve_txn(task, tid, synthetic_txn_id, &plans); self.epoch_system_ms = prev_epoch_ms; resp } diff --git a/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs b/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs index 77582cfd1..ba9fa5731 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_apply/gated.rs @@ -133,6 +133,22 @@ impl CoreLoop { return self.sync_ack_response(task, status, current_hwm); } + // A delta must name the one document it writes. With no target the + // engine admits rows of any document and installs them before the + // write set is known, so the refusal comes before any state change. + if document_id.is_empty() { + return self.sync_reject_response( + task, + ViolationType::ConstraintViolation { + detail: format!( + "a CRDT sync delta into {collection} must name the one document its \ + delta writes; nothing was applied" + ), + }, + prov, + ); + } + // Borrow the engine in a nested block so the &mut borrow is dropped // before sync_commit takes &mut self for sync_hwm. // @@ -223,34 +239,19 @@ impl CoreLoop { ); GateDisposition::Retryable } + // The engine refuses a delta that writes any row but the named + // document as malformed, before it installs. A clean apply wrote + // that document alone. GateOutcome::Applied(ValidatedApplyOutcome::Clean { - write_set, + write_set: _, imported_ops, }) => { - // Enforce the one-document-per-delta sync contract. A delta - // that coalesced multiple documents (or targeted a synthetic - // frame id that matches no written row) cannot be materialized - // past its single surrogate; reject it loudly so the client - // re-pushes one delta per document instead of silently losing - // rows. - match Self::single_document_write_set(collection, document_id, &write_set) { - Err(detail) => { - imported_authoritative = true; - self.checkpoint_coordinator.mark_dirty("crdt", 1); - warn!( - core = self.core_id, - %collection, - %document_id, - detail = %detail, - "crdt sync apply rejected: multi-document delta violates one-document-per-delta contract" - ); - GateDisposition::Terminal(ViolationType::ConstraintViolation { detail }) - } + match imported_ops { // The delta contributed no operations: every one it carried // was already in this document, so nothing was written and // nothing is dirty. Reporting `Applied` here is what let a // peer-id collision retire a write that was discarded. - Ok(()) if imported_ops == 0 => { + 0 => { if !declared_row_present { // The delta imported nothing AND the row it declared // does not exist. A replayed delete looks like this @@ -272,28 +273,27 @@ impl CoreLoop { } GateDisposition::Deduplicated } - Ok(()) => { + _ => { imported_authoritative = true; self.checkpoint_coordinator.mark_dirty("crdt", 1); GateDisposition::Applied } } } + // The candidate was discarded, so authoritative state did not + // move. The record is cancelled and replay never reaches this + // rejection, so the dead-letter entry is stored now. GateOutcome::Applied(ValidatedApplyOutcome::Rejected(vt)) => { - imported_authoritative = true; - self.checkpoint_coordinator.mark_dirty("crdt", 1); - // Replaying this record binds to the same log position, so - // the stored entry stays the only one. if let Err(error) = self.store_crdt_dead_letter(task.request.database_id, tenant_id, task.wal_lsn()) { - warn!( + tracing::error!( core = self.core_id, %collection, %document_id, %error, "crdt sync apply rejected a delta, and its dead-letter entry could not \ - be stored; replay of its record rebuilds it" + be stored; it stays in memory only" ); } GateDisposition::Terminal(vt) @@ -367,8 +367,7 @@ impl CoreLoop { GateDisposition::Terminal(violation) => { // Permanently refused: it will never succeed on a re-push, so // holding the stream for it buys nothing. - self.sync_commit(prov); - self.sync_reject_response(task, violation, prov.seq) + self.sync_reject_response(task, violation, prov) } GateDisposition::Deduplicated => { // The operations are in the document, so the sender is free to @@ -385,3 +384,95 @@ impl CoreLoop { } } } + +#[cfg(test)] +mod tests { + use crate::bridge::envelope::{ErrorCode, Status}; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use nodedb_types::sync::wire::SyncProvenance; + + use super::super::local::tests::{ + dead_letters, install_unique_email, task_at, user_delta, users_params, + }; + + fn provenance(seq: u64) -> SyncProvenance { + SyncProvenance { + producer_id: 9, + epoch: 2, + stream_id: 1, + seq, + } + } + + /// A peer delta a constraint refuses is an error, so its record is + /// cancelled. The stream's mark still advances, the row stays absent, and + /// the dead-letter entry names the producing peer. + #[test] + fn a_terminal_refusal_is_an_error_that_advances_the_mark() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _request_tx, _response_rx) = make_core_with_dir(dir.path()); + let seed = task_at(10); + install_unique_email(&mut core, &seed); + let first = user_delta(2, "a", "x@y.com"); + assert_eq!( + core.execute_crdt_apply(&seed, users_params("a", &first)) + .status, + Status::Ok + ); + + let second = user_delta(3, "b", "x@y.com"); + let prov = provenance(1); + let mut params = users_params("b", &second); + params.peer_id = 3; + params.provenance = Some(&prov); + let response = core.execute_crdt_apply(&task_at(11), params); + + assert_eq!(response.status, Status::Error); + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::SyncRejected { applied_seq: 1, provenance, .. }) + if *provenance == prov + ), + "got {:?}", + response.error_code + ); + assert_eq!(core.sync_hwm_value(9, 1), 1, "the mark advanced"); + let task = task_at(11); + assert!( + core.crdt_engines + .get(&(task.request.database_id, task.request.tenant_id)) + .is_some_and(|engine| !engine.row_exists("users", "b")) + ); + let entries = dead_letters(&mut core); + assert_eq!(entries.len(), 1); + assert_eq!(entries[0].peer_id, 3); + assert_eq!(entries[0].source_lsn, Some(11)); + } + + /// A delta with no target document is refused before anything installs. + #[test] + fn a_delta_without_a_target_document_is_refused_before_it_installs() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _request_tx, _response_rx) = make_core_with_dir(dir.path()); + let delta = user_delta(3, "b", "x@y.com"); + let prov = provenance(1); + let mut params = users_params("", &delta); + params.provenance = Some(&prov); + + let response = core.execute_crdt_apply(&task_at(11), params); + + assert!(matches!( + response.error_code.as_deref(), + Some(ErrorCode::SyncRejected { .. }) + )); + let task = task_at(11); + assert!( + !core + .crdt_engines + .get(&(task.request.database_id, task.request.tenant_id)) + .is_some_and(|engine| engine.row_exists("users", "b")), + "nothing installed" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs b/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs index 1e615c6e8..f6cf69e48 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_apply/local.rs @@ -214,7 +214,7 @@ impl CoreLoop { } #[cfg(test)] -mod tests { +pub(in crate::data::executor::handlers::control::crdt_apply) mod tests { use loro::LoroValue; use nodedb_types::Surrogate; @@ -222,7 +222,10 @@ mod tests { use crate::bridge::envelope::Status; use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; - fn params<'a>(document_id: &'a str, delta: &'a [u8]) -> CrdtApplyParams<'a> { + pub(in crate::data::executor::handlers::control::crdt_apply) fn params<'a>( + document_id: &'a str, + delta: &'a [u8], + ) -> CrdtApplyParams<'a> { CrdtApplyParams { collection: "docs", document_id, @@ -304,14 +307,21 @@ mod tests { assert!(!imported, "a refused delta must not reach the CRDT state"); } - fn users_params<'a>(document_id: &'a str, delta: &'a [u8]) -> CrdtApplyParams<'a> { + pub(in crate::data::executor::handlers::control::crdt_apply) fn users_params<'a>( + document_id: &'a str, + delta: &'a [u8], + ) -> CrdtApplyParams<'a> { CrdtApplyParams { collection: "users", ..params(document_id, delta) } } - fn user_delta(peer: u64, row_id: &str, email: &str) -> Vec { + pub(in crate::data::executor::handlers::control::crdt_apply) fn user_delta( + peer: u64, + row_id: &str, + email: &str, + ) -> Vec { let source = nodedb_crdt::CrdtState::new(peer).expect("source state"); source .upsert( @@ -323,7 +333,9 @@ mod tests { source.export_snapshot().expect("source snapshot") } - fn task_at(lsn: u64) -> ExecutionTask { + pub(in crate::data::executor::handlers::control::crdt_apply) fn task_at( + lsn: u64, + ) -> ExecutionTask { ExecutionTask::with_wal_lsn( make_default_task().request, Some(crate::types::Lsn::new(lsn)), @@ -332,7 +344,10 @@ mod tests { /// Install a UNIQUE email constraint under a strict policy, so a clash /// is a rejection. - fn install_unique_email(core: &mut CoreLoop, task: &ExecutionTask) { + pub(in crate::data::executor::handlers::control::crdt_apply) fn install_unique_email( + core: &mut CoreLoop, + task: &ExecutionTask, + ) { let engine = core .get_crdt_engine(task.request.database_id, task.request.tenant_id) .expect("engine"); @@ -364,7 +379,9 @@ mod tests { core.apply_crdt_local(&task_at(lsn), users_params("b", &second)) } - fn dead_letters(core: &mut CoreLoop) -> Vec { + pub(in crate::data::executor::handlers::control::crdt_apply) fn dead_letters( + core: &mut CoreLoop, + ) -> Vec { let task = make_default_task(); core.get_crdt_engine(task.request.database_id, task.request.tenant_id) .expect("engine") diff --git a/nodedb/src/data/executor/handlers/control/crdt_doc.rs b/nodedb/src/data/executor/handlers/control/crdt_doc.rs index 2e46b2e92..490e39094 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_doc.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_doc.rs @@ -216,7 +216,6 @@ impl CoreLoop { let tid = tenant_id.as_u64(); let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); // The sparse-store removal and its index cascades run in one write txn // this handler owns: on any failure it is dropped un-committed and none // of them land. @@ -237,7 +236,8 @@ impl CoreLoop { database_id: task.request.database_id.as_u64(), tid, collection, - document_id: row_key.as_str(), + // The graph cascade keys nodes by the client key. + document_id, surrogate, user_roles: &task.request.user_roles, enforce: false, @@ -273,7 +273,6 @@ impl CoreLoop { tid, collection, storage_key, - identity: RowIdentity::from_user_key(document_id), }, outcome, ); diff --git a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs index e01ae5e4e..6d69ce49c 100644 --- a/nodedb/src/data/executor/handlers/control/crdt_materialize.rs +++ b/nodedb/src/data/executor/handlers/control/crdt_materialize.rs @@ -194,7 +194,6 @@ impl CoreLoop { tid, collection, storage_key, - identity: RowIdentity::from_user_key(document_id), }, outcome, None, diff --git a/nodedb/src/data/executor/handlers/control/mod.rs b/nodedb/src/data/executor/handlers/control/mod.rs index 9157de584..e036991e7 100644 --- a/nodedb/src/data/executor/handlers/control/mod.rs +++ b/nodedb/src/data/executor/handlers/control/mod.rs @@ -7,6 +7,7 @@ mod calvin_active_verify; mod calvin_overlay_stage; mod calvin_overlay_stage_bulk; mod calvin_passive_read; +pub mod calvin_reply; mod calvin_resolve; mod calvin_txn_id; mod checkpoint_crdt; @@ -25,3 +26,5 @@ pub mod reindex; mod reindex_apply; pub mod snapshot; pub mod synonym_group; + +pub(in crate::data::executor) use calvin_txn_id::calvin_synthetic_txn_id; diff --git a/nodedb/src/data/executor/handlers/control/reindex.rs b/nodedb/src/data/executor/handlers/control/reindex.rs index 94ce7253a..793df132e 100644 --- a/nodedb/src/data/executor/handlers/control/reindex.rs +++ b/nodedb/src/data/executor/handlers/control/reindex.rs @@ -63,6 +63,7 @@ impl CoreLoop { // Reject duplicate concurrent rebuild for same collection. if self + .maintenance .pending_reindex .iter() .any(|p| p.tenant_id == tenant_id && p.collection_key == collection_key) @@ -112,7 +113,7 @@ impl CoreLoop { // Collect completed and failed entries, leaving only still-running ones. // We must separate the poll loop from the apply loop to satisfy the borrow checker: // apply_* functions take &mut self, which conflicts with holding a reference into - // self.pending_reindex at the same time. + // self.maintenance.pending_reindex at the same time. enum Outcome { Done { database_id: nodedb_types::DatabaseId, @@ -129,7 +130,7 @@ impl CoreLoop { let mut outcomes: Vec = Vec::new(); let mut still_running: Vec = Vec::new(); - for pending in self.pending_reindex.drain(..) { + for pending in self.maintenance.pending_reindex.drain(..) { match pending.rx.try_recv() { Ok(Ok(output)) => outcomes.push(Outcome::Done { database_id: pending.database_id, @@ -148,7 +149,7 @@ impl CoreLoop { Err(mpsc::TryRecvError::Empty) => still_running.push(pending), } } - self.pending_reindex = still_running; + self.maintenance.pending_reindex = still_running; for outcome in outcomes { match outcome { @@ -291,7 +292,7 @@ impl CoreLoop { let _ = tx.send(rebuild_hnsw_thread(vectors, dim, params)); }); - self.pending_reindex.push(PendingReindex { + self.maintenance.pending_reindex.push(PendingReindex { database_id: db, tenant_id, collection_key: key.2, @@ -374,7 +375,7 @@ impl CoreLoop { let _ = tx.send(rebuild_fts_thread(input)); }); - self.pending_reindex.push(PendingReindex { + self.maintenance.pending_reindex.push(PendingReindex { database_id, tenant_id, collection_key: collection_key.to_string(), @@ -414,7 +415,7 @@ impl CoreLoop { let _ = tx.send(rebuild_csr_thread(snapshot_bytes, memory)); }); - self.pending_reindex.push(PendingReindex { + self.maintenance.pending_reindex.push(PendingReindex { database_id, tenant_id, collection_key: collection_key.to_string(), diff --git a/nodedb/src/data/executor/handlers/kv/dispatch.rs b/nodedb/src/data/executor/handlers/kv/dispatch.rs index c7ad06c4d..0825d9de5 100644 --- a/nodedb/src/data/executor/handlers/kv/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/dispatch.rs @@ -403,6 +403,22 @@ impl CoreLoop { index_name, primary_key, } => self.execute_kv_sorted_index_score(task, did, tid, index_name, primary_key), + KvOp::SortedIndexTxnRead { + collection, + index_name, + pending, + read, + } => self.execute_kv_sorted_index_txn_read( + task, + super::sorted_txn::SortedIndexTxnReadParams { + did, + tid, + collection: collection.as_str(), + index_name, + pending: pending.as_ref(), + read, + }, + ), KvOp::Transfer { .. } => self.dispatch_kv_transfer(task, did, tid, op), KvOp::TransferItem { .. } => self.dispatch_kv_transfer_item(task, did, tid, op), KvOp::MaterializeScan { diff --git a/nodedb/src/data/executor/handlers/kv/field_compute.rs b/nodedb/src/data/executor/handlers/kv/field_compute.rs index 539382f32..182880aec 100644 --- a/nodedb/src/data/executor/handlers/kv/field_compute.rs +++ b/nodedb/src/data/executor/handlers/kv/field_compute.rs @@ -4,7 +4,7 @@ //! the autocommit handler (`field.rs`), the predicate update, the transaction //! resolver, in-transaction staging, and WAL replay, so a staged value and its //! COMMIT-time durable replay are always computed by the exact same code — -//! mirrors the `engine_atomic_compute` / `stage_kv_atomic` split for +//! mirrors the `nodedb_physical::kv_atomic::compute` / `stage_kv_atomic` split for //! `Incr`/`Cas`/etc. use nodedb_query::msgpack_scan::{KvBodyError, KvBodyShape, kv_body_to_row, row_to_kv_body}; diff --git a/nodedb/src/data/executor/handlers/kv/mod.rs b/nodedb/src/data/executor/handlers/kv/mod.rs index 0c021891b..1215185c6 100644 --- a/nodedb/src/data/executor/handlers/kv/mod.rs +++ b/nodedb/src/data/executor/handlers/kv/mod.rs @@ -18,6 +18,7 @@ pub(in crate::data::executor) mod rls; mod scan; pub(in crate::data::executor) mod sorted; pub(in crate::data::executor) mod sorted_index_compute; +mod sorted_txn; pub(in crate::data::executor) mod transfer; pub(in crate::data::executor) mod ttl; diff --git a/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs b/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs index 9c0e6ce05..c547e6e86 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/atomic_ops.rs @@ -1,10 +1,11 @@ // SPDX-License-Identifier: BUSL-1.1 //! Resolvers for the KV atomics: `Incr`, `IncrFloat`, `Cas`, `GetSet`. Each -//! post-image comes from `engine_atomic_compute`, the same pure functions +//! post-image comes from `nodedb_physical::kv_atomic::compute`, the same pure functions //! `KvEngine::{incr, incr_float, cas, getset}` call — recomputing here would //! let resolve and apply disagree. +use nodedb_physical::kv_atomic::compute; use nodedb_physical::physical_plan::{KvCounterShape, KvResolveOutcome}; use super::context::{ResolveResult, ResolvedPut, expiry_from_ttl, one, put_mutation}; @@ -15,8 +16,6 @@ use crate::data::executor::handlers::kv::atomic::{ }; use crate::data::executor::handlers::kv::rls::admit_kv_row; use crate::data::executor::response_codec; -use crate::engine::kv::current_ms; -use crate::engine::kv::engine_atomic_compute as compute; /// Render a stored body for the `current_value` / `old_value` slot of an /// atomic's reply, exactly as the live handlers do. @@ -49,7 +48,7 @@ impl CoreLoop { let now_ms = self.kv_ttl_now_ms(task); let current = self.kv_resolve_read(did, tid, collection, key, now_ms); let (new_value, new_bytes) = compute::incr(current.as_deref(), delta, shape) - .map_err(|e| atomic_error_code(e, collection))?; + .map_err(|e| atomic_error_code(e.into(), collection))?; admit_kv_row(rls_write_check, &new_bytes, key, tid, collection)?; let expire_at_ms = if ttl_ms > 0 { @@ -93,10 +92,10 @@ impl CoreLoop { if self.kv_engine.is_over_budget() { return Err(ErrorCode::ResourcesExhausted); } - let now_ms = self.kv_atomic_now_ms(); + let now_ms = self.kv_read_now_ms(); let current = self.kv_resolve_read(did, tid, collection, key, now_ms); let (new_value, new_bytes) = compute::incr_float(current.as_deref(), delta, shape) - .map_err(|e| atomic_error_code(e, collection))?; + .map_err(|e| atomic_error_code(e.into(), collection))?; admit_kv_row(rls_write_check, &new_bytes, key, tid, collection)?; let response_payload = @@ -136,10 +135,10 @@ impl CoreLoop { if self.kv_engine.is_over_budget() { return Err(ErrorCode::ResourcesExhausted); } - let now_ms = self.kv_atomic_now_ms(); + let now_ms = self.kv_read_now_ms(); let current = self.kv_resolve_read(did, tid, collection, key, now_ms); let (matches, write_bytes) = compute::cas(current.as_deref(), expected, new_value) - .map_err(|e| atomic_error_code(e, collection))?; + .map_err(|e| atomic_error_code(e.into(), collection))?; // Decided on the image the swap stores, same as `execute_kv_cas`: a // swap into a typed row stores the row, not `new_value` itself. if matches { @@ -191,10 +190,10 @@ impl CoreLoop { if self.kv_engine.is_over_budget() { return Err(ErrorCode::ResourcesExhausted); } - let now_ms = self.kv_atomic_now_ms(); + let now_ms = self.kv_read_now_ms(); let old = self.kv_resolve_read(did, tid, collection, key, now_ms); let write_bytes = compute::getset(old.as_deref(), new_value) - .map_err(|e| atomic_error_code(e, collection))?; + .map_err(|e| atomic_error_code(e.into(), collection))?; // Decided on the image the write stores, same as `execute_kv_getset`. admit_kv_row(rls_write_check, &write_bytes, key, tid, collection)?; @@ -227,14 +226,6 @@ impl CoreLoop { response_payload, )) } - - /// The instant `execute_kv_incr_float` / `execute_kv_cas` / - /// `execute_kv_getset` read for expiry evaluation. - fn kv_atomic_now_ms(&self) -> u64 { - self.epoch_system_ms - .map(|ms| ms as u64) - .unwrap_or_else(current_ms) - } } #[cfg(test)] diff --git a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs index 4857078ea..233c5f57a 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs @@ -337,6 +337,7 @@ fn kv_op_name(op: &KvOp) -> &'static str { KvOp::SortedIndexRange { .. } => "SortedIndexRange", KvOp::SortedIndexCount { .. } => "SortedIndexCount", KvOp::SortedIndexScore { .. } => "SortedIndexScore", + KvOp::SortedIndexTxnRead { .. } => "SortedIndexTxnRead", KvOp::MaterializeScan { .. } => "MaterializeScan", KvOp::ResolveWrite(_) => "ResolveWrite", KvOp::ResolvedWrite { .. } => "ResolvedWrite", diff --git a/nodedb/src/data/executor/handlers/kv/sorted.rs b/nodedb/src/data/executor/handlers/kv/sorted.rs index 46f7ca7e8..58ca30c51 100644 --- a/nodedb/src/data/executor/handlers/kv/sorted.rs +++ b/nodedb/src/data/executor/handlers/kv/sorted.rs @@ -113,23 +113,10 @@ impl CoreLoop { debug!(core = self.core_id, %index_name, "kv sorted index rank"); let now_ms = current_ms(); - match self + let rank = self .kv_engine - .sorted_index_rank(did, tid, index_name, primary_key, now_ms) - { - Some(rank) => { - match response_codec::encode_json_as_msgpack(&serde_json::json!({ "rank": rank })) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error(task, e), - } - } - None => { - match response_codec::encode_json_as_msgpack(&serde_json::json!({ "rank": null })) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error(task, e), - } - } - } + .sorted_index_rank(did, tid, index_name, primary_key, now_ms); + self.sorted_rank_response(task, rank) } pub(in crate::data::executor) fn execute_kv_sorted_index_top_k( @@ -147,21 +134,7 @@ impl CoreLoop { .kv_engine .sorted_index_top_k(did, tid, index_name, k, now_ms) { - Some(entries) => { - let rows: Vec = entries - .into_iter() - .map(|(rank, pk)| { - serde_json::json!({ - "rank": rank, - "key": String::from_utf8_lossy(&pk), - }) - }) - .collect(); - match response_codec::encode_json_vec_as_msgpack(&rows) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error(task, e), - } - } + Some(entries) => self.sorted_rows_response(task, entries), None => self.response_error(task, ErrorCode::NotFound), } } @@ -191,21 +164,7 @@ impl CoreLoop { score_max, now_ms, }) { - Some(entries) => { - let rows: Vec = entries - .into_iter() - .map(|(rank, pk)| { - serde_json::json!({ - "rank": rank, - "key": String::from_utf8_lossy(&pk), - }) - }) - .collect(); - match response_codec::encode_json_vec_as_msgpack(&rows) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error(task, e), - } - } + Some(entries) => self.sorted_rows_response(task, entries), None => self.response_error(task, ErrorCode::NotFound), } } @@ -224,10 +183,7 @@ impl CoreLoop { .kv_engine .sorted_index_count(did, tid, index_name, now_ms) { - Some(count) => match response_codec::encode_count("count", count as usize) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error(task, e), - }, + Some(count) => self.sorted_count_response(task, count), None => self.response_error(task, ErrorCode::NotFound), } } @@ -242,25 +198,9 @@ impl CoreLoop { ) -> Response { debug!(core = self.core_id, %index_name, "kv sorted index score"); - match self + let score = self .kv_engine - .sorted_index_score(did, tid, index_name, primary_key) - { - Some(sort_key) => { - let b64 = - base64::Engine::encode(&base64::engine::general_purpose::STANDARD, &sort_key); - match response_codec::encode_json_as_msgpack(&serde_json::json!({ "score": b64 })) { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error(task, e), - } - } - None => { - match response_codec::encode_json_as_msgpack(&serde_json::json!({ "score": null })) - { - Ok(payload) => self.response_with_payload(task, payload), - Err(e) => self.response_error(task, e), - } - } - } + .sorted_index_score(did, tid, index_name, primary_key); + self.sorted_score_response(task, score) } } diff --git a/nodedb/src/data/executor/handlers/kv/sorted_txn.rs b/nodedb/src/data/executor/handlers/kv/sorted_txn.rs new file mode 100644 index 000000000..4df4d3a44 --- /dev/null +++ b/nodedb/src/data/executor/handlers/kv/sorted_txn.rs @@ -0,0 +1,243 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sorted-index reads inside an explicit transaction. +//! +//! A transaction must see its own DDL and its own writes. An index it created +//! has no tree on this core before COMMIT. A committed index's tree holds none +//! of its staged writes. Either way the read runs over a transaction-local +//! tree, built from the collection's base rows with the transaction's staged +//! writes folded in. A committed index the transaction staged nothing for +//! answers from its registered tree, which already holds the same rows. + +use tracing::debug; + +use nodedb_physical::physical_plan::{SortedIndexRead, SortedIndexSpec}; + +use super::sorted_index_compute::{BuildSortedIndexDefParams, build_sorted_index_def}; +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::response_codec; +use crate::data::executor::task::ExecutionTask; +use crate::engine::kv::current_ms; +use crate::engine::kv::sorted_index::SortedIndex; +use crate::engine::kv::sorted_index::manager::SortedIndexDef; +use crate::types::{DatabaseId, TenantId}; + +/// Parameters for `execute_kv_sorted_index_txn_read`. +pub(in crate::data::executor) struct SortedIndexTxnReadParams<'a> { + pub did: u64, + pub tid: u64, + pub collection: &'a str, + pub index_name: &'a str, + pub pending: Option<&'a SortedIndexSpec>, + pub read: &'a SortedIndexRead, +} + +impl CoreLoop { + pub(in crate::data::executor) fn execute_kv_sorted_index_txn_read( + &self, + task: &ExecutionTask, + params: SortedIndexTxnReadParams<'_>, + ) -> Response { + let SortedIndexTxnReadParams { + did, + tid, + collection, + index_name, + pending, + read, + } = params; + debug!(core = self.core_id, %collection, %index_name, "kv sorted index txn read"); + let now_ms = current_ms(); + let coll_key = ( + DatabaseId::new(did), + TenantId::new(tid), + collection.to_string(), + ); + let txn_id = task.request.txn_id; + let staged = txn_id.is_some_and(|txn_id| { + self.txn_overlays + .get(&txn_id) + .is_some_and(|overlay| overlay.stages_collection(&coll_key)) + }); + + let def = match pending { + Some(spec) => match pending_def(collection, index_name, spec) { + Ok(def) => def, + Err(e) => return self.response_error(task, e), + }, + None if !staged => { + return self.answer_from_registered(task, did, tid, index_name, read, now_ms); + } + None => match self.kv_engine.sorted_index_def(did, tid, index_name) { + Some(def) => def.clone(), + None => return self.response_error(task, ErrorCode::NotFound), + }, + }; + + let mut rows = self.kv_engine.collection_rows(did, tid, collection, now_ms); + if let Some(txn_id) = txn_id { + self.merge_kv_overlay_into_scan( + txn_id, + &coll_key, + &mut rows, + &|_key: &[u8], _value: &[u8]| true, + ); + } + let (index, _) = SortedIndex::build(def, rows.into_iter()); + + match read { + SortedIndexRead::Rank { primary_key } => { + self.sorted_rank_response(task, index.rank(primary_key, now_ms)) + } + SortedIndexRead::TopK { k } => self.sorted_rows_response(task, index.top_k(*k, now_ms)), + SortedIndexRead::Range { + score_min, + score_max, + } => self.sorted_rows_response( + task, + index.range(score_min.as_deref(), score_max.as_deref(), now_ms), + ), + SortedIndexRead::Count => self.sorted_count_response(task, index.count(now_ms)), + SortedIndexRead::Score { primary_key } => { + self.sorted_score_response(task, index.score(primary_key)) + } + } + } + + /// Answer from the tree registered on this core. The transaction staged + /// nothing for the collection, so that tree holds exactly its view. + fn answer_from_registered( + &self, + task: &ExecutionTask, + did: u64, + tid: u64, + index_name: &str, + read: &SortedIndexRead, + now_ms: u64, + ) -> Response { + let engine = &self.kv_engine; + match read { + SortedIndexRead::Rank { primary_key } => { + if engine.sorted_index_def(did, tid, index_name).is_none() { + return self.response_error(task, ErrorCode::NotFound); + } + let rank = engine.sorted_index_rank(did, tid, index_name, primary_key, now_ms); + self.sorted_rank_response(task, rank) + } + SortedIndexRead::TopK { k } => { + match engine.sorted_index_top_k(did, tid, index_name, *k, now_ms) { + Some(entries) => self.sorted_rows_response(task, entries), + None => self.response_error(task, ErrorCode::NotFound), + } + } + SortedIndexRead::Range { + score_min, + score_max, + } => match engine.sorted_index_range(crate::engine::kv::SortedIndexRangeParams { + database_id: did, + tenant_id: tid, + index_name, + score_min: score_min.as_deref(), + score_max: score_max.as_deref(), + now_ms, + }) { + Some(entries) => self.sorted_rows_response(task, entries), + None => self.response_error(task, ErrorCode::NotFound), + }, + SortedIndexRead::Count => { + match engine.sorted_index_count(did, tid, index_name, now_ms) { + Some(count) => self.sorted_count_response(task, count), + None => self.response_error(task, ErrorCode::NotFound), + } + } + SortedIndexRead::Score { primary_key } => { + if engine.sorted_index_def(did, tid, index_name).is_none() { + return self.response_error(task, ErrorCode::NotFound); + } + let score = engine.sorted_index_score(did, tid, index_name, primary_key); + self.sorted_score_response(task, score) + } + } + } + + /// `{"rank": n}`, or `{"rank": null}` for a key the index does not hold. + pub(in crate::data::executor) fn sorted_rank_response( + &self, + task: &ExecutionTask, + rank: Option, + ) -> Response { + match response_codec::encode_json_as_msgpack(&serde_json::json!({ "rank": rank })) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => self.response_error(task, e), + } + } + + /// One `{"rank", "key"}` row per `(rank, primary_key)` entry. + pub(in crate::data::executor) fn sorted_rows_response( + &self, + task: &ExecutionTask, + entries: Vec<(u32, Vec)>, + ) -> Response { + let rows: Vec = entries + .into_iter() + .map(|(rank, pk)| { + serde_json::json!({ + "rank": rank, + "key": String::from_utf8_lossy(&pk), + }) + }) + .collect(); + match response_codec::encode_json_vec_as_msgpack(&rows) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => self.response_error(task, e), + } + } + + /// `{"score": base64}` of the sort key, or `{"score": null}` for a key the + /// index does not hold. + pub(in crate::data::executor) fn sorted_score_response( + &self, + task: &ExecutionTask, + sort_key: Option>, + ) -> Response { + let score = sort_key.map(|sort_key| { + base64::Engine::encode(&base64::engine::general_purpose::STANDARD, &sort_key) + }); + match response_codec::encode_json_as_msgpack(&serde_json::json!({ "score": score })) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => self.response_error(task, e), + } + } + + /// `{"count": n}`. + pub(in crate::data::executor) fn sorted_count_response( + &self, + task: &ExecutionTask, + count: u32, + ) -> Response { + match response_codec::encode_count("count", count as usize) { + Ok(payload) => self.response_with_payload(task, payload), + Err(e) => self.response_error(task, e), + } + } +} + +/// The definition of an index the transaction created, built by the same code +/// that builds a registered one. +fn pending_def( + collection: &str, + index_name: &str, + spec: &SortedIndexSpec, +) -> Result { + build_sorted_index_def(BuildSortedIndexDefParams { + collection, + index_name, + sort_columns: &spec.sort_columns, + key_column: &spec.key_column, + window_type: &spec.window_type, + window_timestamp_column: &spec.window_timestamp_column, + window_start_ms: spec.window_start_ms, + window_end_ms: spec.window_end_ms, + }) +} diff --git a/nodedb/src/data/executor/handlers/kv/transfer_compute.rs b/nodedb/src/data/executor/handlers/kv/transfer_compute.rs index 10afcf117..11f1b3302 100644 --- a/nodedb/src/data/executor/handlers/kv/transfer_compute.rs +++ b/nodedb/src/data/executor/handlers/kv/transfer_compute.rs @@ -4,7 +4,7 @@ //! move), shared by the autocommit handler (`transfer.rs`) and the //! in-transaction staging handler (`stage_kv_transfer.rs`) so a staged //! value and its COMMIT-time durable replay are always computed by the -//! exact same code — mirrors the `engine_atomic_compute` / `stage_kv_atomic` +//! exact same code — mirrors the `nodedb_physical::kv_atomic::compute` / `stage_kv_atomic` //! split for `Incr`/`Cas`/etc. use std::collections::HashMap; diff --git a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs index d8b1f6327..a165eccdb 100644 --- a/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs +++ b/nodedb/src/data/executor/handlers/merge_orchestrated/delete_arms.rs @@ -77,7 +77,6 @@ impl CoreLoop { for del in deletes { let surrogate = del.key.surrogate(); - let row_key = del.key.to_string(); // The identity INSERT minted for this row, from the plan's // captured MessagePack body: the declared primary key, else the // decimal surrogate. @@ -95,7 +94,8 @@ impl CoreLoop { database_id, tid, collection, - document_id: &row_key, + // The graph cascade keys nodes by the client key. + document_id: row_identity.as_str(), surrogate, user_roles: &task.request.user_roles, enforce: true, diff --git a/nodedb/src/data/executor/handlers/point/apply_delete.rs b/nodedb/src/data/executor/handlers/point/apply_delete.rs index 03222dd32..0357613ca 100644 --- a/nodedb/src/data/executor/handlers/point/apply_delete.rs +++ b/nodedb/src/data/executor/handlers/point/apply_delete.rs @@ -26,6 +26,8 @@ pub(in crate::data::executor) struct PointDeleteParams<'a> { pub database_id: u64, pub tid: u64, pub collection: &'a str, + /// The row's client identity. The graph cascade removes the edges of, + /// and marks deleted, the node this names. pub document_id: &'a str, pub surrogate: Surrogate, /// Roles held by the authenticated user. Currently unused by DELETE diff --git a/nodedb/src/data/executor/handlers/timeseries/flush.rs b/nodedb/src/data/executor/handlers/timeseries/flush.rs index bacec0e1b..0e5b13923 100644 --- a/nodedb/src/data/executor/handlers/timeseries/flush.rs +++ b/nodedb/src/data/executor/handlers/timeseries/flush.rs @@ -5,12 +5,9 @@ //! The boot-side counterpart — rebuilding `ts_registries` from the partitions //! this writes — lives in `data::executor::timeseries_checkpoint`. -use std::collections::HashMap; use std::path::Path; use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::handlers::transaction::undo::UndoEntry; -use crate::data::executor::task::ExecutionTask; use crate::engine::timeseries::columnar_segment::ColumnarSegmentWriter; use crate::engine::timeseries::partition_registry::PartitionRegistry; use crate::types::{DatabaseId, TenantId}; @@ -162,82 +159,6 @@ impl CoreLoop { Ok(()) } - /// Finalize the metadata deliberately deferred by transaction-batch - /// timeseries ingestion. At this point all sub-plans, constraints and CRDT - /// application succeeded, so publication is safe. A maintenance flush is - /// post-commit: failure leaves the committed memtable/WAL intact and is - /// logged as retryable backlog. It cannot set `Response::partial`, which - /// means a further stream frame is coming and would strand a COMMIT waiter. - pub(in crate::data::executor) fn finalize_deferred_timeseries_ingests( - &mut self, - task: &ExecutionTask, - undo_log: &[UndoEntry], - ) { - let mut collections = HashMap::new(); - for entry in undo_log { - if let UndoEntry::TimeseriesIngest(token) = entry { - let prior_rows = token - .memtable_before - .as_ref() - .map(|snapshot| snapshot.row_count) - .unwrap_or(0); - collections - .entry(token.collection_key.clone()) - .and_modify(|prior: &mut u64| *prior = (*prior).min(prior_rows)) - .or_insert(prior_rows); - } - } - - if collections.is_empty() { - return; - } - - let mut accepted_any = false; - let mut flush_backlog = false; - for ((database_id, tenant_id, collection), prior_rows) in collections { - let accepted = self - .columnar_memtables - .get(&(database_id, tenant_id, collection.clone())) - .map(|memtable| memtable.row_count().saturating_sub(prior_rows) as usize) - .unwrap_or(0); - accepted_any |= accepted > 0; - self.checkpoint_coordinator - .mark_dirty("timeseries", accepted); - self.note_collection_write_lsn(task, &collection); - self.recharge_ts_memtable_budget(tenant_id, database_id, &collection); - let needs_flush = self - .columnar_memtables - .get(&(database_id, tenant_id, collection.clone())) - .is_some_and(|memtable| { - memtable.memory_bytes() >= self.ts_tuning.memtable_budget_bytes - }); - if needs_flush - && let Err(error) = self.flush_ts_collection( - tenant_id, - database_id, - &collection, - self.epoch_system_ms.unwrap_or(0), - ) - { - flush_backlog = true; - tracing::error!( - collection, - error = %error, - "committed timeseries flush deferred as retryable backlog" - ); - } - } - if accepted_any { - self.last_ts_ingest = Some(std::time::Instant::now()); - } - if flush_backlog { - tracing::warn!( - core = self.core_id, - "committed timeseries rows remain in the retryable flush backlog" - ); - } - } - /// Charge the memtable's current resident footprint, replacing the /// prior charge so the budget tracks `memory_bytes()`, not the sum of /// every recharge. diff --git a/nodedb/src/data/executor/handlers/timeseries/ingest.rs b/nodedb/src/data/executor/handlers/timeseries/ingest.rs index e369b8ca4..63a0a408b 100644 --- a/nodedb/src/data/executor/handlers/timeseries/ingest.rs +++ b/nodedb/src/data/executor/handlers/timeseries/ingest.rs @@ -24,8 +24,8 @@ use super::ingest_dispatch::{TimeseriesApplyMode, TimeseriesIngestParams}; use super::rls_gate; impl CoreLoop { - /// Check every condition that could reject a commit-deferred ILP ingest - /// before it is allowed to cast a Calvin commit vote. The simulation is + /// Check every condition that could reject a staged ILP ingest before the + /// transaction is allowed to commit. The simulation is /// deliberately isolated from live state: schema evolution and dictionary /// probes run against an exact snapshot clone, so this cannot publish a /// schema, consume tag IDs, or create a memtable. @@ -206,12 +206,6 @@ impl CoreLoop { ); } - if mode == TimeseriesApplyMode::CommitDeferred - && let Err(error) = self.prevalidate_deferred_ilp_ingest(task, tid, collection, &lines) - { - return self.response_error(task, error); - } - if mode == TimeseriesApplyMode::RedoInstall && let Err(error) = self.prepare_redo_ts_ingest( task.request.database_id, @@ -258,15 +252,6 @@ impl CoreLoop { // resolves every possible mid-record stop before the first row lands. let soft_limit = self.ts_tuning.memtable_budget_bytes; if self.ts_ingest_needs_flush(&key, &lines) { - if mode == TimeseriesApplyMode::CommitDeferred { - return self.response_error( - task, - ErrorCode::RejectedPrevalidation { - reason: "transactional timeseries ingest requires a flush before mutation" - .into(), - }, - ); - } // A redo install flushed before it took its pre-image. A flush // now would drain rows that pre-image holds, so the install // fails and rolls back instead. diff --git a/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs b/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs index e97143f9a..6a65d9a41 100644 --- a/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs +++ b/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs @@ -15,7 +15,6 @@ use crate::data::executor::task::ExecutionTask; #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub(in crate::data::executor) enum TimeseriesApplyMode { Immediate, - CommitDeferred, /// The install pass of a committed redo record. The memtable flushes the /// rows it already holds when it has no room, then the ingest records /// the pre-image its undo restores. No flush, budget recharge or timer @@ -114,32 +113,6 @@ impl CoreLoop { } let key = (task.request.database_id, tid, collection.to_string()); - if mode == TimeseriesApplyMode::CommitDeferred { - let governor_pressure = self - .governor - .try_reserve( - task.request.database_id, - tid, - nodedb_mem::EngineId::Timeseries, - 0, - ) - .is_err(); - let needs_flush = self.columnar_memtables.get(&key).is_some_and(|memtable| { - memtable.memory_bytes() >= self.ts_tuning.memtable_budget_bytes - || memtable.memory_bytes() >= self.ts_tuning.memtable_hard_limit_bytes - || governor_pressure - }); - if needs_flush { - return self.response_error( - task, - ErrorCode::RejectedPrevalidation { - reason: "transactional timeseries ingest requires a flush before mutation" - .into(), - }, - ); - } - } - let already_flushed = if let Some(lsn) = wal_lsn && let Some(registry) = self.ts_registries.get(&key) { @@ -284,13 +257,12 @@ impl CoreLoop { if let Some(prov) = provenance && ingest_response.status == Status::Ok - && mode != TimeseriesApplyMode::CommitDeferred { self.sync_commit(prov); let applied_seq = self.sync_hwm_value(prov.producer_id, prov.stream_id); return self.sync_ack_response(task, AckStatus::Applied, applied_seq); } - if ingest_response.status == Status::Ok && mode != TimeseriesApplyMode::CommitDeferred { + if ingest_response.status == Status::Ok { self.note_collection_write_lsn(task, collection); } ingest_response diff --git a/nodedb/src/data/executor/handlers/timeseries/mod.rs b/nodedb/src/data/executor/handlers/timeseries/mod.rs index a109d5e27..1b87c7f27 100644 --- a/nodedb/src/data/executor/handlers/timeseries/mod.rs +++ b/nodedb/src/data/executor/handlers/timeseries/mod.rs @@ -16,6 +16,7 @@ pub mod paths; pub mod raw_scan; mod redo_ingest; mod resolve_ingest; +mod returning_preview; mod rls_gate; mod scan; mod sort; diff --git a/nodedb/src/data/executor/handlers/timeseries/returning_preview.rs b/nodedb/src/data/executor/handlers/timeseries/returning_preview.rs new file mode 100644 index 000000000..51b3a8d2f --- /dev/null +++ b/nodedb/src/data/executor/handlers/timeseries/returning_preview.rs @@ -0,0 +1,82 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The rows a staged timeseries ingest will store, rendered as a `SELECT` +//! renders them. +//! +//! A Calvin transaction decides a plan's `RETURNING` rows when the plan +//! stages, before the flush installs anything. The live ingest reads its +//! stored rows back out of the memtable through the raw-scan row emitter. +//! This preview ingests the same stamped lines into a scratch memtable that +//! carries the collection's current schema, and reads them back through the +//! same emitter. The live memtable, its dictionaries and its series catalog +//! are not touched. + +use nodedb_types::timeseries::SeriesCatalog; + +use super::raw_scan::emit_memtable_rows_at; +use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; +use crate::engine::timeseries::columnar_memtable::{ColumnarMemtable, ColumnarMemtableConfig}; +use crate::engine::timeseries::ilp; +use crate::engine::timeseries::ilp_ingest; +use crate::types::TenantId; + +impl CoreLoop { + /// The rows `lines` store in `collection`, each rendered by the raw-scan + /// row emitter. `now_ms` is the ingest instant: the default timestamp of + /// an untimed line and the system time of a bitemporal row. + /// + /// A `RETURNING` ingest takes every row or none, so a line the memtable + /// would reject refuses the whole plan before anything is staged. + pub(in crate::data::executor) fn preview_ilp_ingest_rows( + &self, + task: &ExecutionTask, + tid: TenantId, + collection: &str, + lines: &[ilp::IlpLine<'_>], + now_ms: i64, + ) -> Result, ErrorCode> { + let key = (task.request.database_id, tid, collection.to_string()); + let bitemporal = + self.is_bitemporal(task.request.database_id.as_u64(), tid.as_u64(), collection); + let mut scratch = match self.columnar_memtables.get(&key) { + Some(live) => { + let mut scratch = ColumnarMemtable::new(live.schema().clone(), live.config()); + ilp_ingest::evolve_schema(&mut scratch, lines); + scratch + } + None => { + let mut schema = self.initial_ts_schema(task, tid, collection, lines); + if bitemporal { + ilp_ingest::ensure_bitemporal_columns(&mut schema); + } + ColumnarMemtable::new(schema, ColumnarMemtableConfig::from_tuning(&self.ts_tuning)) + } + }; + let mut catalog = SeriesCatalog::new(); + let outcome = ilp_ingest::ingest_batch_with_lvc(ilp_ingest::IngestBatchArgs { + memtable: &mut scratch, + lines, + catalog: &mut catalog, + default_timestamp_ms: now_ms, + lvc: None, + bitemporal: bitemporal.then_some(ilp_ingest::BitempStamps { system_ms: now_ms }), + collect_row_indices: true, + }); + if outcome.rejected > 0 { + let reason = outcome + .first_rejection + .unwrap_or_else(|| "no reason recorded".to_string()); + return Err(ErrorCode::RejectedPrevalidation { + reason: format!( + "timeseries ingest with RETURNING would reject {} of {} rows, and a row set \ + cannot report a rejected row; first rejection: {reason}", + outcome.rejected, + outcome.accepted + outcome.rejected + ), + }); + } + emit_memtable_rows_at(&scratch, &outcome.accepted_row_indices).map_err(ErrorCode::from) + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/batch.rs b/nodedb/src/data/executor/handlers/transaction/batch.rs deleted file mode 100644 index c19c8312e..000000000 --- a/nodedb/src/data/executor/handlers/transaction/batch.rs +++ /dev/null @@ -1,643 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Transaction batch execution handler. -//! -//! Executes a `PhysicalPlan::TransactionBatch`: on a failure, every write an -//! engine-specific handler made is rolled back. A sub-plan with no such -//! handler runs through the ordinary dispatch path and records no undo, so -//! its write stays (see `batch_irreversible`). CRDT deltas are accumulated in -//! a scratch buffer and only applied on success. - -use std::panic::{AssertUnwindSafe, catch_unwind}; - -use tracing::{debug, error, warn}; - -use crate::bridge::envelope::{ErrorCode, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; -use nodedb_physical::physical_plan::PhysicalPlan; -use nodedb_types::calvin::VersionedReadEntry; - -use crate::data::panic_payload::panic_payload_to_string; - -use super::batch_irreversible::{applied_irreversibly, batch_failure_code, batch_failure_response}; -use super::undo::UndoEntry; - -/// A CRDT delta buffered during a transaction batch: `(delta_bytes, id, -/// collection)`. Deltas accumulate in a scratch buffer and are applied to -/// the LoroDoc only on full-batch success. -pub(super) type CrdtDelta = (Vec, u64, String); - -impl CoreLoop { - /// Execute a transaction batch atomically. - /// - /// All sub-plans are executed in order. If any sub-plan fails, all - /// previous writes are rolled back. CRDT deltas are buffered and only - /// applied to LoroDoc on full success. - /// - /// The Control Plane has already written a single `RecordType::Transaction` - /// WAL record covering all operations before dispatching this batch. - pub(in crate::data::executor) fn execute_transaction_batch( - &mut self, - task: &ExecutionTask, - tid: u64, - plans: &[PhysicalPlan], - versioned_reads: &[VersionedReadEntry], - txn_id: Option, - ) -> Response { - debug!( - core = self.core_id, - plan_count = plans.len(), - "transaction batch begin" - ); - - // Check whether this participant's slice of the transaction's reads was - // still current against the local write versions, observed BEFORE apply. - // Non-gating: the outcome is reported on the response and the apply - // proceeds regardless. Empty read-set (pure-write / autocommit / non-Calvin - // fast path) is vacuously current. - let read_set_current = self.read_set_still_current(task, tid, versioned_reads); - - let undo_log: Vec = Vec::with_capacity(plans.len()); - let crdt_deltas: Vec = Vec::new(); - - // Carry the resolve-time bitemporal stamps this transaction's overlay - // recorded into per-core apply scratch, so `apply_point_put` installs - // each bitemporal document put on the versioned store at the SAME stamp - // the redo carries. Calvin threads its stamps in before the call - // (`txn_id = None` here); a caller applying a session transaction's - // plans passes its `txn_id`. The scratch is consulted ONLY by the - // forward apply below, so it is cleared the moment `run_sub_plans` - // returns (any path). - if let Some(txn_id) = txn_id { - self.load_bitemporal_stamps_for_txn(txn_id); - self.active_graph_system_from = self - .graph_txn_overlays - .get(&txn_id) - .and_then(|overlay| overlay.resolved_system_from()); - } - // Open the transaction's BALANCED accumulator for the duration of the - // sub-plans. Every write handler they reach — including the ones - // reached by passing an ordinary statement plan back through - // `execute` — settles its entries onto this instead of checking them on - // its own, so one leg of a journal per statement is allowed inside an - // explicit transaction and only the whole batch has to balance. - // - // A batch reached from inside another batch (a `MetaOp::TransactionBatch` - // sub-plan passes back through `execute`) does NOT open its own: the - // outer boundary is the one that commits, so the inner batch's entries - // belong to it and are left in place for it to judge. - let owns_scope = self.balanced_txn_entries.is_none(); - if owns_scope { - self.balanced_txn_entries = Some(Vec::new()); - } - let sub = self.run_sub_plans(task, tid, plans, undo_log, crdt_deltas); - // Taken on EVERY exit path, so a failed batch can never leave its - // entries behind for the next transaction on this core to be judged by. - let balanced_entries = if owns_scope { - self.balanced_txn_entries.take().unwrap_or_default() - } else { - Vec::new() - }; - self.active_bitemporal_stamps.clear(); - self.active_graph_system_from = None; - let (last_response, undo_log, crdt_deltas, irreversible) = match sub { - Ok(v) => v, - Err(resp) => return resp, - }; - - let constraint_check = catch_unwind(AssertUnwindSafe(|| { - self.check_balanced_constraints( - task.request.database_id.as_u64(), - tid, - balanced_entries, - ) - })); - match constraint_check { - Ok(Ok(())) => {} - Ok(Err(error_code)) => { - warn!( - core = self.core_id, - "BALANCED constraint violated, rolling back {} operations", - undo_log.len() - ); - let response = self.rollback_transaction_failure( - task, - tid, - undo_log, - self.response_error(task, error_code), - ); - return batch_failure_response(response, irreversible); - } - Err(payload) => { - let detail = panic_payload_to_string(payload.as_ref()); - error!( - core = self.core_id, - panic = %detail, - "BALANCED constraint check panicked; routing through rollback" - ); - return self.rollback_transaction_failure( - task, - tid, - undo_log, - self.response_error( - task, - ErrorCode::Internal { - detail: format!("panic in BALANCED constraint check: {detail}"), - }, - ), - ); - } - } - - let crdt_delta_count = crdt_deltas.len(); - let undo_log = match self.apply_crdt_deltas_or_rollback(task, tid, undo_log, crdt_deltas) { - Ok(undo_log) => undo_log, - Err(response) => return batch_failure_response(response, irreversible), - }; - if crdt_delta_count > 0 { - // Match the direct CRDT apply path: successful transaction-batch - // imports must become checkpoint work, but never before the full - // batch's rollback gates have succeeded. - self.checkpoint_coordinator - .mark_dirty("crdt", crdt_delta_count); - } - - // All transactional gates have succeeded. Only now may timeseries - // publish its write floor/timer/checkpoint/reservation state or start - // best-effort post-commit maintenance. - self.finalize_deferred_timeseries_ingests(task, &undo_log); - self.finalize_timeseries_truncates(&undo_log); - - debug!( - core = self.core_id, - committed = plans.len(), - "transaction batch committed" - ); - - // Record the per-key / per-collection write version for every sub-plan - // in the committed batch. Covers the fast-path commit AND every Calvin - // apply (both funnel through here); one batch WAL LSN for all keys. - self.record_batch_write_versions(task, tid, plans); - self.record_batch_index_write_values(task, tid, &undo_log); - - self.emit_deferred_writes(task, undo_log); - - // Return the last sub-plan payload, but keyed to the outer transaction request. - Response { - request_id: task.request_id(), - status: Status::Ok, - attempt: 1, - // `partial` means another response frame will follow. A failed - // best-effort maintenance flush leaves this committed transaction - // durable in the memtable/WAL, but does not produce another frame. - partial: false, - payload: last_response.payload, - watermark_lsn: self.watermark, - error_code: None, - read_set_valid: Some(read_set_current), - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - } - } - - /// Copy every resolve-time bitemporal stamp recorded in `txn_id`'s staging - /// overlay into the per-core `active_bitemporal_stamps` scratch. Surrogates - /// are globally unique, so all collections' stamps flatten into one map. - /// A no-op when no overlay exists (pure-read / non-bitemporal transaction). - fn load_bitemporal_stamps_for_txn(&mut self, txn_id: crate::types::TxnId) { - let stamps: Vec<_> = match self.txn_overlays.get(&txn_id) { - Some(overlay) => overlay.all_bitemporal_stamps().collect(), - None => return, - }; - for (surrogate, stamp) in stamps { - self.active_bitemporal_stamps.insert(surrogate, stamp); - } - } - - /// Run every sub-plan in order, tracking undo entries and buffered CRDT - /// deltas as it goes. - /// - /// On success, returns the last sub-plan's response plus the accumulated - /// undo log and CRDT delta buffer (for the caller's subsequent commit - /// steps). On failure, rolls back all writes performed so far and - /// returns the terminal error `Response` directly — the caller must - /// return it unchanged. - /// - /// A panic (real or test-injected) during a sub-apply is caught so it - /// routes through the same typed-rollback path instead of unwinding past - /// `undo_log`, which would drop the log without running rollback and - /// leave the shard half-committed. - fn run_sub_plans( - &mut self, - task: &ExecutionTask, - tid: u64, - plans: &[PhysicalPlan], - mut undo_log: Vec, - mut crdt_deltas: Vec, - ) -> Result<(Response, Vec, Vec, bool), Response> { - let mut last_response = self.response_ok(task); - let mut irreversible = false; - - for (i, plan) in plans.iter().enumerate() { - let undo_before = undo_log.len(); - let user_roles = &task.request.user_roles; - let outcome = catch_unwind(AssertUnwindSafe(|| { - let r = self.execute_tx_sub_plan_from_batch( - task, - tid, - plan, - &mut undo_log, - &mut crdt_deltas, - user_roles, - ); - crate::fail_point!("transaction_batch::between_subapply"); - r - })); - let result = match outcome { - Ok(r) => r, - Err(payload) => { - let detail = panic_payload_to_string(payload.as_ref()); - error!( - core = self.core_id, - plan_index = i, - panic = %detail, - "transaction sub-apply panicked; routing through rollback path" - ); - Err(ErrorCode::Internal { - detail: format!("panic in sub-apply at index {i}: {detail}"), - }) - } - }; - - match result { - Ok(resp) => { - irreversible |= applied_irreversibly(plan, undo_before, undo_log.len()); - last_response = resp; - } - Err(error_code) => { - warn!( - core = self.core_id, - plan_index = i, - "transaction sub-plan failed, rolling back {} operations", - undo_log.len() - ); - - // Roll back all previous writes in reverse order. An undo - // panic is no safer than an `Err`: catch it and surface the - // same terminal `RollbackFailed` contract instead of - // unwinding past a half-restored transaction. - let undo_len = undo_log.len(); - let rollback_error_code = match catch_unwind(AssertUnwindSafe(|| { - self.rollback_undo_log(task.request.database_id.as_u64(), tid, undo_log) - })) { - Ok(Ok(())) => batch_failure_code(error_code, irreversible), - Ok(Err((entry_index, detail))) => { - error!( - core = self.core_id, - plan_index = i, - entry_index, - detail = %detail, - "transaction rollback failed; shard state unknown — \ - restart required for WAL replay" - ); - crate::bridge::envelope::ErrorCode::RollbackFailed { - entry_index, - detail, - } - } - Err(payload) => { - let detail = format!( - "panic during transaction rollback: {}", - panic_payload_to_string(payload.as_ref()) - ); - error!( - core = self.core_id, - plan_index = i, - entry_index = undo_len, - detail = %detail, - "transaction rollback panicked; shard state unknown — \ - restart required for WAL replay" - ); - crate::bridge::envelope::ErrorCode::RollbackFailed { - entry_index: undo_len, - detail, - } - } - }; - - // Discard CRDT scratch buffer (never applied). - drop(crdt_deltas); - - return Err(Response { - request_id: task.request_id(), - status: Status::Error, - attempt: 1, - partial: false, - payload: crate::bridge::envelope::Payload::empty(), - watermark_lsn: self.watermark, - error_code: Some(Box::new(rollback_error_code)), - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }); - } - } - } - - Ok((last_response, undo_log, crdt_deltas, irreversible)) - } - - /// Apply all buffered CRDT deltas only after every sub-plan and the - /// `BALANCED` constraint check have succeeded. - /// - /// A failure is returned to the caller, which rolls every forward write - /// back through the same undo log before responding. Never leave CRDT and - /// engine state on different sides of a failed transaction boundary. - pub(super) fn apply_crdt_deltas( - &mut self, - task: &ExecutionTask, - tid: u64, - crdt_deltas: Vec, - ) -> Option { - for (crdt_idx, (delta, _peer_id, collection)) in crdt_deltas.into_iter().enumerate() { - // Crash-injection point — between forward-write commit and CRDT - // apply. WAL replay must roll the CRDT side forward (or roll - // forward writes back) to restore consistency. - crate::fail_point!("transaction_batch::between_crdt_delta"); - - let tenant_id = crate::types::TenantId::new(tid); - match self.get_crdt_engine(task.request.database_id, tenant_id) { - Ok(engine) => { - let outcome = engine.apply_committed_delta_validated( - &collection, - &delta, - nodedb_types::Surrogate::ZERO, - "", - 0, - ); - if !matches!( - outcome, - crate::engine::crdt::tenant_state::ValidatedApplyOutcome::Clean { .. } - ) { - error!( - core = self.core_id, - crdt_delta_index = crdt_idx, - ?outcome, - "CRDT delta validation failed; transaction rollback required" - ); - let code = self.crdt_batch_refusal(task, tenant_id, &collection, outcome); - return Some(Response { - request_id: task.request_id(), - status: Status::Error, - attempt: 1, - partial: false, - payload: crate::bridge::envelope::Payload::empty(), - watermark_lsn: self.watermark, - error_code: Some(Box::new(code)), - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }); - } - // This runs after the import so rollback coverage includes - // panics that occur once an earlier delta is already live. - crate::fail_point!("transaction_batch::after_crdt_delta"); - } - Err(e) => { - error!( - core = self.core_id, - crdt_delta_index = crdt_idx, - error = %e, - "CRDT engine not found; transaction rollback required" - ); - return Some(Response { - request_id: task.request_id(), - status: Status::Error, - attempt: 1, - partial: false, - payload: crate::bridge::envelope::Payload::empty(), - watermark_lsn: self.watermark, - error_code: Some(Box::new(crate::bridge::envelope::ErrorCode::Internal { - detail: format!("CRDT engine not available: {e}"), - })), - read_set_valid: None, - read_version_lsn: crate::types::Lsn::ZERO, - write_set: Vec::new(), - }); - } - } - } - None - } - - /// Emit deferred trigger events for every write recorded in the - /// committed transaction's undo log. `UndoEntry::{PutDocument, - /// DeleteDocument}.identity` is the row's client identity, the same one - /// an immediate trigger sees via `emit_put_event` / - /// `emit_document_delete_event`. - fn emit_deferred_writes(&mut self, task: &ExecutionTask, undo_log: Vec) { - use crate::data::executor::core_loop::deferred::DeferredWrite; - let deferred_writes: Vec = undo_log - .into_iter() - .filter_map(|entry| match entry { - UndoEntry::PutDocument { - collection, - identity, - old_value, - .. - } => Some(DeferredWrite { - collection, - op: if old_value.is_some() { - crate::event::WriteOp::Update - } else { - crate::event::WriteOp::Insert - }, - identity, - new_value: None, - old_value, - }), - UndoEntry::DeleteDocument { - collection, - identity, - old_value, - .. - } => Some(DeferredWrite { - collection, - op: crate::event::WriteOp::Delete, - identity, - new_value: None, - old_value: Some(old_value), - }), - _ => None, // Vector and edge undo entries don't trigger deferred triggers. - }) - .collect(); - - if !deferred_writes.is_empty() { - self.emit_deferred_events( - deferred_writes, - task.request.database_id, - task.request.tenant_id, - task.request.vshard_id, - ); - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::bridge::envelope::Status; - use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; - use crate::data::executor::doc_format; - use crate::data::executor::handlers::point::insert::PointInsertParams; - use crate::engine::document::store::CollectionConfig; - use crate::types::{DatabaseId, TenantId}; - use nodedb_physical::physical_plan::{DocumentOp, ResolvedSumTarget}; - use nodedb_types::{QualifiedCollection, Surrogate}; - - const DB: u64 = 0; - const TID: u64 = 1; - const SOURCE: &str = "point_txns"; - const TARGET: &str = "point_holders"; - - /// The premise every test below rests on. - #[test] - fn the_fixture_is_co_resident() { - assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, TARGET), - "'{SOURCE}' and '{TARGET}' must share a vShard: a cross-shard binding's balance \ - travels on its own task and is never folded into the source write's transaction" - ); - } - const A1: &str = "a1"; - const T1: Surrogate = Surrogate(4001); - - fn binding() -> nodedb_physical::physical_plan::MaterializedSumBinding { - nodedb_physical::physical_plan::MaterializedSumBinding { - target_collection: TARGET.to_string(), - target_column: "balance".to_string(), - join_column: "account_id".to_string(), - value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), - declared_primary_key: None, - } - } - - fn resolved() -> Vec { - vec![ResolvedSumTarget::new(TARGET, A1, T1)] - } - - fn config_key(collection: &str) -> (DatabaseId, TenantId, String) { - ( - DatabaseId::DEFAULT, - TenantId::new(TID), - collection.to_string(), - ) - } - - /// A source collection bound to the sum, and a target row starting at zero. - fn seeded_core(dir: &std::path::Path) -> CoreLoop { - let (mut core, _req, _resp) = make_core_with_dir(dir); - - let mut source = CollectionConfig::new(SOURCE); - source.enforcement.materialized_sum_sources = vec![binding()]; - core.doc_configs.insert(config_key(SOURCE), source); - core.doc_configs - .insert(config_key(TARGET), CollectionConfig::new(TARGET)); - - let seed = serde_json::json!({"id": A1, "balance": "0"}); - core.sparse - .put( - DB, - TID, - TARGET, - &nodedb_types::StorageKey::for_surrogate(T1), - &doc_format::encode_to_msgpack(&seed), - ) - .expect("seed target row"); - core - } - - /// A source row body, in the MessagePack every handler receives. - fn entry(account: &str, amount: i64) -> Vec { - doc_format::encode_to_msgpack(&serde_json::json!({ - "account_id": account, - "amount": amount, - })) - } - - /// The balance the target row currently holds. - fn balance(core: &CoreLoop, surrogate: Surrogate) -> String { - let stored = core - .sparse - .get( - DB, - TID, - TARGET, - &nodedb_types::StorageKey::for_surrogate(surrogate), - ) - .expect("read target") - .expect("target row must exist"); - doc_format::decode_document(&stored) - .expect("target row must decode") - .get("balance") - .and_then(|v| v.as_str()) - .expect("target row must carry a balance") - .to_string() - } - - fn insert( - core: &mut CoreLoop, - task: &ExecutionTask, - surrogate: Surrogate, - body: &[u8], - ) -> Status { - let targets = resolved(); - let document_id = format!("e{}", surrogate.as_u32()); - core.execute_point_insert(PointInsertParams { - task, - tid: TID, - collection: SOURCE, - document_id: &document_id, - surrogate, - value: body, - if_absent: false, - returning: None, - rls_filters: &[], - resolved_sum_targets: &targets, - deferred_sum_targets: &[], - }) - .status - } - - #[test] - fn a_transactional_delete_takes_the_row_back_off_the_total() { - let dir = tempfile::tempdir().expect("tempdir"); - let mut core = seeded_core(dir.path()); - let task = make_default_task(); - - assert_eq!( - insert(&mut core, &task, Surrogate(81), &entry(A1, 90)), - Status::Ok - ); - assert_eq!(balance(&core, T1), "90"); - - let plan = PhysicalPlan::Document(DocumentOp::PointDelete { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, SOURCE), - document_id: "e81".to_string(), - surrogate: Surrogate(81), - pk_bytes: b"e81".to_vec(), - returning: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - resolved_sum_targets: resolved(), - }); - let resp = core.execute_transaction_batch(&task, TID, &[plan], &[], None); - assert_eq!(resp.status, Status::Ok); - assert_eq!( - balance(&core, T1), - "0", - "a delete inside a transaction must debit the target too" - ); - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs b/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs deleted file mode 100644 index 234a59fc3..000000000 --- a/nodedb/src/data/executor/handlers/transaction/batch_crdt.rs +++ /dev/null @@ -1,703 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Panic-safe CRDT gate for transaction-batch apply. - -use std::panic::{AssertUnwindSafe, catch_unwind}; - -use tracing::error; - -use crate::bridge::envelope::{ErrorCode, Response}; -use crate::data::executor::core_loop::{CoreLoop, crdt_rejection}; -use crate::data::executor::task::ExecutionTask; -use crate::data::panic_payload::panic_payload_to_string; -use crate::engine::crdt::tenant_state::ValidatedApplyOutcome; -use crate::types::TenantId; - -use super::batch::CrdtDelta; -use super::undo::UndoEntry; - -impl CoreLoop { - /// The refusal for a buffered CRDT delta that did not apply cleanly. - /// - /// A constraint rejection stores its dead-letter entry first, then - /// refuses with the constraint. The transaction rolls back either way. - pub(super) fn crdt_batch_refusal( - &mut self, - task: &ExecutionTask, - tenant_id: TenantId, - collection: &str, - outcome: ValidatedApplyOutcome, - ) -> ErrorCode { - match outcome { - ValidatedApplyOutcome::Rejected(violation) => match self.store_crdt_dead_letter( - task.request.database_id, - tenant_id, - task.wal_lsn(), - ) { - Ok(()) => crdt_rejection(collection, "transaction batch", &violation), - Err(error) => ErrorCode::Internal { - detail: format!( - "CRDT delta for {collection} violates {violation}, and its \ - dead-letter entry could not be stored: {error}" - ), - }, - }, - outcome @ (ValidatedApplyOutcome::Clean { .. } - | ValidatedApplyOutcome::Malformed - | ValidatedApplyOutcome::PendingDependencies) => ErrorCode::Internal { - detail: format!("CRDT delta validation failed: {outcome:?}"), - }, - } - } -} - -/// A CRDT collection's state before a transaction starts importing its deltas. -struct CrdtCollectionPreimage { - collection: String, - snapshot: Option>, -} - -/// Complete CRDT state required to restore a transaction's pre-image. -/// -/// Keeping the rollback scope together prevents callers from pairing a -/// pre-image with the wrong tenant or database during failure handling. -struct CrdtRollbackScope { - database_id: crate::types::DatabaseId, - tenant_id: TenantId, - engine_existed_before: bool, - preimages: Vec, -} - -/// One failed collection replacement during CRDT transaction rollback. -/// -/// The original engine error is converted at the boundary so every failed -/// collection can be reported together rather than aborting restoration at the -/// first failure. -#[derive(Debug)] -struct CrdtCollectionRestoreFailure { - collection: String, - detail: String, -} - -/// Typed aggregate of every CRDT collection that could not be restored. -/// -/// A rollback failure is fatal, but restoration must still be attempted for -/// all pre-images so the error reports the complete shard-state risk. -#[derive(Debug)] -struct CrdtRestoreFailures { - failures: Vec, -} - -impl std::fmt::Display for CrdtRestoreFailures { - fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - for (index, failure) in self.failures.iter().enumerate() { - if index > 0 { - formatter.write_str("; ")?; - } - write!( - formatter, - "collection {} could not be restored from its pre-image: {}", - failure.collection, failure.detail - )?; - } - Ok(()) - } -} - -impl CoreLoop { - /// Roll back an uncommitted batch failure and surface accounting/restore - /// mismatches as `RollbackFailed` rather than reporting a false abort. - pub(super) fn rollback_transaction_failure( - &mut self, - task: &ExecutionTask, - tid: u64, - undo_log: Vec, - mut response: Response, - ) -> Response { - let undo_len = undo_log.len(); - let rollback = catch_unwind(AssertUnwindSafe(|| { - self.rollback_undo_log(task.request.database_id.as_u64(), tid, undo_log) - })); - let failure = match rollback { - Ok(Ok(())) => None, - Ok(Err((entry_index, detail))) => Some((entry_index, detail)), - Err(payload) => Some(( - undo_len, - format!( - "panic during transaction rollback: {}", - panic_payload_to_string(payload.as_ref()) - ), - )), - }; - if let Some((entry_index, detail)) = failure { - error!( - core = self.core_id, - entry_index, - detail = %detail, - "transaction rollback failed after a gate error or panic; shard state unknown" - ); - response.error_code = Some(Box::new(ErrorCode::RollbackFailed { - entry_index, - detail, - })); - } - response - } - - /// Apply buffered CRDT deltas and roll back every prior forward write on a - /// CRDT error or panic. CRDT imports are also restored exactly: Loro import - /// merges, so an already imported delta must be replaced from a snapshot, - /// not merely imported again. - pub(super) fn apply_crdt_deltas_or_rollback( - &mut self, - task: &ExecutionTask, - tid: u64, - undo_log: Vec, - crdt_deltas: Vec, - ) -> Result, Response> { - let database_id = task.request.database_id; - let tenant_id = TenantId::new(tid); - let engine_key = (database_id, tenant_id); - let engine_existed_before = self.crdt_engines.contains_key(&engine_key); - let preimages = match catch_unwind(AssertUnwindSafe(|| { - self.capture_crdt_preimages(task, tenant_id, &crdt_deltas) - })) { - Ok(Ok(preimages)) => preimages, - Ok(Err(response)) => { - if !engine_existed_before { - self.crdt_engines.remove(&engine_key); - } - return Err(self.rollback_transaction_failure(task, tid, undo_log, response)); - } - Err(payload) => { - if !engine_existed_before { - self.crdt_engines.remove(&engine_key); - } - let response = self.response_error( - task, - ErrorCode::Internal { - detail: format!( - "panic while capturing CRDT transaction pre-images: {}", - panic_payload_to_string(payload.as_ref()) - ), - }, - ); - return Err(self.rollback_transaction_failure(task, tid, undo_log, response)); - } - }; - - let response = match catch_unwind(AssertUnwindSafe(|| { - self.apply_crdt_deltas(task, tid, crdt_deltas) - })) { - Ok(None) => return Ok(undo_log), - Ok(Some(response)) => response, - Err(payload) => self.response_error( - task, - ErrorCode::Internal { - detail: format!( - "panic during CRDT transaction gate: {}", - panic_payload_to_string(payload.as_ref()) - ), - }, - ), - }; - - Err(self.rollback_crdt_transaction_failure( - task, - tid, - undo_log, - response, - CrdtRollbackScope { - database_id, - tenant_id, - engine_existed_before, - preimages, - }, - )) - } - - fn capture_crdt_preimages( - &self, - task: &ExecutionTask, - tenant_id: TenantId, - crdt_deltas: &[CrdtDelta], - ) -> Result, Response> { - // Capture must be read-only. In particular, do not use - // `get_crdt_engine` here: creating an empty engine before every - // pre-commit gate would itself be an avoidable transactional side - // effect. The apply path creates it only after capture succeeds. - let engine = self - .crdt_engines - .get(&(task.request.database_id, tenant_id)); - let mut seen = std::collections::HashSet::with_capacity(crdt_deltas.len()); - let mut preimages = Vec::with_capacity(crdt_deltas.len()); - for (_, _, collection) in crdt_deltas { - if !seen.insert(collection.as_str()) { - continue; - } - let snapshot = match engine - .map(|engine| engine.export_snapshot_bytes(collection)) - .transpose() - { - Ok(snapshot) => snapshot.flatten(), - Err(error) => { - return Err(self.response_error( - task, - ErrorCode::Internal { - detail: format!( - "CRDT pre-image export failed for collection {collection}: {error}" - ), - }, - )); - } - }; - preimages.push(CrdtCollectionPreimage { - collection: collection.clone(), - snapshot, - }); - } - Ok(preimages) - } - - fn rollback_crdt_transaction_failure( - &mut self, - task: &ExecutionTask, - tid: u64, - undo_log: Vec, - response: Response, - rollback_scope: CrdtRollbackScope, - ) -> Response { - let crdt_restore = match catch_unwind(AssertUnwindSafe(|| { - self.restore_crdt_preimages(rollback_scope) - })) { - Ok(result) => result, - Err(payload) => Err(CrdtRestoreFailures { - failures: vec![CrdtCollectionRestoreFailure { - collection: "".into(), - detail: format!( - "panic while restoring CRDT pre-images: {}", - panic_payload_to_string(payload.as_ref()) - ), - }], - }), - }; - let mut response = self.rollback_transaction_failure(task, tid, undo_log, response); - if let Err(restore_failures) = crdt_restore { - error!( - core = self.core_id, - detail = %restore_failures, - "CRDT rollback restore failed; shard state unknown" - ); - let (entry_index, detail) = match response.error_code.as_deref() { - Some(ErrorCode::RollbackFailed { - entry_index, - detail, - }) => ( - *entry_index, - format!( - "forward undo rollback failed at entry {entry_index}: {detail}; \ - CRDT pre-image restoration also failed: {restore_failures}" - ), - ), - _ => ( - 0, - format!("CRDT pre-image restoration failed: {restore_failures}"), - ), - }; - response.error_code = Some(Box::new(ErrorCode::RollbackFailed { - entry_index, - detail, - })); - } - response - } - - fn restore_crdt_preimages( - &mut self, - rollback_scope: CrdtRollbackScope, - ) -> Result<(), CrdtRestoreFailures> { - let CrdtRollbackScope { - database_id, - tenant_id, - engine_existed_before, - preimages, - } = rollback_scope; - let engine_key = (database_id, tenant_id); - if !engine_existed_before { - self.crdt_engines.remove(&engine_key); - return Ok(()); - } - let Some(engine) = self.crdt_engines.get_mut(&engine_key) else { - return Err(CrdtRestoreFailures { - failures: vec![CrdtCollectionRestoreFailure { - collection: "".into(), - detail: "CRDT engine disappeared while rolling back a transaction".into(), - }], - }); - }; - - let mut failures = Vec::new(); - for CrdtCollectionPreimage { - collection, - snapshot, - } in preimages - { - // A single malformed/corrupt pre-image must not keep later - // collections from being restored. The outer gate also catches - // panics, but only this per-collection boundary preserves the - // aggregate failure contract on a panic. - match catch_unwind(AssertUnwindSafe(|| { - engine.restore_collection_snapshot(&collection, snapshot.as_deref()) - })) { - Ok(Ok(())) => {} - Ok(Err(error)) => failures.push(CrdtCollectionRestoreFailure { - collection, - detail: error.to_string(), - }), - Err(payload) => failures.push(CrdtCollectionRestoreFailure { - collection, - detail: format!( - "panic while restoring CRDT pre-image: {}", - panic_payload_to_string(payload.as_ref()) - ), - }), - } - } - if failures.is_empty() { - Ok(()) - } else { - Err(CrdtRestoreFailures { failures }) - } - } -} - -#[cfg(test)] -mod tests { - use loro::LoroValue; - use nodedb_crdt::state::CrdtState; - - use crate::bridge::envelope::{PhysicalPlan, Status}; - use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; - use crate::types::{DatabaseId, TenantId}; - use nodedb_physical::physical_plan::{CrdtOp, TimeseriesOp}; - use nodedb_types::QualifiedCollection; - - fn row_delta(peer: u64, row_id: &str) -> Vec { - let state = CrdtState::new(peer).expect("CRDT state"); - state - .upsert( - "crdt", - row_id, - &[("value", LoroValue::String(row_id.into()))], - ) - .expect("row upsert"); - state.export_snapshot().expect("delta snapshot") - } - - #[cfg(feature = "failpoints")] - #[test] - fn crdt_gate_panic_restores_deferred_timeseries_preimage() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_default_task(); - let plans = [ - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), - payload: b"metrics value=1i 1000000000\n".to_vec(), - format: "ilp".into(), - wal_lsn: None, - surrogates: Vec::new(), - provenance: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), - PhysicalPlan::Crdt(CrdtOp::Apply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "crdt"), - document_id: "row".into(), - delta: Vec::new(), - peer_id: 1, - mutation_id: 1, - surrogate: nodedb_types::Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }), - ]; - let _fail = crate::fail_point::FailGuard::install( - "transaction_batch::between_crdt_delta", - crate::fail_point::FailAction::Panic, - ); - - let response = core.execute_transaction_batch(&task, 1, &plans, &[], None); - - assert_eq!(response.status, Status::Error); - assert!( - !core.columnar_memtables.contains_key(&( - crate::types::DatabaseId::DEFAULT, - TenantId::new(1), - "metrics".to_string(), - )), - "a CRDT-gate panic must remove the deferred timeseries collection" - ); - } - - #[cfg(feature = "failpoints")] - #[test] - fn panic_between_subplans_rolls_back_the_first_timeseries_ingest_completely() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_default_task(); - let tenant_id = task.request.tenant_id; - let key = (task.request.database_id, tenant_id, "metrics".to_string()); - let plans = [ - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), - payload: b"metrics value=1i 1000000000\\n".to_vec(), - format: "ilp".into(), - wal_lsn: None, - surrogates: Vec::new(), - provenance: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), - PhysicalPlan::Crdt(CrdtOp::Apply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "crdt"), - document_id: "after-timeseries".into(), - delta: Vec::new(), - peer_id: 1, - mutation_id: 1, - surrogate: nodedb_types::Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }), - ]; - let _fail = crate::fail_point::FailGuard::install( - "transaction_batch::between_subapply", - crate::fail_point::FailAction::Panic, - ); - - let response = core.execute_transaction_batch(&task, tenant_id.as_u64(), &plans, &[], None); - - assert_eq!(response.status, Status::Error); - assert!(!core.columnar_memtables.contains_key(&key)); - assert!(!core.ts_last_value_caches.contains_key(&key)); - assert!(!core.ts_max_ingested_lsn.contains_key(&key)); - assert!(!core.ts_registries.contains_key(&key)); - assert!(!core.columnar_flushed_surrogates.contains_key(&key)); - assert!(core.last_ts_ingest.is_none()); - } - - #[test] - fn crdt_error_restores_a_previously_applied_delta() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_default_task(); - let tenant_id = task.request.tenant_id; - let original = row_delta(1, "before"); - core.get_crdt_engine(task.request.database_id, tenant_id) - .expect("CRDT engine") - .apply_committed_delta("crdt", &original) - .expect("seed CRDT state"); - - let result = core.apply_crdt_deltas_or_rollback( - &task, - tenant_id.as_u64(), - Vec::new(), - vec![ - (row_delta(2, "during"), 2, "crdt".to_string()), - (b"not a valid Loro delta".to_vec(), 3, "crdt".to_string()), - ], - ); - - assert!( - result.is_err(), - "the invalid second delta must fail the gate" - ); - let engine = core - .get_crdt_engine(task.request.database_id, tenant_id) - .expect("CRDT engine after rollback"); - assert!(engine.row_exists("crdt", "before")); - assert!(!engine.row_exists("crdt", "during")); - } - - #[test] - fn failed_batch_removes_crdt_engine_created_for_earlier_delta() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_default_task(); - let plans = [ - PhysicalPlan::Crdt(CrdtOp::Apply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "crdt"), - document_id: "during".into(), - delta: row_delta(2, "during"), - peer_id: 2, - mutation_id: 2, - surrogate: nodedb_types::Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }), - PhysicalPlan::Crdt(CrdtOp::Apply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "crdt"), - document_id: "bad".into(), - delta: b"not a valid Loro delta".to_vec(), - peer_id: 3, - mutation_id: 3, - surrogate: nodedb_types::Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }), - ]; - - let response = core.execute_transaction_batch(&task, 1, &plans, &[], None); - - assert_eq!(response.status, Status::Error); - assert!( - !core - .crdt_engines - .contains_key(&(crate::types::DatabaseId::DEFAULT, TenantId::new(1),)), - "an aborted first CRDT batch must restore the absent-engine pre-image" - ); - } - - #[test] - fn batch_crdt_error_restores_earlier_delta_and_forward_timeseries_write() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_default_task(); - let tenant_id = task.request.tenant_id; - let original = row_delta(1, "before"); - core.get_crdt_engine(task.request.database_id, tenant_id) - .expect("CRDT engine") - .apply_committed_delta("crdt", &original) - .expect("seed CRDT state"); - let plans = [ - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), - payload: b"metrics value=1i 1000000000\n".to_vec(), - format: "ilp".into(), - wal_lsn: None, - surrogates: Vec::new(), - provenance: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), - PhysicalPlan::Crdt(CrdtOp::Apply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "crdt"), - document_id: "during".into(), - delta: row_delta(2, "during"), - peer_id: 2, - mutation_id: 2, - surrogate: nodedb_types::Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }), - PhysicalPlan::Crdt(CrdtOp::Apply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "crdt"), - document_id: "bad".into(), - delta: b"not a valid Loro delta".to_vec(), - peer_id: 3, - mutation_id: 3, - surrogate: nodedb_types::Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }), - ]; - - let response = core.execute_transaction_batch(&task, tenant_id.as_u64(), &plans, &[], None); - - assert_eq!(response.status, Status::Error); - assert!( - !core.columnar_memtables.contains_key(&( - crate::types::DatabaseId::DEFAULT, - tenant_id, - "metrics".to_string(), - )), - "the forward timeseries write must roll back with a later CRDT failure" - ); - let engine = core - .get_crdt_engine(task.request.database_id, tenant_id) - .expect("CRDT engine after rollback"); - assert!(engine.row_exists("crdt", "before")); - assert!(!engine.row_exists("crdt", "during")); - } - - #[test] - fn restore_attempts_every_preimage_after_an_earlier_failure() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_default_task(); - let tenant_id = task.request.tenant_id; - core.get_crdt_engine(task.request.database_id, tenant_id) - .expect("CRDT engine") - .apply_committed_delta("removed", &row_delta(1, "before")) - .expect("seed CRDT state"); - - let result = core.restore_crdt_preimages(super::CrdtRollbackScope { - database_id: task.request.database_id, - tenant_id, - engine_existed_before: true, - preimages: vec![ - super::CrdtCollectionPreimage { - collection: "broken".into(), - snapshot: Some(b"not a Loro snapshot".to_vec()), - }, - super::CrdtCollectionPreimage { - collection: "removed".into(), - snapshot: None, - }, - ], - }); - - let failures = result.expect_err("an invalid snapshot must fail restore"); - assert_eq!(failures.failures.len(), 1); - assert_eq!(failures.failures[0].collection, "broken"); - assert!( - core.get_crdt_engine(task.request.database_id, tenant_id) - .expect("CRDT engine") - .export_snapshot_bytes("removed") - .expect("snapshot export") - .is_none(), - "a later preimage must still be restored after an earlier failure" - ); - } - - #[cfg(feature = "failpoints")] - #[test] - fn crdt_panic_after_import_restores_the_exact_preimage() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let task = make_default_task(); - let tenant_id = task.request.tenant_id; - let original = row_delta(1, "before"); - core.get_crdt_engine(task.request.database_id, tenant_id) - .expect("CRDT engine") - .apply_committed_delta("crdt", &original) - .expect("seed CRDT state"); - let _fail = crate::fail_point::FailGuard::install( - "transaction_batch::after_crdt_delta", - crate::fail_point::FailAction::Panic, - ); - - let result = core.apply_crdt_deltas_or_rollback( - &task, - tenant_id.as_u64(), - Vec::new(), - vec![(row_delta(2, "during"), 2, "crdt".to_string())], - ); - - assert!(result.is_err(), "the post-import panic must fail the gate"); - let engine = core - .get_crdt_engine(task.request.database_id, tenant_id) - .expect("CRDT engine after rollback"); - assert!(engine.row_exists("crdt", "before")); - assert!(!engine.row_exists("crdt", "during")); - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/batch_irreversible.rs b/nodedb/src/data/executor/handlers/transaction/batch_irreversible.rs deleted file mode 100644 index 823c8222f..000000000 --- a/nodedb/src/data/executor/handlers/transaction/batch_irreversible.rs +++ /dev/null @@ -1,146 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Sub-plans a transaction batch cannot roll back. -//! -//! A sub-plan with no engine-specific transaction handler runs through the -//! ordinary dispatch path. That path records no undo entry, so a later -//! rollback leaves its write in place. A failed batch that ran such a write -//! must not answer with a code that claims nothing applied: the Control -//! Plane cancels every record of a batch refused with such a code, and -//! recovery would then drop the write that stayed. - -use nodedb_physical::physical_plan::PhysicalPlan; - -use crate::bridge::envelope::{ErrorCode, Response}; -use crate::control::server::shared::write_admission::plan_is_write; -use crate::data::executor::handlers::partial_refusal::refusal_after_partial_apply; - -/// Whether `plan` wrote state the undo log does not cover. -/// -/// A tracked write that changed state always adds an undo entry. A write -/// plan that added none either ran untracked or changed nothing. Both count -/// here, which errs toward replaying the batch's records. -pub(super) fn applied_irreversibly( - plan: &PhysicalPlan, - undo_before: usize, - undo_after: usize, -) -> bool { - undo_after == undo_before && plan_is_write(plan) -} - -/// The code a failed batch answers with once its rollback finished. -/// -/// After an irreversible sub-plan applied, a definite refusal becomes -/// `Internal`, which keeps the batch's records for replay. -pub(super) fn batch_failure_code(code: ErrorCode, irreversible: bool) -> ErrorCode { - if irreversible { - refusal_after_partial_apply(code) - } else { - code - } -} - -/// [`batch_failure_code`] applied to a failed batch's response. -pub(super) fn batch_failure_response(mut response: Response, irreversible: bool) -> Response { - if let Some(code) = response.error_code.take() { - response.error_code = Some(Box::new(batch_failure_code(*code, irreversible))); - } - response -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::control::server::dispatch_utils::write_definitely_not_applied; - - #[test] - fn a_definite_refusal_after_an_irreversible_write_keeps_the_records() { - let code = ErrorCode::RejectedConstraint { - constraint: "unique".into(), - detail: "duplicate key".into(), - }; - let answered = batch_failure_code(code, true); - assert!(matches!(answered, ErrorCode::Internal { .. })); - assert!(!write_definitely_not_applied(&answered)); - } - - #[test] - fn a_definite_refusal_with_every_write_rolled_back_stays_definite() { - let code = ErrorCode::RejectedConstraint { - constraint: "unique".into(), - detail: "duplicate key".into(), - }; - assert_eq!(batch_failure_code(code.clone(), false), code); - } - - /// A predicate delete runs untracked. A later sub-plan refuses the batch, - /// the rollback leaves the delete in place, and the answer must not claim - /// that nothing applied. - #[test] - fn a_batch_refused_after_an_untracked_delete_keeps_its_records() { - use std::collections::HashMap; - - use nodedb_physical::physical_plan::{CrdtOp, KvOp}; - use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate, Value}; - - use crate::bridge::envelope::Status; - use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; - - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _req, _resp) = make_core_with_dir(dir.path()); - let task = make_default_task(); - let did = task.request.database_id.as_u64(); - let tid = task.request.tenant_id.as_u64(); - let collection = QualifiedCollection::new(DatabaseId::DEFAULT, "items"); - let value = zerompk::to_msgpack_vec(&Value::Object(HashMap::from([( - "v".to_string(), - Value::Integer(1), - )]))) - .expect("encode value"); - let seed = PhysicalPlan::Kv(KvOp::Put { - collection: collection.clone(), - key: b"k1".to_vec(), - value, - ttl_ms: 0, - surrogate: Surrogate::new(1), - returning: None, - rls_filters: Vec::new(), - }); - let seeded = core.execute_transaction_batch(&task, tid, &[seed], &[], None); - assert_eq!(seeded.status, Status::Ok, "seed put must apply"); - - let delete_all = PhysicalPlan::Kv(KvOp::PredicateDelete { - collection: collection.clone(), - filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }); - let refused = PhysicalPlan::Crdt(CrdtOp::Apply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), - document_id: "doc".into(), - delta: vec![1], - peer_id: 1, - mutation_id: 1, - surrogate: Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }); - let response = - core.execute_transaction_batch(&task, tid, &[delete_all, refused], &[], None); - - assert_eq!(response.status, Status::Error); - let code = response.error_code.map(|code| *code); - assert!( - matches!(code, Some(ErrorCode::Internal { .. })), - "got {code:?}" - ); - let now = crate::engine::kv::current_ms(); - assert_eq!( - core.kv_engine.get(did, tid, "items", b"k1", now), - None, - "the untracked delete stays after the rollback" - ); - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs b/nodedb/src/data/executor/handlers/transaction/index_write_values.rs deleted file mode 100644 index e80c99476..000000000 --- a/nodedb/src/data/executor/handlers/transaction/index_write_values.rs +++ /dev/null @@ -1,323 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Per-index write-VALUE recording for transaction batches — the -//! distributed-Calvin staging carrier and the fast-path/staged recorders. -//! Sibling of `write_version.rs` (per-key/collection versions). - -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; -use crate::types::TenantId; - -use super::undo::UndoEntry; - -/// Hard upper bound on the number of `(epoch, position, vshard)` buckets of -/// distributed-Calvin-flush index tuples staged and awaiting their post-apply -/// `RecordCalvinWriteVersions` drain. Overflow evicts the lowest-keyed bucket. -const MAX_CALVIN_FLUSH_INDEX_TUPLES: usize = 16_384; - -/// Staged index-value tuples for distributed-Calvin flushes, keyed by the batch -/// identity `(epoch, position, vshard)`. Each bucket holds one flushed batch's -/// `(collection, (field, value) pairs)`, awaiting the post-apply drain. -pub(in crate::data::executor) type StagedCalvinIndexTuples = - std::collections::HashMap<(u64, u32, u32), Vec<(String, Vec<(String, String)>)>>; - -/// Extract the `(collection, (field, value) tuples)` a document `UndoEntry` -/// touched, combining every index dimension the write mutated. Returns `None` -/// for non-document entries (whose per-key identity is engine-internal). Shared -/// by the immediate-record and distributed-Calvin-flush staging paths so both -/// derive the identical tuple set. -fn entry_index_tuples(entry: &UndoEntry) -> Option<(String, Vec<(String, String)>)> { - match entry { - UndoEntry::PutDocument { - collection, - secondary_index_added, - secondary_index_removed, - bitemporal_index_tuples, - .. - } => { - let mut tuples = Vec::with_capacity( - secondary_index_added.len() - + secondary_index_removed.len() - + bitemporal_index_tuples.len(), - ); - tuples.extend_from_slice(secondary_index_added); - tuples.extend_from_slice(secondary_index_removed); - tuples.extend_from_slice(bitemporal_index_tuples); - Some((collection.clone(), tuples)) - } - UndoEntry::DeleteDocument { - collection, - secondary_index_tuples, - bitemporal_index_tuples, - .. - } => { - let mut tuples = - Vec::with_capacity(secondary_index_tuples.len() + bitemporal_index_tuples.len()); - tuples.extend_from_slice(secondary_index_tuples); - tuples.extend_from_slice(bitemporal_index_tuples); - Some((collection.clone(), tuples)) - } - UndoEntry::InsertVector { .. } - | UndoEntry::DeleteVector { .. } - | UndoEntry::SpatialInsert { .. } - | UndoEntry::SpatialDelete { .. } - | UndoEntry::EdgeWrite(_) - | UndoEntry::KvPut { .. } - | UndoEntry::KvDelete { .. } - | UndoEntry::KvBatchPut { .. } - | UndoEntry::KvTransfer { .. } - | UndoEntry::KvTransferItem { .. } - | UndoEntry::KvTruncate { .. } - | UndoEntry::KvTtl { .. } - | UndoEntry::SortedIndexDdl { .. } - | UndoEntry::MarkNodeDeleted { .. } - | UndoEntry::NodeLabels { .. } - | UndoEntry::SpatialRow(_) - | UndoEntry::VectorWrite(_) - | UndoEntry::CrdtCollection(_) - | UndoEntry::ArrayTiles { .. } - | UndoEntry::SparseDoc { .. } - | UndoEntry::VectorTruncate(_) - | UndoEntry::FtsDocument(_) - | UndoEntry::SyncHwm { .. } - | UndoEntry::ColumnarInsert { .. } - | UndoEntry::ColumnarUpdate { .. } - | UndoEntry::ColumnarDelete { .. } - | UndoEntry::ColumnarEngineCreated { .. } - | UndoEntry::TimeseriesIngest(_) - | UndoEntry::ColumnarTruncate(_) - | UndoEntry::TimeseriesTruncate(_) - | UndoEntry::StatsRestore { .. } => None, - } -} - -impl CoreLoop { - /// Record the touched secondary-index VALUES of every document write in a - /// committed transaction batch into the per-index write-value substrate. - /// - /// Fast path — the task carries the batch's committed WAL LSN: record each - /// write's tuples immediately at that LSN. Distributed-Calvin-flush path — - /// the apply carries no WAL LSN (the committed LSN is only known post-apply) - /// but a `calvin_flush_key` is scoped: STAGE the tuples under that key for - /// the later `RecordCalvinWriteVersions` drain, in undo-log order. Otherwise - /// (no LSN, no flush key) no-op — the version is never advanced with a wrong - /// value. The undo log is the carrier: it already holds each write's - /// `(field, value)` tuples plus its collection, in deterministic plan order. - pub(in crate::data::executor) fn record_batch_index_write_values( - &mut self, - task: &ExecutionTask, - tid: u64, - undo_log: &[UndoEntry], - ) { - let db = task.request.database_id; - let tenant = TenantId::new(tid); - match (task.wal_lsn(), self.calvin_flush_key) { - (Some(lsn), _) => { - for entry in undo_log { - if let Some((collection, tuples)) = entry_index_tuples(entry) { - self.note_index_write_values(db, tenant, &collection, &tuples, lsn); - } - } - } - (None, Some(key)) => { - let mut staged: Vec<(String, Vec<(String, String)>)> = Vec::new(); - for entry in undo_log { - if let Some(pair) = entry_index_tuples(entry) { - staged.push(pair); - } - } - if staged.is_empty() { - return; - } - self.calvin_flush_index_tuples - .entry(key) - .or_default() - .extend(staged); - self.evict_calvin_flush_index_overflow(); - } - (None, None) => {} - } - } - - /// Bound the staged distributed-Calvin-flush index tuples: while the number - /// of `(epoch, position, vshard)` buckets exceeds - /// `MAX_CALVIN_FLUSH_INDEX_TUPLES`, evict the lowest-keyed bucket. The key - /// order is a total order, so eviction is deterministic across replicas. - fn evict_calvin_flush_index_overflow(&mut self) { - while self.calvin_flush_index_tuples.len() > MAX_CALVIN_FLUSH_INDEX_TUPLES { - let Some(victim) = self.calvin_flush_index_tuples.keys().min().copied() else { - break; - }; - self.calvin_flush_index_tuples.remove(&victim); - } - } - - /// Record the index-value tuples a distributed Calvin flush staged for - /// `(epoch, position, vshard)` at the replicated applied `lsn`, then drop the - /// staged entry. No-op if nothing was staged (single-shard fast path). - pub(in crate::data::executor) fn record_staged_calvin_index_values( - &mut self, - db: crate::types::DatabaseId, - tenant: TenantId, - epoch: u64, - position: u32, - vshard: u32, - lsn: crate::types::Lsn, - ) { - if let Some(entries) = self - .calvin_flush_index_tuples - .remove(&(epoch, position, vshard)) - { - for (collection, tuples) in entries { - self.note_index_write_values(db, tenant, &collection, &tuples, lsn); - } - } - } -} - -#[cfg(test)] -mod tests { - use std::time::{Duration, Instant}; - - use nodedb_types::Surrogate; - - use super::*; - use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request}; - use crate::data::executor::core_loop::tests::make_core_with_dir; - use crate::types::{DatabaseId, Lsn, RequestId, TraceId, VShardId}; - - /// A minimal `ExecutionTask` homing to vShard 0, tenant 1, database DEFAULT, - /// carrying no WAL LSN — the shape a distributed Calvin flush apply dispatches - /// under, so `record_batch_index_write_values` takes the staging branch. - fn make_task() -> ExecutionTask { - let plan = crate::bridge::envelope::PhysicalPlan::Meta( - nodedb_physical::physical_plan::meta::MetaOp::Compact, - ); - let request = Request { - request_id: RequestId::new(1), - tenant_id: TenantId::new(1), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - plan, - deadline: Instant::now() + Duration::from_secs(5), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: crate::types::ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: Admission::Exempt(ExemptReason::Read), - }; - ExecutionTask::new(request) - } - - fn put_entry(collection: &str, field: &str, value: &str) -> UndoEntry { - UndoEntry::PutDocument { - collection: collection.to_string(), - document_id: nodedb_types::StorageKey::for_surrogate(Surrogate::new(1)), - identity: nodedb_types::StorageKey::for_surrogate(Surrogate::new(1)).to_identity(), - old_value: None, - bitemporal_sys_from_ms: None, - bitemporal_index_tuples: Vec::new(), - secondary_index_added: vec![(field.to_string(), value.to_string())], - secondary_index_removed: Vec::new(), - chain_hash_prior: None, - } - } - - fn delete_entry(collection: &str, field: &str, value: &str) -> UndoEntry { - UndoEntry::DeleteDocument { - collection: collection.to_string(), - document_id: nodedb_types::StorageKey::for_surrogate(Surrogate::new(2)), - identity: nodedb_types::StorageKey::for_surrogate(Surrogate::new(2)).to_identity(), - old_value: Vec::new(), - bitemporal_sys_from_ms: None, - bitemporal_index_tuples: Vec::new(), - secondary_index_tuples: vec![(field.to_string(), value.to_string())], - chain_hash_prior: None, - } - } - - /// A distributed Calvin flush stages its index tuples under the flush key - /// (the apply carries `wal_lsn: None`) instead of recording them; the - /// post-apply drain records the staged put- AND delete-tuples at the - /// replicated applied LSN and empties the staging map. - #[test] - fn calvin_flush_stage_then_drain_records_index_values() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - - let task = make_task(); - let tenant = TenantId::new(1); - let db = DatabaseId::DEFAULT; - let key = (7u64, 3u32, 0u32); - - // Flush scope active: the batch's index tuples must STAGE, not record. - core.calvin_flush_key = Some(key); - let undo_log = vec![ - put_entry("orders", "email", "a@b.c"), - delete_entry("orders", "status", "gone"), - ]; - core.record_batch_index_write_values(&task, tenant.as_u64(), &undo_log); - - // Nothing recorded yet (the applied LSN is not known at flush time). - assert_eq!( - core.write_index - .index_values - .value_lsn(db, tenant, "orders", "email", "a@b.c"), - None, - "flush staging must not record into the substrate" - ); - assert!( - core.calvin_flush_index_tuples.contains_key(&key), - "the flush's index tuples must be staged under the flush key" - ); - - // Post-apply drain records both tuples at the replicated applied LSN. - core.record_staged_calvin_index_values(db, tenant, key.0, key.1, key.2, Lsn::new(42)); - - assert_eq!( - core.write_index - .index_values - .value_lsn(db, tenant, "orders", "email", "a@b.c"), - Some(Lsn::new(42)), - "the drain must record the staged PUT tuple at the applied LSN" - ); - assert_eq!( - core.write_index - .index_values - .value_lsn(db, tenant, "orders", "status", "gone"), - Some(Lsn::new(42)), - "the drain must record the staged DELETE tuple at the applied LSN" - ); - assert!( - core.calvin_flush_index_tuples.is_empty(), - "the drain must empty the staging map (cleanup)" - ); - } - - /// Draining a `(epoch, position, vshard)` nothing was staged under is an - /// idempotent no-op — the single-shard fast path never stages. - #[test] - fn drain_without_staging_is_noop() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let tenant = TenantId::new(1); - let db = DatabaseId::DEFAULT; - - core.record_staged_calvin_index_values(db, tenant, 1, 0, 0, Lsn::new(9)); - - assert!(core.calvin_flush_index_tuples.is_empty()); - assert_eq!( - core.write_index - .index_values - .value_lsn(db, tenant, "orders", "email", "a@b.c"), - None, - ); - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/mod.rs b/nodedb/src/data/executor/handlers/transaction/mod.rs index e6982cfc4..a332779eb 100644 --- a/nodedb/src/data/executor/handlers/transaction/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/mod.rs @@ -1,25 +1,11 @@ // SPDX-License-Identifier: BUSL-1.1 -mod batch; -mod batch_crdt; -mod batch_irreversible; -pub(in crate::data::executor) mod index_write_values; pub mod overlay; mod overlay_gauge; pub(in crate::data::executor) mod overlay_reap; pub(in crate::data::executor) mod redo_apply; pub(in crate::data::executor) mod resolve; pub(in crate::data::executor) mod stage_write; -mod sub_plan; -mod sub_plan_columnar; -mod sub_plan_doc; -mod sub_plan_kv; -mod sub_plan_kv_atomics; -mod sub_plan_kv_ops; -mod sub_plan_kv_ttl_sorted; -mod sub_plan_kv_writes; -mod sub_plan_write; -mod sub_request; pub(in crate::data::executor) mod undo; mod write_version; mod write_version_kv; diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/array_staged/txn_overlay.rs b/nodedb/src/data/executor/handlers/transaction/overlay/array_staged/txn_overlay.rs index c08fc41c8..f81085dee 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/array_staged/txn_overlay.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/array_staged/txn_overlay.rs @@ -9,9 +9,9 @@ //! //! Scope: this overlay serves read-your-own-writes for Slice / Project / //! Aggregate / Elementwise reads and the statement-time affected count of -//! `ArrayOp::Put` / `ArrayOp::Delete`. COMMIT durability is unchanged: the -//! buffered `ArrayOp` plan is replayed through the real `handle_array_put` / -//! `handle_array_delete` handlers inside the COMMIT `TransactionBatch`. This +//! `ArrayOp::Put` / `ArrayOp::Delete`. COMMIT serializes the buffered +//! `ArrayOp` plan into the transaction's redo record, which the redo install +//! applies. This //! overlay is in-memory only and is dropped at commit or rollback, same //! lifecycle as `super::TxnOverlay`. //! diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/graph_staged/txn_overlay.rs b/nodedb/src/data/executor/handlers/transaction/overlay/graph_staged/txn_overlay.rs index 9e18f0644..0d07b7095 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/graph_staged/txn_overlay.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/graph_staged/txn_overlay.rs @@ -10,10 +10,9 @@ //! (which is keyed by `u32` surrogate), so this is a parallel, independent //! overlay type held alongside it on `CoreLoop` (`graph_txn_overlays`). //! -//! Scope: this overlay only serves read-your-own-writes for Neighbors / Hop -//! (single-hop reads). COMMIT durability is unchanged -- the buffered -//! `GraphOp` plan is still replayed through the real `execute_edge_put` / -//! `execute_edge_delete` / ... handlers inside the COMMIT `TransactionBatch`. +//! Scope: this overlay serves read-your-own-writes for Neighbors / Hop +//! (single-hop reads), and COMMIT resolves the staged edges and labels into +//! the transaction's redo record, which the redo install applies. //! This overlay is in-memory only and is dropped at commit or rollback, same //! lifecycle as `super::TxnOverlay`. //! diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs index f59b1f4a5..75022a00f 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/merge.rs @@ -39,7 +39,6 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::filter_match::matches_with_resolved_schema; use crate::data::executor::handlers::transaction::overlay::{Staged, StagedTtl}; use crate::engine::document::store::extract_index_values; -use crate::engine::kv::current_ms; use crate::types::{DatabaseId, TenantId, TxnId}; /// Inputs for [`CoreLoop::merge_overlay_into_index_lookup`]. @@ -209,7 +208,7 @@ impl CoreLoop { .map(|(key, _)| super::super::stage_write::kv_row_identity(key)) .collect(); - let now_ms = current_ms(); + let now_ms = self.kv_read_now_ms(); let staged_expired = |doc_id: &RowIdentity| -> bool { matches!( overlay.get_ttl_by_doc_id(coll_key, doc_id), diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/mod.rs b/nodedb/src/data/executor/handlers/transaction/overlay/mod.rs index c56de011c..2a207fb20 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/mod.rs @@ -25,7 +25,7 @@ pub(in crate::data::executor) use fts_merge::FtsMergeParams; pub use graph_staged::{GraphCollKey, GraphTxnOverlay, NodeLabelDelta}; pub(in crate::data::executor) use merge::IndexOverlayMergeParams; pub(in crate::data::executor) use spatial_merge::SpatialOverlayMergeParams; -pub use staged::{CollectionOverlay, MAX_TXN_OVERLAY_BYTES, Staged, TxnOverlay}; +pub use staged::{CollectionOverlay, MAX_TXN_OVERLAY_BYTES, Staged, TouchedSlot, TxnOverlay}; pub use staged_sidecar::{BitemporalStamp, StagedTtl}; pub use staged_vector::StagedVectorRow; pub(in crate::data::executor) use timeseries_merge::TimeseriesOverlayMergeParams; diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs b/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs index 9a013f828..e75267ae4 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/staged.rs @@ -5,7 +5,7 @@ //! Holds the not-yet-durable writes an in-flight transaction has executed at //! statement time (`MetaOp::StageWrite`), so an in-transaction point write //! returns its real command tag and raises constraint violations immediately, -//! while COMMIT's `TransactionBatch` replay remains the sole durable apply. +//! while COMMIT's redo install remains the sole durable apply. //! //! Keying rationale: the real storage key for a document is the SURROGATE //! (`u32`) — `apply_point_put` keys `sparse.versioned_put_in_txn` by @@ -63,6 +63,10 @@ pub struct CollectionOverlay { /// timestamp, keyed by the batch's first surrogate. COMMIT resolve stamps /// the batch's untimed rows with it. Never consulted by other engines. pub(super) ingest_now_by_surrogate: HashMap, + /// The instant each staged unkeyed timeseries ingest read, in stage + /// order. COMMIT resolve stamps the untimed rows of the Nth unkeyed + /// ingest with the Nth instant. + pub(super) unkeyed_ingest_now: Vec, } impl CollectionOverlay { @@ -74,6 +78,7 @@ impl CollectionOverlay { && self.bitemporal_by_surrogate.is_empty() && self.base_pk_by_surrogate.is_empty() && self.ingest_now_by_surrogate.is_empty() + && self.unkeyed_ingest_now.is_empty() } } @@ -84,7 +89,7 @@ impl CollectionOverlay { /// dropping post-savepoint entries would lose an earlier same-slot write. /// Restoring the recorded prior slot rewinds without that loss. #[derive(Debug, Clone)] -struct OverlayUndo { +pub(super) struct OverlayUndo { coll_key: (DatabaseId, TenantId, String), surrogate: u32, doc_id: RowIdentity, @@ -96,9 +101,22 @@ struct OverlayUndo { prev_doc_binding: Option, } +/// One overlay slot a staged mutation touched after a journal marker. +#[derive(Debug)] +pub struct TouchedSlot<'a> { + pub surrogate: u32, + /// The client identity the first mutation after the marker bound. + pub doc_id: &'a RowIdentity, + /// The slot's staged value at the marker. `None` means the row was + /// unstaged, so its value at the marker is its base row. + pub before: Option<&'a Staged>, + /// The slot's staged value now. + pub after: Option<&'a Staged>, +} + /// One undo-journal entry: a slot mutation or a truncate marker. #[derive(Debug, Clone)] -enum JournalEntry { +pub(super) enum JournalEntry { /// A slot's prior state, captured before a staged value/TTL mutation. Slot(OverlayUndo), /// A truncate marker set on `coll_key`. `prev` is the journal position @@ -108,6 +126,10 @@ enum JournalEntry { coll_key: (DatabaseId, TenantId, String), prev: Option, }, + /// An unkeyed timeseries ingest instant appended to `coll_key`. + UnkeyedIngest { + coll_key: (DatabaseId, TenantId, String), + }, } /// Per-transaction staging overlay: holds not-yet-durable writes for every @@ -126,7 +148,7 @@ pub struct TxnOverlay { /// reverse down to a marker. Always appended to by the value/TTL mutators /// and `mark_truncated` so nothing escapes it; dropped with the overlay /// when the transaction resolves. - journal: Vec, + pub(super) journal: Vec, /// Advanced by every staged write AND every in-transaction /// read-your-own-write, so a live transaction's stamp always tracks the /// clock. See [`LeaseStamp`]. @@ -218,6 +240,12 @@ impl TxnOverlay { self.truncated.contains_key(coll_key) } + /// Whether this transaction staged anything for `coll_key`: a value, a + /// tombstone, a TTL delta, or a truncate marker. + pub fn stages_collection(&self, coll_key: &(DatabaseId, TenantId, String)) -> bool { + self.is_truncated(coll_key) || self.collections.contains_key(coll_key) + } + /// Whether a base row of `coll_key` with no staged mutation is visible to /// this transaction: hidden while the collection is truncated. pub fn base_visible(&self, coll_key: &(DatabaseId, TenantId, String)) -> bool { @@ -300,6 +328,33 @@ impl TxnOverlay { self.journal.len() } + /// Every slot of `coll_key` that a staged value or TTL mutation touched + /// after `marker`. Each slot appears once, in the order it was first + /// touched. + pub fn slots_touched_since( + &self, + marker: usize, + coll_key: &(DatabaseId, TenantId, String), + ) -> Vec> { + let mut seen: std::collections::HashSet = std::collections::HashSet::new(); + let mut slots = Vec::new(); + for entry in self.journal.iter().skip(marker) { + let JournalEntry::Slot(undo) = entry else { + continue; + }; + if &undo.coll_key != coll_key || !seen.insert(undo.surrogate) { + continue; + } + slots.push(TouchedSlot { + surrogate: undo.surrogate, + doc_id: &undo.doc_id, + before: undo.prev_value.as_ref(), + after: self.get(coll_key, undo.surrogate), + }); + } + slots + } + /// Revert every staged value/TTL mutation and truncate marker recorded /// after `marker`, restoring each slot to its pre-mutation state (or /// removing it when the prior slot was absent), then truncate the journal @@ -322,6 +377,12 @@ impl TxnOverlay { }; continue; } + JournalEntry::UnkeyedIngest { coll_key } => { + if let Some(overlay) = self.collections.get_mut(&coll_key) { + overlay.unkeyed_ingest_now.pop(); + } + continue; + } }; let Some(overlay) = self.collections.get_mut(&undo.coll_key) else { continue; @@ -469,6 +530,31 @@ mod tests { assert_eq!(collected.len(), 1); } + /// A slot touched twice after the marker is reported once, with its value + /// at the marker and its value now. Other collections and earlier + /// mutations are left out. + #[test] + fn slots_touched_since_reports_each_slot_once_with_its_marker_value() { + let mut overlay = TxnOverlay::new(); + overlay.insert_put(key("users"), 1, &id("a"), vec![1]); + let marker = overlay.journal_len(); + overlay.insert_put(key("users"), 1, &id("a"), vec![2]); + overlay.insert_put(key("users"), 2, &id("b"), vec![3]); + overlay.insert_tombstone(key("users"), 1, &id("a")); + overlay.insert_put(key("orders"), 1, &id("x"), vec![4]); + + let slots = overlay.slots_touched_since(marker, &key("users")); + + assert_eq!(slots.len(), 2); + assert_eq!(slots[0].surrogate, 1); + assert_eq!(slots[0].doc_id, &id("a")); + assert_eq!(slots[0].before, Some(&Staged::Put(vec![1]))); + assert_eq!(slots[0].after, Some(&Staged::Tombstone)); + assert_eq!(slots[1].surrogate, 2); + assert_eq!(slots[1].before, None); + assert_eq!(slots[1].after, Some(&Staged::Put(vec![3]))); + } + #[test] fn insert_tombstone_and_lookup() { let mut overlay = TxnOverlay::new(); diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/staged_sidecar.rs b/nodedb/src/data/executor/handlers/transaction/overlay/staged_sidecar.rs index a23972844..82b442879 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/staged_sidecar.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/staged_sidecar.rs @@ -8,7 +8,7 @@ use nodedb_types::RowIdentity; -use super::staged::TxnOverlay; +use super::staged::{JournalEntry, TxnOverlay}; use crate::types::{DatabaseId, TenantId}; /// A staged TTL delta for one KV row, kept OUTSIDE `Staged` because TTL is @@ -89,6 +89,30 @@ impl TxnOverlay { overlay.ttl_by_surrogate.get(surrogate).copied() } + /// Every staged TTL delta of `coll_key` whose row has no staged value: + /// an `EXPIRE` or `PERSIST` of a base row. Yields the row's client + /// identity and its delta. + pub fn iter_ttl_only_for_collection<'a>( + &'a self, + coll_key: &(DatabaseId, TenantId, String), + ) -> impl Iterator { + self.collections + .get(coll_key) + .into_iter() + .flat_map(|overlay| { + overlay + .doc_id_to_surrogate + .iter() + .filter(move |(_, surrogate)| !overlay.by_surrogate.contains_key(surrogate)) + .filter_map(move |(doc_id, surrogate)| { + overlay + .ttl_by_surrogate + .get(surrogate) + .map(|ttl| (doc_id, *ttl)) + }) + }) + } + /// Record the resolve-time bitemporal stamp for `surrogate` in the given /// collection. Assigned exactly once, at COMMIT resolve, after all /// savepoint activity for the transaction has completed — so no undo @@ -186,15 +210,33 @@ impl TxnOverlay { .copied() } - /// Iterate every `(surrogate, BitemporalStamp)` staged across all - /// collections in this overlay. Surrogates are globally unique, so the - /// commit-time install flattens these into one per-core scratch map. - pub fn all_bitemporal_stamps(&self) -> impl Iterator + '_ { - self.collections.values().flat_map(|overlay| { - overlay - .bitemporal_by_surrogate - .iter() - .map(|(surrogate, stamp)| (*surrogate, *stamp)) - }) + /// Record the instant a staged unkeyed timeseries ingest read as its + /// default row timestamp. A savepoint rollback removes it. + pub fn note_unkeyed_ingest_now( + &mut self, + coll_key: &(DatabaseId, TenantId, String), + now_ms: i64, + ) { + self.collections + .entry(coll_key.clone()) + .or_default() + .unkeyed_ingest_now + .push(now_ms); + self.journal.push(JournalEntry::UnkeyedIngest { + coll_key: coll_key.clone(), + }); + } + + /// The instant the `ordinal`-th unkeyed ingest into `coll_key` read. + pub fn unkeyed_ingest_now( + &self, + coll_key: &(DatabaseId, TenantId, String), + ordinal: usize, + ) -> Option { + self.collections + .get(coll_key)? + .unkeyed_ingest_now + .get(ordinal) + .copied() } } diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/calvin_fold_tests.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/calvin_fold_tests.rs new file mode 100644 index 000000000..12b9745f9 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/calvin_fold_tests.rs @@ -0,0 +1,160 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A Calvin redo record folds its materialized sums from its stamp, the same +//! way live and in restart replay. + +use nodedb_physical::physical_plan::{MaterializedSumBinding, RedoSumTargets, ResolvedSumTarget}; +use nodedb_types::{DatabaseId, Surrogate, TenantId}; +use nodedb_wal::record::{RecordType, WalRecordArgs}; +use nodedb_wal::{TombstoneSet, WalRecord}; + +use super::entry::CommittedRedo; +use super::test_commit::doc_put_sub_record; +use crate::bridge::envelope::Status; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; +use crate::data::executor::doc_format; +use crate::engine::document::store::CollectionConfig; +use crate::types::Lsn; +use crate::wal::{CalvinStamp, RedoRecord}; + +const DB: u64 = 0; +const TID: u64 = 1; +/// Source and target share a vShard, so the inline fold writes the target +/// on this core. +const SOURCE: &str = "local_charges"; +const TARGET: &str = "local_balances"; +const ACCOUNT: &str = "a1"; +const TARGET_SURROGATE: Surrogate = Surrogate(4242); +const LSN: u64 = 70; + +/// A core whose source collection folds `amount` into the target's +/// `balance`, with the target row seeded at 100. +fn core_with_sum(dir: &std::path::Path) -> (CoreLoop, Box) { + let (mut core, req, resp) = make_core_with_dir(dir); + core.doc_configs.insert( + (DatabaseId::DEFAULT, TenantId::new(TID), TARGET.to_string()), + CollectionConfig::new(TARGET), + ); + let mut source = CollectionConfig::new(SOURCE); + source.enforcement.materialized_sum_sources = vec![MaterializedSumBinding { + target_collection: TARGET.to_string(), + target_column: "balance".to_string(), + join_column: "account_id".to_string(), + value_expr: nodedb_query::expr::SqlExpr::Column("amount".to_string()), + declared_primary_key: None, + }]; + core.doc_configs.insert( + (DatabaseId::DEFAULT, TenantId::new(TID), SOURCE.to_string()), + source, + ); + let seed = serde_json::json!({"id": ACCOUNT, "balance": "100"}); + core.sparse + .put( + DB, + TID, + TARGET, + &nodedb_types::StorageKey::for_surrogate(TARGET_SURROGATE), + &doc_format::encode_to_msgpack(&seed), + ) + .expect("seed target row"); + (core, Box::new((req, resp))) +} + +fn sum_targets() -> Vec { + vec![RedoSumTargets { + collection: SOURCE.to_string(), + resolved: vec![ResolvedSumTarget::new(TARGET, ACCOUNT, TARGET_SURROGATE)], + deferred: Vec::new(), + }] +} + +/// A Calvin redo record writing one source row of `amount` 25. +fn calvin_redo() -> Vec { + let body = doc_format::encode_to_msgpack(&serde_json::json!({ + "account_id": ACCOUNT, + "amount": 25, + })); + RedoRecord { + version: 1, + ops: vec![doc_put_sub_record(SOURCE, "c1", &body, 11)], + calvin_stamp: Some(CalvinStamp { + epoch: 3, + position: 0, + vshard_id: 0, + collections: vec![SOURCE.to_string()], + sum_targets: sum_targets(), + }), + } + .to_bytes() + .expect("encode redo") +} + +fn balance(core: &CoreLoop) -> Option { + let stored = core + .sparse + .get( + DB, + TID, + TARGET, + &nodedb_types::StorageKey::for_surrogate(TARGET_SURROGATE), + ) + .expect("read target")?; + doc_format::decode_document(&stored) + .ok()? + .get("balance") + .and_then(|v| v.as_str()) + .map(str::to_string) +} + +#[test] +fn restart_replay_folds_a_calvin_record_to_the_live_total() { + let redo = calvin_redo(); + + let live_dir = tempfile::tempdir().expect("tempdir"); + let (mut live, _live_ends) = core_with_sum(live_dir.path()); + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(LSN)); + let response = live.install_committed_redo( + &task, + TID, + CommittedRedo { + redo: &redo, + collections: &[SOURCE.to_string()], + sum_targets: &sum_targets(), + }, + ); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + assert_eq!(balance(&live).as_deref(), Some("125")); + + let replay_dir = tempfile::tempdir().expect("tempdir"); + let (mut replayed, _replay_ends) = core_with_sum(replay_dir.path()); + let record = WalRecord::new(WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn: LSN, + tenant_id: TID, + vshard_id: 0, + database_id: DB, + payload: redo, + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record"); + replayed + .replay_transaction_redo_wal(std::slice::from_ref(&record), 1, &TombstoneSet::new()) + .expect("replay"); + assert_eq!( + balance(&replayed), + balance(&live), + "restart replay folds to the total the live install produced" + ); + + replayed + .replay_transaction_redo_wal(std::slice::from_ref(&record), 1, &TombstoneSet::new()) + .expect("replay again"); + assert_eq!( + balance(&replayed).as_deref(), + Some("125"), + "a second replay over the installed row folds nothing" + ); +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs index e75f8356b..7c597b53f 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs @@ -38,7 +38,11 @@ impl WrittenEngines { pub(super) fn of(redo: &RedoRecord) -> Self { Self { vectors: redo.ops.iter().any(writes_vector_index), - kv: !kv_ops(&redo.ops).is_empty() || redo.ops.iter().any(is_kv_truncate), + kv: !kv_ops(&redo.ops).is_empty() + || redo + .ops + .iter() + .any(|op| is_kv_truncate(op) || is_kv_ttl(op)), columnar: redo.ops.iter().any(writes_columnar), } } @@ -67,6 +71,16 @@ fn is_kv_truncate(op: &RedoSubRecord) -> bool { .is_ok_and(|(disc, _)| disc == "kv_truncate") } +/// A `kv_expire` or `kv_persist` sub-record: both lead with their +/// discriminator, and the collection follows it. +fn is_kv_ttl(op: &RedoSubRecord) -> bool { + RecordType::from_raw(op.record_type) == Some(RecordType::Put) + && (zerompk::from_msgpack::<(String, String, Vec)>(&op.payload) + .is_ok_and(|(disc, ..)| disc == "kv_persist") + || zerompk::from_msgpack::<(String, String, Vec, u64, u64)>(&op.payload) + .is_ok_and(|(disc, ..)| disc == "kv_expire")) +} + fn writes_columnar(op: &RedoSubRecord) -> bool { match RecordType::from_raw(op.record_type) { Some(RecordType::ColumnarTruncate) => true, diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs index 06a576a06..b631d8b89 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs @@ -6,8 +6,9 @@ //! path. A replica applying a committed record re-executes the write the way //! the transaction batch did: it links the hash chain on an insert and folds //! the row into its materialized-sum targets inside the row's own write -//! transaction. Restart replay does neither, because the chained row and the -//! target rows are already durable by then. +//! transaction. Restart replay runs this path only for a Calvin record whose +//! stamp names sum targets: the fold subtracts the row's prior image, so a +//! row that already holds its post-image folds nothing. //! //! Constraint checks ran before the first write (see `validate`), so the //! writes run with `enforce = false`. In the install pass every written row @@ -51,7 +52,7 @@ impl CoreLoop { value: &[u8], ) -> bool { let result = self.committed_document_put(&row, value); - self.settle_committed_write(result.map(|()| true)) + self.settle_committed_write(row.record_lsn, result.map(|()| true)) } /// Remove one document row of a committed record. Returns whether a row @@ -61,15 +62,23 @@ impl CoreLoop { row: CommittedDocWrite<'_>, ) -> bool { let result = self.committed_document_delete(&row); - self.settle_committed_write(result) + self.settle_committed_write(row.record_lsn, result) } - fn settle_committed_write(&mut self, result: crate::Result) -> bool { + /// Keep a failed write's error on the open scope. Restart replay logs it + /// with the record's LSN and skips the row, as its plain path does. + fn settle_committed_write(&mut self, record_lsn: u64, result: crate::Result) -> bool { match result { Ok(applied) => applied, Err(error) => { - if let Some(scope) = self.redo_apply.scope.as_mut() { - scope.record_error(error); + match self.redo_apply.scope.as_mut() { + Some(scope) => scope.record_error(error), + None => self.replay_record_rejected( + "document", + record_lsn, + None, + &format!("folding a Calvin redo document write failed: {error}"), + ), } false } @@ -81,7 +90,7 @@ impl CoreLoop { row: &CommittedDocWrite<'_>, value: &[u8], ) -> crate::Result<()> { - let (resolved, deferred) = self.committed_sum_targets(row.collection); + let (resolved, deferred) = self.committed_sum_targets(row.collection, row.record_lsn); let surrogate = Surrogate::new(row.surrogate); let storage_key = StorageKey::for_surrogate(surrogate); let wal_lsn = (row.record_lsn != 0).then(|| Lsn::new(row.record_lsn)); @@ -190,7 +199,6 @@ impl CoreLoop { tid: row.tenant_id, collection: row.collection, storage_key, - identity: RowIdentity::from_user_key(row.document_id), }, outcome, chain.prior(), @@ -202,10 +210,9 @@ impl CoreLoop { } fn committed_document_delete(&mut self, row: &CommittedDocWrite<'_>) -> crate::Result { - let (resolved, _) = self.committed_sum_targets(row.collection); + let (resolved, _) = self.committed_sum_targets(row.collection, row.record_lsn); let surrogate = Surrogate::new(row.surrogate); let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); let hook_ctx = HookCtx { database_id: row.database_id, tid: row.tenant_id, @@ -223,7 +230,9 @@ impl CoreLoop { database_id: row.database_id, tid: row.tenant_id, collection: row.collection, - document_id: row_key.as_str(), + // The graph cascade keys nodes by the client key, as the + // autocommit delete does. + document_id: row.document_id, surrogate, user_roles: &[], enforce: false, @@ -262,7 +271,6 @@ impl CoreLoop { tid: row.tenant_id, collection: row.collection, storage_key, - identity: RowIdentity::from_user_key(row.document_id), }, outcome, ); @@ -300,18 +308,24 @@ impl CoreLoop { ); } + /// The sum targets a write to `collection` folds into: the open scope's + /// under a committed-redo apply, else the restart-replay folds of the + /// record at `record_lsn`. fn committed_sum_targets( &self, collection: &str, + record_lsn: u64, ) -> ( Vec, Vec, ) { - self.redo_apply - .scope - .as_ref() - .map(|scope| scope.sum_targets_for(collection)) - .unwrap_or_default() + match self.redo_apply.scope.as_ref() { + Some(scope) => scope.sum_targets_for(collection), + None => self + .redo_apply + .replay_folds_for(record_lsn, collection) + .unwrap_or_default(), + } } fn record_committed_doc_write(&mut self, write: AppliedDocWrite) { diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs index 76c84b8c0..656ca544f 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs @@ -63,6 +63,19 @@ impl CoreLoop { task: &ExecutionTask, tid: u64, committed: CommittedRedo<'_>, + ) -> Response { + self.install_committed_redo(task, tid, committed) + } + + /// Install one committed redo record at the LSN the request carries: + /// validate every sub-record, install with undo, then settle and cover. + /// Every committed transaction installs here, whichever path committed it, + /// and restart replay drives the same arms over the same record. + pub(in crate::data::executor) fn install_committed_redo( + &mut self, + task: &ExecutionTask, + tid: u64, + committed: CommittedRedo<'_>, ) -> Response { let Some(lsn) = task.wal_lsn() else { return self.response_error( @@ -139,8 +152,21 @@ impl CoreLoop { ); return self.response_error(task, error); } - // Events leave only once the record is settled and covered. - for event in std::mem::take(&mut scope.pending_events) { + // Write versions and the watermark move only once the record is + // settled and covered. + for version in std::mem::take(&mut scope.write_versions) { + self.publish_write_version( + version.db, + version.tenant, + &version.collection, + version.key, + version.lsn, + ); + } + // Events leave only once the record is settled and covered. Every + // one names the record's LSN: the install held the watermark back. + for mut event in std::mem::take(&mut scope.pending_events) { + event.lsn = lsn; self.send_write_event(event); } diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs index 0f7b1849e..cb865600f 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs @@ -41,7 +41,7 @@ impl CoreLoop { tid: u64, ops: Vec, ) -> Vec { - let now_ms = crate::engine::kv::current_ms(); + let now_ms = self.kv_read_now_ms(); let mut images = Vec::new(); for op in ops { match op { diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/install_refusal_tests.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/install_refusal_tests.rs new file mode 100644 index 000000000..5db6262e6 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/install_refusal_tests.rs @@ -0,0 +1,534 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One committed transaction per engine kind, installed and then refused. +//! +//! Each kind commits once to show where its write lands, and once with a +//! sub-record after its own that fails while it installs. The refused +//! install rolls the write back, so the core holds none of it. + +use nodedb_array::schema::ArraySchemaBuilder; +use nodedb_array::schema::attr_spec::{AttrSpec, AttrType}; +use nodedb_array::schema::dim_spec::{DimSpec, DimType}; +use nodedb_array::types::ArrayId; +use nodedb_array::types::cell_value::value::CellValue; +use nodedb_array::types::coord::value::CoordValue; +use nodedb_array::types::domain::{Domain, DomainBound}; +use nodedb_physical::physical_plan::{ + ArrayOp, CrdtOp, CrdtWriteVerb, DocumentOp, KvOp, PhysicalPlan, SpatialOp, VectorOp, +}; +use nodedb_types::geometry::Geometry; +use nodedb_types::{DatabaseId, QualifiedCollection, RlsWriteCheck, Surrogate, TenantId, Value}; + +use crate::bridge::envelope::{ErrorCode, Response, Status}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; +use crate::engine::array::wal::ArrayPutCell; + +const TID: u64 = 1; + +fn assert_committed(response: &Response) { + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); +} + +fn assert_refused_at_install(response: &Response) { + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) + ), + "the install fails after the transaction's writes: {:?}", + response.error_code + ); + assert_eq!(response.status, Status::Error); +} + +/// Commit `plans` on a fresh core, or refuse their install when `refuse`. +fn run(plans: &[PhysicalPlan], refuse: bool, setup: impl FnOnce(&mut CoreLoop)) -> CoreLoopHold { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, req, resp) = make_core_with_dir(dir.path()); + setup(&mut core); + let task = make_default_task(); + let response = if refuse { + core.commit_plans_then_refuse_for_test(&task, TID, plans, 200) + } else { + core.commit_plans_for_test(&task, TID, plans, 200) + }; + if refuse { + assert_refused_at_install(&response); + } else { + assert_committed(&response); + } + CoreLoopHold { + core, + _req: Box::new(req), + _resp: Box::new(resp), + _dir: dir, + } +} + +/// A core with its bridge ends and data directory kept alive. +struct CoreLoopHold { + core: CoreLoop, + _req: Box, + _resp: Box, + _dir: tempfile::TempDir, +} + +fn object(fields: &[(&str, Value)]) -> Vec { + let map: std::collections::HashMap = fields + .iter() + .map(|(k, v)| ((*k).to_string(), v.clone())) + .collect(); + zerompk::to_msgpack_vec(&Value::Object(map)).expect("encode object") +} + +// ── Document ───────────────────────────────────────────────────────────── + +fn document_insert() -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), + document_id: "n1".to_string(), + value: object(&[("title", Value::String("kept".into()))]), + if_absent: false, + surrogate: Surrogate::new(41), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }) +} + +fn document_row(core: &CoreLoop) -> Option> { + core.sparse + .get( + 0, + TID, + "notes", + &nodedb_types::StorageKey::for_surrogate(Surrogate::new(41)), + ) + .expect("read row") +} + +#[test] +fn a_document_write_lands_and_a_refused_install_removes_it() { + assert!(document_row(&run(&[document_insert()], false, |_| {}).core).is_some()); + assert!(document_row(&run(&[document_insert()], true, |_| {}).core).is_none()); +} + +// ── KV predicate form ──────────────────────────────────────────────────── + +fn seed_kv(core: &mut CoreLoop) { + core.kv_engine.put(crate::engine::kv::KvPutParams { + database_id: 0, + tenant_id: TID, + collection: "cache", + key: b"k", + value: &object(&[("n", Value::Integer(1))]), + ttl_ms: 0, + now_ms: crate::engine::kv::current_ms(), + surrogate: Surrogate::new(51), + }); +} + +fn kv_predicate_delete() -> PhysicalPlan { + PhysicalPlan::Kv(KvOp::PredicateDelete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), + filters: Vec::new(), + rls_write_check: RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }) +} + +fn kv_present(core: &CoreLoop) -> bool { + core.kv_engine + .get(0, TID, "cache", b"k", crate::engine::kv::current_ms()) + .is_some() +} + +#[test] +fn a_kv_predicate_delete_lands_and_a_refused_install_restores_the_key() { + assert!(!kv_present( + &run(&[kv_predicate_delete()], false, seed_kv).core + )); + assert!(kv_present( + &run(&[kv_predicate_delete()], true, seed_kv).core + )); +} + +// ── Vector (vector-primary direct write) ───────────────────────────────── + +fn vector_direct_insert() -> PhysicalPlan { + PhysicalPlan::Vector(VectorOp::DirectInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vp"), + field: "vec".into(), + surrogate: Surrogate::new(61), + pk_bytes: b"r".to_vec(), + vector: vec![1.0, 0.0], + payload: zerompk::to_msgpack_vec(&std::collections::HashMap::from([( + "id".to_string(), + Value::String("r".into()), + )])) + .expect("encode payload"), + quantization: nodedb_types::VectorQuantization::None, + storage_dtype: nodedb_types::VectorStorageDtype::F32, + payload_indexes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + }) +} + +fn vector_index_present(core: &CoreLoop) -> bool { + core.vector_collections + .contains_key(&CoreLoop::vector_index_key(0, TID, "vp", "vec")) +} + +#[test] +fn a_vector_primary_write_lands_and_a_refused_install_withdraws_it() { + assert!(vector_index_present( + &run(&[vector_direct_insert()], false, |_| {}).core + )); + assert!(!vector_index_present( + &run(&[vector_direct_insert()], true, |_| {}).core + )); +} + +// ── CRDT ───────────────────────────────────────────────────────────────── + +fn crdt_upsert() -> PhysicalPlan { + PhysicalPlan::Crdt(CrdtOp::DocUpsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "tasks"), + document_id: "t1".to_string(), + fields_json: r#"{"title":"kept"}"#.to_string(), + surrogate: Surrogate::new(71), + partial: false, + verb: CrdtWriteVerb::Insert, + returning: None, + rls_filters: Vec::new(), + }) +} + +fn crdt_row_present(core: &CoreLoop) -> bool { + core.crdt_engines + .get(&(DatabaseId::DEFAULT, TenantId::new(TID))) + .and_then(|engine| engine.read_row("tasks", "t1")) + .is_some() +} + +#[test] +fn a_crdt_write_lands_and_a_refused_install_restores_the_document() { + assert!(crdt_row_present(&run(&[crdt_upsert()], false, |_| {}).core)); + assert!(!crdt_row_present(&run(&[crdt_upsert()], true, |_| {}).core)); +} + +// ── Spatial ────────────────────────────────────────────────────────────── + +fn spatial_insert() -> PhysicalPlan { + PhysicalPlan::Spatial(SpatialOp::Insert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "places"), + field: "geom".to_string(), + surrogate: Surrogate::new(81), + geometry: Geometry::Point { + coordinates: [1.0, 2.0], + }, + provenance: None, + }) +} + +fn spatial_entries(core: &CoreLoop) -> usize { + core.spatial_indexes + .get(&( + DatabaseId::DEFAULT, + TenantId::new(TID), + "places".to_string(), + "geom".to_string(), + )) + .map_or(0, |rtree| rtree.len()) +} + +#[test] +fn a_spatial_write_lands_and_a_refused_install_removes_the_entry() { + assert_eq!( + spatial_entries(&run(&[spatial_insert()], false, |_| {}).core), + 1 + ); + assert_eq!( + spatial_entries(&run(&[spatial_insert()], true, |_| {}).core), + 0 + ); +} + +// ── Array ──────────────────────────────────────────────────────────────── + +fn array_id() -> ArrayId { + ArrayId::new(TenantId::new(TID), "grid") +} + +fn open_array(core: &mut CoreLoop) { + let schema = ArraySchemaBuilder::new("grid") + .dim(DimSpec::new( + "x", + DimType::Int64, + Domain::new(DomainBound::Int64(0), DomainBound::Int64(15)), + )) + .attr(AttrSpec::new("v", AttrType::Int64, true)) + .tile_extents(vec![4]) + .build() + .expect("build schema"); + let schema_msgpack = zerompk::to_msgpack_vec(&schema).expect("encode schema"); + let response = + core.handle_array_open(&make_default_task(), &array_id(), &schema_msgpack, 0xA11, 8); + assert_committed(&response); +} + +fn array_put() -> PhysicalPlan { + let cells = vec![ArrayPutCell { + coord: vec![CoordValue::Int64(3)], + attrs: vec![CellValue::Int64(9)], + surrogate: Surrogate::ZERO, + system_from_ms: 1, + valid_from_ms: 0, + valid_until_ms: i64::MAX, + }]; + PhysicalPlan::Array(ArrayOp::Put { + array_id: array_id(), + cells_msgpack: zerompk::to_msgpack_vec(&cells).expect("encode cells"), + wal_lsn: 0, + provenance: None, + }) +} + +fn array_memtable_empty(core: &CoreLoop) -> bool { + core.array_engine + .store(&array_id()) + .map(|store| store.memtable.is_empty()) + .expect("array store") +} + +#[test] +fn an_array_write_lands_and_a_refused_install_restores_the_tiles() { + assert!(!array_memtable_empty( + &run(&[array_put()], false, open_array).core + )); + assert!(array_memtable_empty( + &run(&[array_put()], true, open_array).core + )); +} + +// ── Text (full-text index) ─────────────────────────────────────────────── + +/// A text write stages no row, so this installs its redo sub-record directly, +/// the one resolve serializes from the plan. +fn fts_documents(refuse: bool) -> u32 { + use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; + use crate::types::Lsn; + use crate::wal::{RedoRecord, RedoSubRecord}; + + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let op = nodedb_physical::physical_plan::TextOp::FtsIndexDoc { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), + surrogate: Surrogate::new(91), + text: "hello world".to_string(), + provenance: None, + }; + let (record_type, payload) = crate::control::server::wal_dispatch::encode_text_op_record(&op) + .expect("encode text op") + .expect("an index op journals a record"); + let mut ops = vec![RedoSubRecord { + record_type: record_type as u32, + payload, + }]; + if refuse { + let lines = zerompk::to_msgpack_vec(&vec!["other_probe,host=a value=1 1".to_string()]) + .expect("encode lines"); + ops.push(RedoSubRecord { + record_type: nodedb_wal::record::RecordType::TimeseriesBatch as u32, + payload: + crate::control::server::wal_dispatch::encode_timeseries_batch_payload_with_format( + "refusal_probe", + &lines, + None, + "ilp-msgpack", + ) + .expect("encode ingest"), + }); + } + let redo = RedoRecord { + version: 1, + ops, + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(210)); + let response = core.install_committed_redo( + &task, + TID, + CommittedRedo { + redo: &redo, + collections: &["docs".to_string()], + sum_targets: &[], + }, + ); + if refuse { + assert_refused_at_install(&response); + } else { + assert_committed(&response); + } + core.inverted + .corpus_stats(0, TenantId::new(TID), "docs") + .expect("corpus stats") + .0 +} + +#[test] +fn a_text_write_lands_and_a_refused_install_removes_the_posting() { + assert_eq!(fts_documents(false), 1); + assert_eq!(fts_documents(true), 0); +} + +// ── Index side effects of a document write ─────────────────────────────── + +/// A document write indexes its text and its geometry as side effects. A +/// refused install withdraws the posting and the R-tree entry with the row. +fn indexed_document_put() -> PhysicalPlan { + let location = serde_json::json!({"type": "Point", "coordinates": [10.0, 20.0]}); + let body = nodedb_types::json_to_msgpack(&serde_json::json!({ + "title": "quantum database sentinel", + "location": location, + })) + .expect("encode body"); + PhysicalPlan::Document(DocumentOp::PointPut { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "articles"), + document_id: "a1".to_string(), + value: body, + surrogate: Surrogate::new(101), + pk_bytes: b"a1".to_vec(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }) +} + +fn indexed_side_effects(core: &CoreLoop) -> (u32, usize) { + let postings = core + .inverted + .corpus_stats(0, TenantId::new(TID), "articles") + .expect("corpus stats") + .0; + let entries = core + .spatial_indexes + .get(&( + DatabaseId::DEFAULT, + TenantId::new(TID), + "articles".to_string(), + "location".to_string(), + )) + .map_or(0, |rtree| rtree.len()); + (postings, entries) +} + +#[test] +fn a_refused_install_withdraws_the_index_side_effects_of_a_document_write() { + assert_eq!( + indexed_side_effects(&run(&[indexed_document_put()], false, |_| {}).core), + (1, 1) + ); + assert_eq!( + indexed_side_effects(&run(&[indexed_document_put()], true, |_| {}).core), + (0, 0) + ); +} + +// ── Raw CRDT delta ─────────────────────────────────────────────────────── + +/// A committed record never carries a raw CRDT delta: a rejected delta's +/// dead-letter entry is keyed by its record's LSN, which every sub-record of +/// one committed record shares. The validate pass refuses such a record +/// before anything is written. +#[test] +fn a_committed_record_carrying_a_raw_crdt_delta_is_refused_before_any_write() { + use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; + use crate::types::Lsn; + use crate::wal::{CrdtDeltaWalPayload, RedoRecord, RedoSubRecord}; + + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let delta = CrdtDeltaWalPayload::new( + vec![0u8; 8], + Some("tasks".to_string()), + None, + None, + Some("t1".to_string()), + Some(1), + ) + .encode() + .expect("encode delta"); + let redo = RedoRecord { + version: 1, + ops: vec![RedoSubRecord { + record_type: nodedb_wal::record::RecordType::CrdtDelta as u32, + payload: delta, + }], + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(220)); + + let response = core.install_committed_redo( + &task, + TID, + CommittedRedo { + redo: &redo, + collections: &[], + sum_targets: &[], + }, + ); + + assert!( + matches!( + response.error_code.as_deref(), + Some(ErrorCode::RejectedPrevalidation { reason }) if reason.contains("raw CRDT delta") + ), + "{:?}", + response.error_code + ); + assert!(!crdt_row_present(&core), "nothing was written"); +} + +// ── Write versions and watermark ───────────────────────────────────────── + +#[test] +fn a_refused_install_publishes_no_write_version_and_no_watermark() { + use crate::types::Lsn; + let point = nodedb_types::calvin::ReadKeyIdent::Point(crate::types::KeyRepr::Surrogate(41)); + let committed = run(&[document_insert()], false, |_| {}); + assert_eq!(committed.core.watermark, Lsn::new(200)); + assert!(!committed.core.write_index.read_is_valid( + DatabaseId::DEFAULT, + TenantId::new(TID), + "notes", + &point, + Lsn::new(199), + )); + + let refused = run(&[document_insert()], true, |_| {}); + assert!( + refused.core.watermark < Lsn::new(200), + "the rolled-back install leaves the watermark where it was" + ); + assert!( + refused.core.write_index.read_is_valid( + DatabaseId::DEFAULT, + TenantId::new(TID), + "notes", + &point, + Lsn::new(199), + ), + "the rolled-back install publishes no write version" + ); +} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs index 1e25c3c66..59c80d1a1 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs @@ -12,15 +12,26 @@ //! - [`events`]: the record's Event Plane output. //! - [`sub_ops`]: typed views of the redo sub-records. //! - [`state`]: the per-core state and per-record scope. +//! - [`test_commit`]: a test driver for a session commit on one core. +//! - [`install_refusal_tests`]: one committed and one refused install per +//! engine kind. +//! - [`calvin_fold_tests`]: a Calvin record folds the same live and in +//! restart replay. +#[cfg(test)] +mod calvin_fold_tests; mod cover; mod document; mod entry; mod events; +#[cfg(test)] +mod install_refusal_tests; mod passes; mod settle; mod state; mod sub_ops; +#[cfg(test)] +pub(in crate::data::executor) mod test_commit; mod validate; pub(in crate::data::executor) use document::CommittedDocWrite; diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs index db7724fb0..f22b21e3d 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/passes.rs @@ -11,12 +11,18 @@ //! holds none of the record, which is what restart replay reproduces once //! the funnel cancels the record in the WAL. The refusal is retryable: the //! failure did not come from the record. +//! * A panic in either pass rolls back what the pass wrote and refuses the +//! record as retryable. + +use std::panic::{AssertUnwindSafe, catch_unwind}; use nodedb_physical::physical_plan::RedoSumTargets; use nodedb_wal::WalRecord; use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::transaction::undo::UndoEntry; +use crate::data::panic_payload::panic_payload_to_string; use super::state::{RedoApplyPass, RedoApplyScope}; @@ -64,7 +70,7 @@ impl CoreLoop { sum_targets: &[RedoSumTargets], ) -> Result<(), PassRefusal> { let (applied, scope) = self.run_redo_arms( - target.record, + target, RedoApplyScope::new(RedoApplyPass::Validate, sum_targets.to_vec()), )?; if let Some(code) = applied.err().or(scope.error) { @@ -90,50 +96,72 @@ impl CoreLoop { sum_targets: &[RedoSumTargets], ) -> Result { let (applied, mut scope) = self.run_redo_arms( - target.record, + target, RedoApplyScope::new(RedoApplyPass::Install, sum_targets.to_vec()), )?; let Some(cause) = applied.err().or(scope.error.take()) else { return Ok(scope); }; let undo = std::mem::take(&mut scope.undo); + Err(self.roll_back_redo(target, undo, cause)) + } + + /// Reverse `undo` after a pass failed with `cause`. + fn roll_back_redo( + &mut self, + target: &RedoTarget<'_>, + undo: Vec, + cause: ErrorCode, + ) -> PassRefusal { match self.rollback_undo_log(target.database_id, target.tid, undo) { - Ok(()) => Err(PassRefusal::RolledBack(cause)), - Err((entry_index, detail)) => { - Err(PassRefusal::RollbackFailed(ErrorCode::RollbackFailed { - entry_index, - detail: format!( - "rolling back a committed redo install that failed with {cause:?}: \ - {detail}" - ), - })) - } + Ok(()) => PassRefusal::RolledBack(cause), + Err((entry_index, detail)) => PassRefusal::RollbackFailed(ErrorCode::RollbackFailed { + entry_index, + detail: format!( + "rolling back a committed redo install that failed with {cause:?}: {detail}" + ), + }), } } - /// Drive every replay arm over `record` with `scope` open. Returns the + /// Drive every replay arm over the record with `scope` open. Returns the /// arms' own result and the scope they filled. + /// + /// A panic in an arm rolls back what the pass wrote and refuses the + /// record as retryable: it is no verdict on the record's bytes. fn run_redo_arms( &mut self, - record: &WalRecord, + target: &RedoTarget<'_>, scope: RedoApplyScope, ) -> Result<(Result<(), ErrorCode>, RedoApplyScope), PassRefusal> { // The arms route a record to `vshard_id % num_cores`. A committed // record carries no collection tombstone of its own: a collection // dropped before this entry committed refused the commit instead. self.redo_apply.scope = Some(scope); - let applied = self - .replay_engines_in_lsn_order( - std::slice::from_ref(record), - self.redo_apply.num_cores, + let num_cores = self.redo_apply.num_cores; + let applied = catch_unwind(AssertUnwindSafe(|| { + self.replay_engines_in_lsn_order( + std::slice::from_ref(target.record), + num_cores, &nodedb_wal::TombstoneSet::new(), ) - .map_err(ErrorCode::from); - match self.redo_apply.scope.take() { - Some(scope) => Ok((applied, scope)), + })); + let scope = self.redo_apply.scope.take(); + match (applied, scope) { + (Ok(applied), Some(scope)) => Ok((applied.map_err(ErrorCode::from), scope)), + (Err(payload), Some(mut scope)) => { + let cause = ErrorCode::Internal { + detail: format!( + "panic while applying a committed redo record: {}", + panic_payload_to_string(payload.as_ref()) + ), + }; + let undo = std::mem::take(&mut scope.undo); + Err(self.roll_back_redo(target, undo, cause)) + } // The scope held the undo log. Without it nothing can be rolled // back, so the core's state is unknown. - None => Err(PassRefusal::RollbackFailed(ErrorCode::RollbackFailed { + (_, None) => Err(PassRefusal::RollbackFailed(ErrorCode::RollbackFailed { entry_index: 0, detail: "committed transaction redo lost its apply scope and its undo log".into(), })), diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs index 733597bf0..10c6c19c8 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs @@ -19,9 +19,11 @@ use nodedb_physical::physical_plan::{RedoSumTargets, ResolvedSumTarget}; use nodedb_types::RowIdentity; use crate::bridge::envelope::ErrorCode; +use crate::data::executor::core_loop::write_index::KeyRepr; use crate::data::executor::enforcement::materialized_sum::apply::TargetWrite; use crate::data::executor::handlers::transaction::undo::UndoEntry; use crate::event::WriteOp; +use crate::types::Lsn; /// Committed-redo apply state owned by one core. pub(in crate::data::executor) struct RedoApplyState { @@ -30,19 +32,57 @@ pub(in crate::data::executor) struct RedoApplyState { pub(in crate::data::executor) num_cores: usize, /// `Some` only while one committed redo record applies on this core. pub(in crate::data::executor) scope: Option, + /// During restart replay: the materialized-sum targets each Calvin redo + /// record's stamp carries, keyed by the record's LSN. The document redo + /// arm folds a row at such an LSN into its targets, as the live install + /// did. Empty outside restart replay. + pub(in crate::data::executor) replay_folds: HashMap>, } impl RedoApplyState { + /// The sum targets restart replay folds a write to `collection` at + /// `record_lsn` into, when the record carries any. + pub(in crate::data::executor) fn replay_folds_for( + &self, + record_lsn: u64, + collection: &str, + ) -> Option<(Vec, Vec)> { + self.replay_folds + .get(&record_lsn)? + .iter() + .find(|targets| targets.collection == collection) + .map(|targets| (targets.resolved.clone(), targets.deferred.clone())) + } + /// A single-core default. Every multi-core runtime sets the real count /// through `CoreLoop::set_num_cores` before the core serves requests. pub(in crate::data::executor) fn new() -> Self { Self { num_cores: 1, scope: None, + replay_folds: HashMap::new(), } } } +impl crate::data::executor::core_loop::CoreLoop { + /// Arm restart replay with the sum targets each Calvin record carries. + /// Returns whether this is restart replay: a committed-redo apply folds + /// from its open scope instead, and leaves `folds` unused. + pub(crate) fn begin_replay_folds(&mut self, folds: HashMap>) -> bool { + let restart = self.redo_apply.scope.is_none(); + if restart { + self.redo_apply.replay_folds = folds; + } + restart + } + + /// Drop the restart-replay sum targets once the document redo arm ran. + pub(crate) fn end_replay_folds(&mut self) { + self.redo_apply.replay_folds.clear(); + } +} + /// One document row the committed-redo apply wrote. pub(in crate::data::executor) struct AppliedDocWrite { pub collection: String, @@ -94,6 +134,18 @@ pub(in crate::data::executor) struct RedoApplyScope { pub(in crate::data::executor) timeseries_written: Vec<(CollectionKey, u64)>, /// Events the install's writes raised, sent once it succeeded. pub(in crate::data::executor) pending_events: Vec, + /// Write versions the install's writes produced, published once the + /// record settled. + pub(in crate::data::executor) write_versions: Vec, +} + +/// One write version an install pass holds back until the record settles. +pub(in crate::data::executor) struct DeferredWriteVersion { + pub db: crate::types::DatabaseId, + pub tenant: crate::types::TenantId, + pub collection: String, + pub key: Option, + pub lsn: Lsn, } /// `(database, tenant, collection)`. @@ -120,6 +172,28 @@ impl RedoApplyScope { columnar_written: Vec::new(), timeseries_written: Vec::new(), pending_events: Vec::new(), + write_versions: Vec::new(), + } + } + + /// Hold back one write version until the record settles. The validate + /// pass writes nothing, so it holds nothing. + pub(in crate::data::executor) fn defer_write_version( + &mut self, + db: crate::types::DatabaseId, + tenant: crate::types::TenantId, + collection: &str, + key: Option, + lsn: Lsn, + ) { + if self.pass == RedoApplyPass::Install { + self.write_versions.push(DeferredWriteVersion { + db, + tenant, + collection: collection.to_string(), + key, + lsn, + }); } } diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/test_commit.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/test_commit.rs new file mode 100644 index 000000000..6be556f39 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/test_commit.rs @@ -0,0 +1,238 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Test driver for a session commit on one core: stage each plan, resolve +//! the transaction into its redo record, and install the record. +//! +//! [`CoreLoop::commit_plans_then_refuse_for_test`] appends a sub-record that +//! passes validation and fails while it installs, so the install rolls back +//! every write the transaction's own sub-records made. + +use nodedb_physical::physical_plan::PhysicalPlan; + +use crate::bridge::envelope::{Response, Status}; +use crate::control::wal_replication::transaction_redo::collections::written_collections; +use crate::control::wal_replication::transaction_redo::sum_targets::redo_sum_targets; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; +use crate::types::{Lsn, TxnId}; +use crate::wal::{RedoRecord, RedoSubRecord}; +use nodedb_wal::record::RecordType; + +use super::entry::CommittedRedo; + +impl CoreLoop { + /// Commit `plans` as one session transaction and install its redo record + /// at `lsn`, the path every committed transaction takes on a core. + /// + /// Returns the first refusal: a staging error, a resolve error, or the + /// install's error. The transaction overlay is released on every path. + pub(in crate::data::executor) fn commit_plans_for_test( + &mut self, + task: &ExecutionTask, + tid: u64, + plans: &[PhysicalPlan], + lsn: u64, + ) -> Response { + self.commit_plans_around_for_test(task, tid, plans, lsn, |_| {}) + } + + /// Commit `plans` like [`Self::commit_plans_for_test`], running `between` + /// after the resolve and before the install: a write another session + /// commits while the transaction's record is in flight. + pub(in crate::data::executor) fn commit_plans_around_for_test( + &mut self, + task: &ExecutionTask, + tid: u64, + plans: &[PhysicalPlan], + lsn: u64, + between: impl FnOnce(&mut Self), + ) -> Response { + let response = match self.stage_and_resolve_for_test(task, tid, plans, lsn) { + Ok(redo) => { + between(self); + self.install_redo_for_test(task, tid, plans, &redo, lsn) + } + Err(refusal) => refusal, + }; + // A session COMMIT releases the overlay once the install answered. + self.drop_overlay_entry(TxnId::new(lsn)); + response + } + + /// Commit `plans` like [`Self::commit_plans_for_test`], with one more + /// sub-record after theirs that fails while it installs. + pub(in crate::data::executor) fn commit_plans_then_refuse_for_test( + &mut self, + task: &ExecutionTask, + tid: u64, + plans: &[PhysicalPlan], + lsn: u64, + ) -> Response { + let response = match self.stage_and_resolve_for_test(task, tid, plans, lsn) { + Ok(redo) => { + let mut record = RedoRecord::from_bytes(&redo).expect("decode resolved redo"); + record.ops.push(failing_install_sub_record()); + let redo = record.to_bytes().expect("encode redo"); + self.install_redo_for_test(task, tid, plans, &redo, lsn) + } + Err(refusal) => refusal, + }; + self.drop_overlay_entry(TxnId::new(lsn)); + response + } + + /// Stage `plans` under the transaction `lsn` names and resolve them into + /// the encoded redo record. The overlay stays for the caller to release. + fn stage_and_resolve_for_test( + &mut self, + task: &ExecutionTask, + tid: u64, + plans: &[PhysicalPlan], + lsn: u64, + ) -> Result, Response> { + let txn_id = TxnId::new(lsn); + let mut request = task.request.clone(); + request.txn_id = Some(txn_id); + request.wal_lsn = None; + let stage_task = ExecutionTask::new(request); + for plan in plans { + let staged = self.execute_stage_write(&stage_task, tid, plan); + if staged.status != Status::Ok { + return Err(staged); + } + } + let resolved = self.execute_resolve_txn(&stage_task, tid, txn_id, plans); + if resolved.status != Status::Ok { + return Err(resolved); + } + Ok(resolved.payload.as_bytes().to_vec()) + } + + fn install_redo_for_test( + &mut self, + task: &ExecutionTask, + tid: u64, + plans: &[PhysicalPlan], + redo: &[u8], + lsn: u64, + ) -> Response { + let mut install_request = task.request.clone(); + install_request.wal_lsn = Some(Lsn::new(lsn)); + let install_task = ExecutionTask::new(install_request); + let collections = written_collections(plans); + let sum_targets = redo_sum_targets(plans); + self.install_committed_redo( + &install_task, + tid, + CommittedRedo { + redo, + collections: &collections, + sum_targets: &sum_targets, + }, + ) + } +} + +/// A timeseries batch whose line names another measurement than its +/// collection: it passes validation and fails while it installs. +pub(in crate::data::executor) fn failing_install_sub_record() -> RedoSubRecord { + let lines = zerompk::to_msgpack_vec(&vec!["other_probe,host=a value=1 1".to_string()]) + .expect("encode lines"); + RedoSubRecord { + record_type: RecordType::TimeseriesBatch as u32, + payload: crate::control::server::wal_dispatch::encode_timeseries_batch_payload_with_format( + "refusal_probe", + &lines, + None, + "ilp-msgpack", + ) + .expect("encode ingest"), + } +} + +impl CoreLoop { + /// Run the install pass of a committed redo record carrying `ops` at + /// `lsn`, and return the undo entries that reverse its writes. + /// + /// The forward write lands through the same arms a committed install + /// drives. The caller reverses it with `rollback_undo_log`, as a failed + /// install does. + pub(in crate::data::executor) fn install_with_undo_for_test( + &mut self, + tid: u64, + lsn: u64, + ops: Vec, + ) -> Vec { + let payload = RedoRecord { + version: 1, + ops, + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo"); + let record = nodedb_wal::WalRecord::new(nodedb_wal::record::WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn, + tenant_id: tid, + vshard_id: 0, + database_id: 0, + payload, + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record"); + self.redo_apply.scope = Some(super::state::RedoApplyScope::new( + super::state::RedoApplyPass::Install, + Vec::new(), + )); + let applied = self.replay_engines_in_lsn_order( + std::slice::from_ref(&record), + self.redo_apply.num_cores, + &nodedb_wal::TombstoneSet::new(), + ); + let scope = self.redo_apply.scope.take().expect("install scope"); + applied.expect("install arms"); + assert!(scope.error.is_none(), "install error: {:?}", scope.error); + scope.undo + } +} + +/// The redo sub-record of a document put: `(collection, document_id, body, +/// provenance, surrogate)`. +pub(in crate::data::executor) fn doc_put_sub_record( + collection: &str, + document_id: &str, + body: &[u8], + surrogate: u32, +) -> RedoSubRecord { + RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(&( + collection, + document_id, + body.to_vec(), + None::, + surrogate, + )) + .expect("encode document put"), + } +} + +/// The redo sub-record of a document delete: `(collection, document_id, +/// provenance, surrogate)`. +pub(in crate::data::executor) fn doc_delete_sub_record( + collection: &str, + document_id: &str, + surrogate: u32, +) -> RedoSubRecord { + RedoSubRecord { + record_type: RecordType::Delete as u32, + payload: zerompk::to_msgpack_vec(&( + collection, + document_id, + None::, + surrogate, + )) + .expect("encode document delete"), + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/classify.rs b/nodedb/src/data/executor/handlers/transaction/resolve/classify.rs index 71c782871..074a97ffd 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/classify.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/classify.rs @@ -59,6 +59,7 @@ pub(super) fn classify_kv_op(op: &KvOp, collections: &mut BTreeSet) -> c | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } // Read-only: reports what a governed write would apply, stages // nothing. | KvOp::ResolveWrite(_) => Ok(()), @@ -69,11 +70,12 @@ pub(super) fn classify_kv_op(op: &KvOp, collections: &mut BTreeSet) -> c detail: "kv resolved write is not supported in transaction resolve".to_string(), }), - // A standalone TTL delta has no value post-image, and KV redo carries - // TTL only as part of a value put, so rejecting avoids a silent drop. - KvOp::Expire { .. } | KvOp::Persist { .. } => Err(crate::Error::PlanError { - detail: "kv EXPIRE/PERSIST is not supported in transaction resolve".to_string(), - }), + // A TTL delta resolves to a put of the row's base value carrying the + // staged expiry, so it contributes its collection like a value write. + KvOp::Expire { collection, .. } | KvOp::Persist { collection, .. } => { + collections.insert(collection.to_string()); + Ok(()) + } // Truncate: staged as an overlay marker; the serializer emits the // `kv_truncate` redo ahead of the collection's row entries. @@ -113,7 +115,9 @@ pub(super) fn classify_document_op( | DocumentOp::BulkDelete { collection, .. } // A balance write stages like any other point write: one target row, // one absolute post-image, keyed by the row's own surrogate. - | DocumentOp::ApplyBalanceDelta { collection, .. } => { + | DocumentOp::ApplyBalanceDelta { collection, .. } + // A Calvin batch insert stages each row as a point put. + | DocumentOp::BatchInsert { collection, .. } => { collections.insert(collection.to_string()); Ok(()) } @@ -142,15 +146,15 @@ pub(super) fn classify_document_op( detail: "document resolved write is not supported in transaction resolve".to_string(), }), - // Join/merge have no per-surrogate post-image; `BatchInsert` rides the - // buffered-plan path. None is staged, so rejecting avoids a lossy redo. - DocumentOp::UpdateFromJoin { .. } - | DocumentOp::Merge { .. } - | DocumentOp::BatchInsert { .. } => Err(crate::Error::PlanError { - detail: "document join/merge/batch DML has no staged post-image and is not \ - supported in transaction resolve" - .to_string(), - }), + // Join/merge have no per-surrogate post-image. Neither is staged, so + // rejecting avoids a lossy redo. + DocumentOp::UpdateFromJoin { .. } | DocumentOp::Merge { .. } => { + Err(crate::Error::PlanError { + detail: "document join/merge DML has no staged post-image and is not \ + supported in transaction resolve" + .to_string(), + }) + } // Truncate: staged as an overlay marker; the serializer emits a // `Delete` per removed base row ahead of the collection's overlay diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index 614603353..24513cdeb 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -24,25 +24,15 @@ use crate::wal::{RedoRecord, RedoSubRecord}; use super::classify::{classify_document_op, classify_kv_op}; use super::columnar_image::{ColumnarCollectionImages, ColumnarCollections}; use super::graph::EdgeIdentityKey; -use super::vector_direct::DirectWrites; use super::vector_primary::VectorPrimaryCollections; use super::{array, columnar_image, crdt, document, graph, kv, spatial, text, vector}; -/// Which writes of a transaction its statements staged into the overlay. -#[derive(Clone, Copy, PartialEq, Eq)] -pub(in crate::data::executor) enum StagedWrites { - /// A session transaction: every write is staged at its statement. - Session, - /// A Calvin transaction: document, KV, graph, timeseries and columnar - /// writes are staged; vector-primary direct writes are not, so they - /// resolve from their plan nodes. - Calvin, -} - impl CoreLoop { - /// Resolve a committing session transaction's staged writes into a + /// Resolve a committing transaction's staged writes into a /// [`RedoRecord`] and return its encoded bytes in the response payload. - /// Reads the overlay by `&` and never mutates any base engine. + /// Reads the overlay by `&` and never mutates any base engine. A session + /// transaction and a Calvin transaction stage the same way, so both + /// resolve here. pub(in crate::data::executor) fn execute_resolve_txn( &mut self, task: &ExecutionTask, @@ -50,19 +40,7 @@ impl CoreLoop { txn_id: TxnId, plans: &[PhysicalPlan], ) -> Response { - self.execute_resolve_staged(task, tid, txn_id, plans, StagedWrites::Session) - } - - /// Resolve a transaction whose staging follows `staged`. - pub(in crate::data::executor) fn execute_resolve_staged( - &mut self, - task: &ExecutionTask, - tid: u64, - txn_id: TxnId, - plans: &[PhysicalPlan], - staged: StagedWrites, - ) -> Response { - let ops = match self.resolve_txn_ops(task, tid, txn_id, plans, staged) { + let ops = match self.resolve_txn_ops(task, tid, txn_id, plans) { Ok(ops) => ops, Err(e) => return self.response_error(task, e), }; @@ -86,7 +64,6 @@ impl CoreLoop { tid: u64, txn_id: TxnId, plans: &[PhysicalPlan], - staged: StagedWrites, ) -> crate::Result> { let mut kv_collections: BTreeSet = BTreeSet::new(); let mut doc_collections: BTreeSet = BTreeSet::new(); @@ -94,10 +71,6 @@ impl CoreLoop { let mut edge_surrogates: BTreeMap = BTreeMap::new(); let mut columnar_collections: ColumnarCollections = BTreeMap::new(); let mut vector_primary_collections: VectorPrimaryCollections = BTreeMap::new(); - let mut direct_writes = match staged { - StagedWrites::Session => DirectWrites::Staged(&mut vector_primary_collections), - StagedWrites::Calvin => DirectWrites::Plan, - }; // Plan-driven serializers emit into `ops` during this walk; overlay-driven // serializers only collect collections here, serialized in phase two below. @@ -105,6 +78,8 @@ impl CoreLoop { // The declared primary key of each truncated document collection, // from the plan: names every removed base row in its redo entry. let truncate_primary_keys = truncate_declared_primary_keys(plans); + // Unkeyed timeseries ingests seen per collection, in plan order. + let mut unkeyed_seen = std::collections::HashMap::new(); for plan in plans { match plan { @@ -138,12 +113,17 @@ impl CoreLoop { // errors. A vector-primary direct write only registers its // collection; its staged row is serialized from the overlay. PhysicalPlan::Vector(op) => { - vector::serialize_vector_op(op, &mut ops, &mut direct_writes)? + vector::serialize_vector_op(op, &mut ops, &mut vector_primary_collections)? } PhysicalPlan::Array(op) => array::serialize_array_op(op, &mut ops)?, - PhysicalPlan::Timeseries(op) => { - self.serialize_timeseries_op(task, tid, txn_id, op, &mut ops)? - } + PhysicalPlan::Timeseries(op) => self.serialize_timeseries_op( + task, + tid, + txn_id, + op, + &mut unkeyed_seen, + &mut ops, + )?, // Columnar: every write is staged per surrogate, so the image // the transaction was shown is serialized from the overlay. @@ -165,15 +145,13 @@ impl CoreLoop { } } - if staged == StagedWrites::Session { - let reads_overlay = !kv_collections.is_empty() - || !doc_collections.is_empty() - || !graph_collections.is_empty() - || !columnar_collections.is_empty() - || !vector_primary_collections.is_empty() - || plans.iter().any(graph::is_label_write); - self.require_staging_overlay(txn_id, reads_overlay)?; - } + let reads_overlay = !kv_collections.is_empty() + || !doc_collections.is_empty() + || !graph_collections.is_empty() + || !columnar_collections.is_empty() + || !vector_primary_collections.is_empty() + || plans.iter().any(graph::is_label_write); + self.require_staging_overlay(txn_id, reads_overlay)?; // Pin the resolve-time bitemporal stamp once, in the overlay sidecar, for // every staged put AND tombstone, so the redo carries it and every apply @@ -1056,7 +1034,7 @@ mod tests { } #[test] - fn join_merge_batch_dml_still_yield_typed_error() { + fn join_merge_dml_still_yield_typed_error() { let (mut core, _dir) = make_core(); let task = make_task(); let txn = TxnId::new(45); @@ -1095,15 +1073,6 @@ mod tests { resolved_sum_targets: Vec::new(), declared_primary_key: None, }), - PhysicalPlan::Document(DocumentOp::BatchInsert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), - documents: vec![("d1".to_string(), Vec::new())], - surrogates: vec![Surrogate::ZERO], - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - deferred_sum_targets: Vec::new(), - }), ]; for plan in plans { @@ -1117,6 +1086,50 @@ mod tests { } } + /// A batch insert staged row by row resolves to one document put per row, + /// read from the overlay. + #[test] + fn a_staged_batch_insert_resolves_to_a_put_per_row() { + let (mut core, _dir) = make_core(); + let task = make_task(); + let txn = TxnId::new(46); + let documents = vec![ + ("d1".to_string(), schemaless_body("ann")), + ("d2".to_string(), schemaless_body("bob")), + ]; + let surrogates = vec![Surrogate::new(3), Surrogate::new(4)]; + let staged = core.stage_document_batch_insert( + crate::data::executor::handlers::transaction::stage_write::StageBatchInsertParams { + task: &task, + tid: TID, + txn_id: txn, + collection: "notes", + documents: &documents, + surrogates: &surrogates, + }, + ); + assert_eq!(staged.status, Status::Ok, "{:?}", staged.error_code); + let plan = PhysicalPlan::Document(DocumentOp::BatchInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), + documents, + surrogates, + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + + let resp = core.execute_resolve_txn(&task, TID, txn, &[plan]); + + let redo = decode_redo(&resp); + assert_eq!(redo.ops.len(), 2, "one sub-record per staged row"); + assert!( + redo.ops + .iter() + .all(|op| op.record_type == RecordType::Put as u32) + ); + } + /// The five extended vector writes (`DirectUpsert`, `MultiVectorInsert`, /// `MultiVectorDelete`, `SparseInsert`, `SparseDelete`) must resolve to /// redo sub-records, not a typed error. @@ -3588,25 +3601,47 @@ mod tests { } #[test] - fn spatial_insert_without_provenance_yields_typed_error() { - let (mut core, _dir) = make_core(); + fn spatial_insert_without_provenance_resolves_and_replays() { + let (mut src, _src_dir) = make_core(); let task = make_task(); let txn = TxnId::new(46); + let surrogate = 1u32; let plan = PhysicalPlan::Spatial(SpatialOp::Insert { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "places"), field: "loc".to_string(), - surrogate: Surrogate::new(1), + surrogate: Surrogate::new(surrogate), geometry: spatial_point(0.0, 0.0), provenance: None, }); - let resp = core.execute_resolve_txn(&task, TID, txn, &[plan]); - assert_eq!( - resp.status, - Status::Error, - "a spatial insert with no provenance must raise a typed error, not silently drop" + let resp = src.execute_resolve_txn(&task, TID, txn, &[plan]); + // A plain SQL spatial insert carries no sync producer. It resolves + // with the empty provenance the autocommit spatial WAL path writes. + let redo = decode_redo(&resp); + assert_eq!(redo.ops.len(), 1, "one spatial insert -> one sub-record"); + assert_eq!(redo.ops[0].record_type, RecordType::SpatialPut as u32); + + let record = wrap_redo(&redo); + let (mut dst, _dst_dir) = make_core(); + dst.replay_transaction_redo_wal( + std::slice::from_ref(&record), + 1, + &nodedb_wal::TombstoneSet::new(), + ) + .expect("redo replay must succeed"); + let key = ( + DatabaseId::DEFAULT, + TenantId::new(TID), + "places".to_string(), + "loc".to_string(), ); - assert!(resp.error_code.is_some()); + let entries = dst + .spatial_indexes + .get(&key) + .expect("R-tree index rebuilt by replay") + .entries(); + assert_eq!(entries.len(), 1, "the insert must not be dropped"); + assert_eq!(entries[0].id, spatial_entry_id(surrogate)); } #[test] diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/kv.rs b/nodedb/src/data/executor/handlers/transaction/resolve/kv.rs index 48e4e9f07..3ea375759 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/kv.rs @@ -15,6 +15,10 @@ //! ([`StagedTtl::ExpireAt`]) for the slot it travels verbatim so replay //! installs the exact instant instead of recomputing `now + ttl`; a `Persist` //! (or no TTL delta) emits `None`. +//! * A TTL-only slot (an `EXPIRE` / `PERSIST` of a row the transaction did +//! not otherwise write) → `RecordType::Put`, `encode_kv_expire`'s +//! `("kv_expire", collection, key, ttl_ms, expire_at_ms)` or +//! `encode_kv_persist`'s `("kv_persist", collection, key)`. //! * A staged tombstone ([`Staged::Tombstone`]) → `RecordType::Delete`, //! `("kv_delete", collection, [key])`. //! * A staged TRUNCATE → `RecordType::Delete`, `encode_kv_truncate`'s @@ -41,7 +45,9 @@ use std::collections::BTreeMap; use nodedb_types::RowIdentity; use nodedb_wal::record::RecordType; -use crate::control::server::wal_dispatch_kv::encode::{encode_kv_put, encode_kv_truncate}; +use crate::control::server::wal_dispatch_kv::encode::{ + encode_kv_expire, encode_kv_persist, encode_kv_put, encode_kv_truncate, +}; use crate::data::executor::handlers::transaction::overlay::{Staged, StagedTtl, TxnOverlay}; use crate::data::executor::handlers::transaction::stage_write::unhex_key; use crate::types::{DatabaseId, TenantId}; @@ -67,23 +73,57 @@ pub(super) fn serialize_kv_truncate( Ok(()) } +/// One staged KV row: a value or tombstone, or a TTL delta on a base row. +enum KvSlot<'a> { + Staged(&'a Staged), + TtlOnly(StagedTtl), +} + /// Append the redo sub-records for every KV post-image staged in `overlay` /// for `coll_key` to `ops`, in deterministic doc-id order. +/// +/// An `EXPIRE` or `PERSIST` of a row the transaction did not otherwise +/// write resolves to a `kv_expire` or `kv_persist` sub-record, the shapes +/// the autocommit TTL writes append. The install changes only the expiry of +/// the value the key holds when the record applies. A write committed +/// between stage and install keeps its value. pub(super) fn serialize_kv_collection( overlay: &TxnOverlay, coll_key: &(DatabaseId, TenantId, String), collection: &str, ops: &mut Vec, ) -> crate::Result<()> { - let mut entries: BTreeMap<&RowIdentity, &Staged> = BTreeMap::new(); + let mut entries: BTreeMap<&RowIdentity, KvSlot<'_>> = BTreeMap::new(); for (doc_id, staged) in overlay.iter_doc_entries_for_collection(coll_key) { - entries.insert(doc_id, staged); + entries.insert(doc_id, KvSlot::Staged(staged)); + } + // A staged TRUNCATE hides every base row, so no TTL delta can target one. + if !overlay.is_truncated(coll_key) { + for (doc_id, ttl) in overlay.iter_ttl_only_for_collection(coll_key) { + entries.insert(doc_id, KvSlot::TtlOnly(ttl)); + } } - for (doc_id, staged) in entries { + for (doc_id, slot) in entries { let key = unhex_key(doc_id.as_str()).ok_or_else(|| crate::Error::Internal { detail: format!("kv resolve: overlay doc-id '{doc_id}' is not valid hex"), })?; + let staged = match slot { + KvSlot::Staged(staged) => staged, + KvSlot::TtlOnly(ttl) => { + let payload = match ttl { + StagedTtl::ExpireAt(ms) => { + encode_kv_expire(collection, &key, RESOLVE_TTL_MS, ms)? + } + StagedTtl::Persist => encode_kv_persist(collection, &key)?, + }; + ops.push(RedoSubRecord { + record_type: RecordType::Put as u32, + payload, + }); + continue; + } + }; match staged { Staged::Put(value) => { let expire_at_ms = match overlay.get_ttl_by_doc_id(coll_key, doc_id) { diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/mod.rs b/nodedb/src/data/executor/handlers/transaction/resolve/mod.rs index 2544dcb85..27a5aa637 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/mod.rs @@ -12,7 +12,4 @@ mod spatial; mod text; mod timeseries; mod vector; -mod vector_direct; mod vector_primary; - -pub(in crate::data::executor) use entry::StagedWrites; diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/spatial.rs b/nodedb/src/data/executor/handlers/transaction/resolve/spatial.rs index d8cb66557..ed25672d3 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/spatial.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/spatial.rs @@ -23,16 +23,13 @@ //! append its `SpatialPut` / `SpatialDelete` WAL records, so producer and //! `replay_spatial_wal` never drift. //! -//! ## Provenance is mandatory +//! ## Provenance //! -//! `SpatialOp::Insert` / `Delete` inside a transaction arise ONLY from the -//! Lite sync replication path (see `stage_spatial.rs`'s module docs), which -//! always supplies `provenance: Some(..)`. The WAL wire shape -//! (`SpatialPutPayload` / `SpatialDeletePayload`) carries provenance as a -//! mandatory field, not optional, so a staged spatial op with `provenance: -//! None` is an invariant violation, not a case to invent a zero provenance -//! for — it raises a typed error rather than being silently dropped or -//! fabricated. +//! A spatial write from the Lite sync path carries its producer provenance. A +//! write from SQL carries none, and resolve writes the empty provenance +//! (`producer_id` 0) in its place, as the autocommit spatial WAL path does. +//! Replay and the sync gate treat producer 0 as "no producer": no +//! high-water mark moves and no undo is captured for one. //! //! ## Reads //! @@ -49,8 +46,7 @@ use crate::wal::RedoSubRecord; /// Append the redo sub-record for a single spatial plan op to `ops`. /// /// `Insert` / `Delete` serialize to their engine-native `SpatialPut` / -/// `SpatialDelete` shape; `Scan` emits nothing; either write with no -/// provenance raises a typed error (see module docs). +/// `SpatialDelete` shape. `Scan` emits nothing. pub(super) fn serialize_spatial_op( op: &SpatialOp, ops: &mut Vec, @@ -63,13 +59,14 @@ pub(super) fn serialize_spatial_op( geometry, provenance, } => { - let prov = provenance.as_ref().ok_or_else(|| crate::Error::PlanError { - detail: "spatial insert with no sync provenance has no redo sub-record shape \ - and is not supported in transaction resolve" - .to_string(), - })?; - let payload = - encode_spatial_put_payload(collection.as_str(), field, *surrogate, geometry, prov)?; + let prov = provenance.clone().unwrap_or_default(); + let payload = encode_spatial_put_payload( + collection.as_str(), + field, + *surrogate, + geometry, + &prov, + )?; let bytes = payload.to_bytes().map_err(crate::Error::Wal)?; ops.push(RedoSubRecord { record_type: RecordType::SpatialPut as u32, @@ -83,13 +80,9 @@ pub(super) fn serialize_spatial_op( surrogate, provenance, } => { - let prov = provenance.as_ref().ok_or_else(|| crate::Error::PlanError { - detail: "spatial delete with no sync provenance has no redo sub-record shape \ - and is not supported in transaction resolve" - .to_string(), - })?; + let prov = provenance.clone().unwrap_or_default(); let payload = - encode_spatial_delete_payload(collection.as_str(), field, *surrogate, prov); + encode_spatial_delete_payload(collection.as_str(), field, *surrogate, &prov); let bytes = payload.to_bytes().map_err(crate::Error::Wal)?; ops.push(RedoSubRecord { record_type: RecordType::SpatialDelete as u32, @@ -186,8 +179,10 @@ mod tests { assert!(ops.is_empty(), "read-only scan emits no sub-record"); } + /// A SQL spatial insert carries no provenance. It resolves to a + /// `SpatialPut` with the empty provenance, never a dropped write. #[test] - fn insert_without_provenance_errors_rather_than_dropping() { + fn insert_without_provenance_resolves_with_the_empty_provenance() { let op = SpatialOp::Insert { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "places"), field: "loc".to_string(), @@ -196,16 +191,18 @@ mod tests { provenance: None, }; let mut ops = Vec::new(); - let err = serialize_spatial_op(&op, &mut ops); - assert!( - err.is_err(), - "a spatial insert with no provenance must error, not silently drop" - ); - assert!(ops.is_empty()); + serialize_spatial_op(&op, &mut ops).expect("serialize insert"); + assert_eq!(ops.len(), 1, "the insert is never dropped"); + assert_eq!(ops[0].record_type, RecordType::SpatialPut as u32); + let decoded = nodedb_wal::record::SpatialPutPayload::from_bytes(&ops[0].payload) + .expect("decode SpatialPutPayload"); + assert_eq!(decoded.provenance, SyncProvenance::default()); } + /// A SQL spatial delete carries no provenance. It resolves to a + /// `SpatialDelete` with the empty provenance, never a dropped write. #[test] - fn delete_without_provenance_errors_rather_than_dropping() { + fn delete_without_provenance_resolves_with_the_empty_provenance() { let op = SpatialOp::Delete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "places"), field: "loc".to_string(), @@ -213,11 +210,11 @@ mod tests { provenance: None, }; let mut ops = Vec::new(); - let err = serialize_spatial_op(&op, &mut ops); - assert!( - err.is_err(), - "a spatial delete with no provenance must error, not silently drop" - ); - assert!(ops.is_empty()); + serialize_spatial_op(&op, &mut ops).expect("serialize delete"); + assert_eq!(ops.len(), 1, "the delete is never dropped"); + assert_eq!(ops[0].record_type, RecordType::SpatialDelete as u32); + let decoded = nodedb_wal::record::SpatialDeletePayload::from_bytes(&ops[0].payload) + .expect("decode SpatialDeletePayload"); + assert_eq!(decoded.provenance, SyncProvenance::default()); } } diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/timeseries.rs b/nodedb/src/data/executor/handlers/transaction/resolve/timeseries.rs index d095a1acd..12cabc645 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/timeseries.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/timeseries.rs @@ -37,6 +37,7 @@ impl CoreLoop { tid: u64, txn_id: TxnId, op: &TimeseriesOp, + unkeyed_seen: &mut std::collections::HashMap, ops: &mut Vec, ) -> crate::Result<()> { match op { @@ -59,16 +60,24 @@ impl CoreLoop { tenant, collection.as_str().to_string(), ); - // The instant the statement read. An ingest staged with no - // surrogate recorded none, and resolve reads the clock now. - let now_ms = surrogates - .first() - .and_then(|first| { - self.txn_overlays - .get(&txn_id)? - .ingest_now(&coll_key, first.as_u32()) - }) - .unwrap_or_else(|| self.ingest_now_ms()); + // The instant the statement read. A keyed ingest recorded it + // under its first surrogate. An unkeyed ingest recorded it in + // stage order, so the Nth unkeyed ingest into a collection + // reads the Nth instant. + let overlay = self.txn_overlays.get(&txn_id); + let staged_now = match surrogates.first() { + Some(first) => { + overlay.and_then(|overlay| overlay.ingest_now(&coll_key, first.as_u32())) + } + None => { + let ordinal = unkeyed_seen.entry(coll_key.2.clone()).or_insert(0); + let now = overlay + .and_then(|overlay| overlay.unkeyed_ingest_now(&coll_key, *ordinal)); + *ordinal += 1; + now + } + }; + let now_ms = staged_now.unwrap_or_else(|| self.ingest_now_ms()); let lines = self .stamped_ingest_lines(StampedIngest { database_id: task.request.database_id, diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/vector.rs b/nodedb/src/data/executor/handlers/transaction/resolve/vector.rs index 1618ae7ca..d498e7272 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/vector.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/vector.rs @@ -20,11 +20,9 @@ //! * `DeleteBySurrogate` → `RecordType::VectorDelete`, //! `(collection, surrogate, field_name, provenance)`. //! * `DirectInsert` / `DirectInsertIfAbsent` / `DirectUpsert` / -//! `DirectUpdate` / `DirectDelete` / `DirectTruncate` in a session -//! transaction register their collection only: a vector-primary row is -//! staged whole, so those resolve from the overlay (`vector_primary`). In -//! a Calvin transaction, which stages none of them, each serializes to its -//! autocommit record shape (`vector_direct`). +//! `DirectUpdate` / `DirectDelete` / `DirectTruncate` register their +//! collection only: a vector-primary row is staged whole, so those resolve +//! from the overlay (`vector_primary`). //! * `MultiVectorInsert` → `RecordType::MultiVectorPut`, the 6-element //! flattened multi-vector shape (`replay_multi_vector_put`). //! * `MultiVectorDelete` → `RecordType::MultiVectorDelete`, @@ -58,11 +56,10 @@ //! puts on replay, but `SetParams` is rejected here, so ordering reduces to the //! given plan order. -use nodedb_physical::physical_plan::{VectorDirectWriteIntent, VectorOp}; +use nodedb_physical::physical_plan::VectorOp; use nodedb_wal::record::RecordType; -use super::vector_direct::{DirectInsert, DirectUpdate, DirectWrites}; -use super::vector_primary::VectorPrimarySpec; +use super::vector_primary::{VectorPrimaryCollections, VectorPrimarySpec, note_direct_write}; use crate::control::server::wal_dispatch::{ VectorResolvedDirectWritePayload, encode_multi_vector_delete_payload, encode_multi_vector_put_payload, encode_sparse_vector_delete_payload, @@ -76,13 +73,13 @@ use crate::wal::RedoSubRecord; /// /// Writes serialize to their engine-native record shape (`VectorPut` / /// `VectorDelete` / `MultiVectorPut` / `MultiVectorDelete` / `SparseVectorPut` -/// / `SparseVectorDelete`); a vector-primary direct write goes through -/// `direct`; read and index-maintenance ops emit nothing; vector-index DDL -/// (`SetParams`) raises a typed error (see module docs). +/// / `SparseVectorDelete`); a vector-primary direct write registers its +/// collection in `direct_writes`; read and index-maintenance ops emit nothing; +/// vector-index DDL (`SetParams`) raises a typed error (see module docs). pub(super) fn serialize_vector_op( op: &VectorOp, ops: &mut Vec, - direct: &mut DirectWrites<'_>, + direct_writes: &mut VectorPrimaryCollections, ) -> crate::Result<()> { match op { VectorOp::Insert { @@ -178,116 +175,73 @@ pub(super) fn serialize_vector_op( .to_string(), }), - // Vector-primary direct writes: a session transaction resolves the - // staged rows from the overlay, a Calvin transaction serializes each - // op from its plan node (`vector_direct`). + // Vector-primary direct writes: the overlay holds each staged row, so + // the op only registers its collection (`vector_primary`). VectorOp::DirectUpsert { collection, field, surrogate, pk_bytes, - vector, - payload, quantization, storage_dtype, payload_indexes, - returning: _, - rls_filters: _, - on_conflict_updates, - rls_write_check: _, - } => direct.insert( - DirectInsert { - collection: collection.as_str(), - field, - surrogate: *surrogate, - pk_bytes, - vector, - payload, - spec: spec(*quantization, *storage_dtype, payload_indexes), - intent: VectorDirectWriteIntent::Upsert, - on_conflict_updates, - }, - ops, - ), - VectorOp::DirectInsert { + .. + } + | VectorOp::DirectInsert { collection, field, surrogate, pk_bytes, - vector, - payload, quantization, storage_dtype, payload_indexes, - returning: _, - rls_filters: _, + .. } | VectorOp::DirectInsertIfAbsent { collection, field, surrogate, pk_bytes, - vector, - payload, quantization, storage_dtype, payload_indexes, - returning: _, - rls_filters: _, - } => direct.insert( - DirectInsert { - collection: collection.as_str(), + .. + } => { + note_direct_write( + direct_writes, + collection.as_str(), field, - surrogate: *surrogate, - pk_bytes, - vector, - payload, - spec: spec(*quantization, *storage_dtype, payload_indexes), - intent: if matches!(op, VectorOp::DirectInsertIfAbsent { .. }) { - VectorDirectWriteIntent::InsertIfAbsent - } else { - VectorDirectWriteIntent::Insert - }, - on_conflict_updates: &[], - }, - ops, - ), + Some(spec(*quantization, *storage_dtype, payload_indexes)), + Some((*surrogate, pk_bytes.as_slice())), + ); + Ok(()) + } VectorOp::DirectUpdate { collection, field, - targets, - new_vector, - payload_patch, quantization, storage_dtype, payload_indexes, - returning: _, - rls_filters: _, - rls_write_check: _, - } => direct.update( - DirectUpdate { - collection: collection.as_str(), + .. + } => { + note_direct_write( + direct_writes, + collection.as_str(), field, - targets, - new_vector: new_vector.as_deref(), - payload_patch, - spec: spec(*quantization, *storage_dtype, payload_indexes), - }, - ops, - ), + Some(spec(*quantization, *storage_dtype, payload_indexes)), + None, + ); + Ok(()) + } VectorOp::DirectDelete { - collection, - field, - targets, - returning: _, - rls_filters: _, - rls_write_check: _, - } => direct.delete(collection.as_str(), field, targets, ops), - VectorOp::DirectTruncate { - collection, - field, - restart_identity: _, - } => direct.truncate(collection.as_str(), field, ops), + collection, field, .. + } + | VectorOp::DirectTruncate { + collection, field, .. + } => { + note_direct_write(direct_writes, collection.as_str(), field, None, None); + Ok(()) + } // Resolved vector-primary write, replayed via // `replay_vector_resolved_direct_write`: the same record the // autocommit path appends, carrying every row's stored image. diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/vector_direct.rs b/nodedb/src/data/executor/handlers/transaction/resolve/vector_direct.rs deleted file mode 100644 index bd07dfd92..000000000 --- a/nodedb/src/data/executor/handlers/transaction/resolve/vector_direct.rs +++ /dev/null @@ -1,174 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Vector-primary direct writes in transaction resolve, by staging source. -//! -//! A session transaction stages every direct write into its overlay, so its -//! redo carries the staged rows (`vector_primary`): each op here only -//! registers its collection. A Calvin transaction stages no vector-primary -//! write, so its redo carries each op as the autocommit record shape and -//! replay re-runs it through the live handler in the epoch's deterministic -//! order. - -use nodedb_physical::physical_plan::{UpdateValue, VectorDirectWriteIntent, VectorWriteTargets}; -use nodedb_types::Surrogate; -use nodedb_wal::record::RecordType; - -use super::vector_primary::{VectorPrimaryCollections, VectorPrimarySpec, note_direct_write}; -use crate::control::server::wal_dispatch::{ - VectorDirectUpdatePayload, VectorDirectUpsertPayload, encode_vector_direct_delete_payload, - encode_vector_direct_truncate_payload, encode_vector_direct_update_payload, - encode_vector_direct_upsert_payload, -}; -use crate::wal::RedoSubRecord; - -/// Where a transaction's vector-primary direct writes resolve from. -pub(super) enum DirectWrites<'a> { - /// The overlay holds the staged rows; ops register their collection. - Staged(&'a mut VectorPrimaryCollections), - /// Nothing is staged; each op serializes from its plan node. - Plan, -} - -/// One insert-family direct write (`DirectInsert` / `DirectInsertIfAbsent` / -/// `DirectUpsert`). -pub(super) struct DirectInsert<'a> { - pub collection: &'a str, - pub field: &'a str, - pub surrogate: Surrogate, - pub pk_bytes: &'a [u8], - pub vector: &'a [f32], - pub payload: &'a [u8], - pub spec: VectorPrimarySpec, - pub intent: VectorDirectWriteIntent, - pub on_conflict_updates: &'a [(String, UpdateValue)], -} - -/// One `DirectUpdate`. -pub(super) struct DirectUpdate<'a> { - pub collection: &'a str, - pub field: &'a str, - pub targets: &'a VectorWriteTargets, - pub new_vector: Option<&'a [f32]>, - pub payload_patch: &'a [(String, UpdateValue)], - pub spec: VectorPrimarySpec, -} - -impl DirectWrites<'_> { - pub(super) fn insert( - &mut self, - write: DirectInsert<'_>, - ops: &mut Vec, - ) -> crate::Result<()> { - match self { - Self::Staged(collections) => { - note_direct_write( - collections, - write.collection, - write.field, - Some(write.spec), - Some((write.surrogate, write.pk_bytes)), - ); - Ok(()) - } - Self::Plan => { - let payload = encode_vector_direct_upsert_payload(VectorDirectUpsertPayload { - collection: write.collection, - field: write.field, - surrogate: write.surrogate, - pk_bytes: write.pk_bytes, - vector: write.vector, - payload: write.payload, - quantization: write.spec.quantization, - storage_dtype: write.spec.storage_dtype, - payload_indexes: &write.spec.payload_indexes, - intent: write.intent, - on_conflict_updates: write.on_conflict_updates, - })?; - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectUpsert as u32, - payload, - }); - Ok(()) - } - } - } - - pub(super) fn update( - &mut self, - write: DirectUpdate<'_>, - ops: &mut Vec, - ) -> crate::Result<()> { - match self { - Self::Staged(collections) => { - note_direct_write( - collections, - write.collection, - write.field, - Some(write.spec), - None, - ); - Ok(()) - } - Self::Plan => { - let payload = encode_vector_direct_update_payload(VectorDirectUpdatePayload { - collection: write.collection, - field: write.field, - targets: write.targets, - new_vector: write.new_vector, - payload_patch: write.payload_patch, - quantization: write.spec.quantization, - storage_dtype: write.spec.storage_dtype, - payload_indexes: &write.spec.payload_indexes, - })?; - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectUpdate as u32, - payload, - }); - Ok(()) - } - } - } - - pub(super) fn delete( - &mut self, - collection: &str, - field: &str, - targets: &VectorWriteTargets, - ops: &mut Vec, - ) -> crate::Result<()> { - match self { - Self::Staged(collections) => { - note_direct_write(collections, collection, field, None, None); - Ok(()) - } - Self::Plan => { - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectDelete as u32, - payload: encode_vector_direct_delete_payload(collection, field, targets)?, - }); - Ok(()) - } - } - } - - pub(super) fn truncate( - &mut self, - collection: &str, - field: &str, - ops: &mut Vec, - ) -> crate::Result<()> { - match self { - Self::Staged(collections) => { - note_direct_write(collections, collection, field, None, None); - Ok(()) - } - Self::Plan => { - ops.push(RedoSubRecord { - record_type: RecordType::VectorDirectTruncate as u32, - payload: encode_vector_direct_truncate_payload(collection, field)?, - }); - Ok(()) - } - } - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/vector_primary.rs b/nodedb/src/data/executor/handlers/transaction/resolve/vector_primary.rs index 396840e3c..b3174429e 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/vector_primary.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/vector_primary.rs @@ -23,8 +23,8 @@ //! Rows are emitted in surrogate order, so two resolves of one transaction //! produce byte-identical records. //! -//! This covers session transactions. A Calvin transaction stages no -//! vector-primary write and resolves from its plans instead (`vector_direct`). +//! Session and Calvin transactions stage every direct write, so both +//! resolve here. use std::collections::BTreeMap; diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs index 590c7c139..2761273bb 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/mod.rs @@ -8,9 +8,9 @@ //! statement, not deferred to COMMIT), the real affected-row count is //! computed, and the resulting encoded body (or a tombstone) is recorded in //! the overlay so a later same-transaction read-modify-write observes it. The -//! write is NOT made durable here — the buffered plan is still replayed -//! through the real apply path inside the COMMIT `TransactionBatch`, which -//! remains the sole durable apply. +//! write is NOT made durable here — COMMIT resolves the overlay into the +//! transaction's redo record, and the redo install remains the sole durable +//! apply. mod body; mod constraint; @@ -26,6 +26,7 @@ mod stage_columnar_family; mod stage_columnar_resolved_dml; mod stage_crdt; mod stage_current_body; +mod stage_document_batch; mod stage_graph; mod stage_kv; mod stage_kv_atomic; @@ -38,6 +39,7 @@ mod stage_point_document; mod stage_rls; mod stage_spatial; mod stage_timeseries; +mod stage_timeseries_ilp; mod stage_timeseries_now; mod stage_truncate; mod stage_upsert; @@ -55,6 +57,9 @@ pub(in crate::data::executor) use stage_columnar_dml::{ pub(in crate::data::executor) use stage_columnar_resolved_dml::{ StageColumnarResolvedDeleteParams, StageColumnarResolvedUpdateParams, }; +pub(in crate::data::executor) use stage_document_batch::{ + StageBalanceDeltaParams, StageBatchInsertParams, +}; pub(in crate::data::executor) use stage_graph::GRAPH_LABEL_COLL_KEY; pub(in crate::data::executor) use stage_kv::{kv_row_identity, unhex_key}; pub(in crate::data::executor) use stage_spatial::StageSpatialInsertParams; diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_document_batch.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_document_batch.rs new file mode 100644 index 000000000..1cf03a819 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_document_batch.rs @@ -0,0 +1,398 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Staging for the two document writes only a Calvin transaction stages: +//! `BatchInsert` and `ApplyBalanceDelta`. +//! +//! A multi-shard statement commits through Calvin. Calvin stages every plan +//! into the transaction overlay, and the flush installs the redo record that +//! resolve builds from that overlay. A plan that stages nothing therefore +//! installs nothing. +//! +//! - `BatchInsert` stages each row the way a document `PointPut` stages it. +//! The live batch writes each row through `apply_point_put`, which +//! overwrites a row already stored at its surrogate. +//! - `ApplyBalanceDelta` reads the target row under BASE ∪ OVERLAY, adds the +//! delta to the balance column, and stages the new row. The arithmetic, +//! the encoding and the refusals are those of +//! `CoreLoop::apply_balance_delta`, the live handler's read-modify-write. + +use std::str::FromStr; + +use nodedb_types::{RowIdentity, StorageKey, Surrogate}; +use rust_decimal::Decimal; + +use super::context::StageCtx; +use crate::bridge::envelope::{ErrorCode, Response, Status}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::doc_format; +use crate::data::executor::enforcement::materialized_sum::delta::json_to_decimal; +use crate::data::executor::handlers::document::read::decode::decode_scanned_document; +use crate::data::executor::sparse_body_format::SparseBodyFormat; +use crate::data::executor::task::ExecutionTask; +use crate::types::{TenantId, TxnId}; + +/// Inputs for [`CoreLoop::stage_document_batch_insert`]. +pub(in crate::data::executor) struct StageBatchInsertParams<'a> { + pub task: &'a ExecutionTask, + pub tid: u64, + pub txn_id: TxnId, + pub collection: &'a str, + /// `(client identity, body)` per row, in plan order. + pub documents: &'a [(String, Vec)], + /// One surrogate per row, parallel to `documents`. + pub surrogates: &'a [Surrogate], +} + +/// Inputs for [`CoreLoop::stage_apply_balance_delta`]. +pub(in crate::data::executor) struct StageBalanceDeltaParams<'a> { + pub task: &'a ExecutionTask, + pub tid: u64, + pub txn_id: TxnId, + /// TARGET collection: the one the balance row lives in. + pub collection: &'a str, + /// The target row's client identity. + pub document_id: &'a str, + pub surrogate: Surrogate, + /// The balance column being moved. + pub column: &'a str, + /// Signed amount to add, as an exact decimal string. + pub delta: &'a str, + pub join_column: &'a str, + pub join_value: &'a str, + /// The target collection's declared `PRIMARY KEY` column, when it has one. + pub declared_primary_key: Option<&'a str>, +} + +impl CoreLoop { + /// Stage every row of a document batch insert. Answers the number of + /// rows staged. A row that fails its UNIQUE check fails the whole plan. + pub(in crate::data::executor) fn stage_document_batch_insert( + &mut self, + params: StageBatchInsertParams<'_>, + ) -> Response { + let StageBatchInsertParams { + task, + tid, + txn_id, + collection, + documents, + surrogates, + } = params; + // Every overlay slot and every index is keyed by surrogate, so a row + // without one cannot be staged. + if surrogates.len() != documents.len() { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "document batch insert for '{collection}' carries {} documents but {} \ + surrogates; every row needs its surrogate to be staged", + documents.len(), + surrogates.len(), + ), + }, + ); + } + for ((document_id, value), surrogate) in documents.iter().zip(surrogates) { + let ctx = StageCtx::new( + task, + tid, + txn_id, + collection, + RowIdentity::from_user_key(document_id.as_str()), + *surrogate, + ); + let staged = self.stage_point_put(&ctx, value); + if staged.status == Status::Error { + return staged; + } + } + self.stage_count_response(task, documents.len()) + } + + /// Stage the target row of a materialized-sum balance move with the delta + /// added to its balance column. Answers one affected row. + pub(in crate::data::executor) fn stage_apply_balance_delta( + &mut self, + params: StageBalanceDeltaParams<'_>, + ) -> Response { + let task = params.task; + match self.staged_balance_row(¶ms) { + Ok((identity, body)) => { + let ctx = StageCtx::new( + task, + params.tid, + params.txn_id, + params.collection, + identity, + params.surrogate, + ); + if let Err(e) = self.stage_put_capped(&ctx, body) { + return self.response_error(task, e); + } + self.stage_count_response(task, 1) + } + Err(e) => self.response_error(task, e), + } + } + + /// The target row's identity and stored body after the balance move. + fn staged_balance_row( + &self, + params: &StageBalanceDeltaParams<'_>, + ) -> Result<(RowIdentity, Vec), ErrorCode> { + // A delta that does not parse is a malformed plan. Applying zero + // would report a balance move that never happened. + let delta = Decimal::from_str(params.delta).map_err(|e| ErrorCode::Internal { + detail: format!( + "materialized-sum delta '{}' for {}.{} is not a decimal: {e}", + params.delta, params.collection, params.column + ), + })?; + let database_id = params.task.request.database_id; + let tenant = TenantId::new(params.tid); + let format = self.sparse_body_format(database_id, tenant, params.collection); + // A vector-primary row is a tagged sidecar. A document body written + // over it would read back as tag arrays. + if matches!(format, SparseBodyFormat::VectorSidecar) { + return Err(crate::Error::Storage { + engine: "materialized_sum".into(), + detail: format!( + "target collection '{}' is vector-primary; its rows are metadata \ + sidecars and cannot carry a materialized sum", + params.collection + ), + } + .into()); + } + + let read_ctx = StageCtx::new( + params.task, + params.tid, + params.txn_id, + params.collection, + RowIdentity::from_user_key(params.document_id), + params.surrogate, + ); + let Some(current) = self.stage_current_body(&read_ctx)? else { + return Err(crate::Error::MaterializedSumTargetNotFound { + target_collection: params.collection.to_string(), + join_column: params.join_column.to_string(), + join_value: params.join_value.to_string(), + } + .into()); + }; + + let storage_key = StorageKey::for_surrogate(params.surrogate); + let mut target_doc = decode_scanned_document(¤t, format.as_format_ref())?; + let balance = target_doc + .get(params.column) + .and_then(json_to_decimal) + .unwrap_or(Decimal::ZERO) + + delta; + let Some(object) = target_doc.as_object_mut() else { + return Err(crate::Error::Storage { + engine: "materialized_sum".into(), + detail: format!( + "target row {}/{storage_key} is not an object", + params.collection + ), + } + .into()); + }; + // A balance is stored as text: `f64` loses digits past 15. + object.insert( + params.column.to_string(), + serde_json::Value::String(balance.to_string()), + ); + + let submitted = doc_format::encode_to_msgpack(&target_doc); + let identity = + RowIdentity::of_stored_row(&submitted, params.declared_primary_key, storage_key); + let body = self.stage_encode_put_body( + database_id.as_u64(), + params.tid, + params.collection, + params.surrogate, + &submitted, + )?; + Ok((identity, body)) + } +} + +#[cfg(test)] +mod tests { + use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; + use nodedb_types::{DatabaseId, QualifiedCollection, StorageKey, Surrogate, Value}; + + use crate::bridge::envelope::{Response, Status}; + use crate::data::executor::core_loop::CoreLoop; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::doc_format; + + const TID: u64 = 1; + + fn object(fields: &[(&str, &str)]) -> Vec { + let map: std::collections::HashMap = fields + .iter() + .map(|(k, v)| ((*k).to_string(), Value::String((*v).to_string()))) + .collect(); + zerompk::to_msgpack_vec(&Value::Object(map)).expect("encode object") + } + + fn seed(core: &mut CoreLoop, collection: &str, surrogate: u32, fields: &[(&str, &str)]) { + let body = doc_format::canonicalize_document_for_storage(&object(fields)); + core.sparse + .put( + DatabaseId::DEFAULT.as_u64(), + TID, + collection, + &StorageKey::for_surrogate(Surrogate::new(surrogate)), + &body, + ) + .expect("seed row"); + } + + fn field(core: &CoreLoop, collection: &str, surrogate: u32, name: &str) -> Option { + let body = core + .sparse + .get( + DatabaseId::DEFAULT.as_u64(), + TID, + collection, + &StorageKey::for_surrogate(Surrogate::new(surrogate)), + ) + .expect("read row")?; + match doc_format::decode_document_value(&body).expect("decode row") { + Value::Object(map) => map.get(name).cloned(), + _ => None, + } + } + + fn commit(core: &mut CoreLoop, plans: &[PhysicalPlan]) -> Response { + core.calvin_commit_for_test(&make_default_task(), TID, plans, 1, 100) + } + + fn balance_delta(delta: &str) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::ApplyBalanceDelta { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "accounts"), + document_id: "acc1".to_string(), + surrogate: Surrogate::new(9), + column: "balance".to_string(), + delta: delta.to_string(), + join_column: "id".to_string(), + join_value: "acc1".to_string(), + declared_primary_key: None, + }) + } + + /// A Calvin balance move stages the target row with the delta added, and + /// the flush installs it. + #[test] + fn a_calvin_balance_delta_adds_to_the_stored_balance() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + seed( + &mut core, + "accounts", + 9, + &[("id", "acc1"), ("balance", "10")], + ); + + let response = commit(&mut core, &[balance_delta("5"), balance_delta("-2.5")]); + + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + assert_eq!( + field(&core, "accounts", 9, "balance"), + Some(Value::String("12.5".into())), + "both moves land, the second on the first's staged row" + ); + } + + /// A balance move whose target row is absent refuses the transaction + /// rather than dropping the move. + #[test] + fn a_calvin_balance_delta_on_a_missing_target_is_refused() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + + let response = commit(&mut core, &[balance_delta("5")]); + + assert_eq!(response.status, Status::Error); + assert_eq!(field(&core, "accounts", 9, "balance"), None); + } + + /// A Calvin batch insert installs every row it staged. + #[test] + fn a_calvin_batch_insert_installs_every_row() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let plan = PhysicalPlan::Document(DocumentOp::BatchInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), + documents: vec![ + ("o1".to_string(), object(&[("a", "one")])), + ("o2".to_string(), object(&[("a", "two")])), + ], + surrogates: vec![Surrogate::new(7), Surrogate::new(8)], + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + + let response = commit(&mut core, &[plan]); + + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + assert_eq!( + field(&core, "orders", 7, "a"), + Some(Value::String("one".into())) + ); + assert_eq!( + field(&core, "orders", 8, "a"), + Some(Value::String("two".into())) + ); + } + + /// A batch insert whose surrogates do not pair with its rows is refused + /// and installs nothing. + #[test] + fn a_calvin_batch_insert_without_a_surrogate_per_row_is_refused() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let plan = PhysicalPlan::Document(DocumentOp::BatchInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), + documents: vec![("o1".to_string(), object(&[("a", "one")]))], + surrogates: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + + let response = commit(&mut core, &[plan]); + + assert_eq!(response.status, Status::Error); + assert_eq!(field(&core, "orders", 7, "a"), None); + } + + /// A Calvin truncate removes every base row of the collection. + #[test] + fn a_calvin_truncate_removes_the_base_rows() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + seed(&mut core, "orders", 1, &[("a", "one")]); + seed(&mut core, "orders", 2, &[("a", "two")]); + let plan = PhysicalPlan::Document(DocumentOp::Truncate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }); + + let response = commit(&mut core, &[plan]); + + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + assert_eq!(field(&core, "orders", 1, "a"), None); + assert_eq!(field(&core, "orders", 2, "a"), None); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs index da1404a9f..84d136bfe 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs @@ -15,7 +15,6 @@ use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::Staged; use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::current_ms; use crate::types::{DatabaseId, TenantId, TxnId}; /// Lowercase-hex encode a raw KV key. [`unhex_key`] is the inverse. @@ -198,6 +197,7 @@ impl CoreLoop { | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } | KvOp::MaterializeScan { .. } // Resolve-before-propose is autocommit-only: it decides against // committed state and proposes directly, never through staging. @@ -297,7 +297,7 @@ impl CoreLoop { super::constraint::OverlayPk::Present => true, super::constraint::OverlayPk::Absent => false, super::constraint::OverlayPk::Unstaged => { - let now_ms = current_ms(); + let now_ms = self.kv_read_now_ms(); self.kv_engine .get(ctx.database_id, ctx.tid, ctx.collection, key, now_ms) .is_some() @@ -336,7 +336,7 @@ impl CoreLoop { coll_key.1.as_u64(), coll_key.2.as_str(), key, - current_ms(), + self.kv_read_now_ms(), ) .map(|(_, s)| s), } @@ -355,7 +355,7 @@ impl CoreLoop { Some(Staged::Tombstone) => None, None if !self.stage_base_visible(ctx) => None, None => { - let now_ms = current_ms(); + let now_ms = self.kv_read_now_ms(); self.kv_engine .get(ctx.database_id, ctx.tid, ctx.collection, key, now_ms) } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs index 8436ae538..4f2b0754b 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_atomic.rs @@ -8,7 +8,7 @@ //! `InsertOnConflictUpdate` handler: resolve the current value under //! BASE ∪ OVERLAY via [`CoreLoop::resolve_kv_current`], compute the new value //! with the SAME pure function the autocommit engine methods call -//! (`nodedb::engine::kv::atomic_compute`, see `engine_atomic_compute.rs`) so +//! (`nodedb_physical::kv_atomic::compute`) so //! a staged value and its COMMIT-time durable replay never diverge, then //! stage the new bytes via [`CoreLoop::stage_put_capped`]. //! @@ -36,6 +36,7 @@ //! realistic transaction, and never persisted (COMMIT replay uses the real //! `KvEngine` atomic path, which ignores the overlay's surrogate entirely). +use nodedb_physical::kv_atomic::compute as atomic_compute; use nodedb_physical::physical_plan::{KvCounterShape, KvOp}; use nodedb_types::Surrogate; @@ -47,7 +48,6 @@ use crate::data::executor::handlers::kv::atomic::incr_float_reply; use crate::data::executor::handlers::transaction::overlay::StagedTtl; use crate::data::executor::response_codec; use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::{atomic_compute, current_ms}; use crate::types::TxnId; /// FNV-1a 32-bit hash, used only to derive a stable, collection-local overlay @@ -205,7 +205,7 @@ impl CoreLoop { } self.kv_atomic_json_response(ctx.task, &serde_json::json!({ "value": new_i64 })) } - Err(e) => self.response_atomic_error(ctx.task, ctx.collection, e), + Err(e) => self.response_atomic_error(ctx.task, ctx.collection, e.into()), } } @@ -229,7 +229,7 @@ impl CoreLoop { } self.kv_atomic_json_response(ctx.task, &reply) } - Err(e) => self.response_atomic_error(ctx.task, ctx.collection, e), + Err(e) => self.response_atomic_error(ctx.task, ctx.collection, e.into()), } } @@ -269,7 +269,7 @@ impl CoreLoop { let (matches, write_bytes) = match atomic_compute::cas(current.as_deref(), expected, new_value) { Ok(outcome) => outcome, - Err(e) => return self.response_atomic_error(ctx.task, ctx.collection, e), + Err(e) => return self.response_atomic_error(ctx.task, ctx.collection, e.into()), }; if matches { @@ -306,7 +306,7 @@ impl CoreLoop { let current = self.resolve_kv_current(ctx, key); let write_bytes = match atomic_compute::getset(current.as_deref(), new_value) { Ok(bytes) => bytes, - Err(e) => return self.response_atomic_error(ctx.task, ctx.collection, e), + Err(e) => return self.response_atomic_error(ctx.task, ctx.collection, e.into()), }; if let Err(e) = self.stage_admit_kv_image(ctx, &write_bytes, rls_write_check) { return self.response_error(ctx.task, e); @@ -363,10 +363,7 @@ impl CoreLoop { if ttl_ms == 0 { return; } - let now_ms: u64 = self - .epoch_system_ms - .map(|ms| ms as u64) - .unwrap_or_else(current_ms); + let now_ms = self.kv_read_now_ms(); self.txn_overlay_mut(ctx.txn_id).set_ttl( ctx.coll_key.clone(), ctx.surrogate.0, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs index 34b381c6c..1ef444fb9 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_delete.rs @@ -19,7 +19,6 @@ use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::Staged; use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::current_ms; use crate::types::TxnId; use super::stage_kv::kv_row_identity; @@ -68,7 +67,7 @@ impl CoreLoop { Some(Staged::Put(body)) => Some(body.clone()), Some(Staged::Tombstone) => None, None => { - let now_ms = current_ms(); + let now_ms = self.kv_read_now_ms(); self.kv_engine .get(did.as_u64(), tid, collection, key, now_ms) } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_predicate.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_predicate.rs index e64972413..e6a8f586a 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_predicate.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_predicate.rs @@ -25,7 +25,6 @@ use crate::data::executor::handlers::kv::field_compute::merge_field_updates; use crate::data::executor::response_codec; use crate::data::executor::scan_normalize::kv_row_to_doc; use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::current_ms; use crate::types::{DatabaseId, TenantId, TxnId}; /// Routing identity + payload for one staged `KvOp::PredicateUpdate`. @@ -202,7 +201,7 @@ impl CoreLoop { decode_scan_filters(filter_bytes, "kv predicate dml filters") .map_err(|e| self.response_error(task, e))?; let mut rows = self - .kv_predicate_matches(did, tid, collection, filter_bytes, current_ms()) + .kv_predicate_matches(did, tid, collection, filter_bytes, self.kv_read_now_ms()) .map_err(|e| self.response_error(task, e))?; // `merge_kv_overlay_into_scan` takes an infallible predicate, so a @@ -260,6 +259,7 @@ mod tests { use crate::data::executor::core_loop::tests::make_core_with_dir; use crate::data::executor::handlers::transaction::overlay::Staged; use crate::engine::kv::KvPutParams; + use crate::engine::kv::current_ms; use crate::types::*; fn make_task() -> ExecutionTask { diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs index 80383048b..94734aeb4 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_transfer.rs @@ -11,7 +11,7 @@ //! (`kv::field_compute::merge_field_updates`, `kv::transfer_compute:: //! compute_transfer`), so a staged value and its COMMIT-time durable replay //! are never derived from different code paths -- mirrors `stage_kv_atomic.rs`'s -//! reuse of `engine_atomic_compute`. +//! reuse of `nodedb_physical::kv_atomic::compute`. //! //! Like `Incr` / `IncrFloat` / `Cas` / `GetSet`, these three ops carry a //! planner-assigned cross-engine surrogate on their plan. That surrogate binds diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_ttl.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_ttl.rs index e1928b1b2..f43b764b1 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_ttl.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv_ttl.rs @@ -30,7 +30,6 @@ use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::overlay::StagedTtl; use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::current_ms; use crate::types::TxnId; /// The row a staged TTL mutation targets, plus the policy that decides it. @@ -78,10 +77,7 @@ impl CoreLoop { return self.response_error(task, e); } - let now_ms: u64 = self - .epoch_system_ms - .map(|ms| ms as u64) - .unwrap_or_else(current_ms); + let now_ms = self.kv_read_now_ms(); let coll_key = ctx.coll_key.clone(); let document_id = ctx.document_id.clone(); let surrogate: Surrogate = ctx.surrogate; diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs index 4c250f32c..a689b50d2 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_point_document.rs @@ -8,9 +8,9 @@ //! violations immediately (at the statement, not deferred to COMMIT), computes //! the real affected-row count, and records the resulting encoded body (or a //! tombstone) in the per-transaction overlay so a later same-transaction -//! read-modify-write observes it. The write is NOT made durable here — the -//! buffered plan is still replayed through the real apply path inside the -//! COMMIT `TransactionBatch`, which remains the sole durable apply. +//! read-modify-write observes it. The write is NOT made durable here — +//! COMMIT resolves the overlay into the transaction's redo record, and the +//! redo install remains the sole durable apply. //! //! Split out of `dispatch.rs` (which owns the `MetaOp::StageWrite` routing and //! the shared `stage_overlay_pk` / `stage_put_capped` / `stage_count_response` diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs index 2d1c6fdb4..e0e77bc6d 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_spatial.rs @@ -21,9 +21,8 @@ //! sparse store, built via the same `geometry_to_value` helper and encoded //! with `nodedb_types::value_to_msgpack` -- decoded the same way by //! `merge_overlay_into_spatial_scan` (the `Value::Object` staged-body -//! branch). COMMIT durable replay is unchanged: the buffered `SpatialOp` -//! plan is still replayed through `execute_spatial_insert` / -//! `execute_spatial_delete` inside the COMMIT `TransactionBatch`. +//! branch). COMMIT installs the spatial write from the transaction's redo +//! record, which serializes it from the buffered `SpatialOp` plan node. use nodedb_types::geometry::Geometry; use nodedb_types::{RowIdentity, Surrogate}; diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs index 719f379ad..451e5c88a 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries.rs @@ -73,14 +73,14 @@ pub(in crate::data::executor) struct StageTimeseriesInsertParams<'a> { /// Borrowed inputs for the canonical line-protocol staging path. Bundled /// because the raw parameter list exceeds the project's too-many-arguments /// bound. -struct CanonicalIlpStage<'a> { - task: &'a ExecutionTask, - tid: u64, - txn_id: TxnId, - collection: &'a str, - payload: &'a [u8], - surrogates: &'a [Surrogate], - rls_write_check: &'a nodedb_types::RlsWriteCheck, +pub(super) struct CanonicalIlpStage<'a> { + pub task: &'a ExecutionTask, + pub tid: u64, + pub txn_id: TxnId, + pub collection: &'a str, + pub payload: &'a [u8], + pub surrogates: &'a [Surrogate], + pub rls_write_check: &'a nodedb_types::RlsWriteCheck, } impl CoreLoop { @@ -103,16 +103,25 @@ impl CoreLoop { rls_write_check, } = params; + let stage = CanonicalIlpStage { + task, + tid, + txn_id, + collection, + payload, + surrogates, + rls_write_check, + }; + // The canonical line list always travels with one token per line. if format == "ilp-msgpack" { - return self.stage_canonical_ilp_rows(CanonicalIlpStage { - task, - tid, - txn_id, - collection, - payload, - surrogates, - rls_write_check, - }); + return self.stage_canonical_ilp_rows(stage); + } + // An ingest with no surrogates has no overlay key for its rows. + if surrogates.is_empty() { + return self.stage_unkeyed_timeseries(stage, format); + } + if format == "ilp" { + return self.stage_raw_ilp_rows(stage); } let rows: Vec = match nodedb_types::value_from_msgpack(payload) { @@ -227,7 +236,7 @@ impl CoreLoop { self.stage_count_response(task, staged) } - fn stage_canonical_ilp_rows(&mut self, args: CanonicalIlpStage<'_>) -> Response { + pub(super) fn stage_canonical_ilp_rows(&mut self, args: CanonicalIlpStage<'_>) -> Response { let CanonicalIlpStage { task, tid, diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries_ilp.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries_ilp.rs new file mode 100644 index 000000000..9fb070d76 --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_timeseries_ilp.rs @@ -0,0 +1,216 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Statement-time staging for the timeseries ingests the row-keyed paths in +//! `stage_timeseries` do not take: +//! +//! - A raw line-protocol payload (`format = "ilp"`) that carries one +//! surrogate per line is rewritten into the canonical line list and staged +//! row by row through the canonical path, so a same-transaction read +//! observes its rows. +//! - A payload in any format that carries no surrogates has no overlay key +//! for its rows, as a native bulk ingest sends it. It is decided the way +//! the canonical path decides a batch: normalized into line protocol, +//! parsed, matched to its routed collection, admitted by the write policy +//! and prevalidated against the memtable. It stages no row. COMMIT resolve +//! serializes the ingest from the plan node, so the install still writes +//! every line. + +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::timeseries::StampedIngest; +use crate::types::TenantId; + +use super::stage_timeseries::CanonicalIlpStage; + +impl CoreLoop { + /// Stage a raw line-protocol ingest that carries surrogates. Answers the + /// number of lines. + pub(super) fn stage_raw_ilp_rows(&mut self, args: CanonicalIlpStage<'_>) -> Response { + let task = args.task; + let text = match std::str::from_utf8(args.payload) { + Ok(text) => text, + Err(error) => { + return self.response_error( + task, + ErrorCode::RejectedPrevalidation { + reason: format!("line protocol is not UTF-8: {error}"), + }, + ); + } + }; + let lines: Vec = text + .lines() + .map(str::trim) + .filter(|line| !line.is_empty()) + .map(str::to_string) + .collect(); + if lines.len() != args.surrogates.len() { + return self.response_error( + task, + ErrorCode::RejectedPrevalidation { + reason: format!( + "line-protocol payload carries {} lines but {} surrogates", + lines.len(), + args.surrogates.len() + ), + }, + ); + } + let canonical = match zerompk::to_msgpack_vec(&lines) { + Ok(bytes) => bytes, + Err(error) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("canonical ILP payload encode failed: {error}"), + }, + ); + } + }; + self.stage_canonical_ilp_rows(CanonicalIlpStage { + payload: &canonical, + ..args + }) + } + + /// Decide an ingest that carries no surrogates, in `format`. Stages no + /// row and answers the number of lines it holds. + pub(super) fn stage_unkeyed_timeseries( + &mut self, + args: CanonicalIlpStage<'_>, + format: &str, + ) -> Response { + match self.decide_unkeyed_timeseries(&args, format) { + Ok((lines, now_ms)) => { + // COMMIT resolve stamps the untimed rows with the instant the + // statement read, not its own. + let coll_key = ( + args.task.request.database_id, + TenantId::new(args.tid), + args.collection.to_string(), + ); + self.txn_overlay_mut(args.txn_id) + .note_unkeyed_ingest_now(&coll_key, now_ms); + self.stage_count_response(args.task, lines) + } + Err(error) => self.response_error(args.task, error), + } + } + + /// Normalize, parse, route-check, admit and prevalidate an unkeyed + /// ingest. Mutates nothing. Returns the number of lines and the instant + /// the ingest read as its default row timestamp. + fn decide_unkeyed_timeseries( + &self, + args: &CanonicalIlpStage<'_>, + format: &str, + ) -> Result<(usize, i64), ErrorCode> { + let tenant = TenantId::new(args.tid); + let now_ms = self.ingest_now_ms(); + let lines = self.stamped_ingest_lines(StampedIngest { + database_id: args.task.request.database_id, + tid: tenant, + collection: args.collection, + payload: args.payload, + format, + now_ms, + })?; + let source = lines.join("\n"); + let parsed = match crate::engine::timeseries::ilp::parse_batch(&source) { + Ok(parsed) if parsed.lines().len() == lines.len() => parsed, + _ => { + return Err(ErrorCode::RejectedPrevalidation { + reason: "invalid line-protocol row".into(), + }); + } + }; + let measurement = args + .collection + .split_once(':') + .map(|(_, name)| name) + .unwrap_or(args.collection); + if parsed + .lines() + .iter() + .any(|row| row.measurement.as_ref() != measurement) + { + return Err(ErrorCode::RejectedPrevalidation { + reason: "line-protocol measurement does not match routed collection".into(), + }); + } + crate::data::executor::handlers::timeseries::admit_ilp_lines( + args.rls_write_check, + parsed.lines(), + self.declared_ts_time_key(args.task.request.database_id, tenant, args.collection), + now_ms, + args.tid, + args.collection, + )?; + self.prevalidate_deferred_ilp_ingest(args.task, tenant, args.collection, parsed.lines())?; + Ok((lines.len(), now_ms)) + } +} + +#[cfg(test)] +mod tests { + use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp}; + use nodedb_types::{DatabaseId, QualifiedCollection}; + + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::CoreLoop; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::task::ExecutionTask; + use crate::types::TxnId; + + const TID: u64 = 1; + + fn unkeyed_ingest() -> PhysicalPlan { + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), + payload: b"metrics,host=a value=1".to_vec(), + format: "ilp".to_owned(), + wal_lsn: None, + surrogates: Vec::new(), + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }) + } + + /// Stage the unkeyed ingest at `stage_ms` and resolve it at + /// `resolve_ms`. Returns the resolved redo record. + fn stage_then_resolve(stage_ms: i64, resolve_ms: i64) -> Vec { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx): (CoreLoop, _, _) = make_core_with_dir(dir.path()); + let txn_id = TxnId::new(5); + let mut request = make_default_task().request; + request.txn_id = Some(txn_id); + let task = ExecutionTask::new(request); + let plan = unkeyed_ingest(); + + core.epoch_system_ms = Some(stage_ms); + let staged = core.execute_stage_write(&task, TID, &plan); + assert_eq!(staged.status, Status::Ok, "{:?}", staged.error_code); + + core.epoch_system_ms = Some(resolve_ms); + let resolved = core.execute_resolve_txn(&task, TID, txn_id, std::slice::from_ref(&plan)); + assert_eq!(resolved.status, Status::Ok, "{:?}", resolved.error_code); + resolved.payload.as_bytes().to_vec() + } + + #[test] + fn an_unkeyed_ingest_resolves_with_the_instant_its_statement_read() { + let staged_early = stage_then_resolve(1_000_000, 9_000_000); + assert_eq!( + staged_early, + stage_then_resolve(1_000_000, 1_000_000), + "the untimed row carries the stage instant" + ); + assert_ne!( + staged_early, + stage_then_resolve(9_000_000, 9_000_000), + "the commit instant does not reach the row" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_truncate.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_truncate.rs index e50fc444d..e87e3755f 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_truncate.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_truncate.rs @@ -6,10 +6,10 @@ //! transaction's overlay (`TxnOverlay::mark_truncated`). Every row staged //! earlier in the same transaction is tombstoned through the undo journal, //! and every base row with no newer overlay entry is hidden from the -//! transaction's own reads and existence probes. Nothing touches base: the -//! buffered plan replays through the live truncate inside the COMMIT -//! `TransactionBatch`, in statement order, and `ROLLBACK` / `ROLLBACK TO -//! SAVEPOINT` drop the marker. +//! transaction's own reads and existence probes. Nothing touches base: COMMIT +//! resolves the truncate into the transaction's redo record ahead of the +//! rows staged after it, and `ROLLBACK` / `ROLLBACK TO SAVEPOINT` drop the +//! marker. //! //! The response carries no payload: the statement's tag is the bare //! `TRUNCATE` autocommit answers with, so there is no row count to report. diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan.rs deleted file mode 100644 index 79c0b1691..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan.rs +++ /dev/null @@ -1,460 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Per-sub-plan dispatch within a transaction batch. -//! -//! Write-op execution helpers (the pieces that actually mutate engine state -//! and record undo entries) live in `sub_plan_write.rs`; this file only -//! routes each `PhysicalPlan` variant to its engine-specific handler. - -use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Request, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; -use crate::types::{RequestId, TenantId}; -use nodedb_physical::physical_plan::{CrdtOp, DocumentOp, GraphOp, MetaOp, TimeseriesOp, VectorOp}; - -use super::sub_plan_doc::{TxPointDelete, TxPointPut}; -use super::sub_plan_write::{TxEdgeDeleteParams, TxEdgePutParams, TxVectorInsertParams}; -use super::sub_request::SubRequestScope; -use super::undo::UndoEntry; - -impl CoreLoop { - /// Execute a single sub-plan within a transaction, recording undo info. - /// - /// CRDT deltas are NOT applied immediately — they are buffered in - /// `crdt_deltas` and only applied after all sub-plans succeed. - /// - /// Dispatches by outer `PhysicalPlan` variant to a per-engine helper. - /// Each helper handles that engine's write sub-ops (pushing an - /// `UndoEntry`) and routes every other sub-op through the standard - /// read-only / DDL dispatch path. - #[cfg(test)] - pub(super) fn execute_tx_sub_plan( - &mut self, - tid: u64, - plan: &PhysicalPlan, - undo_log: &mut Vec, - crdt_deltas: &mut Vec<(Vec, u64, String)>, - user_roles: &[String], - ) -> Result { - let task = Self::build_dummy_task(tid); - self.execute_tx_sub_plan_with_task(&task, tid, plan, undo_log, crdt_deltas, user_roles) - } - - /// Replay a sub-plan with the parent batch's database and vShard identity. - /// Calvin executes the same logical batch on each participant, so erasing - /// this routing identity makes every replica appear to be vShard zero and - /// breaks canonical-owner side effects such as graph statistics. - pub(super) fn execute_tx_sub_plan_from_batch( - &mut self, - parent: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - undo_log: &mut Vec, - crdt_deltas: &mut Vec<(Vec, u64, String)>, - user_roles: &[String], - ) -> Result { - // A deferred timeseries ingest must receive the enclosing transaction - // record's WAL LSN. The synthetic task below intentionally has no LSN - // (the batch records ordinary write versions only after commit), but - // the timeseries partition stamp is its own replay floor. Losing the - // enclosing LSN here would let a later flush stamp zero and replay the - // committed transaction on top of its partition after restart. - if let PhysicalPlan::Timeseries(op) = plan { - return self.exec_tx_timeseries(parent, tid, plan, op, undo_log); - } - - // The sub-plan is part of the parent statement, so it runs on what is - // left of the parent's budget rather than a fresh one. - let task = Self::build_dummy_task_at(tid, &parent.request); - self.execute_tx_sub_plan_with_task(&task, tid, plan, undo_log, crdt_deltas, user_roles) - } - - fn execute_tx_sub_plan_with_task( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - undo_log: &mut Vec, - crdt_deltas: &mut Vec<(Vec, u64, String)>, - user_roles: &[String], - ) -> Result { - match plan { - PhysicalPlan::Document(op) => { - self.exec_tx_document(dummy_task, tid, plan, op, user_roles, undo_log) - } - PhysicalPlan::Vector(op) => self.exec_tx_vector(dummy_task, tid, plan, op, undo_log), - PhysicalPlan::Graph(op) => self.exec_tx_graph(dummy_task, tid, plan, op, undo_log), - PhysicalPlan::Crdt(op) => self.exec_tx_crdt(dummy_task, tid, plan, op, crdt_deltas), - PhysicalPlan::Kv(kv_op) => self.execute_tx_kv(dummy_task, tid, plan, kv_op, undo_log), - PhysicalPlan::Columnar(op) => { - self.exec_tx_columnar(dummy_task, tid, plan, op, undo_log) - } - PhysicalPlan::Timeseries(op) => { - self.exec_tx_timeseries(dummy_task, tid, plan, op, undo_log) - } - PhysicalPlan::Spatial(_) - | PhysicalPlan::Text(_) - | PhysicalPlan::Query(_) - | PhysicalPlan::Meta(_) - | PhysicalPlan::Array(_) - | PhysicalPlan::ClusterArray(_) - | PhysicalPlan::ClusterEvent(_) => { - self.exec_tx_passthrough(tid, plan, &dummy_task.request) - } - } - } - - /// Build the ephemeral task used for sub-plan response construction. - /// - /// no-determinism: the deadline is ephemeral, not written to WAL. The - /// placeholder `plan` (a no-op `Meta::Cancel`) is never executed; it - /// only carries request metadata for response building. - #[cfg(test)] - pub(super) fn build_dummy_task(tid: u64) -> ExecutionTask { - use crate::bridge::envelope::{Admission, ExemptReason}; - use crate::types::{DatabaseId, VShardId}; - - let scope = SubRequestScope { - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - // no-determinism: test-only dummy deadline, never written to Calvin state - deadline: std::time::Instant::now() + std::time::Duration::from_secs(60), - admission: Admission::Exempt(ExemptReason::Read), - }; - ExecutionTask::new(scope.request(tid, Self::dummy_plan())) - } - - /// The dummy task runs under [`SubRequestScope::of`] `parent`: the - /// parent's database, vShard, deadline, and admission. Every sub-plan - /// this task carries stops when the statement does. - fn build_dummy_task_at(tid: u64, parent: &Request) -> ExecutionTask { - ExecutionTask::new(SubRequestScope::of(parent).request(tid, Self::dummy_plan())) - } - - fn dummy_plan() -> PhysicalPlan { - PhysicalPlan::Meta(MetaOp::Cancel { - target_request_id: RequestId::new(0), - }) - } - - /// Document engine: point writes are undo-tracked; everything else - /// (point reads, scans, DDL) passes through the standard dispatch path. - fn exec_tx_document( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - op: &DocumentOp, - user_roles: &[String], - undo_log: &mut Vec, - ) -> Result { - match op { - DocumentOp::PointPut { - collection, - document_id, - value, - surrogate, - resolved_sum_targets, - .. - } => self.tx_point_put( - TxPointPut { - task: dummy_task, - tid, - collection: collection.as_str(), - document_id, - surrogate: *surrogate, - value, - user_roles, - insert_if_absent: None, - resolved_sum_targets, - // A put carries no deferral list of its own: its balance is - // settled from row images and deferred by OMISSION from the - // resolution just above, which travels with it. - deferred_sum_targets: &[], - }, - undo_log, - ), - - DocumentOp::PointInsert { - collection, - document_id, - value, - if_absent, - surrogate, - resolved_sum_targets, - // An insert's rows are new by construction, so its cross-shard - // balance is settled at plan time and marked here rather than - // omitted from the resolution. Forwarding it is not optional: - // this arm is on the CALVIN apply path, which is the only path - // a deferral-carrying write ever takes. - deferred_sum_targets, - .. - } => self.tx_point_put( - TxPointPut { - task: dummy_task, - tid, - collection: collection.as_str(), - document_id, - surrogate: *surrogate, - value, - user_roles, - insert_if_absent: Some(*if_absent), - resolved_sum_targets, - deferred_sum_targets, - }, - undo_log, - ), - - DocumentOp::PointDelete { - collection, - document_id, - surrogate, - resolved_sum_targets, - .. - } => self.tx_point_delete( - TxPointDelete { - task: dummy_task, - tid, - collection: collection.as_str(), - document_id, - surrogate: *surrogate, - user_roles, - resolved_sum_targets, - }, - undo_log, - ), - - _ => self.exec_tx_passthrough(tid, plan, &dummy_task.request), - } - } - - /// Vector engine: primary-vector insert/delete are undo-tracked; - /// everything else passes through the standard dispatch path. - fn exec_tx_vector( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - op: &VectorOp, - undo_log: &mut Vec, - ) -> Result { - match op { - VectorOp::Insert { - collection, - vector, - dim, - field_name, - surrogate, - pk_bytes: _, - provenance: _, - } => self.exec_tx_vector_insert( - dummy_task, - tid, - TxVectorInsertParams { - collection: collection.as_str(), - vector, - dim: *dim, - field_name, - surrogate: *surrogate, - }, - undo_log, - ), - - VectorOp::Delete { - collection, - vector_id, - } => Ok(self.exec_tx_vector_delete( - dummy_task, - tid, - collection.as_str(), - *vector_id, - undo_log, - )), - - _ => self.exec_tx_passthrough(tid, plan, &dummy_task.request), - } - } - - /// Graph engine: edge put/delete are undo-tracked; everything else - /// passes through the standard dispatch path. - fn exec_tx_graph( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - op: &GraphOp, - undo_log: &mut Vec, - ) -> Result { - match op { - GraphOp::EdgePut { - collection, - src_id, - label, - dst_id, - properties, - src_surrogate, - dst_surrogate, - } => self.exec_tx_edge_put( - dummy_task, - tid, - TxEdgePutParams { - collection: collection.as_str(), - src_id, - label, - dst_id, - properties, - src_surrogate: *src_surrogate, - dst_surrogate: *dst_surrogate, - }, - undo_log, - ), - - GraphOp::EdgeDelete { - collection, - src_id, - label, - dst_id, - rls_write_check, - .. - } => self.exec_tx_edge_delete( - dummy_task, - tid, - TxEdgeDeleteParams { - collection: collection.as_str(), - src_id, - label, - dst_id, - rls_write_check, - }, - undo_log, - ), - - GraphOp::EdgePutBatch { edges } => { - let mut response = self.response_ok(dummy_task); - for edge in edges { - response = self.exec_tx_edge_put( - dummy_task, - tid, - TxEdgePutParams { - collection: edge.collection.as_str(), - src_id: &edge.src_id, - label: &edge.label, - dst_id: &edge.dst_id, - properties: &[], - src_surrogate: edge.src_surrogate, - dst_surrogate: edge.dst_surrogate, - }, - undo_log, - )?; - } - Ok(response) - } - - GraphOp::EdgeDeleteBatch { edges } => { - let mut response = self.response_ok(dummy_task); - for edge in edges { - response = self.exec_tx_edge_delete( - dummy_task, - tid, - TxEdgeDeleteParams { - collection: edge.collection.as_str(), - src_id: &edge.src_id, - label: &edge.label, - dst_id: &edge.dst_id, - // A batched edge carries no property image, so the - // planner refuses the batch outright while a write - // policy applies — nothing reaches here to decide. - rls_write_check: &nodedb_types::RlsWriteCheck::NoPolicyApplies, - }, - undo_log, - )?; - } - Ok(response) - } - - _ => self.exec_tx_passthrough(tid, plan, &dummy_task.request), - } - } - - /// CRDT raw deltas cannot be part of a transaction batch: their exact - /// post-merge authorization must be evaluated at the serialized admission - /// boundary before any durable proposal. Document operations remain - /// transaction-capable through their own staged handlers. - fn exec_tx_crdt( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - op: &CrdtOp, - _crdt_deltas: &mut Vec<(Vec, u64, String)>, - ) -> Result { - match op { - CrdtOp::Apply { .. } | CrdtOp::ApplyAuthenticated { .. } => { - Err(ErrorCode::Unsupported { - detail: "CRDT Apply is not supported inside transaction batches".into(), - }) - } - _ => self.exec_tx_passthrough(tid, plan, &dummy_task.request), - } - } - - /// Timeseries engine: ingest is undo-tracked; everything else passes - /// through the standard dispatch path. - fn exec_tx_timeseries( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - op: &TimeseriesOp, - undo_log: &mut Vec, - ) -> Result { - match op { - TimeseriesOp::Ingest { - collection, - payload, - format, - wal_lsn, - rls_write_check, - .. - } => self.execute_tx_timeseries_ingest( - dummy_task, - super::sub_plan_kv::TxTimeseriesIngestParams { - tid: TenantId::new(tid), - collection: collection.as_str(), - payload, - format, - rls_write_check, - // The enclosing transaction record, when present, is - // the durable identity of this ingest. A plan-local LSN is - // only a compatibility fallback for direct unit callers. - wal_lsn: dummy_task.wal_lsn().map(|lsn| lsn.as_u64()).or(*wal_lsn), - }, - undo_log, - ), - - // Staged as an overlay marker at statement time; the live truncate - // records its whole pre-image (memory state plus the partition - // directory renamed aside) for atomic rollback. - TimeseriesOp::Truncate { - collection, - restart_identity: _, - } => { - let resp = self.execute_timeseries_truncate( - dummy_task, - collection.as_str(), - Some(undo_log), - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "timeseries truncate failed".into(), - })); - } - Ok(resp) - } - - TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_) => { - self.exec_tx_passthrough(tid, plan, &dummy_task.request) - } - } - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_columnar.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_columnar.rs deleted file mode 100644 index a3380b74e..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_columnar.rs +++ /dev/null @@ -1,144 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Columnar engine sub-plan dispatch within a transaction batch. -//! -//! Split out of `sub_plan.rs` to keep that file under the size limit; this -//! is still the columnar arm of the same per-sub-plan dispatcher. - -use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; -use nodedb_physical::physical_plan::ColumnarOp; - -use super::undo::UndoEntry; - -impl CoreLoop { - /// Columnar engine: insert, predicate update / delete, their resolved - /// forms, and truncate are undo-tracked; everything else passes through - /// the standard dispatch path. - /// - /// Predicate update/delete are staged at statement time; this is the - /// durable COMMIT replay. Undo is captured here so a sibling sub-plan - /// failing later in the same COMMIT batch reverses this mutation — - /// without it the columnar change would survive an atomic-rollback - /// (partial commit). - pub(super) fn exec_tx_columnar( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - op: &ColumnarOp, - undo_log: &mut Vec, - ) -> Result { - match op { - ColumnarOp::Insert { - collection, - payload, - format, - intent, - on_conflict_updates, - surrogates, - schema_bytes, - provenance: _, - wal_lsn: _, - rls_write_check, - // A row-returning write is refused before it can be staged into - // a transaction, so neither the projection nor the read gate - // that bounds it can be set on a plan reaching this path. - returning: _, - rls_filters: _, - } => self.execute_tx_columnar_insert( - dummy_task, - super::sub_plan_kv::TxColumnarInsertParams { - collection: collection.as_str(), - payload, - format, - intent: *intent, - on_conflict_updates, - surrogates, - schema_bytes, - rls_write_check, - }, - undo_log, - ), - - ColumnarOp::Update { - collection, - filters, - updates, - rls_write_check, - } => self.exec_tx_columnar_update( - dummy_task, - collection.as_str(), - filters, - updates, - rls_write_check, - undo_log, - ), - - ColumnarOp::Delete { - collection, - filters, - rls_write_check, - } => self.exec_tx_columnar_delete( - dummy_task, - collection.as_str(), - filters, - rls_write_check, - undo_log, - ), - - // Resolved-row-set forms: the Control Plane already resolved the - // predicate and decided the write policy against the exact rows, - // so — unlike `Update`/`Delete` above — there is no filter scan - // here either; the apply handler itself does the drift check - // against the current PK index before mutating anything. - ColumnarOp::ResolvedUpdate { - collection, - rows, - rls_write_check, - } => self.exec_tx_columnar_resolved_update( - dummy_task, - collection.as_str(), - rows, - rls_write_check, - undo_log, - ), - - ColumnarOp::ResolvedDelete { - collection, - pks, - rls_write_check, - } => self.exec_tx_columnar_resolved_delete( - dummy_task, - collection.as_str(), - pks, - rls_write_check, - undo_log, - ), - - // Staged as an overlay marker at statement time; the live truncate - // wipes every row replayed before it in this batch and records - // its whole pre-image for atomic rollback. - ColumnarOp::Truncate { - collection, - restart_identity: _, - } => { - let resp = - self.execute_columnar_truncate(dummy_task, collection.as_str(), Some(undo_log)); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "columnar truncate failed".into(), - })); - } - Ok(resp) - } - - ColumnarOp::Scan { .. } - | ColumnarOp::MaterializeScan { .. } - | ColumnarOp::ResolveDml { .. } => { - self.exec_tx_passthrough(tid, plan, &dummy_task.request) - } - } - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs deleted file mode 100644 index 4e4717170..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/delete.rs +++ /dev/null @@ -1,155 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Document PointDelete helper for transaction sub-plans. - -use crate::bridge::envelope::{ErrorCode, Response}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::enforcement::funnel::WriteEnforcementOutcome; -use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, WriteImages}; -use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; -use crate::data::executor::handlers::transaction::undo::UndoEntry; -use crate::data::executor::handlers::transaction::undo::document_outcome::{ - DocumentRow, push_delete_undo, push_target_undo, -}; -use crate::data::executor::task::ExecutionTask; - -/// Parameters for [`CoreLoop::tx_point_delete`]. -pub(in crate::data::executor::handlers::transaction) struct TxPointDelete<'a> { - pub task: &'a ExecutionTask, - pub tid: u64, - pub collection: &'a str, - pub document_id: &'a str, - pub surrogate: nodedb_types::Surrogate, - pub user_roles: &'a [String], - /// Join-key VALUE → target row surrogate for every materialized-sum target - /// this delete must debit, resolved on the Control Plane at plan time. - pub resolved_sum_targets: &'a [nodedb_physical::physical_plan::ResolvedSumTarget], -} - -impl CoreLoop { - /// Execute a PointDelete within a transaction. - pub(in crate::data::executor::handlers::transaction) fn tx_point_delete( - &mut self, - p: TxPointDelete<'_>, - undo_log: &mut Vec, - ) -> Result { - let TxPointDelete { - task: dummy_task, - tid, - collection, - document_id, - surrogate, - user_roles, - resolved_sum_targets, - } = p; - let database_id = dummy_task.request.database_id.as_u64(); - let hook_ctx = HookCtx { - database_id, - tid, - collection, - resolved_targets: resolved_sum_targets, - // A delete carries no deferral list, and needs none: `PointDelete` - // has no such field because its balance is settled from the stored - // row's image and deferred by OMISSION from the resolution above, - // which this helper already forwards. Empty here is complete, not a - // dropped field. - deferred_sum_targets: &[], - wal_lsn: dummy_task.wal_lsn(), - }; - - // Core delete path shared with the autocommit caller: bitemporal-vs-plain - // primary tombstone/delete (including versioned index tombstones), - // FTS/inverted removal, secondary-index cascade, graph-edge cascade, - // spatial R-tree removal, `mark_node_deleted` bookkeeping, doc_cache - // invalidation, and stateless DELETE enforcement. Every side-effect is - // captured in the outcome and reversed via the undo log below, so the - // transactional delete is identical to autocommit and fully - // rollback-safe. - // - // Each transaction sub-plan owns its own per-row redb write txn; the - // batch is stitched together by the undo log, not one big txn. A - // failure inside `apply_point_delete` returns before the commit, so the - // txn is dropped and every sparse-database write it staged is rolled - // back. - let txn = self.sparse.begin_write().map_err(|e| ErrorCode::Internal { - detail: e.to_string(), - })?; - let outcome = self.apply_point_delete( - &txn, - PointDeleteParams { - database_id, - tid, - collection, - document_id, - surrogate, - user_roles, - enforce: true, - resolved_targets: resolved_sum_targets, - }, - )?; - - // Whether a row was actually removed, captured before `prior_value` is - // moved into the undo entry below. A delete against an absent key is the - // same plan as one that removed a row, so the count is only knowable - // here. - let removed = outcome.prior_value.is_some(); - - // Image-folding enforcement, inside the SAME transaction the removal was - // staged in, so a materialized-sum debit and the row's removal land or - // roll back together. The pre-image is the only image a delete has, and - // it is what a running total has to subtract; a delete that matched - // nothing folds nothing. - let enforcement = match outcome.prior_value { - Some(ref old) => write_hook::run( - self, - &txn, - &hook_ctx, - WriteImages::Delete { - old: ImageBody::Stored(old), - }, - )?, - None => WriteEnforcementOutcome::default(), - }; - let WriteEnforcementOutcome { - target_writes, - balanced_entries, - } = enforcement; - - // A removal SUBTRACTS the row's amount from its group, so it is - // accumulated onto the open batch like any other write: a transaction - // that deletes one leg of a balanced journal leaves the group - // unbalanced and is refused at the batch's commit boundary. - self.settle_balanced_entries(database_id, tid, collection, balanced_entries)?; - - txn.commit().map_err(|e| ErrorCode::Internal { - detail: format!("commit: {e}"), - })?; - self.checkpoint_coordinator.mark_dirty("sparse", 1); - - // A target write is a full document write with side effects of its - // own. The delete's own entries reverse the row, its index entries, - // vectors, R-tree entries, the node tombstone, and every edge the - // cascade removed. - push_target_undo(undo_log, &target_writes); - push_delete_undo( - undo_log, - DocumentRow { - database_id, - tid, - collection, - storage_key: nodedb_types::StorageKey::for_surrogate(surrogate), - // The plan's `document_id` is the row's client identity. - identity: nodedb_types::RowIdentity::from_user_key(document_id), - }, - outcome, - ); - - // `PointDelete` renders a `DELETE ` command tag, so its response - // carries the count — 0 when the key was absent — exactly as the - // autocommit handler (`handlers/point/delete.rs`) reports it. A bare - // `Ok` here leaves a Calvin-flushed delete with no count for the - // coordinator to render, since `execute_transaction_batch` hands the - // last sub-plan's payload back as the participant's applied response. - Ok(self.response_affected(dummy_task, u64::from(removed))) - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/mod.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/mod.rs deleted file mode 100644 index 64a75365f..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/mod.rs +++ /dev/null @@ -1,7 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -pub(in crate::data::executor::handlers::transaction) mod delete; -pub(in crate::data::executor::handlers::transaction) mod put; - -pub(in crate::data::executor::handlers::transaction) use delete::TxPointDelete; -pub(in crate::data::executor::handlers::transaction) use put::TxPointPut; diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs deleted file mode 100644 index f1a98d86a..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_doc/put.rs +++ /dev/null @@ -1,324 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Document PointPut helper for transaction sub-plans. - -use crate::bridge::envelope::{ErrorCode, Response}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::enforcement::chain_guard::{self, ChainGuard}; -use crate::data::executor::enforcement::funnel::WriteEnforcementOutcome; -use crate::data::executor::enforcement::write_hook::{self, HookCtx, ImageBody, WriteImages}; -use crate::data::executor::handlers::point::apply_put::PointPutParams; -use crate::data::executor::handlers::transaction::undo::UndoEntry; -use crate::data::executor::handlers::transaction::undo::document_outcome::{ - DocumentRow, push_put_undo, push_target_undo, -}; -use crate::data::executor::task::ExecutionTask; - -/// Parameters for [`CoreLoop::tx_point_put`]. -pub(in crate::data::executor::handlers::transaction) struct TxPointPut<'a> { - pub task: &'a ExecutionTask, - pub tid: u64, - pub collection: &'a str, - pub document_id: &'a str, - pub surrogate: nodedb_types::Surrogate, - pub value: &'a [u8], - pub user_roles: &'a [String], - /// Insert-vs-upsert semantics. `None` = PUT/upsert (overwrite is allowed, - /// no existence probe). `Some(if_absent)` = INSERT semantics: probe for an - /// existing primary key under the same write txn and, if present, either - /// silently skip (`if_absent = true`, `INSERT ... ON CONFLICT DO NOTHING`) - /// or reject with a `unique` constraint violation (`if_absent = false`). - pub insert_if_absent: Option, - /// Join-key VALUE → target row surrogate, resolved on the Control Plane at - /// plan time for every materialized-sum target this write may touch. The - /// Data Plane addresses target rows with these and never derives them: the - /// primary-key → surrogate map is Control-Plane catalog state. - pub resolved_sum_targets: &'a [nodedb_physical::physical_plan::ResolvedSumTarget], - /// Materialized-sum TARGET collections whose delta the Control Plane - /// settled at plan time and shipped on its own `ApplyBalanceDelta` task. - /// This write must not apply them as well. - /// - /// It travels this far because the CALVIN apply path runs through here: - /// `execute_calvin_flush` replays every staged plan via - /// `execute_transaction_batch`, which routes `PointInsert` into this - /// helper. A cross-shard statement is the only kind that ever carries a - /// deferral AND the only kind that commits through Calvin, so dropping it - /// here dropped it on every write that has one — the source core folded - /// the balance inline and the sibling task folded it again. - /// - /// Empty for a PUT: `PointPut` carries no deferral list, because its - /// balance is settled from row images and deferred by OMISSION from - /// `resolved_sum_targets` above, which this struct already forwards. - pub deferred_sum_targets: &'a [String], -} - -impl CoreLoop { - /// Execute a PointPut within a transaction. - pub(in crate::data::executor::handlers::transaction) fn tx_point_put( - &mut self, - p: TxPointPut<'_>, - undo_log: &mut Vec, - ) -> Result { - let TxPointPut { - task: dummy_task, - tid, - collection, - document_id, - surrogate, - value, - user_roles, - insert_if_absent, - resolved_sum_targets, - deferred_sum_targets, - } = p; - let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); - let database_id = dummy_task.request.database_id.as_u64(); - - // Pre-read the plain-table value: it decides insert-vs-update for the - // hash chain, and it is the PRE-IMAGE the enforcement funnel folds — an - // enforcement that only sees the post-image cannot tell an update from - // an insert, which is how a running total came to double-count one. - // The authoritative prior value for the undo entry comes from - // `apply_point_put`'s outcome, which is bitemporal-aware. - // - // Read only when something needs it: a collection that declares neither - // a chain nor an image-folding constraint must not pay for a read whose - // result nothing consults. - let hook_ctx = HookCtx { - database_id, - tid, - collection, - resolved_targets: resolved_sum_targets, - deferred_sum_targets, - wal_lsn: dummy_task.wal_lsn(), - }; - let mut chain = ChainGuard::begin(self, database_id, tid, collection); - let folds_images = write_hook::folds_images(self, &hook_ctx); - let prior_bytes = if chain.enabled() || folds_images { - self.sparse - .get(database_id, tid, collection, &storage_key) - .ok() - .flatten() - } else { - None - }; - let is_insert = prior_bytes.is_none(); - - // Hash-chain wraps the document with a `_chain_hash` field on insert; - // feed that wrapped value into `apply_point_put` so it stores/indexes - // the chained form. - let chained: Option> = if is_insert { - chain - .chain_insert(self, database_id, tid, document_id, value) - .map_err(|e| ErrorCode::Internal { - detail: format!("hash chain: {e}"), - })? - } else { - None - }; - let effective_value: &[u8] = chained.as_deref().unwrap_or(value); - - // Each transaction sub-plan owns its own per-row redb write txn; the - // batch is stitched together by the undo log, not one big txn. - let txn = self.sparse.begin_write().map_err(|e| ErrorCode::Internal { - detail: e.to_string(), - })?; - - // INSERT semantics: probe for an existing primary key under the SAME - // write txn we will commit through — linearizable with the write, so no - // concurrent writer can slip a row in between the probe and the commit. - // Mirrors autocommit `execute_point_insert`. PUT/upsert (`None`) skips - // this entirely and keeps overwrite behaviour. - if let Some(if_absent) = insert_if_absent { - let exists_result = if self.is_bitemporal(database_id, tid, collection) { - self.sparse.versioned_exists_current_in_txn( - &txn, - database_id, - tid, - collection, - &storage_key, - ) - } else { - self.sparse - .exists_in_txn(&txn, database_id, tid, collection, &storage_key) - }; - let exists = match exists_result { - Ok(exists) => exists, - Err(e) => { - // Restore any chain-head pre-image mutated above before bailing. - chain.restore(self); - return Err(ErrorCode::from(e)); - } - }; - if exists { - // No write, no undo push — drop the txn without committing. - chain.restore(self); - if if_absent { - // `INSERT ... ON CONFLICT DO NOTHING`: silent skip, which - // affected NO row. The count is reported, not omitted — - // see the count contract at the end of this function. - return Ok(self.response_affected(dummy_task, 0)); - } - return Err(ErrorCode::from(crate::Error::RejectedConstraint { - collection: collection.to_string(), - constraint: "unique".to_string(), - detail: format!( - "duplicate key value '{document_id}' violates primary-key \ - uniqueness on '{collection}'" - ), - })); - } - } - - // Core write path shared with the autocommit callers: bitemporal-vs-plain - // primary doc write, FTS/inverted, doc_cache, aggregate-cache - // invalidation, UNIQUE enforcement, generated columns, stateless PUT - // enforcement, and the side indexes (secondary/spatial/vector/stats). - // Every side-effect is captured in the outcome and reversed via the undo - // log below, so the transactional write is identical to autocommit and - // fully rollback-safe. - let outcome = match self.apply_point_put( - &txn, - PointPutParams { - database_id, - tid, - collection, - storage_key, - surrogate, - value: effective_value, - index_text: true, - user_roles, - enforce: true, - wal_lsn: dummy_task.wal_lsn(), - resolved_targets: resolved_sum_targets, - }, - ) { - Ok(o) => o, - Err(e) => { - // `apply_point_put` rejected the write (e.g. UNIQUE violation) - // after we mutated the chain head and, on the later rejections, - // after it had already cached the row. Reverse both so the - // aborted op leaves no trace, then propagate the typed error. - chain_guard::abort_after_apply( - self, - &chain, - database_id, - tid, - collection, - &storage_key, - ); - return Err(e.into()); - } - }; - - // Persist the advanced chain head inside the SAME write transaction the - // chained row lands in. Every abort path above returns before this point - // and drops `txn` uncommitted, so a rejected insert never leaves a head - // behind on disk either. - if let Err(e) = chain.persist_head(self, &txn) { - chain_guard::abort_after_apply( - self, - &chain, - database_id, - tid, - collection, - &storage_key, - ); - return Err(ErrorCode::from(e)); - } - - // Write-path enforcement runs one level ABOVE `apply_point_put`, and - // inside THIS transaction: a materialized-sum target write is itself an - // `apply_point_put`, so every derived write lands or rolls back with the - // row that caused it. On failure the chain-head pre-image is restored - // and `txn` is dropped uncommitted, leaving neither the row nor any - // target it credited behind. - // - // The post-image is the SUBMITTED body, not the chained one: - // `_chain_hash` is a wrapper the hash chain adds around the row, and no - // constraint is declared over it. - let images = match prior_bytes { - Some(ref old) => WriteImages::Update { - old: ImageBody::Stored(old), - new: ImageBody::Submitted(value), - }, - None => WriteImages::Insert { - new: ImageBody::Submitted(value), - }, - }; - let enforcement = match write_hook::run(self, &txn, &hook_ctx, images) { - Ok(outcome) => outcome, - Err(e) => { - chain_guard::abort_after_apply( - self, - &chain, - database_id, - tid, - collection, - &storage_key, - ); - return Err(ErrorCode::from(e)); - } - }; - let WriteEnforcementOutcome { - target_writes, - balanced_entries, - } = enforcement; - - // The BALANCED check spans the whole transaction — debits and credits - // arrive on different rows — so this row's signed contributions are - // accumulated onto the open batch, which judges them all at its commit - // boundary. Nothing is checked here: one leg per statement is legal - // inside an explicit transaction. - if let Err(e) = self.settle_balanced_entries(database_id, tid, collection, balanced_entries) - { - chain_guard::abort_after_apply( - self, - &chain, - database_id, - tid, - collection, - &storage_key, - ); - return Err(ErrorCode::from(e)); - } - - txn.commit().map_err(|e| ErrorCode::Internal { - detail: format!("commit: {e}"), - })?; - self.checkpoint_coordinator.mark_dirty("sparse", 1); - - // A target write is a full document write with side effects of its - // own, reversed with the same entries the source row uses. - push_target_undo(undo_log, &target_writes); - push_put_undo( - undo_log, - DocumentRow { - database_id, - tid, - collection, - storage_key, - // The plan's `document_id` is the row's client identity. - identity: nodedb_types::RowIdentity::from_user_key(document_id), - }, - outcome, - chain.prior(), - ); - - // One row was written, and the count is REPORTED — `PointPut` and - // `PointInsert` both render an `INSERT ` command tag, so their - // response must carry the count `response_affected` documents as - // mandatory for every count-bearing plan. The autocommit handlers - // (`handlers/point/put.rs`, `handlers/point/insert.rs`) already do. - // - // This path is not only the explicit-transaction commit, where the tag - // is `COMMIT` and the count is discarded: `execute_calvin_flush` replays - // a participant's staged plans through `execute_transaction_batch`, - // which returns the LAST sub-plan's payload as the whole participant's - // applied response — and that response is what the scheduler deposits - // and what the coordinator shapes a cross-shard statement's tag from. - // Returning a bare `Ok` here left an autocommit cross-shard INSERT with - // no count to render at all. - Ok(self.response_affected(dummy_task, 1)) - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs deleted file mode 100644 index 3c6e8785a..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv.rs +++ /dev/null @@ -1,267 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Columnar and Timeseries write tracking for transaction batches. -//! -//! These handlers capture prior state before each write so the undo log -//! can reverse the operation on batch failure. -//! -//! KV operation dispatch lives in `sub_plan_kv_ops`. - -use nodedb_columnar::pk_index::RowLocation; - -use crate::bridge::envelope::{ErrorCode, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::handlers::timeseries::{TimeseriesApplyMode, TimeseriesIngestExec}; -use crate::data::executor::task::ExecutionTask; -use crate::types::TenantId; -use nodedb_physical::physical_plan::ColumnarInsertIntent; -use nodedb_physical::physical_plan::document::UpdateValue; - -use super::undo::UndoEntry; - -/// Captured undo state for a pending columnar insert: the list of new PK bytes -/// to insert, paired with the prior `RowLocation` of any displaced memtable rows. -pub(in crate::data::executor) type ColumnarUndoState = (Vec>, Vec<(Vec, RowLocation)>); - -/// Parameters for [`CoreLoop::execute_tx_columnar_insert`]. -pub(super) struct TxColumnarInsertParams<'a> { - pub collection: &'a str, - pub payload: &'a [u8], - pub format: &'a str, - pub intent: ColumnarInsertIntent, - pub on_conflict_updates: &'a [(String, UpdateValue)], - pub surrogates: &'a [nodedb_types::Surrogate], - pub schema_bytes: &'a [u8], - /// Compiled row-level-security WRITE predicate carried by the buffered - /// plan. COMMIT replay is the sole durable apply for an in-transaction - /// columnar write, so the predicate has to survive the buffering. - pub rls_write_check: &'a nodedb_types::RlsWriteCheck, -} - -/// Parameters for [`CoreLoop::execute_tx_timeseries_ingest`]. -pub(super) struct TxTimeseriesIngestParams<'a> { - pub tid: TenantId, - pub collection: &'a str, - pub payload: &'a [u8], - pub format: &'a str, - pub wal_lsn: Option, - /// Compiled row-level-security WRITE predicate carried by the buffered - /// plan; see [`TxColumnarInsertParams::rls_write_check`]. - pub rls_write_check: &'a nodedb_types::RlsWriteCheck, -} - -impl CoreLoop { - // ── Columnar insert ────────────────────────────────────────────────────── - - /// Execute a columnar insert in a transaction context. - /// - /// Captures `row_count_before`, inserted PK bytes, and displaced prior-row - /// locations before the insert so the undo log can reverse the operation. - pub(super) fn execute_tx_columnar_insert( - &mut self, - task: &ExecutionTask, - params: TxColumnarInsertParams<'_>, - undo_log: &mut Vec, - ) -> Result { - let TxColumnarInsertParams { - collection, - payload, - format, - intent, - on_conflict_updates, - surrogates, - schema_bytes, - rls_write_check, - } = params; - let collection_key = ( - task.request.database_id, - task.request.tenant_id, - collection.to_string(), - ); - - let row_count_before = self - .columnar_engines - .get(&collection_key) - .map(|e| e.memtable().row_count()) - .unwrap_or(0); - - let (inserted_pks, displaced) = - self.capture_columnar_insert_undo_state(&collection_key, payload, intent); - - let resp = self.execute_columnar_insert( - task, - crate::data::executor::handlers::columnar_write::ColumnarInsertParams { - collection, - payload, - format, - intent, - on_conflict_updates, - surrogates, - schema_bytes, - provenance: None, - rls_write_check, - // A row-returning write inside a transaction is refused on the - // Control Plane before it reaches any sub-plan, so this path - // never carries a projection or the read gate that bounds one. - returning: None, - rls_filters: &[], - // R-tree entries are in-memory, so a rollback must un-index - // the rows it removes. - spatial_undo: Some(&mut *undo_log), - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "columnar insert failed".into(), - })); - } - - undo_log.push(UndoEntry::ColumnarInsert { - collection_key, - row_count_before, - inserted_pks, - displaced, - }); - Ok(resp) - } - - /// Capture the PK bytes and displaced prior-row locations for a pending - /// columnar insert, without executing the insert. - fn capture_columnar_insert_undo_state( - &self, - collection_key: &(nodedb_types::DatabaseId, TenantId, String), - payload: &[u8], - intent: ColumnarInsertIntent, - ) -> ColumnarUndoState { - let Some(engine) = self.columnar_engines.get(collection_key) else { - // Engine doesn't exist yet; execute_columnar_insert will create it. - // row_count_before will be 0, so truncate_to(0) handles rollback. - return (Vec::new(), Vec::new()); - }; - let ndb_rows: Vec = match nodedb_types::value_from_msgpack(payload) { - Ok(nodedb_types::Value::Array(arr)) => arr, - Ok(v @ nodedb_types::Value::Object(_)) => vec![v], - _ => return (Vec::new(), Vec::new()), - }; - let schema = engine.schema(); - let rows: Vec> = ndb_rows - .iter() - .filter_map(|row| match row { - nodedb_types::Value::Object(obj) => Some( - schema - .columns - .iter() - .map(|col| { - obj.get(&col.name) - .cloned() - .unwrap_or(nodedb_types::Value::Null) - }) - .collect(), - ), - _ => None, - }) - .collect(); - self.columnar_insert_undo_state(collection_key, &rows, intent) - } - - /// The PK bytes a columnar insert of `rows` (schema-ordered values) will - /// bind, and the memtable rows it will displace, captured before the - /// insert runs. - pub(in crate::data::executor) fn columnar_insert_undo_state( - &self, - collection_key: &(nodedb_types::DatabaseId, TenantId, String), - rows: &[Vec], - intent: ColumnarInsertIntent, - ) -> ColumnarUndoState { - let mut inserted_pks: Vec> = Vec::new(); - let mut displaced: Vec<(Vec, RowLocation)> = Vec::new(); - let Some(engine) = self.columnar_engines.get(collection_key) else { - return (inserted_pks, displaced); - }; - for values in rows { - let Ok(pk_bytes) = engine.encode_pk_from_row(values) else { - continue; - }; - - match intent { - // `InsertUnique` never displaces a prior row. A real - // conflict fails the statement before this undo entry is - // used. On the success path it matches `InsertIfAbsent`: - // record the new PK only when nothing occupies it. - ColumnarInsertIntent::InsertIfAbsent | ColumnarInsertIntent::InsertUnique => { - if !engine.pk_index().contains(&pk_bytes) { - inserted_pks.push(pk_bytes); - } - } - ColumnarInsertIntent::Insert | ColumnarInsertIntent::Put => { - if let Some(prior_loc) = engine.pk_index().get(&pk_bytes).copied() - && prior_loc.segment_id == engine.memtable_segment_id() - { - displaced.push((pk_bytes.clone(), prior_loc)); - } - inserted_pks.push(pk_bytes); - } - } - } - - (inserted_pks, displaced) - } - - // ── Timeseries ingest ──────────────────────────────────────────────────── - - /// Execute a timeseries ingest in a transaction context. - /// - /// Captures the complete in-memory pre-image before deferred ingest. - /// - /// Transactional ingest may evolve schema and symbol dictionaries, create - /// a memtable, update the last-value cache and advance an LSN. A row count - /// cannot restore those mutations, so the undo token owns snapshots of all - /// mutable state before any ingest code runs. - pub(super) fn execute_tx_timeseries_ingest( - &mut self, - task: &ExecutionTask, - params: TxTimeseriesIngestParams<'_>, - undo_log: &mut Vec, - ) -> Result { - let TxTimeseriesIngestParams { - tid, - collection, - payload, - format, - wal_lsn, - rls_write_check, - } = params; - let collection_key = (task.request.database_id, tid, collection.to_string()); - - let undo = self.capture_timeseries_ingest_undo(&collection_key); - - // Push before mutation. A panic in ingest is caught by the batch - // driver, which can then restore this exact pre-image. - undo_log.push(UndoEntry::TimeseriesIngest(undo)); - let resp = self.execute_timeseries_ingest(TimeseriesIngestExec { - task, - tid, - collection, - payload, - format, - wal_lsn, - provenance: None, - mode: TimeseriesApplyMode::CommitDeferred, - rls_write_check, - // A row-returning write inside a transaction is refused on the - // Control Plane before it can be staged, so no plan reaching this - // sub-plan path carries a projection or the read gate that bounds - // one. Blanking them here is a statement of that invariant, not a - // convenience. - returning: None, - rls_filters: &[], - }); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "timeseries ingest failed".into(), - })); - } - - Ok(resp) - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs deleted file mode 100644 index 7f47b89cf..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_atomics.rs +++ /dev/null @@ -1,196 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! COMMIT-time execution of the KV read-modify-write atomics: `Incr`, -//! `IncrFloat`, `Cas`, `GetSet`. -//! -//! Split out of `sub_plan_kv_writes.rs` to keep that file under the per-file -//! line budget. All four share one shape: read the prior value, delegate to -//! the SAME live handler autocommit uses — so the row-level-security write -//! gate, the TTL resolution, and the event emission are the autocommit ones -//! rather than a second implementation — then record the prior value so a -//! later sibling sub-plan's failure can restore it. - -use crate::bridge::envelope::{ErrorCode, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::current_ms; -use nodedb_physical::physical_plan::KvOp; - -use super::undo::UndoEntry; - -impl CoreLoop { - /// Execute one KV atomic in a transaction context. - /// - /// Caller invariant: `op` is `Incr`, `IncrFloat`, `Cas`, or `GetSet` — - /// `execute_tx_kv_write` routes nothing else here. - pub(super) fn execute_tx_kv_atomic( - &mut self, - task: &ExecutionTask, - did: u64, - tid: u64, - op: &KvOp, - undo_log: &mut Vec, - ) -> Result { - match op { - KvOp::Incr { - collection, - key, - delta, - ttl_ms, - surrogate, - rls_write_check, - shape, - } => { - let now_ms = current_ms(); - let prior = self - .kv_engine - .entry_image(did, tid, collection.as_str(), key, now_ms); - let resp = self.execute_kv_incr( - crate::data::executor::handlers::kv::atomic::KvAtomicCtx { - task, - did, - tid, - collection: collection.as_str(), - key, - surrogate: *surrogate, - rls_write_check, - }, - *delta, - *ttl_ms, - shape, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv incr failed".into(), - })); - } - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior, - }); - Ok(resp) - } - - KvOp::IncrFloat { - collection, - key, - delta, - surrogate, - rls_write_check, - shape, - } => { - let now_ms = current_ms(); - let prior = self - .kv_engine - .entry_image(did, tid, collection.as_str(), key, now_ms); - let resp = self.execute_kv_incr_float( - crate::data::executor::handlers::kv::atomic::KvAtomicCtx { - task, - did, - tid, - collection: collection.as_str(), - key, - surrogate: *surrogate, - rls_write_check, - }, - delta, - shape, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv incr float failed".into(), - })); - } - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior, - }); - Ok(resp) - } - - KvOp::Cas { - collection, - key, - expected, - new_value, - surrogate, - rls_write_check, - } => { - let now_ms = current_ms(); - let prior = self - .kv_engine - .entry_image(did, tid, collection.as_str(), key, now_ms); - let resp = self.execute_kv_cas( - crate::data::executor::handlers::kv::atomic::KvAtomicCtx { - task, - did, - tid, - collection: collection.as_str(), - key, - surrogate: *surrogate, - rls_write_check, - }, - expected, - new_value, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv cas failed".into(), - })); - } - // CAS only mutates on success (which we verified above). - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior, - }); - Ok(resp) - } - - KvOp::GetSet { - collection, - key, - new_value, - surrogate, - rls_filters, - rls_write_check, - } => { - let now_ms = current_ms(); - let prior = self - .kv_engine - .entry_image(did, tid, collection.as_str(), key, now_ms); - let resp = self.execute_kv_getset( - crate::data::executor::handlers::kv::atomic::KvAtomicCtx { - task, - did, - tid, - collection: collection.as_str(), - key, - surrogate: *surrogate, - rls_write_check, - }, - new_value, - rls_filters, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv get-set failed".into(), - })); - } - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior, - }); - Ok(resp) - } - - // Routed here only by `execute_tx_kv_write`'s atomic arm. - other => Err(ErrorCode::Internal { - detail: format!("execute_tx_kv_atomic called with a non-atomic KvOp: {other:?}"), - }), - } - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ops.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ops.rs deleted file mode 100644 index f3b02b8c5..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ops.rs +++ /dev/null @@ -1,168 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! KV operation dispatch for transaction batches. - -use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; -use nodedb_physical::physical_plan::KvOp; - -use super::undo::UndoEntry; - -impl CoreLoop { - /// Execute a KV operation in a transaction context. Write ops (including - /// TTL and sorted-index DDL) capture prior state and push an - /// `UndoEntry`; reads execute without undo tracking. - pub(super) fn execute_tx_kv( - &mut self, - task: &ExecutionTask, - tid: u64, - plan: &PhysicalPlan, - op: &KvOp, - undo_log: &mut Vec, - ) -> Result { - let did = task.request.database_id.as_u64(); - match op { - // ── Read-only KV ops — no undo needed ─────────────────────────── - KvOp::Get { .. } - | KvOp::Scan { .. } - | KvOp::MaterializeScan { .. } - | KvOp::BatchGet { .. } - | KvOp::GetTtl { .. } - | KvOp::FieldGet { .. } - | KvOp::SortedIndexRank { .. } - | KvOp::SortedIndexTopK { .. } - | KvOp::SortedIndexRange { .. } - | KvOp::SortedIndexCount { .. } - | KvOp::SortedIndexScore { .. } => { - let resp = self.execute_kv(task, did, tid, op); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv read failed".into(), - })); - } - Ok(resp) - } - - // ── DDL — reject inside TransactionBatch ── - // `plan_requires_txn_buffering` classifies these unbuffered, so a - // client statement never replays through this arm at commit; it - // guards a hypothetical direct-dispatch route. - KvOp::RegisterIndex { .. } | KvOp::DropIndex { .. } => Err(ErrorCode::Internal { - detail: "KV secondary-index DDL is not permitted inside a TransactionBatch".into(), - }), - - // ── Truncate — replayed live, in statement order ── - // Staged as an overlay marker at statement time; the live truncate - // wipes every row replayed before it in this batch. Like the - // Document truncate passthrough, it pushes no undo entry. - KvOp::Truncate { collection, .. } => { - let resp = self.execute_kv_truncate(task, did, tid, collection.as_str()); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv truncate failed".into(), - })); - } - Ok(resp) - } - - // ── TTL ops — capture prior expiry, execute, push undo ─────────── - KvOp::Expire { - collection, - key, - ttl_ms, - rls_write_check, - } => self.execute_tx_kv_expire( - task, - crate::data::executor::handlers::kv::ttl::KvTtlTarget { - did, - tid, - collection: collection.as_str(), - key, - rls_write_check, - }, - *ttl_ms, - undo_log, - ), - - KvOp::Persist { - collection, - key, - rls_write_check, - } => self.execute_tx_kv_persist( - task, - crate::data::executor::handlers::kv::ttl::KvTtlTarget { - did, - tid, - collection: collection.as_str(), - key, - rls_write_check, - }, - undo_log, - ), - - // ── Sorted-index DDL — capture prior def, execute, push undo ───── - KvOp::RegisterSortedIndex { - collection, - index_name, - sort_columns, - key_column, - window_type, - window_timestamp_column, - window_start_ms, - window_end_ms, - } => self.execute_tx_kv_register_sorted_index( - task, - super::sub_plan_kv_ttl_sorted::TxRegisterSortedIndexParams { - did, - tid, - collection: collection.as_str(), - index_name, - sort_columns, - key_column, - window_type, - window_timestamp_column, - window_start_ms: *window_start_ms, - window_end_ms: *window_end_ms, - }, - undo_log, - ), - - KvOp::DropSortedIndex { index_name } => { - self.execute_tx_kv_drop_sorted_index(task, did, tid, index_name, undo_log) - } - - // ── Write ops — delegated to `sub_plan_kv_writes::execute_tx_kv_write`. - KvOp::Put { .. } - | KvOp::Insert { .. } - | KvOp::InsertIfAbsent { .. } - | KvOp::InsertOnConflictUpdate { .. } - | KvOp::Delete { .. } - | KvOp::BatchPut { .. } - | KvOp::FieldSet { .. } - | KvOp::Incr { .. } - | KvOp::IncrFloat { .. } - | KvOp::Cas { .. } - | KvOp::GetSet { .. } - | KvOp::Transfer { .. } - | KvOp::TransferItem { .. } => self.execute_tx_kv_write(task, did, tid, op, undo_log), - - // ── Resolve-before-propose — reject inside TransactionBatch ── - // Both are autocommit-only: resolution is decided against - // committed state and proposed straight through Raft. - KvOp::ResolveWrite(_) | KvOp::ResolvedWrite { .. } => Err(ErrorCode::Internal { - detail: "KV resolve-before-propose is not permitted inside a TransactionBatch" - .into(), - }), - - // ── Predicate DML — replayed live, in statement order ── - // Staged per matched row at statement time; the live handler - // re-evaluates the predicate against the batch-ordered base at - // COMMIT, the same passthrough Document `BulkUpdate`/`BulkDelete` - // take in `exec_tx_document`. - KvOp::PredicateUpdate { .. } | KvOp::PredicateDelete { .. } => { - self.exec_tx_passthrough(tid, plan, &task.request) - } - } - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ttl_sorted.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ttl_sorted.rs deleted file mode 100644 index 0aba7c443..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_ttl_sorted.rs +++ /dev/null @@ -1,216 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! KV TTL (`Expire`/`Persist`) and sorted-index DDL (`RegisterSortedIndex`/ -//! `DropSortedIndex`) execution for transaction batches. -//! -//! Split out of `sub_plan_kv_ops.rs` (the main `KvOp` dispatcher) once these -//! four arms pushed that file over this crate's per-file line budget. Each -//! handler here captures the prior state needed for `rollback_undo_log` -//! before delegating to the same live-path handler autocommit uses -//! (`execute_kv_expire` / `execute_kv_persist` / `sorted.rs`'s register/drop), -//! so a COMMIT-time replay and a live autocommit statement always produce the -//! identical result. - -use crate::bridge::envelope::{ErrorCode, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::handlers::kv::sorted::KvRegisterSortedIndexParams; -use crate::data::executor::handlers::kv::ttl::KvTtlTarget; -use crate::data::executor::task::ExecutionTask; - -use super::undo::UndoEntry; - -/// Parameters for [`CoreLoop::execute_tx_kv_register_sorted_index`], bundled -/// so the method stays under the `too_many_arguments` clippy threshold. -pub(super) struct TxRegisterSortedIndexParams<'a> { - pub did: u64, - pub tid: u64, - pub collection: &'a str, - pub index_name: &'a str, - pub sort_columns: &'a [(String, String)], - pub key_column: &'a str, - pub window_type: &'a str, - pub window_timestamp_column: &'a str, - pub window_start_ms: u64, - pub window_end_ms: u64, -} - -impl CoreLoop { - // ── TTL: Expire / Persist ──────────────────────────────────────────────── - - /// Execute `EXPIRE` in a transaction context. - /// - /// Captures the key's prior TTL metadata (`has_ttl` + `expire_at_ms`) - /// before the write so a later sibling sub-plan's failure can restore the - /// exact prior instant via `UndoEntry::KvTtl`. The absolute expiry instant - /// itself is resolved by `execute_kv_expire` via `kv_ttl_now_ms`, so a - /// COMMIT-time replay (where `task.resolved_now_ms()` is absent) installs - /// an instant resolved AT COMMIT rather than at original statement time -- - /// the correct semantics for when the write becomes visible, but one that - /// can differ from what a same-transaction `GET_TTL` observed against the - /// staging overlay before COMMIT. - /// - /// `target` is the same bundle the autocommit handler takes and is handed - /// straight through, so the row this addresses and the row the write policy - /// decides cannot drift apart. `undo_log` stays a separate trailing `&mut` - /// parameter -- bundling a mutable borrow alongside borrowed fields of the - /// same struct fights the borrow checker at the call site for no benefit. - pub(super) fn execute_tx_kv_expire( - &mut self, - task: &ExecutionTask, - target: KvTtlTarget<'_>, - ttl_ms: u64, - undo_log: &mut Vec, - ) -> Result { - let KvTtlTarget { - did, - tid, - collection, - key, - .. - } = target; - let prior_meta = self.kv_engine.get_ttl_meta(did, tid, collection, key); - let resp = self.execute_kv_expire(task, target, ttl_ms); - if resp.status == Status::Error { - // Mirrors the live handler: EXPIRE on an absent key returns - // `ErrorCode::NotFound` (see `handlers/kv/ttl.rs`), not a - // synthesized default. - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv expire failed".into(), - })); - } - let prior_expiry = prior_meta.and_then(|m| m.has_ttl.then_some(m.expire_at_ms)); - undo_log.push(UndoEntry::KvTtl { - collection: collection.to_string(), - key: key.to_vec(), - prior_expiry, - }); - Ok(resp) - } - - /// Execute `PERSIST` in a transaction context. See `execute_tx_kv_expire` - /// for the undo-capture shape; `Persist` resolves no instant of its own - /// (`KvEngine::persist` takes no `now_ms`), so there is no COMMIT-vs- - /// statement-time divergence to note here. - pub(super) fn execute_tx_kv_persist( - &mut self, - task: &ExecutionTask, - target: KvTtlTarget<'_>, - undo_log: &mut Vec, - ) -> Result { - let KvTtlTarget { - did, - tid, - collection, - key, - .. - } = target; - let prior_meta = self.kv_engine.get_ttl_meta(did, tid, collection, key); - let resp = self.execute_kv_persist(task, target); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv persist failed".into(), - })); - } - let prior_expiry = prior_meta.and_then(|m| m.has_ttl.then_some(m.expire_at_ms)); - undo_log.push(UndoEntry::KvTtl { - collection: collection.to_string(), - key: key.to_vec(), - prior_expiry, - }); - Ok(resp) - } - - // ── Sorted index DDL: Register / Drop ─────────────────────────────────── - - /// Execute `RegisterSortedIndex` in a transaction context. - /// - /// Captures whether an index already existed under this name (and its - /// definition, if so) before registering, via the shared pure builder - /// `build_sorted_index_def` (through `execute_kv_register_sorted_index`, - /// the same path autocommit uses) -- never a hand-rolled `SortedIndexDef`. - pub(super) fn execute_tx_kv_register_sorted_index( - &mut self, - task: &ExecutionTask, - params: TxRegisterSortedIndexParams<'_>, - undo_log: &mut Vec, - ) -> Result { - let TxRegisterSortedIndexParams { - did, - tid, - collection, - index_name, - sort_columns, - key_column, - window_type, - window_timestamp_column, - window_start_ms, - window_end_ms, - } = params; - let prior_def = self - .kv_engine - .sorted_index_def(did, tid, index_name) - .cloned(); - let resp = self.execute_kv_register_sorted_index( - task, - KvRegisterSortedIndexParams { - did, - tid, - collection, - index_name, - sort_columns, - key_column, - window_type, - window_timestamp_column, - window_start_ms, - window_end_ms, - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv register sorted index failed".into(), - })); - } - undo_log.push(UndoEntry::SortedIndexDdl { - database_id: did, - tenant_id: tid, - index_name: index_name.to_string(), - prior_def, - }); - Ok(resp) - } - - /// Execute `DropSortedIndex` in a transaction context. - /// - /// Captures the dropped index's definition so a later sibling sub-plan's - /// failure can restore it -- `rollback_undo_log` re-registers it, which - /// rebuilds the order-statistic tree by backfilling from the KV - /// collection's CURRENT contents at undo time (see `UndoEntry:: - /// SortedIndexDdl` doc comment for why this is correct regardless of - /// undo-log ordering). - pub(super) fn execute_tx_kv_drop_sorted_index( - &mut self, - task: &ExecutionTask, - did: u64, - tid: u64, - index_name: &str, - undo_log: &mut Vec, - ) -> Result { - let prior_def = self - .kv_engine - .sorted_index_def(did, tid, index_name) - .cloned(); - let resp = self.execute_kv_drop_sorted_index(task, did, tid, index_name); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv drop sorted index failed".into(), - })); - } - undo_log.push(UndoEntry::SortedIndexDdl { - database_id: did, - tenant_id: tid, - index_name: index_name.to_string(), - prior_def, - }); - Ok(resp) - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs deleted file mode 100644 index fb6688420..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_kv_writes.rs +++ /dev/null @@ -1,430 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! KV read-modify-write op execution for transaction batches. Each handler -//! captures the prior value(s) before the write so a later sibling -//! sub-plan's failure can restore them via `rollback_undo_log`. - -use crate::bridge::envelope::{ErrorCode, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; -use crate::engine::kv::current_ms; -use nodedb_physical::physical_plan::KvOp; - -use super::undo::UndoEntry; - -impl CoreLoop { - /// Execute a KV read-modify-write operation in a transaction context. - /// Only called for the `KvOp` write variants that capture-then-execute; - /// every other variant is handled directly in `execute_tx_kv`. - pub(super) fn execute_tx_kv_write( - &mut self, - task: &ExecutionTask, - did: u64, - tid: u64, - op: &KvOp, - undo_log: &mut Vec, - ) -> Result { - match op { - KvOp::Put { - collection, - key, - value, - ttl_ms, - surrogate, - .. - } => { - let now_ms = current_ms(); - let prior = self - .kv_engine - .entry_image(did, tid, collection.as_str(), key, now_ms); - let resp = self.execute_kv_put( - task, - crate::data::executor::handlers::kv::crud::KvWriteParams { - did, - tid, - collection: collection.as_str(), - key, - value, - ttl_ms: *ttl_ms, - surrogate: *surrogate, - returning: None, - rls_filters: &[], - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv put failed".into(), - })); - } - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior, - }); - Ok(resp) - } - - KvOp::Insert { - collection, - key, - value, - ttl_ms, - surrogate, - .. - } => { - let resp = self.execute_kv_insert( - task, - crate::data::executor::handlers::kv::crud::KvWriteParams { - did, - tid, - collection: collection.as_str(), - key, - value, - ttl_ms: *ttl_ms, - surrogate: *surrogate, - returning: None, - rls_filters: &[], - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv insert failed".into(), - })); - } - // Insert only succeeds when key was absent; prior_value is None. - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior: None, - }); - Ok(resp) - } - - KvOp::InsertIfAbsent { - collection, - key, - value, - ttl_ms, - surrogate, - .. - } => { - let now_ms = current_ms(); - let was_absent = self - .kv_engine - .get(did, tid, collection.as_str(), key, now_ms) - .is_none(); - let resp = self.execute_kv_insert_if_absent( - task, - crate::data::executor::handlers::kv::crud::KvWriteParams { - did, - tid, - collection: collection.as_str(), - key, - value, - ttl_ms: *ttl_ms, - surrogate: *surrogate, - returning: None, - rls_filters: &[], - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv insert-if-absent failed".into(), - })); - } - // Only push undo if the key was actually written (was absent). - if was_absent { - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior: None, - }); - } - Ok(resp) - } - - KvOp::InsertOnConflictUpdate { - collection, key, .. - } => { - let now_ms = current_ms(); - let prior = self - .kv_engine - .entry_image(did, tid, collection.as_str(), key, now_ms); - let resp = self.execute_kv(task, did, tid, op); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv insert-on-conflict-update failed".into(), - })); - } - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior, - }); - Ok(resp) - } - - KvOp::Delete { - collection, - keys, - rls_write_check, - .. - } => { - let now_ms = current_ms(); - // Capture prior values for all keys that exist before deleting. - let priors: Vec<(Vec, crate::engine::kv::KvEntryImage)> = keys - .iter() - .filter_map(|k| { - let image = - self.kv_engine - .entry_image(did, tid, collection.as_str(), k, now_ms)?; - Some((k.clone(), image)) - }) - .collect(); - // In-transaction writes never carry `RETURNING`: the Control - // Plane refuses the clause before staging (see `BatchPut`). - let resp = self.execute_kv_delete( - task, - crate::data::executor::handlers::kv::crud::KvDeleteParams { - did, - tid, - collection: collection.as_str(), - keys, - rls_write_check, - returning: None, - rls_filters: &[], - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv delete failed".into(), - })); - } - for (key, prior) in priors { - undo_log.push(UndoEntry::KvDelete { - collection: collection.to_string(), - key, - prior, - }); - } - Ok(resp) - } - - KvOp::BatchPut { - collection, - entries, - ttl_ms, - surrogates, - .. - } => { - let now_ms = current_ms(); - let prior_entries: Vec<(Vec, Option)> = - entries - .iter() - .map(|(k, _v)| { - let prior = self.kv_engine.entry_image( - did, - tid, - collection.as_str(), - k, - now_ms, - ); - (k.clone(), prior) - }) - .collect(); - let resp = self.execute_kv_batch_put( - task, - crate::data::executor::handlers::kv::batch::KvBatchPutArgs { - did, - tid, - collection: collection.as_str(), - entries, - ttl_ms: *ttl_ms, - surrogates, - returning: None, - rls_filters: &[], - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv batch put failed".into(), - })); - } - undo_log.push(UndoEntry::KvBatchPut { - collection: collection.to_string(), - entries: prior_entries, - }); - Ok(resp) - } - - KvOp::FieldSet { - collection, - key, - updates, - surrogate, - if_present, - rls_write_check, - .. - } => { - let now_ms = current_ms(); - let prior = self - .kv_engine - .entry_image(did, tid, collection.as_str(), key, now_ms); - let resp = self.execute_kv_field_set( - crate::data::executor::handlers::kv::atomic::KvAtomicCtx { - task, - did, - tid, - collection: collection.as_str(), - key, - surrogate: *surrogate, - rls_write_check, - }, - crate::data::executor::handlers::kv::field::KvFieldSetArgs { - updates, - if_present: *if_present, - returning: None, - rls_filters: &[], - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv field set failed".into(), - })); - } - undo_log.push(UndoEntry::KvPut { - collection: collection.to_string(), - key: key.clone(), - prior, - }); - Ok(resp) - } - - // The four atomics capture/restore identically, sharing one handler. - KvOp::Incr { .. } | KvOp::IncrFloat { .. } | KvOp::Cas { .. } | KvOp::GetSet { .. } => { - self.execute_tx_kv_atomic(task, did, tid, op, undo_log) - } - - KvOp::Transfer { - collection, - source_key, - dest_key, - .. - } => { - let now_ms = current_ms(); - let source_prior = - self.kv_engine - .entry_image(did, tid, collection.as_str(), source_key, now_ms); - let dest_prior = - self.kv_engine - .entry_image(did, tid, collection.as_str(), dest_key, now_ms); - let resp = self.execute_kv(task, did, tid, op); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv transfer failed".into(), - })); - } - let Some(source_bytes) = source_prior else { - // Transfer requires source to exist; it would have failed above. - return Err(ErrorCode::Internal { - detail: "kv transfer: source prior missing after success".into(), - }); - }; - undo_log.push(UndoEntry::KvTransfer { - collection: collection.to_string(), - source_key: source_key.clone(), - source_prior: source_bytes, - dest_key: dest_key.clone(), - dest_prior, - }); - Ok(resp) - } - - KvOp::TransferItem { - source_collection, - dest_collection, - item_key, - dest_key, - surrogate, - source_rls_write_check, - dest_rls_write_check, - } => { - let now_ms = current_ms(); - let source_prior = self.kv_engine.entry_image( - did, - tid, - source_collection.as_str(), - item_key, - now_ms, - ); - let dest_prior = self.kv_engine.entry_image( - did, - tid, - dest_collection.as_str(), - dest_key, - now_ms, - ); - let resp = self.execute_kv_transfer_item( - task, - crate::data::executor::handlers::kv::transfer::TransferItemParams { - did, - tid, - source_collection: source_collection.as_str(), - dest_collection: dest_collection.as_str(), - item_key, - dest_key, - surrogate: *surrogate, - source_rls_write_check, - dest_rls_write_check, - }, - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "kv transfer-item failed".into(), - })); - } - let Some(source_bytes) = source_prior else { - return Err(ErrorCode::Internal { - detail: "kv transfer-item: source prior missing after success".into(), - }); - }; - undo_log.push(UndoEntry::KvTransferItem { - source_collection: source_collection.to_string(), - dest_collection: dest_collection.to_string(), - item_key: item_key.clone(), - dest_key: dest_key.clone(), - source_prior: source_bytes, - dest_prior, - }); - Ok(resp) - } - - // Every non-write-RMW variant is handled directly by - // `sub_plan_kv_ops::execute_tx_kv` before dispatching here. - KvOp::Get { .. } - | KvOp::Scan { .. } - | KvOp::MaterializeScan { .. } - | KvOp::BatchGet { .. } - | KvOp::GetTtl { .. } - | KvOp::FieldGet { .. } - | KvOp::SortedIndexRank { .. } - | KvOp::SortedIndexTopK { .. } - | KvOp::SortedIndexRange { .. } - | KvOp::SortedIndexCount { .. } - | KvOp::SortedIndexScore { .. } - | KvOp::RegisterIndex { .. } - | KvOp::DropIndex { .. } - | KvOp::Truncate { .. } - | KvOp::Expire { .. } - | KvOp::Persist { .. } - | KvOp::RegisterSortedIndex { .. } - | KvOp::DropSortedIndex { .. } - | KvOp::ResolveWrite(_) - | KvOp::ResolvedWrite { .. } - | KvOp::PredicateUpdate { .. } - | KvOp::PredicateDelete { .. } => Err(ErrorCode::Internal { - detail: "execute_tx_kv_write called with a non-write-RMW KvOp variant".into(), - }), - } - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs b/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs deleted file mode 100644 index 34cffb8f6..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_plan_write.rs +++ /dev/null @@ -1,353 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Write-op execution helpers for transactional sub-plans. -//! -//! Each function here performs one engine's tracked write (recording an -//! `UndoEntry` for rollback) or the shared read-only / DDL passthrough -//! dispatch. Routing from `PhysicalPlan` variants to these helpers lives in -//! `sub_plan.rs`. - -use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; - -use super::sub_request::SubRequestScope; -use super::undo::UndoEntry; - -/// Fields for a transactional primary-vector insert (see `VectorOp::Insert`). -pub(super) struct TxVectorInsertParams<'a> { - pub collection: &'a str, - pub vector: &'a [f32], - pub dim: usize, - pub field_name: &'a str, - pub surrogate: nodedb_types::Surrogate, -} - -/// Fields for a transactional graph edge put (see `GraphOp::EdgePut`). -pub(super) struct TxEdgePutParams<'a> { - pub collection: &'a str, - pub src_id: &'a str, - pub label: &'a str, - pub dst_id: &'a str, - pub properties: &'a [u8], - pub src_surrogate: nodedb_types::Surrogate, - pub dst_surrogate: nodedb_types::Surrogate, -} - -/// Edge identity for a transaction-scoped edge delete. -pub(super) struct TxEdgeDeleteParams<'a> { - pub collection: &'a str, - pub src_id: &'a str, - pub label: &'a str, - pub dst_id: &'a str, - /// Compiled RLS write-policy filters the staged plan carried. Decided - /// against the edge's pre-image inside the batch, so a rejected delete - /// fails the whole transaction instead of applying unchecked. - pub rls_write_check: &'a nodedb_types::RlsWriteCheck, -} - -impl CoreLoop { - /// Insert into a primary-vector collection, recording an undo entry. - pub(super) fn exec_tx_vector_insert( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - params: TxVectorInsertParams<'_>, - undo_log: &mut Vec, - ) -> Result { - let TxVectorInsertParams { - collection, - vector, - dim, - field_name, - surrogate, - } = params; - - let index_key = Self::vector_index_key( - dummy_task.request.database_id.as_u64(), - tid, - collection, - field_name, - ); - let vp = self - .vector_params - .get(&index_key) - .cloned() - .unwrap_or_default(); - let index = self - .vector_collections - .entry(index_key.clone()) - .or_insert_with(|| crate::engine::vector::collection::VectorCollection::new(dim, vp)); - - if vector.len() != index.dim() { - return Err(ErrorCode::Internal { - detail: format!( - "dimension mismatch: expected {}, got {}", - index.dim(), - vector.len() - ), - }); - } - - let vector_id = index.len() as u32; - index.insert_with_surrogate(vector.to_vec(), surrogate); - // Advance the checkpoint watermark with this transaction's WAL LSN so a - // later vector checkpoint records the write as absorbed; the redo replay - // (which carries the same enclosing record LSN) is then gated instead of - // appending a duplicate node. - if let Some(lsn) = dummy_task.wal_lsn() { - index.note_checkpoint_lsn(lsn.as_u64()); - } - // This is the direct primary-vector write path (VectorOp), not - // the document auto-index cascade — it never populates - // `vector_doc_map` (that reverse map is keyed by storage key, - // which this path doesn't have). `None` `doc_id` tells - // `apply_undo_vector` to skip the `vector_doc_map` mutation. - undo_log.push(UndoEntry::InsertVector { - index_key, - vector_id, - collection: collection.to_string(), - field: field_name.to_string(), - doc_id: None, - }); - Ok(self.response_ok(dummy_task)) - } - - /// Delete from a primary-vector collection, recording an undo entry. - pub(super) fn exec_tx_vector_delete( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - collection: &str, - vector_id: u32, - undo_log: &mut Vec, - ) -> Response { - let index_key = - Self::vector_index_key(dummy_task.request.database_id.as_u64(), tid, collection, ""); - if let Some(index) = self.vector_collections.get_mut(&index_key) - && index.delete(vector_id) - { - // Same direct primary-vector path as `VectorOp::Insert` - // above — no `vector_doc_map` entry to restore, so a - // `None` `doc_id` skips that mutation in `apply_undo_vector`. - undo_log.push(UndoEntry::DeleteVector { - index_key, - vector_id, - collection: collection.to_string(), - field: String::new(), - doc_id: None, - }); - } - self.response_ok(dummy_task) - } - - /// Upsert a graph edge, recording an undo entry with the prior properties. - pub(super) fn exec_tx_edge_put( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - params: TxEdgePutParams<'_>, - undo_log: &mut Vec, - ) -> Result { - let TxEdgePutParams { - collection, - src_id, - label, - dst_id, - properties, - src_surrogate, - dst_surrogate, - } = params; - - // The compensation entry is recorded inside `execute_edge_put_with_undo` - // at the only safe point — after the edge-store version is durably - // written and before the fallible CSR mutation. Recording it here, up - // front, would leave a phantom undo entry when dangling-endpoint - // validation or the edge-store write itself rejects, corrupting - // bitemporal history on rollback. - let resp = self.execute_edge_put_with_undo( - dummy_task, - crate::data::executor::handlers::graph::EdgePutParams { - tid, - collection, - src_id, - label, - dst_id, - properties, - src_surrogate, - dst_surrogate, - }, - Some(undo_log), - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "edge put failed".into(), - })); - } - Ok(resp) - } - - /// Delete a graph edge, recording an undo entry with the prior properties. - pub(super) fn exec_tx_edge_delete( - &mut self, - dummy_task: &ExecutionTask, - tid: u64, - params: TxEdgeDeleteParams<'_>, - undo_log: &mut Vec, - ) -> Result { - let TxEdgeDeleteParams { - collection, - src_id, - label, - dst_id, - rls_write_check, - } = params; - - // Compensation is recorded inside `execute_edge_delete_with_undo` only - // after the tombstone is durably written (and only when a live - // pre-image existed), so a rejected/failed delete leaves no phantom - // re-insert entry behind. - let resp = self.execute_edge_delete_with_undo( - dummy_task, - crate::data::executor::handlers::graph::EdgeDeleteParams { - tid, - collection, - src_id, - label, - dst_id, - rls_write_check, - }, - Some(undo_log), - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "edge delete failed".into(), - })); - } - Ok(resp) - } - - /// Apply a columnar predicate `UPDATE`, recording undo for atomic rollback. - pub(super) fn exec_tx_columnar_update( - &mut self, - dummy_task: &ExecutionTask, - collection: &str, - filters: &[u8], - updates: &[(String, Vec)], - rls_write_check: &nodedb_types::RlsWriteCheck, - undo_log: &mut Vec, - ) -> Result { - let resp = self.execute_columnar_update( - dummy_task, - collection, - filters, - updates, - rls_write_check, - Some(undo_log), - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "columnar update failed".into(), - })); - } - Ok(resp) - } - - /// Apply a columnar predicate `DELETE`, recording undo for atomic rollback. - pub(super) fn exec_tx_columnar_delete( - &mut self, - dummy_task: &ExecutionTask, - collection: &str, - filters: &[u8], - rls_write_check: &nodedb_types::RlsWriteCheck, - undo_log: &mut Vec, - ) -> Result { - let resp = self.execute_columnar_delete( - dummy_task, - collection, - filters, - rls_write_check, - Some(undo_log), - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "columnar delete failed".into(), - })); - } - Ok(resp) - } - - /// Apply a columnar `ResolvedUpdate` (Control-Plane-resolved row set), - /// recording undo for atomic rollback. Mirrors `exec_tx_columnar_update` - /// exactly — the underlying handler decides `{"affected": N}` vs a - /// drift-check `OllpRetryRequired` the same way regardless of caller. - pub(super) fn exec_tx_columnar_resolved_update( - &mut self, - dummy_task: &ExecutionTask, - collection: &str, - rows: &[(nodedb_types::Value, Vec)], - rls_write_check: &nodedb_types::RlsWriteCheck, - undo_log: &mut Vec, - ) -> Result { - let resp = self.execute_columnar_resolved_update( - dummy_task, - collection, - rows, - rls_write_check, - Some(undo_log), - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "columnar resolved update failed".into(), - })); - } - Ok(resp) - } - - /// Apply a columnar `ResolvedDelete` (Control-Plane-resolved row set), - /// recording undo for atomic rollback. - pub(super) fn exec_tx_columnar_resolved_delete( - &mut self, - dummy_task: &ExecutionTask, - collection: &str, - pks: &[nodedb_types::Value], - rls_write_check: &nodedb_types::RlsWriteCheck, - undo_log: &mut Vec, - ) -> Result { - let resp = self.execute_columnar_resolved_delete( - dummy_task, - collection, - pks, - rls_write_check, - Some(undo_log), - ); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "columnar resolved delete failed".into(), - })); - } - Ok(resp) - } - - /// Execute a sub-plan with no undo-tracked arm via the standard dispatch - /// path. No undo entry is recorded. - /// - /// The sub-plan runs under [`SubRequestScope::of`] `parent`: the parent's - /// database, vShard, deadline, and admission. A fresh budget per sub-plan - /// lets a transaction outlive its client's `statement_timeout`. - pub(super) fn exec_tx_passthrough( - &mut self, - tid: u64, - plan: &PhysicalPlan, - parent: &crate::bridge::envelope::Request, - ) -> Result { - let request = SubRequestScope::of(parent).request(tid, plan.clone()); - let resp = self.execute(&ExecutionTask::new(request)); - if resp.status == Status::Error { - return Err(resp.error_code.map(|c| *c).unwrap_or(ErrorCode::Internal { - detail: "sub-plan execution failed".into(), - })); - } - Ok(resp) - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/sub_request.rs b/nodedb/src/data/executor/handlers/transaction/sub_request.rs deleted file mode 100644 index 3240f092c..000000000 --- a/nodedb/src/data/executor/handlers/transaction/sub_request.rs +++ /dev/null @@ -1,114 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! The request envelope a transaction sub-plan runs under. - -use crate::bridge::envelope::{Admission, PhysicalPlan, Priority, Request}; -use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; - -/// Request fields a sub-plan inherits from the batch that carries it. -pub(super) struct SubRequestScope { - /// Data Plane handlers key storage and collection metadata by database. - pub database_id: DatabaseId, - /// Handlers decide shard-owner side effects by vShard. - pub vshard_id: VShardId, - pub deadline: std::time::Instant, - pub admission: Admission, -} - -impl SubRequestScope { - /// Copy `parent`'s database, vShard, `deadline`, and `admission`. - /// - /// A sub-request's - /// [`execution_deadline`](Request::execution_deadline) then equals the - /// parent's. A sub-plan is part of the statement that spawned it, so it - /// runs on that statement's remaining budget. An already-ordered parent, - /// such as a Calvin apply, has no execution deadline, and neither does - /// its sub-plan. - pub(super) fn of(parent: &Request) -> Self { - Self { - database_id: parent.database_id, - vshard_id: parent.vshard_id, - // no-determinism: sub-plan deadline is ephemeral, not written to WAL - deadline: parent.deadline, - admission: parent.admission, - } - } - - /// Build the request that runs `plan` for tenant `tid` in this scope. - pub(super) fn request(&self, tid: u64, plan: PhysicalPlan) -> Request { - Request { - request_id: RequestId::new(0), - tenant_id: TenantId::new(tid), - database_id: self.database_id, - vshard_id: self.vshard_id, - plan, - // no-determinism: sub-plan deadline is ephemeral, not written to WAL - deadline: self.deadline, - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: self.admission, - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::bridge::envelope::ExemptReason; - use nodedb_physical::physical_plan::MetaOp; - - fn cancel_plan() -> PhysicalPlan { - PhysicalPlan::Meta(MetaOp::Cancel { - target_request_id: RequestId::new(0), - }) - } - - #[test] - fn sub_request_runs_in_parent_database_and_vshard() { - let parent = SubRequestScope { - database_id: DatabaseId::new(7), - vshard_id: VShardId::new(3), - deadline: std::time::Instant::now() + std::time::Duration::from_secs(5), - admission: Admission::Admitted, - } - .request(9, cancel_plan()); - - let sub = SubRequestScope::of(&parent).request(9, PhysicalPlan::Meta(MetaOp::Compact)); - - assert_eq!(sub.database_id, DatabaseId::new(7)); - assert_eq!(sub.vshard_id, VShardId::new(3)); - assert_eq!(sub.tenant_id, TenantId::new(9)); - assert!(matches!(sub.plan, PhysicalPlan::Meta(MetaOp::Compact))); - } - - #[test] - fn sub_request_shares_parent_execution_deadline() { - for admission in [ - Admission::Admitted, - Admission::Exempt(ExemptReason::Read), - Admission::Exempt(ExemptReason::AlreadyOrdered), - ] { - let parent = SubRequestScope { - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - deadline: std::time::Instant::now() + std::time::Duration::from_secs(5), - admission, - } - .request(1, cancel_plan()); - - let sub = SubRequestScope::of(&parent).request(1, cancel_plan()); - - assert_eq!(sub.admission, parent.admission); - assert_eq!(sub.execution_deadline(), parent.execution_deadline()); - } - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs index a509d51ce..a741f3481 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs @@ -490,7 +490,7 @@ mod tests { }), PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), - payload: b"other_measurement value=2i 2000000000\n".to_vec(), + payload: b"metrics value=2i 2000000000\n".to_vec(), format: "ilp".into(), wal_lsn: None, surrogates: Vec::new(), @@ -501,16 +501,23 @@ mod tests { }), ]; - let response = core.execute_transaction_batch(&task, TID, &plans, &[], None); + let response = core.commit_plans_then_refuse_for_test(&task, TID, &plans, 70); - assert_eq!(response.status, crate::bridge::envelope::Status::Error); + assert!( + matches!( + response.error_code.as_deref(), + Some(crate::bridge::envelope::ErrorCode::RetryableRefusal { .. }) + ), + "the install fails after the transaction's writes: {:?}", + response.error_code + ); assert!( !core.columnar_memtables.contains_key(&( crate::types::DatabaseId::DEFAULT, TenantId::new(TID), "metrics".to_string(), )), - "reverse-order rollback must restore the pre-transaction absence after repeated ingests" + "the rolled-back install must restore the pre-transaction absence after repeated ingests" ); assert!( !core.ts_last_value_caches.contains_key(&( @@ -573,8 +580,8 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "metrics"), payload: b"metrics value=1i 1000000000\n".to_vec(), format: "ilp".into(), - // Buffered transaction plans normally have no per-op LSN. The - // transaction record's LSN above must become the partition stamp. + // Buffered transaction plans carry no per-op LSN. The + // transaction record's LSN must become the partition stamp. wal_lsn: None, surrogates: Vec::new(), provenance: None, @@ -583,8 +590,8 @@ mod tests { rls_filters: Vec::new(), })]; - let response = core.execute_transaction_batch(&task, TID, &plans, &[], None); - assert_eq!(response.status, Status::Ok); + let response = core.commit_plans_for_test(&task, TID, &plans, lsn); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); let key = ( crate::types::DatabaseId::new(DB), TenantId::new(TID), @@ -612,17 +619,9 @@ mod tests { // ── Columnar predicate UPDATE / DELETE undo ───────────────────────────── // // A columnar predicate UPDATE / DELETE is staged at statement time and - // replayed durably at COMMIT through `execute_tx_sub_plan`. Before the undo - // parity fix, that replay hit the undo-less passthrough arm, so a SIBLING - // sub-plan failing later in the same COMMIT batch left the columnar mutation - // applied — a partial, non-atomic commit. These tests drive the real capture - // path (`execute_tx_sub_plan`) then reverse via `rollback_undo_log` — the same - // reverse-order driver `execute_transaction_batch` runs on a sibling failure — - // and assert the columnar state is fully restored. - // - // PRE-FIX the `undo_log.len() == 1` assertion fails (the passthrough pushed no - // undo entry), and the post-rollback state assertion fails (the mutation - // survived the aborted batch). + // installed at COMMIT from the transaction's redo record. A sub-record + // failing later in the same record rolls the columnar mutation back with + // every other write of the record. use nodedb_physical::physical_plan::{ColumnarOp, PhysicalPlan}; @@ -685,7 +684,7 @@ mod tests { seed_columnar_engine(&mut core, &[(1, 10), (2, 20)]); assert_eq!(columnar_rows(&core), vec![(1, 10), (2, 20)]); - // Durable COMMIT replay of `UPDATE m SET v = 999` (empty filter = all rows). + // COMMIT of `UPDATE m SET v = 999` (empty filter = all rows). let updates = vec![( "v".to_string(), nodedb_types::value_to_msgpack(&Value::Integer(999)).unwrap(), @@ -697,24 +696,17 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, }); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("columnar update sub-plan must succeed"); + let task = make_default_task(); + let response = core.commit_plans_then_refuse_for_test(&task, TID, &[plan], 71); - // The mutation applied, and — critically — an undo entry was captured. - assert_eq!(columnar_rows(&core), vec![(1, 999), (2, 999)]); - assert_eq!( - undo_log.len(), - 1, - "columnar UPDATE must push exactly one undo entry (pre-fix: 0, on the undo-less passthrough)" + assert!( + matches!( + response.error_code.as_deref(), + Some(crate::bridge::envelope::ErrorCode::RetryableRefusal { .. }) + ), + "the install fails after the transaction's writes: {:?}", + response.error_code ); - assert!(matches!(undo_log[0], UndoEntry::ColumnarUpdate { .. })); - - // A sibling sub-plan fails later in the same COMMIT: reverse the batch. - core.rollback_undo_log(nodedb_types::DatabaseId::DEFAULT.as_u64(), TID, undo_log) - .expect("rollback must succeed"); - assert_eq!( columnar_rows(&core), vec![(1, 10), (2, 20)], @@ -730,33 +722,24 @@ mod tests { seed_columnar_engine(&mut core, &[(1, 10), (2, 20), (3, 30)]); assert_eq!(columnar_rows(&core), vec![(1, 10), (2, 20), (3, 30)]); - // Durable COMMIT replay of `DELETE FROM m` (empty filter = all rows). + // COMMIT of `DELETE FROM m` (empty filter = all rows). let plan = PhysicalPlan::Columnar(ColumnarOp::Delete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "m"), filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, }); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("columnar delete sub-plan must succeed"); + let task = make_default_task(); + let response = core.commit_plans_then_refuse_for_test(&task, TID, &[plan], 72); assert!( - columnar_rows(&core).is_empty(), - "all rows must be deleted by the durable replay" - ); - assert_eq!( - undo_log.len(), - 1, - "columnar DELETE must push exactly one undo entry (pre-fix: 0, on the undo-less passthrough)" + matches!( + response.error_code.as_deref(), + Some(crate::bridge::envelope::ErrorCode::RetryableRefusal { .. }) + ), + "the install fails after the transaction's writes: {:?}", + response.error_code ); - assert!(matches!(undo_log[0], UndoEntry::ColumnarDelete { .. })); - - // A sibling sub-plan fails later in the same COMMIT: reverse the batch. - core.rollback_undo_log(nodedb_types::DatabaseId::DEFAULT.as_u64(), TID, undo_log) - .expect("rollback must succeed"); - assert_eq!( columnar_rows(&core), vec![(1, 10), (2, 20), (3, 30)], diff --git a/nodedb/src/data/executor/handlers/transaction/undo/balanced.rs b/nodedb/src/data/executor/handlers/transaction/undo/balanced.rs deleted file mode 100644 index c31cc3048..000000000 --- a/nodedb/src/data/executor/handlers/transaction/undo/balanced.rs +++ /dev/null @@ -1,53 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! BALANCED constraint check at a transaction's commit boundary. -//! -//! The entries checked here are the signed contributions every write in the -//! transaction handed to -//! [`settle_balanced_entries`](crate::data::executor::core_loop::CoreLoop::settle_balanced_entries) -//! as it ran — an insert's post-image added, a delete's pre-image subtracted, -//! an update's both. They are NOT re-derived from the undo log, and that is the -//! point: -//! -//! * the undo log records a delete as an entry to be REVERSED, not as an amount -//! the transaction removed, so a transaction that deleted one leg of a -//! balanced journal contributed nothing and passed; -//! * `old_value: None` is not "this was an insert" — a `PointPut` onto an -//! absent row inside a rolled-back savepoint carries the same shape; -//! * re-reading each row from the store to recover its body repeats a read for -//! bytes the write itself already held and decoded. - -use std::collections::HashMap; - -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::enforcement::balanced::{self, BalancedEntry}; - -impl CoreLoop { - /// Check BALANCED constraints across everything this transaction wrote. - /// - /// `entries` is the transaction's accumulated `(collection, entry)` set, - /// grouped here so each collection is judged against its own definition. - pub(in crate::data::executor::handlers::transaction) fn check_balanced_constraints( - &self, - database_id: u64, - tid: u64, - entries: Vec<(String, BalancedEntry)>, - ) -> crate::Result<()> { - let mut by_collection: HashMap> = HashMap::new(); - for (collection, entry) in entries { - by_collection.entry(collection).or_default().push(entry); - } - - for (collection, collection_entries) in &by_collection { - // A collection with entries but no definition can only happen if - // the constraint was dropped mid-transaction; there is then no rule - // left to judge those entries against. - let Some(def) = self.balanced_def(database_id, tid, collection) else { - continue; - }; - balanced::check_balanced(collection, &def, collection_entries)?; - } - - Ok(()) - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/columnar_insert.rs b/nodedb/src/data/executor/handlers/transaction/undo/columnar_insert.rs new file mode 100644 index 000000000..96c2f99ce --- /dev/null +++ b/nodedb/src/data/executor/handlers/transaction/undo/columnar_insert.rs @@ -0,0 +1,58 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Pre-image capture for a columnar insert that runs under an undo log. + +use nodedb_columnar::pk_index::RowLocation; +use nodedb_physical::physical_plan::ColumnarInsertIntent; + +use crate::data::executor::core_loop::CoreLoop; +use crate::types::TenantId; + +/// The PK bytes an insert binds, and the memtable rows it displaces as +/// `(pk_bytes, prior_location)`. +pub(in crate::data::executor) type ColumnarUndoState = (Vec>, Vec<(Vec, RowLocation)>); + +impl CoreLoop { + /// The PK bytes a columnar insert of `rows` (schema-ordered values) will + /// bind, and the memtable rows it will displace, captured before the + /// insert runs. + pub(in crate::data::executor) fn columnar_insert_undo_state( + &self, + collection_key: &(nodedb_types::DatabaseId, TenantId, String), + rows: &[Vec], + intent: ColumnarInsertIntent, + ) -> ColumnarUndoState { + let mut inserted_pks: Vec> = Vec::new(); + let mut displaced: Vec<(Vec, RowLocation)> = Vec::new(); + let Some(engine) = self.columnar_engines.get(collection_key) else { + return (inserted_pks, displaced); + }; + for values in rows { + let Ok(pk_bytes) = engine.encode_pk_from_row(values) else { + continue; + }; + + match intent { + // `InsertUnique` never displaces a prior row. A real conflict + // fails the statement before this undo entry is used. On the + // success path it matches `InsertIfAbsent`: record the new PK + // only when nothing occupies it. + ColumnarInsertIntent::InsertIfAbsent | ColumnarInsertIntent::InsertUnique => { + if !engine.pk_index().contains(&pk_bytes) { + inserted_pks.push(pk_bytes); + } + } + ColumnarInsertIntent::Insert | ColumnarInsertIntent::Put => { + if let Some(prior_loc) = engine.pk_index().get(&pk_bytes).copied() + && prior_loc.segment_id == engine.memtable_segment_id() + { + displaced.push((pk_bytes.clone(), prior_loc)); + } + inserted_pks.push(pk_bytes); + } + } + } + + (inserted_pks, displaced) + } +} diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document.rs b/nodedb/src/data/executor/handlers/transaction/undo/document.rs index 6e7eddf13..bcc16db7e 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document.rs @@ -37,9 +37,6 @@ impl CoreLoop { UndoEntry::PutDocument { collection, document_id, - // Rollback restores prior storage state; it emits no event, so - // the row's client identity has no reader here. - identity: _, old_value, bitemporal_sys_from_ms, bitemporal_index_tuples, @@ -131,9 +128,6 @@ impl CoreLoop { UndoEntry::DeleteDocument { collection, document_id, - // Rollback restores prior storage state; it emits no event, so - // the row's client identity has no reader here. - identity: _, old_value, bitemporal_sys_from_ms, bitemporal_index_tuples, @@ -428,7 +422,6 @@ mod tests { let entry = UndoEntry::PutDocument { collection: "c".into(), document_id: d1, - identity: d1.to_identity(), old_value: None, bitemporal_sys_from_ms: Some(t), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -483,7 +476,6 @@ mod tests { let entry = UndoEntry::DeleteDocument { collection: "c".into(), document_id: d1, - identity: d1.to_identity(), old_value: b"v1".to_vec(), bitemporal_sys_from_ms: Some(2_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -519,7 +511,6 @@ mod tests { let restore = UndoEntry::PutDocument { collection: "c".into(), document_id: storage_key(0), - identity: storage_key(0).to_identity(), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -537,7 +528,6 @@ mod tests { let genesis = UndoEntry::PutDocument { collection: "c".into(), document_id: storage_key(0), - identity: storage_key(0).to_identity(), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -562,7 +552,6 @@ mod tests { let overwrite = UndoEntry::PutDocument { collection: "c".into(), document_id: key1, - identity: key1.to_identity(), old_value: Some(b"old".to_vec()), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -582,7 +571,6 @@ mod tests { let insert = UndoEntry::PutDocument { collection: "c".into(), document_id: key2, - identity: key2.to_identity(), old_value: None, bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), @@ -604,7 +592,6 @@ mod tests { let entry = UndoEntry::DeleteDocument { collection: "c".into(), document_id: key1, - identity: key1.to_identity(), old_value: b"prior".to_vec(), bitemporal_sys_from_ms: None, bitemporal_index_tuples: Vec::new(), diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document_fts.rs b/nodedb/src/data/executor/handlers/transaction/undo/document_fts.rs index a509ab14c..064c13bd1 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document_fts.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document_fts.rs @@ -90,19 +90,17 @@ impl CoreLoop { /// searchable again; a rolled-back PUT must remove the postings it wrote. #[cfg(test)] mod tests { - use std::time::{Duration, Instant}; - - use nodedb_physical::physical_plan::{DocumentOp, StorageMode}; + use nodedb_physical::physical_plan::StorageMode; + use nodedb_types::DatabaseId; use nodedb_types::columnar::{ColumnDef, ColumnType, StrictSchema}; - use nodedb_types::{DatabaseId, Surrogate}; use super::*; - use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; use crate::data::executor::core_loop::tests::make_core_with_dir; - use crate::data::executor::handlers::transaction::sub_plan_doc::{TxPointDelete, TxPointPut}; - use crate::data::executor::task::ExecutionTask; + use crate::data::executor::handlers::transaction::redo_apply::test_commit::{ + doc_delete_sub_record, doc_put_sub_record, + }; use crate::engine::document::store::CollectionConfig; - use crate::types::{ReadConsistency, RequestId, TenantId, TraceId, VShardId}; + use crate::types::TenantId; const DB: u64 = 0; const TID: u64 = 1; @@ -154,62 +152,14 @@ mod tests { .is_empty() } - fn dummy_task() -> ExecutionTask { - ExecutionTask::new(Request { - request_id: RequestId::new(1), - tenant_id: TenantId::new(TID), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLL), - document_id: PK.into(), - surrogate: Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - // no-determinism: test-only deadline is not written to Calvin state. - deadline: Instant::now() + Duration::from_secs(30), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::Read, - ), - }) - } - - /// Commit an insert by driving `tx_point_put` and discarding its undo log - /// (the txn commits internally). + /// Commit an insert through the committed-redo install and discard its + /// undo log. fn commit_put(core: &mut CoreLoop) { - let task = dummy_task(); - let value = doc_bytes(); - let mut throwaway = Vec::new(); - core.tx_point_put( - TxPointPut { - task: &task, - tid: TID, - collection: COLL, - document_id: PK, - surrogate: Surrogate::new(1), - value: &value, - user_roles: &[], - insert_if_absent: None, - resolved_sum_targets: &[], - deferred_sum_targets: &[], - }, - &mut throwaway, - ) - .unwrap(); + core.install_with_undo_for_test( + TID, + 10, + vec![doc_put_sub_record(COLL, PK, &doc_bytes(), 1)], + ); } #[test] @@ -224,21 +174,8 @@ mod tests { "strict body must be searchable after insert" ); - let task = dummy_task(); - let mut undo_log = Vec::new(); - core.tx_point_delete( - TxPointDelete { - task: &task, - tid: TID, - collection: COLL, - document_id: PK, - surrogate: Surrogate::new(1), - user_roles: &[], - resolved_sum_targets: &[], - }, - &mut undo_log, - ) - .unwrap(); + let undo_log = + core.install_with_undo_for_test(TID, 20, vec![doc_delete_sub_record(COLL, PK, 1)]); assert!( !fts_searchable(&core), "delete cascade must remove strict FTS postings" @@ -260,25 +197,11 @@ mod tests { assert!(!fts_searchable(&core)); - let task = dummy_task(); - let value = doc_bytes(); - let mut undo_log = Vec::new(); - core.tx_point_put( - TxPointPut { - task: &task, - tid: TID, - collection: COLL, - document_id: PK, - surrogate: Surrogate::new(1), - value: &value, - user_roles: &[], - insert_if_absent: None, - resolved_sum_targets: &[], - deferred_sum_targets: &[], - }, - &mut undo_log, - ) - .unwrap(); + let undo_log = core.install_with_undo_for_test( + TID, + 20, + vec![doc_put_sub_record(COLL, PK, &doc_bytes(), 1)], + ); assert!(fts_searchable(&core), "strict body searchable mid-tx"); core.rollback_undo_log(DB, TID, undo_log) diff --git a/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs b/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs index 78316ff42..d9e8c403f 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/document_outcome.rs @@ -12,7 +12,7 @@ //! Entries are pushed in the order the writes happened. Rollback runs the log //! in reverse. -use nodedb_types::{RowIdentity, StorageKey}; +use nodedb_types::StorageKey; use crate::data::executor::enforcement::materialized_sum::apply::TargetWrite; use crate::data::executor::handlers::point::apply_delete::PointDeleteOutcome; @@ -27,8 +27,6 @@ pub(in crate::data::executor::handlers) struct DocumentRow<'a> { pub tid: u64, pub collection: &'a str, pub storage_key: StorageKey, - /// The row's client identity, as its event names it. - pub identity: RowIdentity, } /// Push the undo entries that reverse every materialized-sum target row the @@ -44,7 +42,6 @@ pub(in crate::data::executor::handlers) fn push_target_undo( undo_log.push(UndoEntry::PutDocument { collection: target.collection.clone(), document_id: StorageKey::for_surrogate(target.surrogate), - identity: target.identity.clone(), old_value: outcome.prior_value.clone(), bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: outcome.bitemporal_index_tuples.clone(), @@ -73,7 +70,6 @@ pub(in crate::data::executor::handlers) fn push_put_undo( undo_log.push(UndoEntry::PutDocument { collection: row.collection.to_string(), document_id: row.storage_key, - identity: row.identity, old_value: outcome.prior_value, bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: outcome.bitemporal_index_tuples, @@ -100,7 +96,6 @@ pub(in crate::data::executor::handlers) fn push_delete_undo( undo_log.push(UndoEntry::DeleteDocument { collection: row.collection.to_string(), document_id: row.storage_key, - identity: row.identity, old_value, bitemporal_sys_from_ms: outcome.bitemporal_sys_from_ms, bitemporal_index_tuples: outcome.bitemporal_index_tuples, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs index 21f84a5c8..18fecc1cc 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs @@ -84,8 +84,6 @@ pub(in crate::data::executor) enum UndoEntry { /// The redb storage key. `.surrogate()` recovers the numeric surrogate /// FTS index rollback needs. document_id: nodedb_types::StorageKey, - /// The row's client identity, as the deferred event names it. - identity: nodedb_types::RowIdentity, /// `None` if the document didn't exist before (inserted); `Some(bytes)` /// if it was overwritten (updated). old_value: Option>, @@ -117,8 +115,6 @@ pub(in crate::data::executor) enum UndoEntry { /// delete cascade removed this document's postings, and a /// rolled-back delete recomputes and re-inserts them under it. document_id: nodedb_types::StorageKey, - /// The row's client identity, as the deferred event names it. - identity: nodedb_types::RowIdentity, old_value: Vec, /// System-time key of the versioned tombstone row this op appended on a /// bitemporal collection. `None` = plain op → re-insert via the @@ -213,70 +209,21 @@ pub(in crate::data::executor) enum UndoEntry { key: Vec, prior: crate::engine::kv::KvEntryImage, }, - /// Undo a KV BatchPut by reinstating the prior state of every key. + /// Undo a KV `EXPIRE` / `PERSIST` by putting back the key's prior expiry. + /// The value is untouched: a TTL change writes only the expiry. /// - /// Each element is `(key, prior)` where `prior == None` means the key was - /// newly inserted. - KvBatchPut { - collection: String, - entries: Vec<(Vec, Option)>, - }, - /// Undo a KV Transfer (fungible) by reinstating the source and destination - /// prior state. - KvTransfer { + /// `prior_expire_at_ms` is the absolute instant the key expired at, or + /// [`NO_EXPIRY`](crate::engine::kv::entry::NO_EXPIRY) when it had none. + KvTtl { collection: String, - source_key: Vec, - source_prior: crate::engine::kv::KvEntryImage, - dest_key: Vec, - dest_prior: Option, - }, - /// Undo a KV TransferItem by reinstating the source and destination prior - /// state. - KvTransferItem { - source_collection: String, - dest_collection: String, - item_key: Vec, - dest_key: Vec, - source_prior: crate::engine::kv::KvEntryImage, - dest_prior: Option, + key: Vec, + prior_expire_at_ms: u64, }, /// Undo a KV `TRUNCATE` by reinstalling every row the collection held. KvTruncate { collection: String, rows: Vec, }, - /// Undo a KV `Expire`/`Persist` by restoring the key's prior TTL state. - /// - /// `prior_expiry == None` means the key had no TTL (persistent) before - /// the forward op; undo calls `KvEngine::persist`. `prior_expiry == - /// Some(expire_at_ms)` means the key had a TTL expiring at that exact - /// absolute instant; undo calls `KvEngine::expire_with_absolute_expiry` - /// with it verbatim (not a freshly-derived `now_ms + ttl_ms`, which - /// would drift from the original instant by the elapsed time). - KvTtl { - collection: String, - key: Vec, - prior_expiry: Option, - }, - /// Undo a KV `RegisterSortedIndex`/`DropSortedIndex` by restoring the - /// index name's prior definition state. - /// - /// `prior_def == None` means no index existed under this name before - /// the forward op (a fresh `RegisterSortedIndex`); undo drops it. - /// `prior_def == Some(def)` means an index existed under this name - /// before the forward op (either overwritten by `RegisterSortedIndex`, - /// or removed by `DropSortedIndex`); undo re-registers `def`, which - /// rebuilds the order-statistic tree by backfilling from the KV - /// collection's CURRENT contents at undo time -- correct regardless of - /// undo-log ordering relative to sibling KV-write undos, since - /// `SortedIndexManager::register` always derives the tree fresh from - /// live table state rather than from a point-in-time snapshot. - SortedIndexDdl { - database_id: u64, - tenant_id: u64, - index_name: String, - prior_def: Option, - }, /// Undo a `mark_node_deleted` by removing the node from the in-memory /// deleted-nodes set (edge referential-integrity tracker). /// diff --git a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs index c43efe33e..b2738e8d5 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/graph_node.rs @@ -134,17 +134,12 @@ pub(super) struct NodeLabelsUndo { #[cfg(test)] mod tests { - use std::time::{Duration, Instant}; - use super::*; - use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::data::executor::handlers::point::apply_put::PointPutParams; - use crate::data::executor::handlers::transaction::sub_plan_doc::TxPointDelete; - use crate::data::executor::task::ExecutionTask; + use crate::data::executor::handlers::transaction::redo_apply::test_commit::doc_delete_sub_record; use crate::engine::document::store::CollectionConfig; - use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; - use nodedb_physical::physical_plan::DocumentOp; + use crate::types::TenantId; use nodedb_types::Surrogate; const DB: u64 = 0; @@ -201,41 +196,6 @@ mod tests { txn.commit().unwrap(); } - /// A throwaway `ExecutionTask` (DEFAULT database id, inert `PointGet` plan) — - /// the only fields the tx doc helpers read are `database_id` and `request_id`. - fn dummy_task() -> ExecutionTask { - ExecutionTask::new(Request { - request_id: RequestId::new(1), - tenant_id: TenantId::new(TID), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLL), - document_id: PK.into(), - surrogate: Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - deadline: Instant::now() + Duration::from_secs(30), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::Read, - ), - }) - } - #[test] fn mark_node_returns_true_only_on_first_insert() { let dir = tempfile::tempdir().unwrap(); @@ -265,21 +225,8 @@ mod tests { assert!(core.mark_node_deleted(DB, TID, PK)); assert!(core.is_node_deleted(DB, TID, PK)); - let task = dummy_task(); - let mut undo_log = Vec::new(); - core.tx_point_delete( - TxPointDelete { - task: &task, - tid: TID, - collection: COLL, - document_id: PK, - surrogate: Surrogate::new(1), - user_roles: &[], - resolved_sum_targets: &[], - }, - &mut undo_log, - ) - .unwrap(); + let undo_log = + core.install_with_undo_for_test(TID, 20, vec![doc_delete_sub_record(COLL, PK, 1)]); // The delete's mark was a no-op (already marked) → no MarkNodeDeleted undo // was captured, so rollback must leave the tombstone intact. assert!( diff --git a/nodedb/src/data/executor/handlers/transaction/undo/kv.rs b/nodedb/src/data/executor/handlers/transaction/undo/kv.rs index e2031622f..f8a65b399 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/kv.rs @@ -1,12 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 //! KV undo entry application logic. -//! -//! Split out of `apply.rs` (which grouped every engine family in one file) -//! once the `KvTtl` / `SortedIndexDdl` arms pushed the KV family over this -//! crate's per-file line budget. - -use tracing::error; use crate::data::executor::core_loop::CoreLoop; use crate::engine::kv::current_ms; @@ -60,65 +54,24 @@ impl CoreLoop { ); Ok(()) } - UndoEntry::KvBatchPut { + UndoEntry::KvTtl { collection, - entries, + key, + prior_expire_at_ms, } => { - let now_ms = current_ms(); - for (key, prior) in entries { - self.kv_engine.reinstate_entry( - kv_key(did, tid, &collection, &key), - prior.as_ref(), - now_ms, + if prior_expire_at_ms == crate::engine::kv::entry::NO_EXPIRY { + self.kv_engine.persist(did, tid, &collection, &key); + } else { + self.kv_engine.expire_with_absolute_expiry( + did, + tid, + &collection, + &key, + prior_expire_at_ms, ); } Ok(()) } - UndoEntry::KvTransfer { - collection, - source_key, - source_prior, - dest_key, - dest_prior, - } => { - let now_ms = current_ms(); - self.kv_engine.restore_entry_image( - kv_key(did, tid, &collection, &source_key), - &source_prior, - now_ms, - ); - self.kv_engine.reinstate_entry( - kv_key(did, tid, &collection, &dest_key), - dest_prior.as_ref(), - now_ms, - ); - Ok(()) - } - UndoEntry::KvTransferItem { - source_collection, - dest_collection, - item_key, - dest_key, - source_prior, - dest_prior, - } => { - // Cross-collection move: the forward op deleted `item_key` from - // `source_collection` and wrote `dest_key` in `dest_collection`. - // Both halves are reinstated: the source row always existed, - // the destination key may have been absent. - let now_ms = current_ms(); - self.kv_engine.restore_entry_image( - kv_key(did, tid, &source_collection, &item_key), - &source_prior, - now_ms, - ); - self.kv_engine.reinstate_entry( - kv_key(did, tid, &dest_collection, &dest_key), - dest_prior.as_ref(), - now_ms, - ); - Ok(()) - } UndoEntry::KvTruncate { collection, rows } => { // Every write after the truncate was reversed first, so the // collection holds what the truncate left. Empty it and @@ -138,90 +91,6 @@ impl CoreLoop { } Ok(()) } - UndoEntry::KvTtl { - collection, - key, - prior_expiry, - } => { - // The forward `Expire`/`Persist` only succeeds when the key - // exists (mirrors the live handler's `NotFound` on an absent - // key), so this undo entry is only ever pushed after that - // precondition held. If a sibling undo already applied and - // this key is now genuinely missing, that is a broken - // invariant, not a soft "nothing to do" case. - let restored = match prior_expiry { - Some(expire_at_ms) => self.kv_engine.expire_with_absolute_expiry( - did, - tid, - &collection, - &key, - expire_at_ms, - ), - None => self.kv_engine.persist(did, tid, &collection, &key), - }; - if restored { - Ok(()) - } else { - let detail = format!( - "kv ttl undo: key missing in {collection} during rollback of Expire/Persist" - ); - error!( - core = self.core_id, - entry_index, - error = %detail, - "transaction undo: kv ttl restore failed; shard state unknown" - ); - Err((entry_index, detail)) - } - } - UndoEntry::SortedIndexDdl { - database_id, - tenant_id, - index_name, - prior_def, - } => { - match prior_def { - // An index existed under this name before the forward op - // (an overwritten `RegisterSortedIndex`, or the index a - // `DropSortedIndex` removed) -- restore it. `register` - // rebuilds the order-statistic tree by backfilling from - // the KV collection's CURRENT contents, which is correct - // regardless of where this undo entry falls relative to - // sibling KV-write undos in the log (see `UndoEntry` - // doc comment). - Some(def) => { - let collection = def.collection.clone(); - self.kv_engine.register_sorted_index( - database_id, - tenant_id, - &collection, - def, - ); - Ok(()) - } - // No index existed under this name before the forward op - // (a fresh `RegisterSortedIndex`) -- undo removes it. - None => { - if self - .kv_engine - .drop_sorted_index(database_id, tenant_id, &index_name) - { - Ok(()) - } else { - let detail = format!( - "sorted index undo: '{index_name}' missing during rollback of RegisterSortedIndex" - ); - error!( - core = self.core_id, - entry_index, - error = %detail, - "transaction undo: sorted index drop failed; shard state unknown" - ); - Err((entry_index, detail)) - } - } - } - } _ => Err(( entry_index, "apply_undo_kv called with non-kv entry".to_string(), @@ -230,42 +99,11 @@ impl CoreLoop { } } -/// Unit tests for `Expire`/`Persist`/`RegisterSortedIndex`/`DropSortedIndex` -/// at the COMMIT-replay level (`execute_tx_sub_plan` -> `execute_tx_kv`). -/// -/// `plan_requires_txn_buffering` -/// (`control/server/shared/write_admission/predicate/txn_buffering.rs`) -/// classifies all four `true` (buffered), so a client statement issued -/// inside `BEGIN ... COMMIT` replays through this exact call at COMMIT. -/// Pre-fix, `execute_tx_kv`'s reject arm returned -/// `ErrorCode::Internal { detail: "KV DDL / TTL operations are not -/// permitted inside a TransactionBatch" }` for all four, so every -/// `.expect(...)` below on the sub-plan's result is the assertion that used -/// to fail with that error. -/// -/// Two of the four have no SQL surface that reaches this path today, which -/// is why these tests drive `execute_tx_sub_plan` directly rather than a -/// `sql_transactions_*.rs` pgwire integration test: -/// -/// - `Expire`/`Persist` have NO pgwire SQL surface at all (no `EXPIRE(...)` -/// / `PERSIST(...)` SQL function is wired in -/// `control/server/shared/ddl/neutral/router/string_engine_ops.rs`). The -/// RESP protocol's `EXPIRE`/`PERSIST` commands exist but RESP has no -/// `MULTI`/`EXEC` transaction support, so they can never reach a `BEGIN` -/// block. Only the native binary protocol threads a `txn_id` through -/// (`control/server/native/dispatch/direct_ops.rs`) for these ops. -/// - `RegisterSortedIndex`/`DropSortedIndex` DO have pgwire SQL syntax -/// (`CREATE SORTED INDEX` / `DROP SORTED INDEX`, -/// `control/server/shared/ddl/neutral/kv_sorted_index.rs`), but that -/// handler dispatches immediately via `dispatch_to_data_plane` without -/// ever consulting the connection's transaction state -- a separate, -/// pre-existing gap where SQL sorted-index DDL is never staged at all, so -/// it cannot replay at COMMIT via that surface regardless of this fix. #[cfg(test)] mod tests { use super::*; - use crate::bridge::envelope::PhysicalPlan; - use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; use crate::engine::kv::current_ms; use nodedb_physical::physical_plan::KvOp; use nodedb_types::{DatabaseId, QualifiedCollection}; @@ -291,327 +129,160 @@ mod tests { .get_ttl_ms(DB, TID, collection, key, current_ms()) } - // ── Expire ─────────────────────────────────────────────────────────────── - - #[test] - fn kv_expire_in_tx_commit_replay_sets_ttl() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - - put_kv(&mut core, "cache", b"k", b"v", 0); - assert_eq!( - ttl_ms(&core, "cache", b"k"), - Some(-1), - "key must start persistent (no TTL)" - ); + fn expire_plan(ttl_ms: u64) -> PhysicalPlan { + PhysicalPlan::Kv(KvOp::Expire { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), + key: b"k".to_vec(), + ttl_ms, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + }) + } - let plan = PhysicalPlan::Kv(KvOp::Expire { + fn persist_plan() -> PhysicalPlan { + PhysicalPlan::Kv(KvOp::Persist { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), key: b"k".to_vec(), - ttl_ms: 5_000, rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - }); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("EXPIRE sub-plan must succeed at COMMIT replay"); + }) + } - let remaining = ttl_ms(&core, "cache", b"k").expect("key must still exist after EXPIRE"); - assert!( - remaining > 0 && remaining <= 5_000, - "TTL must be set by the COMMIT replay, got {remaining}" - ); - assert_eq!(undo_log.len(), 1, "EXPIRE must push exactly one undo entry"); + fn assert_refused_at_install(response: &Response) { + assert_eq!(response.status, Status::Error); assert!( matches!( - undo_log[0], - UndoEntry::KvTtl { - prior_expiry: None, - .. - } + response.error_code.as_deref(), + Some(ErrorCode::RetryableRefusal { .. }) ), - "prior state (no TTL) must be captured for rollback" + "the install fails after the transaction's writes: {:?}", + response.error_code ); } + // ── Expire / Persist ───────────────────────────────────────────────────── + #[test] - fn kv_expire_in_tx_rollback_reverts_ttl() { + fn a_committed_expire_sets_the_ttl() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - put_kv(&mut core, "cache", b"k", b"v", 0); - - let plan = PhysicalPlan::Kv(KvOp::Expire { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), - key: b"k".to_vec(), - ttl_ms: 5_000, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - }); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("EXPIRE sub-plan must succeed"); - assert!(ttl_ms(&core, "cache", b"k").unwrap() > 0); - - // A sibling sub-plan fails later in the same COMMIT: reverse the batch. - core.rollback_undo_log(DB, TID, undo_log) - .expect("rollback must succeed"); - assert_eq!( ttl_ms(&core, "cache", b"k"), Some(-1), - "rollback must revert the key to its pre-EXPIRE persistent state" + "key starts persistent" ); - } - // ── Persist ────────────────────────────────────────────────────────────── + let response = + core.commit_plans_for_test(&make_default_task(), TID, &[expire_plan(5_000)], 30); + + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + let remaining = ttl_ms(&core, "cache", b"k").expect("key exists after EXPIRE"); + assert!( + remaining > 0 && remaining <= 5_000, + "the committed EXPIRE sets the TTL, got {remaining}" + ); + } #[test] - fn kv_persist_in_tx_commit_replay_clears_ttl() { + fn a_refused_install_keeps_the_ttl_an_expire_changed() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + put_kv(&mut core, "cache", b"k", b"v", 0); - put_kv(&mut core, "cache", b"k", b"v", 60_000); - assert!( - ttl_ms(&core, "cache", b"k").unwrap() > 0, - "key must start with a TTL" + let response = core.commit_plans_then_refuse_for_test( + &make_default_task(), + TID, + &[expire_plan(5_000)], + 31, ); - let plan = PhysicalPlan::Kv(KvOp::Persist { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), - key: b"k".to_vec(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - }); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("PERSIST sub-plan must succeed at COMMIT replay"); - + assert_refused_at_install(&response); assert_eq!( ttl_ms(&core, "cache", b"k"), Some(-1), - "TTL must be cleared by the COMMIT replay" - ); - assert_eq!(undo_log.len(), 1); - assert!( - matches!( - undo_log[0], - UndoEntry::KvTtl { - prior_expiry: Some(_), - .. - } - ), - "prior TTL instant must be captured for rollback" + "the key stays persistent" ); } #[test] - fn kv_persist_in_tx_rollback_restores_ttl() { + fn an_expire_keeps_a_value_written_between_resolve_and_install() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + put_kv(&mut core, "cache", b"k", b"before", 0); - put_kv(&mut core, "cache", b"k", b"v", 60_000); - let before = ttl_ms(&core, "cache", b"k").unwrap(); - - let plan = PhysicalPlan::Kv(KvOp::Persist { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), - key: b"k".to_vec(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - }); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("PERSIST sub-plan must succeed"); - assert_eq!(ttl_ms(&core, "cache", b"k"), Some(-1)); - - core.rollback_undo_log(DB, TID, undo_log) - .expect("rollback must succeed"); - - let after = ttl_ms(&core, "cache", b"k").expect("key must still exist"); - assert!( - after > 0 && after <= before, - "rollback must restore a TTL close to the pre-PERSIST value \ - (before={before}, after={after})" + let response = core.commit_plans_around_for_test( + &make_default_task(), + TID, + &[expire_plan(5_000)], + 34, + |core| put_kv(core, "cache", b"k", b"after", 0), ); - } - - // ── RegisterSortedIndex ────────────────────────────────────────────────── - - fn seed_players(core: &mut CoreLoop) { - for (key, score) in [("p1", 10i64), ("p2", 30), ("p3", 20)] { - let value = nodedb_types::json_to_msgpack(&serde_json::json!({ - "player_id": key, - "score": score, - })) - .unwrap(); - put_kv(core, "players", key.as_bytes(), &value, 0); - } - } - - fn register_plan() -> PhysicalPlan { - PhysicalPlan::Kv(KvOp::RegisterSortedIndex { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "players"), - index_name: "lb".to_string(), - sort_columns: vec![("score".to_string(), "DESC".to_string())], - key_column: "player_id".to_string(), - window_type: "none".to_string(), - window_timestamp_column: String::new(), - window_start_ms: 0, - window_end_ms: 0, - }) - } - - #[test] - fn kv_register_sorted_index_in_tx_commit_replay_is_queryable() { - let dir = tempfile::tempdir().unwrap(); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - seed_players(&mut core); - - let plan = register_plan(); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("RegisterSortedIndex sub-plan must succeed at COMMIT replay"); - - assert_eq!(undo_log.len(), 1); - assert!(matches!( - undo_log[0], - UndoEntry::SortedIndexDdl { - prior_def: None, - .. - } - )); - let top = core - .kv_engine - .sorted_index_top_k(DB, TID, "lb", 3, current_ms()) - .expect("index must be queryable immediately after COMMIT replay"); - let ranked_keys: Vec> = top.into_iter().map(|(_, pk)| pk).collect(); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); assert_eq!( - ranked_keys, - vec![b"p2".to_vec(), b"p3".to_vec(), b"p1".to_vec()], - "DESC top-3 must rank by score: p2(30) > p3(20) > p1(10)" + core.kv_engine.get(DB, TID, "cache", b"k", current_ms()), + Some(b"after".to_vec()), + "the install changes only the expiry of the value the key holds" + ); + let remaining = ttl_ms(&core, "cache", b"k").expect("key exists after EXPIRE"); + assert!( + remaining > 0 && remaining <= 5_000, + "the committed EXPIRE sets the TTL, got {remaining}" ); } #[test] - fn kv_register_sorted_index_in_tx_rollback_removes_index() { + fn a_persist_keeps_a_value_written_between_resolve_and_install() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - seed_players(&mut core); + put_kv(&mut core, "cache", b"k", b"before", 60_000); - let plan = register_plan(); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("RegisterSortedIndex sub-plan must succeed"); - assert!( - core.kv_engine - .sorted_index_top_k(DB, TID, "lb", 3, current_ms()) - .is_some() + let response = core.commit_plans_around_for_test( + &make_default_task(), + TID, + &[persist_plan()], + 35, + |core| put_kv(core, "cache", b"k", b"after", 60_000), ); - core.rollback_undo_log(DB, TID, undo_log) - .expect("rollback must succeed"); - - assert!( - core.kv_engine - .sorted_index_top_k(DB, TID, "lb", 3, current_ms()) - .is_none(), - "rollback must remove the index a fresh RegisterSortedIndex created" + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + assert_eq!( + core.kv_engine.get(DB, TID, "cache", b"k", current_ms()), + Some(b"after".to_vec()) ); - } - - // ── DropSortedIndex ────────────────────────────────────────────────────── - - /// Register `lb` live (outside a transaction), exactly as - /// `execute_kv_register_sorted_index` would -- the def this seeds is what - /// the `DropSortedIndex` undo entry must capture and restore. - fn seed_live_index(core: &mut CoreLoop) { - seed_players(core); - let def = crate::data::executor::handlers::kv::sorted_index_compute::build_sorted_index_def( - crate::data::executor::handlers::kv::sorted_index_compute::BuildSortedIndexDefParams { - collection: "players", - index_name: "lb", - sort_columns: &[("score".to_string(), "DESC".to_string())], - key_column: "player_id", - window_type: "", - window_timestamp_column: "", - window_start_ms: 0, - window_end_ms: 0, - }, - ) - .expect("build sorted index def"); - core.kv_engine - .register_sorted_index(DB, TID, "players", def); + assert_eq!(ttl_ms(&core, "cache", b"k"), Some(-1)); } #[test] - fn kv_drop_sorted_index_in_tx_commit_replay_removes_it() { + fn a_committed_persist_clears_the_ttl() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - seed_live_index(&mut core); - assert!( - core.kv_engine - .sorted_index_top_k(DB, TID, "lb", 3, current_ms()) - .is_some() - ); + put_kv(&mut core, "cache", b"k", b"v", 60_000); - let plan = PhysicalPlan::Kv(KvOp::DropSortedIndex { - index_name: "lb".to_string(), - }); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("DropSortedIndex sub-plan must succeed at COMMIT replay"); + let response = core.commit_plans_for_test(&make_default_task(), TID, &[persist_plan()], 32); - assert!( - core.kv_engine - .sorted_index_top_k(DB, TID, "lb", 3, current_ms()) - .is_none(), - "index must be gone after COMMIT replay" - ); - assert_eq!(undo_log.len(), 1); - assert!(matches!( - undo_log[0], - UndoEntry::SortedIndexDdl { - prior_def: Some(_), - .. - } - )); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + assert_eq!(ttl_ms(&core, "cache", b"k"), Some(-1)); } #[test] - fn kv_drop_sorted_index_in_tx_rollback_restores_it() { + fn a_refused_install_keeps_the_ttl_a_persist_cleared() { let dir = tempfile::tempdir().unwrap(); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - seed_live_index(&mut core); + let before = seed_with_expiry_and_surrogate(&mut core); - let plan = PhysicalPlan::Kv(KvOp::DropSortedIndex { - index_name: "lb".to_string(), - }); - let mut undo_log = Vec::new(); - let mut crdt_deltas = Vec::new(); - core.execute_tx_sub_plan(TID, &plan, &mut undo_log, &mut crdt_deltas, &[]) - .expect("DropSortedIndex sub-plan must succeed"); - assert!( - core.kv_engine - .sorted_index_top_k(DB, TID, "lb", 3, current_ms()) - .is_none() + let response = core.commit_plans_then_refuse_for_test( + &make_default_task(), + TID, + &[persist_plan()], + 33, ); - core.rollback_undo_log(DB, TID, undo_log) - .expect("rollback must succeed"); - - let top = core - .kv_engine - .sorted_index_top_k(DB, TID, "lb", 3, current_ms()) - .expect("rollback must restore the dropped index, rebuilt from live KV data"); - let ranked_keys: Vec> = top.into_iter().map(|(_, pk)| pk).collect(); + assert_refused_at_install(&response); assert_eq!( - ranked_keys, - vec![b"p2".to_vec(), b"p3".to_vec(), b"p1".to_vec()], - "restored index must rank identically to the original" + core.kv_engine + .entry_image(DB, TID, "cache", b"k", current_ms()), + Some(before), + "the key keeps its value, expiry instant and surrogate" ); } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/mod.rs b/nodedb/src/data/executor/handlers/transaction/undo/mod.rs index 387c2d86a..d0cb26b99 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/mod.rs @@ -3,7 +3,7 @@ //! Undo log types and rollback logic for transaction batches. pub(super) mod apply; -pub(super) mod balanced; +pub(super) mod columnar_insert; pub(in crate::data::executor) mod crdt_collection; pub(super) mod document; pub(super) mod document_fts; diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index 2c2ffcdce..17cffe40d 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -57,12 +57,8 @@ impl CoreLoop { UndoEntry::EdgeWrite(undo) => self.apply_undo_edge_write(entry_index, *undo), UndoEntry::KvPut { .. } | UndoEntry::KvDelete { .. } - | UndoEntry::KvBatchPut { .. } - | UndoEntry::KvTransfer { .. } - | UndoEntry::KvTransferItem { .. } - | UndoEntry::KvTruncate { .. } | UndoEntry::KvTtl { .. } - | UndoEntry::SortedIndexDdl { .. } => self.apply_undo_kv(did, tid, entry_index, entry), + | UndoEntry::KvTruncate { .. } => self.apply_undo_kv(did, tid, entry_index, entry), UndoEntry::ColumnarInsert { .. } | UndoEntry::ColumnarUpdate { .. } | UndoEntry::ColumnarDelete { .. } => self.apply_undo_columnar(entry_index, entry), @@ -148,21 +144,18 @@ impl CoreLoop { #[cfg(test)] mod tests { - use std::time::{Duration, Instant}; - use super::*; - use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; use crate::data::executor::core_loop::tests::make_core_with_dir; use crate::data::executor::handlers::point::apply_delete::PointDeleteParams; use crate::data::executor::handlers::point::apply_put::PointPutParams; - use crate::data::executor::handlers::transaction::sub_plan_doc::{TxPointDelete, TxPointPut}; - use crate::data::executor::task::ExecutionTask; + use crate::data::executor::handlers::transaction::redo_apply::test_commit::{ + doc_delete_sub_record, doc_put_sub_record, + }; use crate::engine::document::store::CollectionConfig; use crate::engine::graph::csr::Direction; use crate::engine::graph::edge_store::EdgeRef; use crate::engine::sparse::btree_versioned::{VersionedIndexEntry, VersionedPut}; - use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; - use nodedb_physical::physical_plan::DocumentOp; + use crate::types::TenantId; use nodedb_types::Surrogate; const DB: u64 = 0; @@ -209,8 +202,8 @@ mod tests { /// Scenario 4 (unit level): a rolled-back transaction that does a /// bitemporal PUT followed by a bitemporal DELETE (tombstone) must, via - /// `rollback_undo_log` — the same reverse-order driver `execute_transaction_batch` - /// uses on abort — restore `core.sparse.versioned_get_current` to its + /// `rollback_undo_log` — the same reverse-order driver a failed redo + /// install runs — restore `core.sparse.versioned_get_current` to its /// pre-transaction state (nothing) with the version rows and index entries /// physically gone, not merely hidden. #[test] @@ -257,7 +250,6 @@ mod tests { UndoEntry::PutDocument { collection: "c".into(), document_id: d1, - identity: d1.to_identity(), old_value: None, bitemporal_sys_from_ms: Some(1_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -268,7 +260,6 @@ mod tests { UndoEntry::DeleteDocument { collection: "c".into(), document_id: d1, - identity: d1.to_identity(), old_value: b"v1".to_vec(), bitemporal_sys_from_ms: Some(2_000), bitemporal_index_tuples: vec![("status".into(), "active".into())], @@ -277,8 +268,8 @@ mod tests { }, ]; - // Abort: roll back in reverse order, exactly as `execute_transaction_batch` - // does when a sub-plan fails. + // Abort: roll back in reverse order, exactly as a redo install does + // when a sub-record fails. core.rollback_undo_log(DB, TID, undo_log) .expect("rollback must succeed"); @@ -499,42 +490,6 @@ mod tests { txn.commit().unwrap(); } - /// A throwaway `ExecutionTask` (DEFAULT database id, inert `PointGet` plan) — - /// the only fields the tx doc helpers read are `database_id` and `request_id`. - fn dummy_task() -> ExecutionTask { - ExecutionTask::new(Request { - request_id: RequestId::new(1), - tenant_id: TenantId::new(TID), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - plan: PhysicalPlan::Document(DocumentOp::PointGet { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLL), - document_id: PK.into(), - surrogate: Surrogate::ZERO, - pk_bytes: Vec::new(), - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }), - // no-determinism: test-only deadline is not written to Calvin state. - deadline: Instant::now() + Duration::from_secs(30), - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::Read, - ), - }) - } - fn seed_edge(core: &mut Core) { let tenant = nodedb_types::TenantId::new(TID); let ord = core.hlc.next_ordinal(); @@ -571,25 +526,7 @@ mod tests { let dir_b = tempfile::tempdir().unwrap(); let (mut b, _tb, _rb) = make_core_with_dir(dir_b.path()); register(&mut b); - let task = dummy_task(); - let mut undo_log = Vec::new(); - let value = doc_bytes(); - b.tx_point_put( - TxPointPut { - task: &task, - tid: TID, - collection: COLL, - document_id: PK, - surrogate: Surrogate::new(1), - value: &value, - user_roles: &[], - insert_if_absent: None, - resolved_sum_targets: &[], - deferred_sum_targets: &[], - }, - &mut undo_log, - ) - .unwrap(); + b.install_with_undo_for_test(TID, 20, vec![doc_put_sub_record(COLL, PK, &doc_bytes(), 1)]); // Identical index state across the autocommit and committed-tx paths. assert_eq!(secondary_index_docs(&a), secondary_index_docs(&b)); @@ -617,25 +554,11 @@ mod tests { assert!(!vector_searchable(&core)); assert!(!fts_searchable(&core)); - let task = dummy_task(); - let mut undo_log = Vec::new(); - let value = doc_bytes(); - core.tx_point_put( - TxPointPut { - task: &task, - tid: TID, - collection: COLL, - document_id: PK, - surrogate: Surrogate::new(1), - value: &value, - user_roles: &[], - insert_if_absent: None, - resolved_sum_targets: &[], - deferred_sum_targets: &[], - }, - &mut undo_log, - ) - .unwrap(); + let undo_log = core.install_with_undo_for_test( + TID, + 20, + vec![doc_put_sub_record(COLL, PK, &doc_bytes(), 1)], + ); // Mid-tx: side-effects landed. assert_eq!(secondary_index_docs(&core), vec![row_key()]); assert!(spatial_entry_present(&core)); @@ -685,21 +608,7 @@ mod tests { register(&mut b); autocommit_put(&mut b); seed_edge(&mut b); - let task = dummy_task(); - let mut undo_log = Vec::new(); - b.tx_point_delete( - TxPointDelete { - task: &task, - tid: TID, - collection: COLL, - document_id: PK, - surrogate: Surrogate::new(1), - user_roles: &[], - resolved_sum_targets: &[], - }, - &mut undo_log, - ) - .unwrap(); + b.install_with_undo_for_test(TID, 20, vec![doc_delete_sub_record(COLL, PK, 1)]); // Both paths wiped every index identically. assert_eq!(secondary_index_docs(&a), secondary_index_docs(&b)); @@ -738,21 +647,8 @@ mod tests { assert!(edge_present(&mut core)); assert!(!core.is_node_deleted(DB, TID, PK)); - let task = dummy_task(); - let mut undo_log = Vec::new(); - core.tx_point_delete( - TxPointDelete { - task: &task, - tid: TID, - collection: COLL, - document_id: PK, - surrogate: Surrogate::new(1), - user_roles: &[], - resolved_sum_targets: &[], - }, - &mut undo_log, - ) - .unwrap(); + let undo_log = + core.install_with_undo_for_test(TID, 20, vec![doc_delete_sub_record(COLL, PK, 1)]); // Mid-tx: the delete cascaded. assert!(!spatial_entry_present(&core)); assert!(!vector_searchable(&core)); diff --git a/nodedb/src/data/executor/handlers/transaction/write_version_kv.rs b/nodedb/src/data/executor/handlers/transaction/write_version_kv.rs index 8e97e6a54..5e8712292 100644 --- a/nodedb/src/data/executor/handlers/transaction/write_version_kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/write_version_kv.rs @@ -105,7 +105,8 @@ impl CoreLoop { | KvOp::SortedIndexTopK { .. } | KvOp::SortedIndexRange { .. } | KvOp::SortedIndexCount { .. } - | KvOp::SortedIndexScore { .. } => {} + | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } => {} // Index DDL: no row key written. KvOp::RegisterIndex { .. } | KvOp::DropIndex { .. } diff --git a/nodedb/src/data/executor/handlers/truncate.rs b/nodedb/src/data/executor/handlers/truncate.rs index 6ca60567f..dd061dc31 100644 --- a/nodedb/src/data/executor/handlers/truncate.rs +++ b/nodedb/src/data/executor/handlers/truncate.rs @@ -229,9 +229,10 @@ impl CoreLoop { collection: None, }); } - // On an error neither edge store changed: the edges stay in - // both, and the dangling-edge sweep retries them. - if let Err(e) = self.cascade_node_edges(database_id, tid, &doc_id) { + // The graph keys a row's node by its client key. On an error + // neither edge store changed: the edges stay in both, and the + // dangling-edge sweep retries them. + if let Err(e) = self.cascade_node_edges(database_id, tid, row_identity.as_str()) { warn!(core = self.core_id, %doc_id, error = %e, "truncate: edge cascade failed"); } self.doc_cache.invalidate( diff --git a/nodedb/src/data/executor/handlers/write_batch.rs b/nodedb/src/data/executor/handlers/write_batch.rs index 4be2e97df..8f86de4d9 100644 --- a/nodedb/src/data/executor/handlers/write_batch.rs +++ b/nodedb/src/data/executor/handlers/write_batch.rs @@ -42,7 +42,9 @@ impl CoreLoop { .task_queue .front() .is_some_and(|t| is_batchable_put(t) && !t.is_expired()); - if !front_is_put { + // While a staged Calvin transaction owns rows, every write passes the + // fence in `poll_one` one at a time. + if !front_is_put || !self.calvin.commit_pending.is_empty() { return 0; } diff --git a/nodedb/src/data/executor/sync_gate.rs b/nodedb/src/data/executor/sync_gate.rs index 18e81f3cf..d5e8d926b 100644 --- a/nodedb/src/data/executor/sync_gate.rs +++ b/nodedb/src/data/executor/sync_gate.rs @@ -187,20 +187,33 @@ impl CoreLoop { self.sync_outcome_response(task, SyncAckResult::acked(status, applied_seq)) } - /// Build the gate reply for a frame the validator refused **permanently**. + /// Refuse a frame **permanently**, before anything of it installed. /// - /// The high-water-mark still advances: the same bytes will fail identically - /// on a re-push, so holding the stream for them buys nothing. A refusal the - /// sender *should* retry is not this — it reports an + /// The high-water-mark advances: the same bytes will fail identically on a + /// re-push, so holding the stream for them buys nothing. The refusal is an + /// error response, so the Control Plane cancels the frame's record and + /// journals the mark in a `SyncSeqAdvance` record of its own. Restart + /// replay then restores the mark and never applies the frame. + /// + /// A refusal the sender *should* retry is not this: it reports an /// [`AckStatus::Gap`] through [`Self::sync_ack_response`] and holds the /// mark, which is what keeps the re-push admissible. pub(in crate::data::executor) fn sync_reject_response( - &self, + &mut self, task: &ExecutionTask, violation: nodedb_types::sync::violation::ViolationType, - applied_seq: u64, + prov: &SyncProvenance, ) -> Response { - self.sync_outcome_response(task, SyncAckResult::rejected(violation, applied_seq)) + self.sync_commit(prov); + let applied_seq = self.sync_hwm_value(prov.producer_id, prov.stream_id); + self.response_error( + task, + ErrorCode::SyncRejected { + violation, + applied_seq, + provenance: prov.clone(), + }, + ) } fn sync_outcome_response(&self, task: &ExecutionTask, gate_result: SyncAckResult) -> Response { diff --git a/nodedb/src/data/executor/wal_replay/crdt.rs b/nodedb/src/data/executor/wal_replay/crdt.rs index 7f8e82225..aac6ffa68 100644 --- a/nodedb/src/data/executor/wal_replay/crdt.rs +++ b/nodedb/src/data/executor/wal_replay/crdt.rs @@ -151,7 +151,7 @@ impl CoreLoop { &payload.bytes, nodedb_types::Surrogate::new(surrogate), document_id, - 0, + payload.peer_id, crate::engine::crdt::tenant_state::DeltaSigningAdmission { auth: nodedb_crdt::CrdtAuthContext { user_id: signing.auth_user_id, @@ -169,7 +169,7 @@ impl CoreLoop { &payload.bytes, nodedb_types::Surrogate::new(surrogate), document_id, - 0, + payload.peer_id, ), }, Err(e) => { @@ -271,7 +271,7 @@ impl CoreLoop { &payload.bytes, nodedb_types::Surrogate::ZERO, "", - 0, + payload.peer_id, ) { crate::engine::crdt::tenant_state::ValidatedApplyOutcome::Clean { .. @@ -761,7 +761,8 @@ mod crdt_replay_tests { None, Some(row_id.to_owned()), Some(0), - ); + ) + .with_peer_id(peer); nodedb_wal::WalRecord::new(nodedb_wal::WalRecordArgs { record_type: RecordType::CrdtDelta as u32, lsn, @@ -818,6 +819,7 @@ mod crdt_replay_tests { .collect(); assert_eq!(entries.len(), 1); assert_eq!(entries[0].source_lsn, Some(20)); + assert_eq!(entries[0].peer_id, 3, "the entry names the producing peer"); assert_eq!( h.core .sparse diff --git a/nodedb/src/data/executor/wal_replay/crdt_ordered.rs b/nodedb/src/data/executor/wal_replay/crdt_ordered.rs index e4c93d317..051f6ba82 100644 --- a/nodedb/src/data/executor/wal_replay/crdt_ordered.rs +++ b/nodedb/src/data/executor/wal_replay/crdt_ordered.rs @@ -44,7 +44,24 @@ impl CoreLoop { /// In the install pass it records the collection's Loro pre-image and /// returns whether the write proceeds. A record naming no collection is /// left unclaimed, so the validate pass refuses the record. + /// + /// A raw delta is refused. A rejected delta's dead-letter entry is keyed + /// by the LSN of the record that carried it, and every sub-record of a + /// committed record shares that record's LSN, so two rejected deltas in + /// one record would claim one key. fn redo_crdt_prelude(&mut self, record: &nodedb_wal::WalRecord) -> bool { + if RecordType::from_raw(record.logical_record_type()) == Some(RecordType::CrdtDelta) { + if let Some(scope) = self.redo_apply.scope.as_mut() { + scope.record_error(crate::Error::Internal { + detail: format!( + "committed redo record at lsn {} carries a raw CRDT delta; a \ + transaction journals CRDT row intents only", + record.header.lsn + ), + }); + } + return false; + } let Some(collection) = crdt_record_collection(record) else { return false; }; diff --git a/nodedb/src/data/executor/wal_replay/kv.rs b/nodedb/src/data/executor/wal_replay/kv.rs index c9f91ec0d..ce995135a 100644 --- a/nodedb/src/data/executor/wal_replay/kv.rs +++ b/nodedb/src/data/executor/wal_replay/kv.rs @@ -82,9 +82,36 @@ impl CoreLoop { puts += applied; continue; } + // kv_expire / kv_persist change only the expiry of the value + // the key holds when the record applies — see + // `wal_replay_kv_expiry.rs`. A committed redo record carries + // them for a transaction's TTL-only writes. + if let Some(applied) = self.try_replay_kv_expire( + &record.payload, + tenant_id, + database_id, + now_ms, + record_lsn, + tombstones, + ) { + puts += applied; + continue; + } + if let Some(applied) = self.try_replay_kv_persist( + &record.payload, + tenant_id, + database_id, + now_ms, + record_lsn, + tombstones, + ) { + puts += applied; + continue; + } // A committed redo record carries only absolute `kv_put` - // post-images. Every other shape is left unclaimed, so the - // validate pass refuses the record before any arm writes. + // post-images and TTL changes. Every other shape is left + // unclaimed, so the validate pass refuses the record before + // any arm writes. if self.applying_committed_redo() { continue; } @@ -190,30 +217,6 @@ impl CoreLoop { continue; } - // kv_expire — see `wal_replay_kv_expiry.rs`. - if let Some(applied) = self.try_replay_kv_expire( - &record.payload, - tenant_id, - database_id, - record_lsn, - tombstones, - ) { - puts += applied; - continue; - } - - // kv_persist — see `wal_replay_kv_expiry.rs`. - if let Some(applied) = self.try_replay_kv_persist( - &record.payload, - tenant_id, - database_id, - record_lsn, - tombstones, - ) { - puts += applied; - continue; - } - // kv_incr (delta): re-runs the same integer increment // against current state. if let Some(applied) = self.try_replay_kv_incr( diff --git a/nodedb/src/data/executor/wal_replay_columnar_dml.rs b/nodedb/src/data/executor/wal_replay_columnar_dml.rs index 05319b563..548584098 100644 --- a/nodedb/src/data/executor/wal_replay_columnar_dml.rs +++ b/nodedb/src/data/executor/wal_replay_columnar_dml.rs @@ -140,9 +140,10 @@ impl CoreLoop { Some(Lsn::new(record_lsn)), ); - // Re-execute via the same live handlers the autocommit dispatch used — - // `undo_log: None` mirrors the autocommit path (no transaction batch - // to roll back). + // Re-execute via the same live handlers the autocommit dispatch used. + // Inside a committed-redo install the handlers capture the pre-image + // of every row they change, so a later sub-record's failure rolls the + // mutation back with the rest of the record. // // Replay carries no predicate. The policy decided these rows when the // record was written, and the identity that wrote it is not present at @@ -150,6 +151,8 @@ impl CoreLoop { // replay: its record is cancelled by a `WriteAborted` marker before // the refusal is acknowledged. let replay_check = nodedb_types::RlsWriteCheck::already_decided_elsewhere(); + let mut undo = Vec::new(); + let recording = self.recording_redo_undo(); let response = if record.is_update { self.execute_columnar_update( &task, @@ -157,7 +160,7 @@ impl CoreLoop { &record.filters, &record.updates, &replay_check, - None, + recording.then_some(&mut undo), ) } else { self.execute_columnar_delete( @@ -165,9 +168,10 @@ impl CoreLoop { &record.collection, &record.filters, &replay_check, - None, + recording.then_some(&mut undo), ) }; + self.record_redo_undo(undo); if response.status != Status::Ok { self.replay_record_rejected( diff --git a/nodedb/src/data/executor/wal_replay_kv_expiry.rs b/nodedb/src/data/executor/wal_replay_kv_expiry.rs index 38d8f200a..2dd6fd604 100644 --- a/nodedb/src/data/executor/wal_replay_kv_expiry.rs +++ b/nodedb/src/data/executor/wal_replay_kv_expiry.rs @@ -1,14 +1,15 @@ // SPDX-License-Identifier: BUSL-1.1 -//! WAL replay for the KV `Expire` / `Persist` TTL-mutation records. +//! WAL replay and committed-redo install for the KV `Expire` / `Persist` +//! TTL-mutation records. //! -//! Both records were durably WAL-appended (`wal_append_kv_op`'s `KvOp::Expire` -//! / `KvOp::Persist` arms) but had no decode arm in `replay_kv_wal` before this -//! module existed, so both were silently lost on every crash-restart. WAL replay -//! is the recovery path for these writes above the KV checkpoint's replay floor; -//! at or below that floor the checkpoint already carries each row's resolved -//! absolute `expire_at_ms`, so the gate in `skip_kv_replay_record` is what stops -//! a TTL mutation from being applied twice. +//! `wal_append_kv_op` appends both records for autocommit writes. A +//! transaction's TTL-only writes reach its committed redo record in the same +//! shapes. Each record changes only the expiry of the value the key holds +//! when it applies, so a write committed between stage and install keeps its +//! value. At or below the KV checkpoint's replay floor the checkpoint already +//! carries each row's absolute `expire_at_ms`. The gate in +//! `skip_kv_replay_record` stops a TTL mutation from applying twice. //! //! `kv_expire` always carries the Control-Plane-resolved absolute //! `expire_at_ms` (see `encode_kv_expire`'s doc comment for why `EXPIRE` has @@ -30,6 +31,7 @@ use tracing::warn; use super::core_loop::CoreLoop; use crate::data::executor::core_loop::write_index::KeyRepr; +use crate::data::executor::handlers::transaction::undo::UndoEntry; impl CoreLoop { /// Decode + tombstone-gate + replay one `kv_expire` WAL record. @@ -50,6 +52,7 @@ impl CoreLoop { payload: &[u8], tenant_id: u64, database_id: u64, + now_ms: u64, record_lsn: u64, tombstones: &nodedb_wal::TombstoneSet, ) -> Option { @@ -62,6 +65,10 @@ impl CoreLoop { if self.skip_kv_replay_record(tombstones, tenant_id, &collection, record_lsn) { return Some(0); } + if self.claim_for_validation() { + return Some(0); + } + self.record_kv_ttl_undo(database_id, tenant_id, &collection, &key, now_ms); let applied = self.kv_engine.expire_with_absolute_expiry( database_id, @@ -102,6 +109,7 @@ impl CoreLoop { payload: &[u8], tenant_id: u64, database_id: u64, + now_ms: u64, record_lsn: u64, tombstones: &nodedb_wal::TombstoneSet, ) -> Option { @@ -114,6 +122,10 @@ impl CoreLoop { if self.skip_kv_replay_record(tombstones, tenant_id, &collection, record_lsn) { return Some(0); } + if self.claim_for_validation() { + return Some(0); + } + self.record_kv_ttl_undo(database_id, tenant_id, &collection, &key, now_ms); let applied = self .kv_engine @@ -138,6 +150,36 @@ impl CoreLoop { } } +impl CoreLoop { + /// In the install pass of a committed-redo apply, record the undo that + /// puts the key's current expiry back. The value is not touched, so the + /// undo restores only the expiry. A key that is absent now records + /// nothing: the TTL change will find no key either. + fn record_kv_ttl_undo( + &mut self, + database_id: u64, + tenant_id: u64, + collection: &str, + key: &[u8], + now_ms: u64, + ) { + if !self.recording_redo_undo() { + return; + } + let Some(image) = + self.kv_engine + .entry_image(database_id, tenant_id, collection, key, now_ms) + else { + return; + }; + self.record_redo_undo([UndoEntry::KvTtl { + collection: collection.to_string(), + key: key.to_vec(), + prior_expire_at_ms: image.expire_at_ms, + }]); + } +} + #[cfg(test)] mod tests { use std::sync::Arc; diff --git a/nodedb/src/data/executor/wal_replay_redo_document.rs b/nodedb/src/data/executor/wal_replay_redo_document.rs index c2afb5323..0a0c1ac80 100644 --- a/nodedb/src/data/executor/wal_replay_redo_document.rs +++ b/nodedb/src/data/executor/wal_replay_redo_document.rs @@ -47,7 +47,8 @@ //! redb-synchronous-durable: by the time this replay runs, the target balance //! the original write produced is already on disk, and the derived target write //! carries its own redo record naming the target collection. Folding again on -//! replay would add the same amount a second time. +//! replay would add the same amount a second time. A Calvin record is the +//! exception: see "Calvin records in restart replay" below. //! //! ### Why Raft replication does the opposite //! @@ -76,6 +77,14 @@ //! their targets and link their hash chain, exactly as replication does (see //! `handlers::transaction::redo_apply`). With no scope open this arm is plain //! restart replay. +//! +//! ### Calvin records in restart replay +//! +//! A Calvin redo record's stamp carries the sum targets its slice folds, and +//! no later record carries the target rows. Restart replay runs its document +//! rows through the committed path, which folds them at the record's LSN. +//! The fold subtracts the row's prior image, so a row that already holds its +//! post-image folds nothing. use nodedb_types::Surrogate; use nodedb_types::sync::wire::SyncProvenance; @@ -187,7 +196,8 @@ impl CoreLoop { self.observe_bitemporal_stamp(s.sys_from_ms); self.active_bitemporal_stamps.insert(surrogate_u32, s); } - let applied = if self.redo_apply.scope.is_some() { + let folds = self.redo_folds_at(record_lsn, &collection); + let applied = if folds { self.apply_committed_document_put( CommittedDocWrite { database_id, @@ -223,9 +233,10 @@ impl CoreLoop { ); } } else { - // Replay keys the row by its surrogate; the record's text - // `document_id` is the client key, read only by a committed - // redo apply for the row's event identity. + // Replay keys the row by its surrogate. The record's text + // `document_id` is the client key. The graph cascade keys + // nodes by it, and a committed redo apply names the row's + // event by it. // A `bitemporal=true` collection's delete carries its // resolve-time system time as a fifth element; the plain form // has four. The stamp forces the versioned tombstone at that @@ -263,7 +274,8 @@ impl CoreLoop { }, ); } - let removed = if self.redo_apply.scope.is_some() { + let folds = self.redo_folds_at(record_lsn, &collection); + let removed = if folds { self.apply_committed_document_delete(CommittedDocWrite { database_id, tenant_id, @@ -273,7 +285,13 @@ impl CoreLoop { record_lsn, }) } else { - self.apply_document_delete(database_id, tenant_id, &collection, surrogate_u32) + self.apply_document_delete( + database_id, + tenant_id, + &collection, + &document_id, + surrogate_u32, + ) }; if sys_from_ms.is_some() { self.active_bitemporal_stamps.remove(&surrogate_u32); @@ -301,6 +319,20 @@ impl CoreLoop { } } + /// Whether a document write at `record_lsn` to `collection` runs the + /// committed path, which folds materialized sums: always under a + /// committed-redo apply, and in restart replay when the record's Calvin + /// stamp names sum targets for `collection`. Folding reads the row's + /// prior image, so a replay over a row that already holds the post-image + /// folds nothing. + fn redo_folds_at(&self, record_lsn: u64, collection: &str) -> bool { + self.redo_apply.scope.is_some() + || self + .redo_apply + .replay_folds_for(record_lsn, collection) + .is_some() + } + /// Apply one document PUT through the shared `apply_point_put` core write /// path in its own redb write transaction. `enforce = false`: replayed /// writes were admission-checked when first committed, so re-running @@ -383,11 +415,10 @@ impl CoreLoop { database_id: u64, tenant_id: u64, collection: &str, + document_id: &str, surrogate_u32: u32, ) -> bool { let surrogate = Surrogate::new(surrogate_u32); - let storage_key = StorageKey::for_surrogate(surrogate); - let row_key = storage_key.to_string(); let txn = match self.sparse.begin_write() { Ok(t) => t, Err(e) => { @@ -406,7 +437,8 @@ impl CoreLoop { database_id, tid: tenant_id, collection, - document_id: row_key.as_str(), + // The graph cascade keys nodes by the client key. + document_id, surrogate, user_roles: &[], enforce: false, diff --git a/nodedb/src/engine/kv/engine_atomic.rs b/nodedb/src/engine/kv/engine_atomic.rs index e76dcbcd3..f067ed6ef 100644 --- a/nodedb/src/engine/kv/engine_atomic.rs +++ b/nodedb/src/engine/kv/engine_atomic.rs @@ -6,10 +6,10 @@ //! hash slot). No cross-core coordination is needed because each key maps //! to exactly one core. +use nodedb_physical::kv_atomic::{AtomicComputeError, compute}; use nodedb_physical::physical_plan::KvCounterShape; use super::engine::KvEngine; -use super::engine_atomic_compute as compute; use super::engine_helpers::{expiry_key, table_key}; use super::entry::NO_EXPIRY; use super::hash_table::KvHashTable; @@ -79,6 +79,16 @@ pub enum AtomicError { Rejected(Box), } +impl From for AtomicError { + fn from(error: AtomicComputeError) -> Self { + match error { + AtomicComputeError::TypeMismatch { detail } => Self::TypeMismatch { detail }, + AtomicComputeError::Counter(fault) => Self::Counter(fault), + AtomicComputeError::Encode { detail } => Self::Encode { detail }, + } + } +} + /// A gate consulted with the computed post-image before an atomic commits. /// /// Every atomic computes the value it stores from the stored one: INCR runs diff --git a/nodedb/src/engine/kv/engine_atomic_compute.rs b/nodedb/src/engine/kv/engine_atomic_compute.rs deleted file mode 100644 index 49ab2c406..000000000 --- a/nodedb/src/engine/kv/engine_atomic_compute.rs +++ /dev/null @@ -1,543 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Pure value computation for `INCR`/`INCR_FLOAT`/`CAS`/`GETSET`, shared by -//! the autocommit `KvEngine` methods (`engine_atomic.rs`), the in-transaction -//! staging handlers (`stage_kv_atomic.rs`), the resolve handlers, and WAL -//! replay. Every path computes a stored value with the same function, so all -//! of them store the same bytes. -//! -//! A body has one of two shapes ([`kv_body_shape`]). A typed row (a msgpack -//! map) keeps its typed column semantics. A raw body (the single-`value` SQL -//! form, RESP `SET`) is a byte string. `INCR` and `INCR_FLOAT` read it as -//! decimal text by the Redis rules and store the result as decimal text. - -use std::collections::HashMap; - -use nodedb_query::msgpack_scan::{KvBodyShape, kv_body_shape, row_to_kv_body}; -use nodedb_types::Value; - -use nodedb_physical::physical_plan::KvCounterShape; - -use super::engine_atomic::AtomicError; -use super::float_text; -use crate::bridge::envelope::CounterFault; - -/// The field of a typed row an atomic never targets. -const KEY_FIELD: &str = "key"; - -/// Decode a map-shaped body into its typed columns. Returns `Ok(None)` for a -/// raw body, and `TypeMismatch` for a map-shaped body that does not decode. -fn typed_row(bytes: &[u8]) -> Result>, AtomicError> { - if kv_body_shape(bytes) != KvBodyShape::Map { - return Ok(None); - } - match nodedb_types::value_from_msgpack(bytes) { - Ok(Value::Object(map)) => Ok(Some(map)), - Ok(other) => Err(AtomicError::TypeMismatch { - detail: format!("stored row is {}, not an object", other.type_name()), - }), - Err(e) => Err(AtomicError::TypeMismatch { - detail: format!("stored row does not decode: {e}"), - }), - } -} - -/// Encode typed columns back into a map-shaped body. The fields are written -/// in key order, so every replica and every WAL replay stores the same -/// bytes. -fn encode_map(map: HashMap) -> Result, AtomicError> { - row_to_kv_body(&Value::Object(map), KvBodyShape::Map).map_err(|e| AtomicError::Encode { - detail: format!("typed row re-encode: {e}"), - }) -} - -/// The column an atomic reads and writes in a typed row: the first column -/// in key order that `pick` accepts, never the `key` column. -/// -/// Key order is the order the row is stored in. A `HashMap` iterates in a -/// per-process random order, so choosing by iteration order lets two -/// replicas move two different columns. -fn target_field( - map: &HashMap, - pick: impl Fn(&Value) -> Option, -) -> Option<(String, T)> { - let mut chosen: Option<(&String, T)> = None; - for (name, value) in map { - if name == KEY_FIELD { - continue; - } - if chosen.as_ref().is_some_and(|(best, _)| *best <= name) { - continue; - } - if let Some(picked) = pick(value) { - chosen = Some((name, picked)); - } - } - chosen.map(|(name, picked)| (name.clone(), picked)) -} - -/// The i64 an `INCR` reads from a typed column. -fn column_i64(value: &Value) -> Option { - match value { - Value::Integer(i) => Some(*i), - Value::Float(f) => integral_f64_to_i64(*f), - _ => None, - } -} - -/// The f64 an `INCR_FLOAT` reads from a typed column. -fn column_f64(value: &Value) -> Option { - match value { - Value::Float(f) => Some(*f), - Value::Integer(i) => Some(*i as f64), - _ => None, - } -} - -/// The string a `CAS` or `GETSET` addresses in a typed column. -fn column_string(value: &Value) -> Option { - match value { - Value::String(s) => Some(s.clone()), - _ => None, - } -} - -/// A whole `f64` inside the i64 range, as an i64. -fn integral_f64_to_i64(v: f64) -> Option { - (v.fract() == 0.0 && v >= i64::MIN as f64 && v <= i64::MAX as f64).then_some(v as i64) -} - -fn not_an_integer_column() -> AtomicError { - AtomicError::TypeMismatch { - detail: "row has no integer column".into(), - } -} - -fn not_a_numeric_column() -> AtomicError { - AtomicError::TypeMismatch { - detail: "row has no numeric column".into(), - } -} - -/// Read a raw body as a decimal i64 by the Redis rule. -fn parse_raw_i64(bytes: &[u8]) -> Result { - std::str::from_utf8(bytes) - .ok() - .filter(|text| is_canonical_integer(text)) - .and_then(|text| text.parse::().ok()) - .ok_or(AtomicError::Counter(CounterFault::NotAnInteger)) -} - -/// The Redis integer grammar: `0`, or an optional `-` then digits with no -/// leading zero. A `+` sign, whitespace, and an empty body are refused. -fn is_canonical_integer(text: &str) -> bool { - let digits = text.strip_prefix('-').unwrap_or(text); - text == "0" - || (digits - .bytes() - .next() - .is_some_and(|b| (b'1'..=b'9').contains(&b)) - && digits.bytes().all(|b| b.is_ascii_digit())) -} - -/// The raw body for an integer: its decimal text, the text -/// `scalar_to_raw_bytes` writes for the same value. -fn raw_decimal(v: i64) -> Vec { - v.to_string().into_bytes() -} - -/// The row an absent key becomes under a typed [`KvCounterShape`]: the -/// template with `column` set to `value`. -fn fresh_typed_row( - column: &Option, - template: &[u8], - value: Value, - missing_column: AtomicError, -) -> Result, AtomicError> { - let column = column.as_ref().ok_or(missing_column)?; - let mut map = typed_row(template)?.ok_or(AtomicError::TypeMismatch { - detail: "fresh row template is not a typed row".into(), - })?; - map.insert(column.clone(), value); - encode_map(map) -} - -/// Compute the new value for `INCR`, given the current body (if any). -/// Returns `(new_i64, new_bytes)`. -/// -/// A typed row keeps its shape: the integer column [`target_field`] picks -/// moves, and every other column stays. A raw body is decimal text in and -/// decimal text out. An absent key starts at 0 and takes `shape`. -pub fn incr( - current: Option<&[u8]>, - delta: i64, - shape: &KvCounterShape, -) -> Result<(i64, Vec), AtomicError> { - let overflow = AtomicError::Counter(CounterFault::IntegerOverflow); - let Some(bytes) = current else { - let written = match shape { - KvCounterShape::Raw => raw_decimal(delta), - KvCounterShape::Typed { column, template } => fresh_typed_row( - column, - template, - Value::Integer(delta), - not_an_integer_column(), - )?, - }; - return Ok((delta, written)); - }; - if let Some(mut map) = typed_row(bytes)? { - let (field, old_i64) = target_field(&map, column_i64).ok_or(not_an_integer_column())?; - let new_i64 = old_i64.checked_add(delta).ok_or(overflow)?; - map.insert(field, Value::Integer(new_i64)); - return Ok((new_i64, encode_map(map)?)); - } - let new_i64 = parse_raw_i64(bytes)?.checked_add(delta).ok_or(overflow)?; - Ok((new_i64, raw_decimal(new_i64))) -} - -/// Compute the new value for `INCR_FLOAT`. `delta` is the client's decimal -/// text. Returns `(new_f64, new_bytes)`. -/// -/// A typed row keeps its shape, as in [`incr`], and its column adds in -/// `f64`. A raw body is decimal text in and decimal text out, added exactly -/// by the Redis rules (see `float_text`). An absent key starts at 0 and takes -/// `shape`. -pub fn incr_float( - current: Option<&[u8]>, - delta: &str, - shape: &KvCounterShape, -) -> Result<(f64, Vec), AtomicError> { - let Some(bytes) = current else { - return match shape { - KvCounterShape::Raw => float_text::fresh(delta), - KvCounterShape::Typed { column, template } => { - let value = float_text::delta_to_f64(delta)?; - let written = fresh_typed_row( - column, - template, - Value::Float(value), - not_a_numeric_column(), - )?; - Ok((value, written)) - } - }; - }; - let Some(mut map) = typed_row(bytes)? else { - return float_text::add(bytes, delta); - }; - let delta = float_text::delta_to_f64(delta)?; - let (field, old_f64) = target_field(&map, column_f64).ok_or(not_a_numeric_column())?; - let new_f64 = old_f64 + delta; - if !new_f64.is_finite() { - return Err(AtomicError::Counter(CounterFault::NonFinite)); - } - map.insert(field, Value::Float(new_f64)); - Ok((new_f64, encode_map(map)?)) -} - -/// Write `new_value` into the string column of the typed row `row` and -/// encode it. `column` is the column [`target_field`] picked. -fn swap_string_column( - mut row: HashMap, - column: String, - new_value: &[u8], -) -> Result, AtomicError> { - row.insert( - column, - Value::String(String::from_utf8_lossy(new_value).into_owned()), - ); - encode_map(row) -} - -/// A typed row and its string column, when `current` is a typed row with -/// one. [`cas`] and [`getset`] address the same column. -fn string_column(current: Option<&[u8]>) -> Option<(HashMap, String, String)> { - let row = typed_row(current?).ok().flatten()?; - let (column, text) = target_field(&row, column_string)?; - Some((row, column, text)) -} - -/// Compute the CAS outcome: whether `expected` matches the current value, -/// and the bytes to write when it does. -/// -/// The current value matches when its bytes equal `expected`, or when it is -/// a typed row whose string column holds `expected`. A typed row with a -/// string column keeps its shape: only that column is swapped. -pub fn cas( - current: Option<&[u8]>, - expected: &[u8], - new_value: &[u8], -) -> Result<(bool, Vec), AtomicError> { - let Some(cur) = current else { - return Ok(if expected.is_empty() { - (true, new_value.to_vec()) - } else { - (false, Vec::new()) - }); - }; - let typed = string_column(current); - let column_matches = typed - .as_ref() - .is_some_and(|(_, _, text)| *text == String::from_utf8_lossy(expected)); - if cur != expected && !column_matches { - return Ok((false, Vec::new())); - } - let write_bytes = match typed { - Some((row, column, _)) => swap_string_column(row, column, new_value)?, - None => new_value.to_vec(), - }; - Ok((true, write_bytes)) -} - -/// Compute the bytes to write for `GETSET`: the string column of a typed -/// row swapped in place, or a plain overwrite. -pub fn getset(current: Option<&[u8]>, new_value: &[u8]) -> Result, AtomicError> { - match string_column(current) { - Some((row, column, _)) => swap_string_column(row, column, new_value), - None => Ok(new_value.to_vec()), - } -} - -#[cfg(test)] -mod tests { - use super::*; - - static RAW: KvCounterShape = KvCounterShape::Raw; - - /// A typed shape moving `column`, with `rest` as the other stored columns. - fn typed_shape(column: Option<&str>, rest: &[(&str, Value)]) -> KvCounterShape { - KvCounterShape::Typed { - column: column.map(str::to_string), - template: row(rest), - } - } - - fn row(fields: &[(&str, Value)]) -> Vec { - let map: HashMap = fields - .iter() - .map(|(k, v)| ((*k).to_string(), v.clone())) - .collect(); - nodedb_types::value_to_msgpack(&Value::Object(map)).expect("encode row") - } - - fn columns(bytes: &[u8]) -> HashMap { - typed_row(bytes) - .expect("a typed row decodes") - .expect("a typed row stays a typed row") - } - - #[test] - fn incr_on_a_one_column_typed_row_keeps_the_row() { - let current = row(&[("n", Value::Integer(5))]); - let (new_i64, bytes) = incr(Some(¤t), 3, &RAW).expect("incr"); - assert_eq!(new_i64, 8); - assert_eq!(columns(&bytes).get("n"), Some(&Value::Integer(8))); - } - - #[test] - fn incr_moves_the_first_numeric_column_in_key_order() { - let current = row(&[ - ("b", Value::Integer(100)), - ("a", Value::Integer(1)), - ("label", Value::String("x".into())), - ]); - let (new_i64, bytes) = incr(Some(¤t), 1, &RAW).expect("incr"); - assert_eq!(new_i64, 2); - let cols = columns(&bytes); - assert_eq!(cols.get("a"), Some(&Value::Integer(2))); - assert_eq!(cols.get("b"), Some(&Value::Integer(100))); - assert_eq!(cols.get("label"), Some(&Value::String("x".into()))); - } - - #[test] - fn incr_on_a_typed_row_encodes_the_same_bytes_every_time() { - let current = row(&[ - ("a", Value::Integer(1)), - ("b", Value::Integer(2)), - ("c", Value::Integer(3)), - ]); - let (_, first) = incr(Some(¤t), 1, &RAW).expect("incr"); - for _ in 0..16 { - let (_, again) = incr(Some(¤t), 1, &RAW).expect("incr"); - assert_eq!(again, first); - } - } - - #[test] - fn incr_on_a_typed_row_without_a_numeric_column_is_a_type_mismatch() { - let current = row(&[("label", Value::String("x".into()))]); - assert!(matches!( - incr(Some(¤t), 1, &RAW), - Err(AtomicError::TypeMismatch { .. }) - )); - } - - #[test] - fn incr_on_a_raw_body_reads_and_writes_decimal_text() { - let (new_i64, bytes) = incr(Some(b"5"), 1, &RAW).expect("incr"); - assert_eq!(new_i64, 6); - assert_eq!(bytes, b"6".to_vec()); - - let (new_i64, bytes) = incr(Some(b"-10"), 3, &RAW).expect("incr"); - assert_eq!(new_i64, -7); - assert_eq!(bytes, b"-7".to_vec()); - - let (fresh, bytes) = incr(None, 4, &RAW).expect("incr"); - assert_eq!(fresh, 4); - assert_eq!(bytes, b"4".to_vec()); - } - - #[test] - fn incr_on_non_integer_raw_text_is_not_an_integer() { - for body in [ - b"abc".as_slice(), - b"", - b"1.5", - b"+5", - b"05", - b"-0", - b" 5", - b"5 ", - b"99999999999999999999", - ] { - assert!( - matches!( - incr(Some(body), 1, &RAW), - Err(AtomicError::Counter(CounterFault::NotAnInteger)) - ), - "{:?}", - String::from_utf8_lossy(body) - ); - } - } - - #[test] - fn incr_past_the_i64_range_is_an_overflow() { - let max = i64::MAX.to_string(); - assert!(matches!( - incr(Some(max.as_bytes()), 1, &RAW), - Err(AtomicError::Counter(CounterFault::IntegerOverflow)) - )); - let min = i64::MIN.to_string(); - assert!(matches!( - incr(Some(min.as_bytes()), -1, &RAW), - Err(AtomicError::Counter(CounterFault::IntegerOverflow)) - )); - let (value, bytes) = incr(Some(min.as_bytes()), 0, &RAW).expect("i64::MIN parses"); - assert_eq!(value, i64::MIN); - assert_eq!(bytes, min.into_bytes()); - } - - #[test] - fn incr_float_on_a_raw_body_reads_and_writes_decimal_text() { - let (new_f64, bytes) = incr_float(Some(b"1.5"), "1", &RAW).expect("incr_float"); - assert_eq!(new_f64, 2.5); - assert_eq!(bytes, b"2.5".to_vec()); - - let (new_f64, bytes) = incr_float(Some(b"10.5"), "0.5", &RAW).expect("incr_float"); - assert_eq!(new_f64, 11.0); - assert_eq!(bytes, b"11".to_vec()); - - let (_, bytes) = incr_float(Some(b"5"), "0.25", &RAW).expect("incr_float"); - assert_eq!(bytes, b"5.25".to_vec()); - - for (stored, delta, expected) in [ - ("0.1", "0.2", "0.3"), - ("10.5", "0.1", "10.6"), - ("5.0e3", "200", "5200"), - ("3.0", "0", "3"), - ("-1.5", "1.5", "0"), - ("1", "0.12345678901234567891", "1.12345678901234567891"), - ] { - let (_, bytes) = incr_float(Some(stored.as_bytes()), delta, &RAW).expect("incr_float"); - assert_eq!(bytes, expected.as_bytes().to_vec(), "{stored} + {delta}"); - } - } - - #[test] - fn incr_float_on_non_numeric_raw_text_is_not_a_float() { - for body in [b"abc".as_slice(), b"", b"NaN", b" 1.5"] { - assert!( - matches!( - incr_float(Some(body), "1", &RAW), - Err(AtomicError::Counter(CounterFault::NotAFloat)) - ), - "{:?}", - String::from_utf8_lossy(body) - ); - } - } - - #[test] - fn incr_float_to_infinity_is_non_finite() { - let max = f64::MAX.to_string(); - assert!(matches!( - incr_float(Some(max.as_bytes()), &max, &RAW), - Err(AtomicError::Counter(CounterFault::NonFinite)) - )); - } - - #[test] - fn incr_float_on_a_one_column_typed_row_keeps_the_row() { - let current = row(&[("score", Value::Float(1.5))]); - let (new_f64, bytes) = incr_float(Some(¤t), "1", &RAW).expect("incr_float"); - assert_eq!(new_f64, 2.5); - assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); - } - - #[test] - fn incr_on_an_absent_key_under_a_typed_shape_creates_the_typed_row() { - let shape = typed_shape(Some("n"), &[("status", Value::String("new".into()))]); - let (value, bytes) = incr(None, 7, &shape).expect("incr"); - assert_eq!(value, 7); - let cols = columns(&bytes); - assert_eq!(cols.get("n"), Some(&Value::Integer(7))); - assert_eq!(cols.get("status"), Some(&Value::String("new".into()))); - } - - #[test] - fn incr_float_on_an_absent_key_under_a_typed_shape_creates_the_typed_row() { - let shape = typed_shape(Some("score"), &[]); - let (value, bytes) = incr_float(None, "2.5", &shape).expect("incr_float"); - assert_eq!(value, 2.5); - assert_eq!(columns(&bytes).get("score"), Some(&Value::Float(2.5))); - } - - #[test] - fn an_absent_key_under_a_typed_shape_without_a_column_is_a_type_mismatch() { - let shape = typed_shape(None, &[]); - assert!(matches!( - incr(None, 1, &shape), - Err(AtomicError::TypeMismatch { .. }) - )); - assert!(matches!( - incr_float(None, "1", &shape), - Err(AtomicError::TypeMismatch { .. }) - )); - } - - #[test] - fn cas_on_a_one_column_typed_row_swaps_the_column() { - let current = row(&[("state", Value::String("idle".into()))]); - let (matched, bytes) = cas(Some(¤t), b"idle", b"busy").expect("cas"); - assert!(matched); - assert_eq!( - columns(&bytes).get("state"), - Some(&Value::String("busy".into())) - ); - let (matched, _) = cas(Some(¤t), b"busy", b"idle").expect("cas"); - assert!(!matched); - } - - #[test] - fn getset_on_a_one_column_typed_row_swaps_the_column() { - let current = row(&[("token", Value::String("old".into()))]); - let bytes = getset(Some(¤t), b"new").expect("getset"); - assert_eq!( - columns(&bytes).get("token"), - Some(&Value::String("new".into())) - ); - assert_eq!(getset(None, b"raw").expect("getset"), b"raw".to_vec()); - } -} diff --git a/nodedb/src/engine/kv/engine_sorted.rs b/nodedb/src/engine/kv/engine_sorted.rs index 973f9f1f2..8bf4378da 100644 --- a/nodedb/src/engine/kv/engine_sorted.rs +++ b/nodedb/src/engine/kv/engine_sorted.rs @@ -52,9 +52,22 @@ impl KvEngine { .entry(tkey) .or_insert_with(|| collection.to_string()); - // Collect existing entries from the hash table for backfill. - let entries: Vec<(Vec, Vec)> = self - .tables + let entries = self.collection_rows(database_id, tenant_id, collection, now_ms); + self.sorted_indexes + .register(database_id, tenant_id, def, entries.into_iter()) + } + + /// Every live `(key, value)` row of one collection, the rows a sorted + /// index is built from. Empty when the collection holds no rows. + pub fn collection_rows( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + now_ms: u64, + ) -> Vec<(Vec, Vec)> { + let tkey = table_key(database_id, tenant_id, collection); + self.tables .get(&tkey) .map(|t| { let (entries, _) = t.scan(0, usize::MAX, now_ms, None); @@ -63,10 +76,7 @@ impl KvEngine { .map(|(k, v)| (k.to_vec(), v.to_vec())) .collect() }) - .unwrap_or_default(); - - self.sorted_indexes - .register(database_id, tenant_id, def, entries.into_iter()) + .unwrap_or_default() } /// Drop a sorted index. Returns `true` if it existed. diff --git a/nodedb/src/engine/kv/float_text.rs b/nodedb/src/engine/kv/float_text.rs deleted file mode 100644 index bb80a302a..000000000 --- a/nodedb/src/engine/kv/float_text.rs +++ /dev/null @@ -1,224 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! `INCRBYFLOAT` on a raw KV body: decimal text in, decimal text out. -//! -//! Redis adds in `long double` and prints the sum with 17 fractional digits, -//! trailing zeros trimmed, so `"0.1"` plus `0.2` stores `"0.3"`. Rust has no -//! `long double`. Exact decimal addition gives the same text for every sum -//! that fits a [`Decimal`]: 28 significant digits, magnitude below 7.9e28. -//! An operand or a sum outside that range is added in `f64` instead. - -use std::str::FromStr; - -use rust_decimal::Decimal; -use rust_decimal::prelude::ToPrimitive; - -use super::engine_atomic::AtomicError; -use crate::bridge::envelope::CounterFault; - -/// Add `delta` to the raw body `stored`. Returns the new value and the text -/// to store. -/// -/// `stored` and `delta` must both be decimal numbers (see -/// [`is_decimal_number`]). Anything else is `Counter(NotAFloat)`. A sum that -/// is not finite is `Counter(NonFinite)`. -pub(super) fn add(stored: &[u8], delta: &str) -> Result<(f64, Vec), AtomicError> { - let text = std::str::from_utf8(stored) - .ok() - .filter(|text| is_decimal_number(text)) - .ok_or(AtomicError::Counter(CounterFault::NotAFloat))?; - let delta_f64 = delta_to_f64(delta)?; - if let (Some(base), Some(step)) = (parse_decimal(text), parse_decimal(delta)) - && let Some(sum) = base.checked_add(step) - { - let sum = sum.normalize(); - let value = sum - .to_f64() - .ok_or(AtomicError::Counter(CounterFault::NonFinite))?; - return Ok((value, sum.to_string().into_bytes())); - } - let base: f64 = text - .parse() - .map_err(|_| AtomicError::Counter(CounterFault::NotAFloat))?; - let value = base + delta_f64; - if !value.is_finite() { - return Err(AtomicError::Counter(CounterFault::NonFinite)); - } - Ok((value, float_text(value))) -} - -/// The text for a fresh float counter: `0` plus `delta`. -pub(super) fn fresh(delta: &str) -> Result<(f64, Vec), AtomicError> { - add(b"0", delta) -} - -/// `delta` as a finite `f64`, for a typed column. A delta that is not a -/// decimal number is `Counter(NotAFloat)`. One outside the `f64` range is -/// `Counter(NonFinite)`. -pub fn delta_to_f64(delta: &str) -> Result { - if !is_decimal_number(delta) { - return Err(AtomicError::Counter(CounterFault::NotAFloat)); - } - let value: f64 = delta - .parse() - .map_err(|_| AtomicError::Counter(CounterFault::NotAFloat))?; - if value.is_finite() { - Ok(value) - } else { - Err(AtomicError::Counter(CounterFault::NonFinite)) - } -} - -/// The exact value of `text`, or `None` when it does not fit a [`Decimal`]. -fn parse_decimal(text: &str) -> Option { - if text.contains(['e', 'E']) { - Decimal::from_scientific(text).ok() - } else { - Decimal::from_str(text).ok() - } -} - -/// The text for an `f64` sum outside the [`Decimal`] range: plain decimal -/// digits with no exponent, the form Redis prints. -fn float_text(value: f64) -> Vec { - let text = value.to_string(); - if text == "-0" { - b"0".to_vec() - } else { - text.into_bytes() - } -} - -/// The number grammar `INCRBYFLOAT` accepts, for a stored body and for the -/// client's increment: an optional sign, then digits with an optional point -/// (at least one digit), then an optional exponent `e` or `E` with an -/// optional sign and at least one digit. No whitespace, digit separators, -/// `inf`, or `nan`. -pub fn is_decimal_number(text: &str) -> bool { - let bytes = text.as_bytes(); - let mut i = 0; - if matches!(bytes.first(), Some(b'+' | b'-')) { - i += 1; - } - let int_digits = count_digits(&bytes[i..]); - i += int_digits; - let mut frac_digits = 0; - if bytes.get(i) == Some(&b'.') { - i += 1; - frac_digits = count_digits(&bytes[i..]); - i += frac_digits; - } - if int_digits + frac_digits == 0 { - return false; - } - if matches!(bytes.get(i), Some(b'e' | b'E')) { - i += 1; - if matches!(bytes.get(i), Some(b'+' | b'-')) { - i += 1; - } - let exp_digits = count_digits(&bytes[i..]); - if exp_digits == 0 { - return false; - } - i += exp_digits; - } - i == bytes.len() -} - -fn count_digits(bytes: &[u8]) -> usize { - bytes.iter().take_while(|b| b.is_ascii_digit()).count() -} - -#[cfg(test)] -mod tests { - use super::*; - - fn text_of(stored: &str, delta: &str) -> String { - let (_, bytes) = add(stored.as_bytes(), delta).expect("add"); - String::from_utf8(bytes).expect("UTF-8") - } - - #[test] - fn decimal_text_adds_exactly_like_redis() { - assert_eq!(text_of("0.1", "0.2"), "0.3"); - assert_eq!(text_of("10.5", "0.1"), "10.6"); - assert_eq!(text_of("5.0e3", "200"), "5200"); - assert_eq!(text_of("3.0", "0"), "3"); - assert_eq!(text_of("-1.5", "1.5"), "0"); - assert_eq!(text_of("1.5", "1"), "2.5"); - assert_eq!(text_of("+2", "-0.5"), "1.5"); - assert_eq!(text_of("1E-2", "0"), "0.01"); - assert_eq!(text_of("1", "1e1"), "11"); - } - - #[test] - fn a_twenty_digit_delta_adds_exactly() { - assert_eq!( - text_of("1", "0.12345678901234567891"), - "1.12345678901234567891" - ); - assert_eq!(text_of("10000000000000000000", "1"), "10000000000000000001"); - } - - #[test] - fn the_returned_value_matches_the_stored_text() { - let (value, bytes) = add(b"0.1", "0.2").expect("add"); - assert_eq!(value, 0.3); - assert_eq!(bytes, b"0.3".to_vec()); - } - - #[test] - fn a_fresh_counter_stores_the_delta_text() { - assert_eq!(fresh("2.5").expect("fresh").1, b"2.5".to_vec()); - assert_eq!(fresh("0").expect("fresh").1, b"0".to_vec()); - assert_eq!(fresh("-0.0").expect("fresh").1, b"0".to_vec()); - } - - #[test] - fn a_sum_outside_the_decimal_range_adds_in_f64() { - let (value, bytes) = add(b"1e300", "1").expect("add"); - assert_eq!(value, 1e300); - assert_eq!(bytes, 1e300f64.to_string().into_bytes()); - } - - #[test] - fn text_that_is_not_a_number_is_not_a_float() { - for stored in [ - "abc", "", "NaN", "inf", " 1.5", "1.5 ", "1_000", ".", "1e", "e5", "0x10", - ] { - assert!( - matches!( - add(stored.as_bytes(), "1"), - Err(AtomicError::Counter(CounterFault::NotAFloat)) - ), - "{stored:?}" - ); - } - } - - #[test] - fn a_non_finite_sum_is_refused() { - let max = f64::MAX.to_string(); - assert!(matches!( - add(max.as_bytes(), &max), - Err(AtomicError::Counter(CounterFault::NonFinite)) - )); - assert!(matches!( - add(b"1", "1e400"), - Err(AtomicError::Counter(CounterFault::NonFinite)) - )); - } - - #[test] - fn a_delta_that_is_not_a_number_is_not_a_float() { - for delta in ["abc", "", "inf", "NaN", " 1"] { - assert!( - matches!( - add(b"1", delta), - Err(AtomicError::Counter(CounterFault::NotAFloat)) - ), - "{delta:?}" - ); - } - } -} diff --git a/nodedb/src/engine/kv/mod.rs b/nodedb/src/engine/kv/mod.rs index 5af11ece0..fe7a3f9c1 100644 --- a/nodedb/src/engine/kv/mod.rs +++ b/nodedb/src/engine/kv/mod.rs @@ -4,7 +4,6 @@ mod batch_put; mod clock; pub mod engine; pub mod engine_atomic; -pub mod engine_atomic_compute; mod engine_helpers; mod engine_index; mod engine_rename; @@ -13,7 +12,6 @@ mod engine_stats; mod engine_write; pub mod entry; pub mod expiry_wheel; -pub mod float_text; mod hash_helpers; pub mod hash_table; pub mod index; @@ -30,7 +28,6 @@ pub use engine_atomic::{ AtomicAdmission, AtomicError, AtomicKeyCtx, CasResult, GetSetResult, IncrStep, Incremented, admit_any, }; -pub use engine_atomic_compute as atomic_compute; pub use engine_index::RegisterIndexParams; pub use engine_rename::RenameCollectionParams; pub use engine_sorted::SortedIndexRangeParams; diff --git a/nodedb/src/engine/kv/sorted_index/checkpoint.rs b/nodedb/src/engine/kv/sorted_index/checkpoint.rs index 4a8e7e588..5929b5e79 100644 --- a/nodedb/src/engine/kv/sorted_index/checkpoint.rs +++ b/nodedb/src/engine/kv/sorted_index/checkpoint.rs @@ -18,7 +18,8 @@ //! the tree that was actually live. use super::super::engine_helpers::table_key; -use super::manager::{SortedIndex, SortedIndexDef, SortedIndexManager, index_key}; +use super::index::SortedIndex; +use super::manager::{SortedIndexDef, SortedIndexManager, index_key}; use super::tree::OrderStatTree; /// One sorted index as a checkpoint sees it: its definition plus its full tree diff --git a/nodedb/src/engine/kv/sorted_index/index.rs b/nodedb/src/engine/kv/sorted_index/index.rs new file mode 100644 index 000000000..4b73880d2 --- /dev/null +++ b/nodedb/src/engine/kv/sorted_index/index.rs @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One sorted index: its definition, its order-statistic tree, and the reads +//! it answers. +//! +//! The manager holds one of these per registered index. A transaction builds +//! a private one from the collection's base rows with its staged writes +//! folded in. Both answer through the same methods, so a read inside a +//! transaction ranks, counts and windows exactly like an autocommit read. + +use super::key::SortKeyEncoder; +use super::manager::SortedIndexDef; +use super::tree::OrderStatTree; +use super::windowed_query::{self, SortedIndexRef}; + +/// A sorted index: definition plus tree. +pub struct SortedIndex { + pub(super) def: SortedIndexDef, + pub(super) tree: OrderStatTree, +} + +impl std::fmt::Debug for SortedIndex { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("SortedIndex") + .field("name", &self.def.name) + .field("collection", &self.def.collection) + .field("count", &self.tree.count()) + .finish() + } +} + +impl SortedIndex { + /// Build an index over `rows`, `(primary_key_bytes, value_bytes)` pairs. + /// + /// A row that lacks a sort column is left out. Returns the index and the + /// number of rows it holds. + pub fn build( + def: SortedIndexDef, + rows: impl Iterator, Vec)>, + ) -> (Self, u32) { + let mut tree = OrderStatTree::new(); + let mut indexed = 0u32; + for (pk_bytes, value_bytes) in rows { + if let Some(sort_key) = extract_sort_key_from_value(&def, &value_bytes) { + tree.insert(sort_key, pk_bytes); + indexed += 1; + } + } + (Self { def, tree }, indexed) + } + + /// The definition this index was built from. + pub fn def(&self) -> &SortedIndexDef { + &self.def + } + + fn index_ref(&self) -> SortedIndexRef<'_> { + SortedIndexRef { + def: &self.def, + tree: &self.tree, + } + } + + /// The 1-based rank of `primary_key`. + /// + /// A windowed index counts only the entries inside the current window. + pub fn rank(&self, primary_key: &[u8], now_ms: u64) -> Option { + if self.def.window.is_unwindowed() { + return self.tree.rank(primary_key); + } + windowed_query::windowed_rank(&self.index_ref(), primary_key, now_ms) + } + + /// The top `k` entries as `(rank, primary_key)` pairs. + pub fn top_k(&self, k: u32, now_ms: u64) -> Vec<(u32, Vec)> { + if self.def.window.is_unwindowed() { + return self + .tree + .top_k(k) + .into_iter() + .enumerate() + .map(|(i, (_, pk))| (i as u32 + 1, pk.to_vec())) + .collect(); + } + windowed_query::windowed_top_k(&self.index_ref(), k, now_ms) + } + + /// The entries in a score range as `(rank, primary_key)` pairs. + /// + /// `score_min` and `score_max` are the raw value bytes of the index's + /// LEADING sort column (as [`extract_sort_key_from_value`] produces them), + /// not encoded tree keys. The caller names a score. Only the index's own + /// encoder knows the framing and direction that turn it into a bound the + /// tree compares against. + pub fn range( + &self, + score_min: Option<&[u8]>, + score_max: Option<&[u8]>, + now_ms: u64, + ) -> Vec<(u32, Vec)> { + let (lower, upper) = self + .def + .encoder + .first_column_range_bounds(score_min, score_max); + let entries = self.tree.range(lower.as_deref(), upper.as_deref()); + + if self.def.window.is_unwindowed() { + return entries + .into_iter() + .filter_map(|(_, pk)| { + let rank = self.tree.rank(pk)?; + Some((rank, pk.to_vec())) + }) + .collect(); + } + windowed_query::windowed_range(&self.index_ref(), &entries, now_ms) + } + + /// The number of entries, inside the current window for a windowed index. + pub fn count(&self, now_ms: u64) -> u32 { + if self.def.window.is_unwindowed() { + return self.tree.count(); + } + windowed_query::windowed_count(&self.index_ref(), now_ms) + } + + /// The sort key of `primary_key` (ZSCORE equivalent). + pub fn score(&self, primary_key: &[u8]) -> Option> { + self.tree.get_sort_key(primary_key).map(|s| s.to_vec()) + } +} + +/// Extract the sort columns from a MessagePack-encoded KV value and build a +/// sort key. `None` when the value is not a map or lacks a sort column. +pub(super) fn extract_sort_key_from_value( + def: &SortedIndexDef, + value_bytes: &[u8], +) -> Option> { + let doc: serde_json::Value = nodedb_types::json_from_msgpack(value_bytes).ok()?; + let obj = doc.as_object()?; + + let mut values: Vec> = Vec::with_capacity(def.encoder.column_count()); + for col in def.encoder.columns() { + let field_val = obj.get(&col.name)?; + values.push(field_value_to_sort_bytes(field_val)); + } + + let refs: Vec<&[u8]> = values.iter().map(|v| v.as_slice()).collect(); + Some(def.encoder.encode(&refs)) +} + +/// Convert a JSON field value to sortable bytes. +fn field_value_to_sort_bytes(val: &serde_json::Value) -> Vec { + match val { + serde_json::Value::Number(n) => { + if let Some(i) = n.as_i64() { + SortKeyEncoder::encode_i64(i).to_vec() + } else if let Some(f) = n.as_f64() { + SortKeyEncoder::encode_f64(f).to_vec() + } else { + Vec::new() + } + } + serde_json::Value::String(s) => s.as_bytes().to_vec(), + _ => Vec::new(), + } +} diff --git a/nodedb/src/engine/kv/sorted_index/manager.rs b/nodedb/src/engine/kv/sorted_index/manager.rs index 80229c710..4cb59c1a5 100644 --- a/nodedb/src/engine/kv/sorted_index/manager.rs +++ b/nodedb/src/engine/kv/sorted_index/manager.rs @@ -11,10 +11,9 @@ use std::collections::{BTreeSet, HashMap}; +use super::index::SortedIndex; use super::key::SortKeyEncoder; -use super::tree::OrderStatTree; use super::window::WindowConfig; -use super::windowed_query::{self, SortedIndexRef}; /// Definition of a sorted index (metadata). #[derive(Debug, Clone)] @@ -31,12 +30,6 @@ pub struct SortedIndexDef { pub window: WindowConfig, } -/// A live sorted index: definition + data. -pub(super) struct SortedIndex { - pub(super) def: SortedIndexDef, - pub(super) tree: OrderStatTree, -} - /// Manages all sorted indexes on a single TPC core. /// /// Key: `(tenant_hash, index_name)` where tenant_hash is the same hash @@ -58,16 +51,6 @@ pub struct SortedIndexManager { pub(super) collection_indexes: HashMap>, } -impl std::fmt::Debug for SortedIndex { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.debug_struct("SortedIndex") - .field("name", &self.def.name) - .field("collection", &self.def.collection) - .field("count", &self.tree.count()) - .finish() - } -} - impl SortedIndexManager { pub fn new() -> Self { Self { @@ -112,23 +95,14 @@ impl SortedIndexManager { let tbl_key = super::super::engine_helpers::table_key(database_id, tenant_id, &def.collection); - let mut tree = OrderStatTree::new(); - let mut backfilled = 0u32; - - // Backfill from existing data. - for (pk_bytes, value_bytes) in existing_entries { - if let Some(sort_key) = extract_sort_key_from_value(&def, &value_bytes) { - tree.insert(sort_key, pk_bytes); - backfilled += 1; - } - } + let (index, backfilled) = SortedIndex::build(def, existing_entries); self.collection_indexes .entry(tbl_key) .or_default() .insert(idx_key.clone()); - self.indexes.insert(idx_key, SortedIndex { def, tree }); + self.indexes.insert(idx_key, index); backfilled } @@ -261,19 +235,8 @@ impl SortedIndexManager { primary_key: &[u8], now_ms: u64, ) -> Option { - let idx = self.get_index(database_id, tenant_id, index_name)?; - - if idx.def.window.is_unwindowed() { - return idx.tree.rank(primary_key); - } - - // Windowed: need to count how many entries with a lower sort key - // are within the current window. This is the expensive path. - let idx_ref = SortedIndexRef { - def: &idx.def, - tree: &idx.tree, - }; - windowed_query::windowed_rank(&idx_ref, primary_key, now_ms) + self.get_index(database_id, tenant_id, index_name)? + .rank(primary_key, now_ms) } /// Get the top K entries from a sorted index. @@ -287,33 +250,14 @@ impl SortedIndexManager { k: u32, now_ms: u64, ) -> Option)>> { - let idx = self.get_index(database_id, tenant_id, index_name)?; - - if idx.def.window.is_unwindowed() { - let entries = idx.tree.top_k(k); - return Some( - entries - .into_iter() - .enumerate() - .map(|(i, (_, pk))| (i as u32 + 1, pk.to_vec())) - .collect(), - ); - } - - let idx_ref = SortedIndexRef { - def: &idx.def, - tree: &idx.tree, - }; - Some(windowed_query::windowed_top_k(&idx_ref, k, now_ms)) + Some( + self.get_index(database_id, tenant_id, index_name)? + .top_k(k, now_ms), + ) } - /// Get entries in a score range from a sorted index. - /// - /// `score_min` and `score_max` are the raw value bytes of the index's - /// LEADING sort column (as [`extract_sort_key_from_value`] produces them), - /// not encoded tree keys: the caller names a score, and only the index's - /// own encoder knows the framing and direction that turn it into a bound - /// the tree can be compared against. + /// Get entries in a score range from a sorted index. See + /// [`SortedIndex::range`] for what the bounds hold. /// /// Returns `(rank, primary_key)` pairs. pub fn range( @@ -325,32 +269,10 @@ impl SortedIndexManager { score_max: Option<&[u8]>, now_ms: u64, ) -> Option)>> { - let idx = self.get_index(database_id, tenant_id, index_name)?; - - let (lower, upper) = idx - .def - .encoder - .first_column_range_bounds(score_min, score_max); - let entries = idx.tree.range(lower.as_deref(), upper.as_deref()); - - if idx.def.window.is_unwindowed() { - // Compute rank for each entry. - return Some( - entries - .into_iter() - .filter_map(|(_, pk)| { - let rank = idx.tree.rank(pk)?; - Some((rank, pk.to_vec())) - }) - .collect(), - ); - } - - let idx_ref = SortedIndexRef { - def: &idx.def, - tree: &idx.tree, - }; - Some(windowed_query::windowed_range(&idx_ref, &entries, now_ms)) + Some( + self.get_index(database_id, tenant_id, index_name)? + .range(score_min, score_max, now_ms), + ) } /// Get the total count of entries in a sorted index. @@ -361,17 +283,10 @@ impl SortedIndexManager { index_name: &str, now_ms: u64, ) -> Option { - let idx = self.get_index(database_id, tenant_id, index_name)?; - - if idx.def.window.is_unwindowed() { - return Some(idx.tree.count()); - } - - let idx_ref = SortedIndexRef { - def: &idx.def, - tree: &idx.tree, - }; - Some(windowed_query::windowed_count(&idx_ref, now_ms)) + Some( + self.get_index(database_id, tenant_id, index_name)? + .count(now_ms), + ) } /// Get the sort key for a primary key in a sorted index (ZSCORE equivalent). @@ -382,8 +297,8 @@ impl SortedIndexManager { index_name: &str, primary_key: &[u8], ) -> Option> { - let idx = self.get_index(database_id, tenant_id, index_name)?; - idx.tree.get_sort_key(primary_key).map(|s| s.to_vec()) + self.get_index(database_id, tenant_id, index_name)? + .score(primary_key) } /// Get the index definition. @@ -393,8 +308,7 @@ impl SortedIndexManager { tenant_id: u64, index_name: &str, ) -> Option<&SortedIndexDef> { - let idx = self.get_index(database_id, tenant_id, index_name)?; - Some(&idx.def) + Some(self.get_index(database_id, tenant_id, index_name)?.def()) } fn get_index( @@ -420,22 +334,6 @@ pub(super) fn index_key(database_id: u64, tenant_id: u64, index_name: &str) -> S format!("{database_id}:{tenant_id}:{index_name}") } -/// Extract field values from a MessagePack-encoded KV value and build a sort key. -fn extract_sort_key_from_value(def: &SortedIndexDef, value_bytes: &[u8]) -> Option> { - let doc: serde_json::Value = nodedb_types::json_from_msgpack(value_bytes).ok()?; - let obj = doc.as_object()?; - - let mut values: Vec> = Vec::with_capacity(def.encoder.column_count()); - for col in def.encoder.columns() { - let field_val = obj.get(&col.name)?; - let bytes = field_value_to_sort_bytes(field_val); - values.push(bytes); - } - - let refs: Vec<&[u8]> = values.iter().map(|v| v.as_slice()).collect(); - Some(def.encoder.encode(&refs)) -} - /// Build a sort key from pre-extracted field name/value pairs. fn build_sort_key_from_fields( def: &SortedIndexDef, @@ -459,23 +357,6 @@ fn build_sort_key_from_fields( Some(def.encoder.encode(&refs)) } -/// Convert a JSON field value to sortable bytes. -fn field_value_to_sort_bytes(val: &serde_json::Value) -> Vec { - match val { - serde_json::Value::Number(n) => { - if let Some(i) = n.as_i64() { - SortKeyEncoder::encode_i64(i).to_vec() - } else if let Some(f) = n.as_f64() { - SortKeyEncoder::encode_f64(f).to_vec() - } else { - Vec::new() - } - } - serde_json::Value::String(s) => s.as_bytes().to_vec(), - _ => Vec::new(), - } -} - #[cfg(test)] mod tests { use super::super::key::{SortColumn, SortDirection}; diff --git a/nodedb/src/engine/kv/sorted_index/mod.rs b/nodedb/src/engine/kv/sorted_index/mod.rs index ea658082e..2d94a0a5b 100644 --- a/nodedb/src/engine/kv/sorted_index/mod.rs +++ b/nodedb/src/engine/kv/sorted_index/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 pub mod checkpoint; +pub mod index; pub mod key; pub mod manager; pub mod tree; @@ -8,6 +9,7 @@ pub mod window; mod windowed_query; pub use checkpoint::SortedIndexSnapshot; +pub use index::SortedIndex; pub use key::{SortDirection, SortKeyEncoder}; pub use manager::SortedIndexManager; pub use tree::OrderStatTree; diff --git a/nodedb/src/error/types.rs b/nodedb/src/error/types.rs index 58c686df5..724aa3157 100644 --- a/nodedb/src/error/types.rs +++ b/nodedb/src/error/types.rs @@ -210,6 +210,11 @@ pub enum Error { #[error("CRDT Apply is not supported inside explicit transactions")] CrdtApplyForbiddenInTransaction, + /// A statement that applies at once and cannot be rolled back ran + /// inside an explicit transaction block. SQLSTATE 25001. + #[error("{statement} cannot run inside a transaction block")] + NotInTransactionBlock { statement: String }, + #[error("CRDT admission timed out on {vshard_id} after {timeout_ms}ms")] CrdtAdmissionTimeout { vshard_id: VShardId, diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index 48da6ab87..4d0d5439d 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -141,6 +141,7 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { | Error::CrdtApplyForbiddenInTransaction => { NodeDbError::bad_request("invalid CRDT admission request".to_owned()) } + Error::NotInTransactionBlock { .. } => NodeDbError::bad_request(e.to_string()), Error::CrdtAdmissionTimeout { .. } => NodeDbError::deadline_exceeded(), Error::NoLeader { vshard_id } => { NodeDbError::no_leader(format!("vshard {vshard_id} has no serving leader")) diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index 9efae4486..4d421eaf3 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -11,7 +11,7 @@ use nodedb_types::error::{ErrorCode as PublicCode, NodeDbError}; -use crate::bridge::envelope::{CounterFault, ErrorCode}; +use crate::bridge::envelope::ErrorCode; /// Convert a deterministic Data-Plane code into the public error a client /// can classify. @@ -35,6 +35,11 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { ErrorCode::RejectedPrevalidation { reason } => { NodeDbError::prevalidation_rejected("data plane", reason) } + // A sync frame the validator refused is a constraint verdict on the + // frame. + ErrorCode::SyncRejected { violation, .. } => { + NodeDbError::constraint_violation("", "sync", violation.to_string()) + } // Nothing was applied and the identical frame is expected to succeed // once the transient precondition resolves, so it presents as the // retriable class rather than a permanent refusal. @@ -111,14 +116,9 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { } // The same text the SQL surfaces send, with the collection in the // details. RESP renders the bare Redis text from the code itself. - ErrorCode::CounterFault { collection, fault } => NodeDbError::kv_counter_fault( - collection, - fault.message(), - matches!( - fault, - CounterFault::IntegerOverflow | CounterFault::NonFinite - ), - ), + ErrorCode::CounterFault { collection, fault } => { + NodeDbError::kv_counter_fault(collection, fault.message(), fault.is_out_of_range()) + } ErrorCode::InsufficientBalance { collection, detail } => { NodeDbError::insufficient_balance(collection, detail) } @@ -166,6 +166,7 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { #[cfg(test)] mod tests { use super::*; + use crate::bridge::envelope::CounterFault; #[test] fn constraint_code_classifies_as_constraint_violation() { diff --git a/nodedb/src/wal/crdt_payload.rs b/nodedb/src/wal/crdt_payload.rs index 7a051bf3b..fb3b70227 100644 --- a/nodedb/src/wal/crdt_payload.rs +++ b/nodedb/src/wal/crdt_payload.rs @@ -30,6 +30,9 @@ pub(crate) struct CrdtDeltaWalPayload { pub document_id: Option, pub surrogate: Option, pub signing: Option, + /// The peer that produced the delta. Replay validates under it, so a + /// dead-letter entry replay stores names the same peer the live apply did. + pub peer_id: u64, } #[derive(zerompk::ToMessagePack, zerompk::FromMessagePack)] @@ -41,6 +44,7 @@ struct CrdtDeltaWalPayloadV4 { expected_frontier_digest: Option<[u8; 32]>, document_id: Option, surrogate: Option, + peer_id: u64, auth_user_id: u64, auth_device_id: u64, auth_seq_no: u64, @@ -57,6 +61,7 @@ struct CrdtDeltaWalPayloadV3 { expected_frontier_digest: Option<[u8; 32]>, document_id: Option, surrogate: Option, + peer_id: u64, } /// Exact fenced wire shape emitted before sparse replay identity was retained. @@ -94,6 +99,7 @@ impl CrdtDeltaWalPayload { document_id, surrogate, signing: None, + peer_id: 0, } } @@ -102,6 +108,11 @@ impl CrdtDeltaWalPayload { self } + pub(crate) fn with_peer_id(mut self, peer_id: u64) -> Self { + self.peer_id = peer_id; + self + } + /// Encode the current explicit wire format. pub(crate) fn encode(&self) -> Result, zerompk::Error> { if let Some(signing) = self.signing { @@ -113,6 +124,7 @@ impl CrdtDeltaWalPayload { expected_frontier_digest: self.expected_frontier_digest, document_id: self.document_id.clone(), surrogate: self.surrogate, + peer_id: self.peer_id, auth_user_id: signing.auth_user_id, auth_device_id: signing.auth_device_id, auth_seq_no: signing.auth_seq_no, @@ -128,6 +140,7 @@ impl CrdtDeltaWalPayload { expected_frontier_digest: self.expected_frontier_digest, document_id: self.document_id.clone(), surrogate: self.surrogate, + peer_id: self.peer_id, }) } @@ -151,7 +164,8 @@ impl CrdtDeltaWalPayload { auth_seq_no: v4.auth_seq_no, delta_signature: v4.delta_signature, required: v4.signing_required, - })); + }) + .with_peer_id(v4.peer_id)); } if let Ok(v3) = zerompk::from_msgpack::(bytes) && v3.format == CRDT_DELTA_WAL_FORMAT_V3 @@ -163,7 +177,8 @@ impl CrdtDeltaWalPayload { v3.expected_frontier_digest, v3.document_id, v3.surrogate, - )); + ) + .with_peer_id(v3.peer_id)); } if let Ok(v2) = zerompk::from_msgpack::(bytes) && v2.format == CRDT_DELTA_WAL_FORMAT_V2 @@ -275,4 +290,25 @@ mod tests { CrdtDeltaWalPayload::decode(&payload.encode().expect("encode")).expect("decode v3"); assert_eq!(decoded, payload); } + + #[test] + fn the_producing_peer_round_trips_in_both_current_formats() { + let unsigned = + CrdtDeltaWalPayload::new(vec![1], Some("docs".into()), None, None, None, None) + .with_peer_id(0xFEED); + let decoded = + CrdtDeltaWalPayload::decode(&unsigned.encode().expect("encode")).expect("decode v3"); + assert_eq!(decoded.peer_id, 0xFEED); + + let signed = unsigned.clone().with_signing(CrdtDeltaSigning { + auth_user_id: 1, + auth_device_id: 2, + auth_seq_no: 3, + delta_signature: [4; 32], + required: true, + }); + let decoded = + CrdtDeltaWalPayload::decode(&signed.encode().expect("encode")).expect("decode v4"); + assert_eq!(decoded.peer_id, 0xFEED); + } } diff --git a/nodedb/src/wal/manager/appender.rs b/nodedb/src/wal/manager/appender.rs index 3981fb451..f2d7e54ed 100644 --- a/nodedb/src/wal/manager/appender.rs +++ b/nodedb/src/wal/manager/appender.rs @@ -30,13 +30,26 @@ pub struct RecordedAppend { pub database_id: DatabaseId, } +/// Where a recording appender reports each record it writes. The report +/// runs while the WAL lock is held, so no reader sees the record before its +/// recorder does. +pub trait AppendSink: Sync { + fn record(&self, append: RecordedAppend); +} + +impl AppendSink for Mutex> { + fn record(&self, append: RecordedAppend) { + self.lock().unwrap_or_else(|p| p.into_inner()).push(append); + } +} + /// Appends WAL records that all carry one apply key. #[derive(Clone, Copy)] pub struct WalAppender<'a> { wal: &'a WalManager, apply_key: u64, /// Collects every record this appender writes, when set. - sink: Option<&'a Mutex>>, + sink: Option<&'a dyn AppendSink>, } impl WalManager { @@ -51,12 +64,12 @@ impl WalManager { } } - /// An appender that also pushes every record it writes onto `sink`, in + /// An appender that also reports every record it writes to `sink`, in /// append order. pub fn recording_appender<'a>( &'a self, apply_key: u64, - sink: &'a Mutex>, + sink: &'a dyn AppendSink, ) -> WalAppender<'a> { WalAppender { wal: self, @@ -94,18 +107,16 @@ impl WalAppender<'_> { self.apply_key, ) .map_err(crate::Error::Wal)?; - drop(wal); let lsn = Lsn::new(lsn); if let Some(sink) = self.sink { - sink.lock() - .unwrap_or_else(|p| p.into_inner()) - .push(RecordedAppend { - lsn, - tenant_id, - vshard_id, - database_id, - }); + sink.record(RecordedAppend { + lsn, + tenant_id, + vshard_id, + database_id, + }); } + drop(wal); Ok(lsn) } } diff --git a/nodedb/src/wal/manager/mod.rs b/nodedb/src/wal/manager/mod.rs index 75438420e..b75490149 100644 --- a/nodedb/src/wal/manager/mod.rs +++ b/nodedb/src/wal/manager/mod.rs @@ -15,5 +15,5 @@ pub mod encryption; pub mod ops; pub mod replay; -pub use appender::{NO_APPLY_KEY, RecordedAppend, WalAppender}; +pub use appender::{AppendSink, NO_APPLY_KEY, RecordedAppend, WalAppender}; pub use core::WalManager; diff --git a/nodedb/src/wal/redo/record.rs b/nodedb/src/wal/redo/record.rs index 77f85f61f..86b910895 100644 --- a/nodedb/src/wal/redo/record.rs +++ b/nodedb/src/wal/redo/record.rs @@ -69,11 +69,11 @@ pub struct RedoSubRecord { } /// Calvin sequencer coordinates that a [`RedoRecord`] may carry to double as an -/// applied-marker. Mirrors `nodedb_wal::CalvinAppliedPayload`. +/// applied-marker, and what the vShard's slice folds. Mirrors +/// `nodedb_wal::CalvinAppliedPayload` in its coordinates. #[derive( Debug, Clone, - Copy, PartialEq, Eq, Serialize, @@ -89,6 +89,17 @@ pub struct CalvinStamp { pub position: u32, /// The vshard that applied this transaction. pub vshard_id: u32, + /// Every collection the vShard's slice writes. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + #[msgpack(default)] + pub collections: Vec, + /// The materialized-sum targets the slice's document writes fold into, + /// keyed by source collection. The live install and restart replay both + /// fold from them at the record's LSN, so no later record carries the + /// target rows. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + #[msgpack(default)] + pub sum_targets: Vec, } /// Redo sub-record payload for a graph edge upsert — the payload bytes of a @@ -215,6 +226,8 @@ mod tests { epoch: 42, position: 7, vshard_id: 3, + collections: Vec::new(), + sum_targets: Vec::new(), }), }; let bytes = record.to_bytes().expect("encode"); diff --git a/nodedb/src/wal/redo/replay.rs b/nodedb/src/wal/redo/replay.rs index bfafdd93f..8cfae0cc4 100644 --- a/nodedb/src/wal/redo/replay.rs +++ b/nodedb/src/wal/redo/replay.rs @@ -76,13 +76,22 @@ use crate::data::executor::core_loop::CoreLoop; /// Reconstituted records are always plaintext (`encryption_key: None`): the /// enclosing record was already decrypted when the WAL was read into memory, so /// its sub-payloads are cleartext and these records never touch disk. -fn reconstitute_redo_records(records: &[WalRecord]) -> crate::Result> { +/// +/// Also returns the materialized-sum targets every Calvin record's stamp +/// carries, keyed by the record's LSN. +fn reconstitute_with_folds(records: &[WalRecord]) -> crate::Result<(Vec, RedoFolds)> { let mut out = Vec::new(); + let mut folds = RedoFolds::new(); for record in records { if RecordType::from_raw(record.logical_record_type()) != Some(RecordType::TransactionRedo) { continue; } let redo = RedoRecord::from_bytes(&record.payload)?; + if let Some(stamp) = redo.calvin_stamp + && !stamp.sum_targets.is_empty() + { + folds.insert(record.header.lsn, stamp.sum_targets); + } for sub in redo.ops { out.push(WalRecord::new(WalRecordArgs { record_type: sub.record_type, @@ -96,9 +105,13 @@ fn reconstitute_redo_records(records: &[WalRecord]) -> crate::Result>; + /// Merge the standalone records with the reconstituted redo sub-records into /// one LSN-ordered sequence. /// @@ -161,8 +174,11 @@ impl CoreLoop { num_cores: usize, tombstones: &nodedb_wal::TombstoneSet, ) -> crate::Result<()> { - let redo_ops = reconstitute_redo_records(records)?; + let (redo_ops, folds) = reconstitute_with_folds(records)?; let ordered = merge_by_lsn(records, &redo_ops); + // A committed-redo apply folds from its open scope. Restart replay + // folds from each Calvin record's stamp. + let restart = self.begin_replay_folds(folds); self.replay_vector_wal(&ordered, num_cores, tombstones); crate::fail_point!("replay::between_engine_passes"); @@ -197,6 +213,9 @@ impl CoreLoop { // `apply_point_put` rebuilds any secondary vector index inline, so no // separate `replay_document_vector_wal` pass is needed for redo puts. self.replay_document_redo(&redo_ops, num_cores, tombstones); + if restart { + self.end_replay_folds(); + } self.replay_graph_redo(&redo_ops, num_cores, tombstones); // Node-label deltas staged inside a transaction resolve to the same // `GraphNodeLabelSet` / `GraphNodeLabelRemove` sub-record shape the @@ -231,6 +250,11 @@ mod tests { use super::*; use crate::wal::{RedoRecord, RedoSubRecord}; + /// The reconstituted records alone, without the fold targets. + fn reconstitute_redo_records(records: &[WalRecord]) -> crate::Result> { + Ok(super::reconstitute_with_folds(records)?.0) + } + fn redo_wal_record(lsn: u64, tenant_id: u64, vshard_id: u32, record: &RedoRecord) -> WalRecord { WalRecord::new(WalRecordArgs { record_type: RecordType::TransactionRedo as u32, diff --git a/nodedb/tests/crash_core_stall.rs b/nodedb/tests/crash_core_stall.rs index 2bd0cefc0..e6d2519ed 100644 --- a/nodedb/tests/crash_core_stall.rs +++ b/nodedb/tests/crash_core_stall.rs @@ -31,9 +31,21 @@ const WEDGE_SLEEP_MILLIS: u64 = 12_000; /// windows, with slack for a loaded runner. const STALL_DETECT_DEADLINE: Duration = Duration::from_secs(60); -/// Deadline for `/healthz` to return to 200. The marker clears one sampling -/// window after the core resumes. -const RECOVERY_DEADLINE: Duration = Duration::from_secs(20); +/// Static stage calls the wedge write runs on the one core. The edge's two +/// endpoints sit on distinct vShards, and each vShard stages separately, so +/// the failpoint sleeps once per vShard. +const WEDGED_STAGES: u64 = 2; + +/// Monitor sampling window. +const SAMPLE_WINDOW_MILLIS: u64 = 5_000; + +/// Deadline for `/healthz` to return to 200, counted from the first 503. +/// The 503 can appear as early as two sampling windows into the first sleep. +/// The core then stays frozen for the rest of every stage's sleep. The marker +/// clears one sampling window after the core resumes. The deadline covers the +/// whole freeze plus two sampling windows of slack for a loaded runner. +const RECOVERY_DEADLINE: Duration = + Duration::from_millis(WEDGE_SLEEP_MILLIS * WEDGED_STAGES + 2 * SAMPLE_WINDOW_MILLIS); /// Deadline for the write to terminate, so a hung write fails with a clear /// message instead of running out nextest's kill budget. diff --git a/nodedb/tests/crash_harness/boot.rs b/nodedb/tests/crash_harness/boot.rs new file mode 100644 index 000000000..844db350d --- /dev/null +++ b/nodedb/tests/crash_harness/boot.rs @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Waiting for a spawned server to report ready. +//! +//! `free_port` releases each port before the child binds it. Another process +//! can take the port in between, and every protocol bind is boot-fatal. The +//! wait therefore watches the child as well as `/healthz`. A child that exits +//! on a bind collision is respawned on fresh ports. Any other exit, or a boot +//! that outlives its budget, panics with this boot's server output. + +use std::time::{Duration, Instant}; + +use super::{BOOT_READY_TIMEOUT, BOOT_READY_TIMEOUT_EXTENDED, CrashHarness, check_healthz}; +use super::{diagnostics, free_port}; + +/// Spawns one boot can use before a bind collision becomes a panic. +/// Each one allocates six fresh ports, so repeated collisions are unlikely. +const BIND_COLLISION_ATTEMPTS: u32 = 3; + +/// The OS error text every listener bind failure carries in the boot error. +const ADDRESS_IN_USE: &str = "Address already in use"; + +/// How one wait for a boot ended. +enum BootEnd { + Ready, + Exited(std::process::ExitStatus), + TimedOut, +} + +impl CrashHarness { + /// Block until `/healthz` reports ready. + /// + /// Panics with the server output when the child exits or the budget + /// runs out. + pub fn wait_ready(&mut self) { + self.wait_ready_within(BOOT_READY_TIMEOUT); + } + + /// [`CrashHarness::wait_ready`] for a test whose nextest kill budget is raised. + pub fn wait_ready_extended(&mut self) { + self.wait_ready_within(BOOT_READY_TIMEOUT_EXTENDED); + } + + fn wait_ready_within(&mut self, budget: Duration) { + let mut attempt = 1; + loop { + match self.poll_boot(budget) { + BootEnd::Ready => return, + BootEnd::Exited(status) => { + let boot_log = self.current_boot_log(); + if boot_log.contains(ADDRESS_IN_USE) && attempt < BIND_COLLISION_ATTEMPTS { + attempt += 1; + self.reallocate_ports(); + self.spawn(); + continue; + } + panic!( + "nodedb exited during boot {} with {status} before reporting ready.\n{}{}{}", + self.boot_count, + self.keep_data_dir_note(), + diagnostics::faultbox_report_section(self.data_dir()), + diagnostics::log_tail_section(&self.server_log()), + ); + } + BootEnd::TimedOut => panic!( + "nodedb did not become ready within {budget:?} on boot {} and is still running.\n{}{}{}", + self.boot_count, + self.keep_data_dir_note(), + diagnostics::faultbox_report_section(self.data_dir()), + diagnostics::log_tail_section(&self.server_log()), + ), + } + } + } + + /// Poll the child and `/healthz` until one of them settles the boot. + fn poll_boot(&mut self, budget: Duration) -> BootEnd { + let deadline = Instant::now() + budget; + loop { + // A dead child never answers, so waiting out the budget hides its error. + if let Some(child) = self.child.as_mut() { + match child.try_wait() { + Ok(Some(status)) => { + self.child = None; + return BootEnd::Exited(status); + } + Ok(None) => {} + Err(e) => panic!("failed to poll the nodedb child process: {e}"), + } + } + if check_healthz(self.http_port) { + return BootEnd::Ready; + } + if Instant::now() >= deadline { + return BootEnd::TimedOut; + } + std::thread::sleep(Duration::from_millis(100)); + } + } + + /// Server output written since the marker of the latest boot. + fn current_boot_log(&self) -> String { + let log = self.server_log(); + let marker = format!("=== crash harness boot {} (pid", self.boot_count); + match log.rfind(&marker) { + Some(start) => log[start..].to_string(), + None => log, + } + } + + /// Give every protocol a fresh port for the next spawn. + fn reallocate_ports(&mut self) { + self.http_port = free_port(); + self.pgwire_port = free_port(); + self.native_port = free_port(); + self.sync_port = free_port(); + self.resp_port = free_port(); + self.ilp_port = free_port(); + } +} diff --git a/nodedb/tests/crash_harness/mod.rs b/nodedb/tests/crash_harness/mod.rs index aebe7f074..c356cb731 100644 --- a/nodedb/tests/crash_harness/mod.rs +++ b/nodedb/tests/crash_harness/mod.rs @@ -17,6 +17,8 @@ use std::time::{Duration, Instant}; // `pub` so a crash test can read faultbox reports directly, not just via // the panic-path diagnostics wired into `pgwire.rs`. pub mod diagnostics; +// `wait_ready` and its bind-collision respawn. +mod boot; // The ILP client helper lives in `nodedb-test-support` and is imported // directly by tests that need it, not re-exported here. mod pgwire; @@ -314,23 +316,6 @@ impl CrashHarness { self.child = Some(child); } - /// Block until `/healthz` reports ready, panicking on timeout. - pub fn wait_ready(&self) { - self.wait_ready_within(BOOT_READY_TIMEOUT); - } - - /// [`wait_ready`] for a test whose nextest kill budget is raised. - pub fn wait_ready_extended(&self) { - self.wait_ready_within(BOOT_READY_TIMEOUT_EXTENDED); - } - - fn wait_ready_within(&self, budget: Duration) { - assert!( - wait_for_healthz(self.http_port, budget), - "nodedb did not become ready within {budget:?}" - ); - } - /// Spawn the server and assert that boot fails-stop rather than coming up. /// The server must never report `/healthz`-ready and must exit non-zero /// within `timeout`. Panics otherwise. diff --git a/nodedb/tests/inproc/cases/calvin_executor_apply.rs b/nodedb/tests/inproc/cases/calvin_executor_apply.rs index 9ed445679..35720daa6 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_apply.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_apply.rs @@ -1,10 +1,11 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Data Plane apply and rollback coverage for `MetaOp::TransactionBatch`. +//! Data Plane apply and rollback coverage for a committed Calvin transaction. //! -//! These tests exercise the cross-shard apply path using -//! `MetaOp::TransactionBatch` (CalvinExecuteStatic) directly — no -//! `ReplicatedWrite::CrossShardForward`, no `decode_forwarded_plans`. +//! These drive the participant's Data Plane steps directly: stage +//! (`CalvinExecuteStatic`), resolve (`CalvinResolve`) and flush the resolved +//! redo record at its LSN (`CalvinFlush`) — the steps the scheduler dispatches +//! once the global verdict is commit. use std::sync::Arc; use std::time::{Duration, Instant}; @@ -13,8 +14,9 @@ use nodedb::bridge::dispatch::{BridgeRequest, BridgeResponse}; use nodedb::bridge::envelope::{ErrorCode, Priority, Request, Status}; use nodedb::data::executor::core_loop::CoreLoop; use nodedb::types::*; +use nodedb::wal::{RedoRecord, RedoSubRecord}; use nodedb_bridge::buffer::{Consumer, Producer, RingBuffer}; -use nodedb_physical::physical_plan::{KvOp, MetaOp, PhysicalPlan, VectorOp}; +use nodedb_physical::physical_plan::{KvOp, MetaOp, PhysicalPlan}; use nodedb_types::{OrdinalClock, QualifiedCollection}; // ── Helpers ───────────────────────────────────────────────────────────────── @@ -112,122 +114,144 @@ fn kv_get(coll: &str, key: &[u8]) -> PhysicalPlan { }) } -// ── Test 1: successful Calvin static-set apply ─────────────────────────────── +/// Stage `plans` as the Calvin transaction at `(epoch, 0)` and resolve it. +/// Returns the resolved redo record. +fn stage_and_resolve( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + epoch: u64, + plans: Vec, +) -> RedoRecord { + send_ok( + core, + tx, + rx, + PhysicalPlan::Meta(MetaOp::CalvinExecuteStatic { + epoch, + position: 0, + tenant_id: TenantId::new(1), + plans, + epoch_system_ms: 0, + is_group_leader: true, + versioned_reads: Vec::new(), + }), + ); + let redo = send_ok( + core, + tx, + rx, + PhysicalPlan::Meta(MetaOp::CalvinResolve { epoch, position: 0 }), + ); + RedoRecord::from_bytes(&redo).expect("decode resolved redo") +} + +/// Flush `redo` as the Calvin transaction at `(epoch, 0)` with its record at +/// `lsn`. +fn flush( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + epoch: u64, + redo: &RedoRecord, + lsn: u64, +) -> nodedb::bridge::envelope::Response { + let mut request = make_request(PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch, + position: 0, + redo: redo.to_bytes().expect("encode redo"), + collections: vec!["orders".into(), "rollback_coll".into()], + sum_targets: Vec::new(), + })); + request.wal_lsn = Some(Lsn::new(lsn)); + tx.try_push(BridgeRequest::unfloored(request)).unwrap(); + core.tick(); + rx.try_pop().unwrap().inner +} -/// A `MetaOp::TransactionBatch` dispatched directly (CalvinExecuteStatic) must -/// apply atomically. After a successful apply the written key must be readable. +/// A timeseries batch whose line names another measurement than its +/// collection: it passes validation and fails while it installs. +fn failing_install_sub_record() -> RedoSubRecord { + let lines = zerompk::to_msgpack_vec(&vec!["other_probe,host=a value=1 1".to_string()]) + .expect("encode lines"); + RedoSubRecord { + record_type: nodedb_wal::record::RecordType::TimeseriesBatch as u32, + payload: zerompk::to_msgpack_vec(&( + "timeseries", + "refusal_probe", + lines.as_slice(), + None::<&nodedb_types::sync::wire::SyncProvenance>, + "ilp-msgpack", + )) + .expect("encode ingest"), + } +} + +// ── Test 1: successful Calvin flush ────────────────────────────────────────── + +/// A committed Calvin transaction installs its redo record at the flush. The +/// written key is readable after it. #[test] fn calvin_static_apply_success() { let (mut core, mut tx, mut rx, _dir) = make_core(); - let batch_plan = PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![kv_put("orders", b"k1", b"v1")], - }); - - let resp = send_raw(&mut core, &mut tx, &mut rx, batch_plan); + let redo = stage_and_resolve( + &mut core, + &mut tx, + &mut rx, + 1, + vec![kv_put("orders", b"k1", b"v1")], + ); + let resp = flush(&mut core, &mut tx, &mut rx, 1, &redo, 10); assert_eq!( resp.status, Status::Ok, - "Calvin apply must succeed; got {:?}", + "Calvin flush must succeed; got {:?}", resp.error_code ); let payload = send_ok(&mut core, &mut tx, &mut rx, kv_get("orders", b"k1")); assert!( !payload.is_empty(), - "row must exist after successful Calvin apply" + "row must exist after a successful Calvin flush" ); } -// ── Test 2: failing sub-plan → error without RollbackFailed ────────────────── +// ── Test 2: failing install → error without RollbackFailed ─────────────────── -/// When a sub-plan in the batch fails, the batch must roll back cleanly. -/// The response must be `Status::Error` and must NOT be `RollbackFailed` -/// (the rollback itself must succeed). -/// -/// We trigger the failure via a `VectorOp::Insert` with a dimension mismatch: -/// the vector index is configured for dim=3 but we insert a dim=2 vector. +/// When a sub-record of the redo record fails while it installs, the flush +/// rolls every write back. The response is `Status::Error` and is NOT +/// `RollbackFailed` (the rollback itself succeeded), and the transaction's +/// write is gone. #[test] fn calvin_static_apply_failure_rolls_back_cleanly() { let (mut core, mut tx, mut rx, _dir) = make_core(); - // Configure and seed the vector collection (dim=3). - send_ok( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Vector(VectorOp::SetParams { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vec_coll"), - field_name: String::new(), - dim: 3, - m: 16, - ef_construction: 200, - metric: "cosine".into(), - index_type: String::new(), - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - }), - ); - // Seed one valid vector so the index knows dim=3. - send_ok( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Vector(VectorOp::Insert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vec_coll"), - vector: vec![1.0, 2.0, 3.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ); - - let plans = vec![ - // First sub-plan succeeds. - kv_put("rollback_coll", b"should_be_gone", b"present"), - // Second sub-plan fails: dim mismatch (index expects 3, we provide 2). - PhysicalPlan::Vector(VectorOp::Insert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vec_coll"), - vector: vec![1.0, 2.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::new(99), - pk_bytes: None, - provenance: None, - }), - ]; - - let resp = send_raw( + let mut redo = stage_and_resolve( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans, - }), + 2, + vec![kv_put("rollback_coll", b"should_be_gone", b"present")], ); + redo.ops.push(failing_install_sub_record()); + let resp = flush(&mut core, &mut tx, &mut rx, 2, &redo, 20); assert_eq!( resp.status, Status::Error, - "a failing sub-plan must cause the batch to fail; got {:?}", + "a failing sub-record must fail the flush; got {:?}", resp.error_code ); - // Rollback must succeed — the engine must not be in an unknown state. assert!( - !matches!( + matches!( resp.error_code.as_deref(), - Some(ErrorCode::RollbackFailed { .. }) + Some(ErrorCode::RetryableRefusal { .. }) ), - "rollback must succeed on failure path; got {:?}", + "the install rolled back every write; got {:?}", resp.error_code ); - // The first sub-plan's write ("should_be_gone") must have been rolled back. let get_resp = send_raw( &mut core, &mut tx, diff --git a/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs b/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs index 8ec994ca0..1dae5af34 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs @@ -1,33 +1,29 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Executor panic-recovery test for `MetaOp::CalvinExecuteStatic`. +//! Executor panic-recovery test for a committed Calvin transaction's flush. //! -//! Compiled only with `--features failpoints`. Tests the Calvin executor's -//! panic-recovery path: when a panic fires mid-execution inside -//! `execute_transaction_batch` (called by `execute_calvin_execute_static`), -//! the `catch_unwind` in the batch handler must: +//! Compiled only with `--features failpoints`. Tests the flush's +//! panic-recovery path: when a panic fires while the flush installs the +//! transaction's redo record (`replay::between_standalone_and_redo`, after +//! the KV writes landed), the install must: //! -//! - Catch the panic and route through the typed-rollback path. -//! - Return `Status::Error` with `ErrorCode::Internal { detail }` naming -//! the panic site. -//! - Leave the WAL in a recoverable state (all previous sub-plan writes -//! rolled back before the response is returned). -//! - Allow normal operation to resume after the fail point is disabled. +//! - Catch the panic and roll back every write the install made. +//! - Return `Status::Error` with a retryable refusal naming the panic. +//! - Leave the core in a state normal operation resumes from. //! //! ## Failure model alignment //! -//! Per the Calvin failure model: an executor panic during `CalvinExecuteStatic` -//! causes the shard to return `Status::Error`. The lock-manager invariant -//! ("locks NOT released on panic") is enforced by the scheduler layer in -//! production; the executor's contract is: +//! Per the Calvin failure model: a panic while the flush installs causes the +//! shard to return `Status::Error`. The lock-manager invariant ("locks NOT +//! released on a failed flush") is enforced by the scheduler layer, which +//! halts. The executor's contract is: //! -//! 1. Rolled-back writes are not visible after the failed batch. -//! 2. The error response carries `ErrorCode::Internal` naming the panic site. +//! 1. Rolled-back writes are not visible after the failed flush. +//! 2. The error response is a retryable refusal naming the panic. //! 3. Subsequent operations on the same `CoreLoop` succeed (no state //! corruption from the unwind). -//! 4. WAL replay correctness: a fresh `CoreLoop` opened at the same data -//! directory sees only writes that were committed before the panic batch -//! (the rolled-back writes were never durably committed). +//! 4. A fresh `CoreLoop` opened at the same data directory does not see the +//! rolled-back writes. //! //! ## Note on test scope //! @@ -40,6 +36,8 @@ #[allow(unused_imports)] use nodedb_test_support::tx_batch_helpers::*; +#[cfg(feature = "failpoints")] +use nodedb::bridge::dispatch::BridgeRequest; #[cfg(feature = "failpoints")] use nodedb::bridge::envelope::{ErrorCode, Status}; #[cfg(feature = "failpoints")] @@ -67,30 +65,18 @@ fn calvin_static(epoch: u64, plans: Vec) -> PhysicalPlan { }) } -/// Build a `MetaOp::TransactionBatch` with the given sub-plans. -#[cfg(feature = "failpoints")] -fn tx_batch(plans: Vec) -> PhysicalPlan { - PhysicalPlan::Meta(MetaOp::TransactionBatch { - plans, - txn_id: None, - }) -} - -/// Build the `MetaOp::CalvinFlush` that resolves a staged transaction to -/// commit. `CalvinExecuteStatic` stages the plans; the flush replays them -/// through `execute_transaction_batch` (where the panic fail point fires). +/// The fail point the install passes between the KV and the document arms. #[cfg(feature = "failpoints")] -fn calvin_flush(epoch: u64) -> PhysicalPlan { - PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch, position: 0 }) -} +const INSTALL_FAIL_POINT: &str = "replay::between_standalone_and_redo"; -/// Stage a static Calvin batch (validate + buffer) then flush it to base, -/// returning the flush response (where any apply-time panic surfaces). The -/// stage step must always return `Status::Ok`. +/// Stage a static Calvin transaction (validate + stage), resolve it into its +/// redo record, and flush the record at `lsn`, returning the flush response +/// (where any install-time panic surfaces). The stage and resolve steps must +/// always return `Status::Ok`. #[cfg(feature = "failpoints")] fn stage_then_flush( core: &mut nodedb::data::executor::core_loop::CoreLoop, - tx: &mut nodedb_bridge::buffer::Producer, + tx: &mut nodedb_bridge::buffer::Producer, rx: &mut nodedb_bridge::buffer::Consumer, epoch: u64, plans: Vec, @@ -99,10 +85,47 @@ fn stage_then_flush( assert_eq!( staged.status, Status::Ok, - "stage must succeed (validate + buffer, no apply); got {:?}", + "stage must succeed (validate + stage, no apply); got {:?}", staged.error_code ); - send_raw(core, tx, rx, calvin_flush(epoch)) + let resolved = send_raw( + core, + tx, + rx, + PhysicalPlan::Meta(MetaOp::CalvinResolve { epoch, position: 0 }), + ); + assert_eq!( + resolved.status, + Status::Ok, + "resolve must succeed; got {:?}", + resolved.error_code + ); + let mut request = make_request(PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch, + position: 0, + redo: resolved.payload.to_vec(), + collections: Vec::new(), + sum_targets: Vec::new(), + })); + request.wal_lsn = Some(nodedb::types::Lsn::new(epoch * 10)); + tx.try_push(BridgeRequest::unfloored(request)).unwrap(); + core.tick(); + rx.try_pop().unwrap().inner +} + +/// Assert `resp` is the retryable refusal a panic mid-install answers with. +#[cfg(feature = "failpoints")] +fn assert_panic_refusal(resp: &nodedb::bridge::envelope::Response) { + assert_eq!(resp.status, Status::Error, "got {:?}", resp.error_code); + match resp.error_code.as_deref() { + Some(ErrorCode::RetryableRefusal { reason }) => { + assert!( + reason.contains("panic"), + "the refusal must name the panic: {reason}" + ); + } + other => panic!("expected ErrorCode::RetryableRefusal, got {other:?}"), + } } /// Build a KV Put plan for the given collection. @@ -130,22 +153,21 @@ fn kv_get_in(coll: &str, key: &[u8]) -> PhysicalPlan { }) } -// ── Test 1: CalvinExecuteStatic panic caught, typed response returned ───────── +// ── Test 1: install panic caught, typed response returned ───────────────────── -/// Panic injected between sub-applies inside a `CalvinExecuteStatic` batch. +/// Panic injected while the flush installs the transaction's redo record. /// -/// The batch has two sub-plans: the first KV put succeeds, then the fail -/// point fires. The handler must catch the unwind and return a typed -/// `ErrorCode::Internal` response. The first sub-plan's write must be rolled -/// back before the response is returned (all-or-nothing guarantee). +/// The record carries two KV puts, which land before the fail point fires. +/// The install must catch the unwind, roll both writes back, and answer with +/// a retryable refusal naming the panic. #[cfg(feature = "failpoints")] #[test] fn calvin_static_panic_returns_internal_error() { let (mut core, mut tx, mut rx, _dir) = make_core(); - let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); + let _guard = FailGuard::install(INSTALL_FAIL_POINT, FailAction::Panic); - // Stage succeeds (no apply); the panic fires when the flush replays the plans. + // Stage and resolve write nothing; the panic fires in the flush's install. let resp = stage_then_flush( &mut core, &mut tx, @@ -156,27 +178,12 @@ fn calvin_static_panic_returns_internal_error() { kv_put_in("orders", b"panic_key2", b"should_not_persist"), ], ); - - assert_eq!( - resp.status, - Status::Error, - "expected Status::Error after Calvin executor panic; got {:?}", - resp.status - ); - match resp.error_code.as_deref() { - Some(ErrorCode::Internal { detail }) => { - assert!( - detail.contains("panic in sub-apply"), - "error detail must name the panic site: {detail}" - ); - } - other => panic!("expected ErrorCode::Internal, got {other:?}"), - } + assert_panic_refusal(&resp); } // ── Test 2: rolled-back writes not visible after Calvin panic ───────────────── -/// After a Calvin executor panic, writes from the failed batch must not be +/// After a Calvin executor panic, writes from the failed flush must not be /// visible. The CoreLoop remains functional for subsequent requests. #[cfg(feature = "failpoints")] #[test] @@ -188,15 +195,11 @@ fn calvin_static_panic_rollback_not_visible() { &mut core, &mut tx, &mut rx, - tx_batch(vec![kv_put_in( - "orders", - b"committed_key", - b"committed_val", - )]), + kv_put_in("orders", b"committed_key", b"committed_val"), ); // Inject a panic on the second sub-apply. - let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); + let _guard = FailGuard::install(INSTALL_FAIL_POINT, FailAction::Panic); let resp = stage_then_flush( &mut core, @@ -211,7 +214,7 @@ fn calvin_static_panic_rollback_not_visible() { assert_eq!( resp.status, Status::Error, - "panic batch must return Error; got {:?}", + "a panicking flush must return Error; got {:?}", resp.status ); @@ -253,17 +256,17 @@ fn calvin_static_panic_rollback_not_visible() { // ── Test 3: normal operation resumes after fail point disabled ──────────────── -/// After clearing the fail point, `CalvinExecuteStatic` batches must commit +/// After clearing the fail point, Calvin transactions must commit /// successfully. This confirms there is no state corruption from the earlier -/// panicked batch. +/// panicking flush. #[cfg(feature = "failpoints")] #[test] fn calvin_static_normal_operation_resumes_after_panic() { let (mut core, mut tx, mut rx, _dir) = make_core(); - // Trigger a panic batch. + // Trigger a panicking flush. { - let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); + let _guard = FailGuard::install(INSTALL_FAIL_POINT, FailAction::Panic); let resp = stage_then_flush( &mut core, &mut tx, @@ -277,13 +280,13 @@ fn calvin_static_normal_operation_resumes_after_panic() { assert_eq!( resp.status, Status::Error, - "panic batch must return Error; got {:?}", + "a panicking flush must return Error; got {:?}", resp.status ); // Guard drops here, clearing the fail point. } - // Normal staged Calvin batch after the fail point is cleared: stage + flush. + // A normal Calvin transaction after the fail point is cleared. let success_resp = stage_then_flush( &mut core, &mut tx, @@ -294,7 +297,7 @@ fn calvin_static_normal_operation_resumes_after_panic() { assert_eq!( success_resp.status, Status::Ok, - "Calvin batch after fail-point clear must succeed; got {:?}", + "a Calvin flush after the fail point is cleared must succeed; got {:?}", success_resp.error_code ); @@ -319,13 +322,14 @@ fn calvin_static_normal_operation_resumes_after_panic() { // ── Test 4: WAL replay correctness — fresh CoreLoop sees only committed data ── -/// After a panicked Calvin batch, create a fresh `CoreLoop` at the same data +/// After a panicking Calvin flush, create a fresh `CoreLoop` at the same data /// directory. The fresh core must see only data that was committed before the -/// panic batch — the rolled-back writes must not appear after replay. +/// panicking flush — the rolled-back writes must not appear after replay. /// -/// This verifies the WAL-recoverability invariant: the panic batch's sub-plan -/// writes were rolled back before the response was returned, so they were never -/// durably committed to any WAL record. A fresh core therefore starts clean. +/// The install rolled its writes back before the response returned, so a +/// fresh core opened over the same directory holds none of them. The redo +/// record itself lives in the WAL the scheduler appends, which this +/// core-level test does not write. #[cfg(feature = "failpoints")] #[test] fn calvin_static_replay_sees_only_committed_data() { @@ -354,7 +358,7 @@ fn calvin_static_replay_sees_only_committed_data() { (core, req_tx, resp_rx) }; - // Commit a reference write before the panic batch. + // Commit a reference write before the panicking flush. { use nodedb::bridge::dispatch::BridgeRequest; use nodedb::bridge::envelope::{Priority, Request}; @@ -382,10 +386,12 @@ fn calvin_static_replay_sees_only_committed_data() { admission: nodedb::bridge::envelope::Admission::Admitted, }; - // Commit a value before the panic batch. - tx.try_push(BridgeRequest::unfloored(make_req(tx_batch(vec![ - kv_put_in("replay_coll", b"pre_commit", b"alive"), - ])))) + // Commit a value before the panicking flush. + tx.try_push(BridgeRequest::unfloored(make_req(kv_put_in( + "replay_coll", + b"pre_commit", + b"alive", + )))) .unwrap(); core.tick(); let pre_resp = rx.try_pop().unwrap().inner; @@ -396,37 +402,20 @@ fn calvin_static_replay_sees_only_committed_data() { pre_resp.error_code ); - // Panic batch — writes must not persist. Stage first (no apply, always - // Ok), then flush (where the panic fires during the replay). - let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); - - tx.try_push(BridgeRequest::unfloored(make_req(calvin_static( + // Panicking install — writes must not persist. Stage and resolve + // write nothing; the panic fires while the flush installs. + let _guard = FailGuard::install(INSTALL_FAIL_POINT, FailAction::Panic); + let panic_resp = stage_then_flush( + &mut core, + &mut tx, + &mut rx, 1, vec![ kv_put_in("replay_coll", b"should_not_exist", b"gone"), kv_put_in("replay_coll", b"should_not_exist2", b"gone"), ], - )))) - .unwrap(); - core.tick(); - let stage_resp = rx.try_pop().unwrap().inner; - assert_eq!( - stage_resp.status, - Status::Ok, - "stage must succeed (validate + buffer); got {:?}", - stage_resp.error_code - ); - - tx.try_push(BridgeRequest::unfloored(make_req(calvin_flush(1)))) - .unwrap(); - core.tick(); - let panic_resp = rx.try_pop().unwrap().inner; - assert_eq!( - panic_resp.status, - Status::Error, - "flush of panic batch must return Error; got {:?}", - panic_resp.status ); + assert_panic_refusal(&panic_resp); // Guard drops, clearing the fail point. } @@ -485,7 +474,7 @@ fn calvin_static_replay_sees_only_committed_data() { // on a single-CoreLoop reopen. WAL-driven KV state recovery is a property // of the cluster apply path (replicated entries → applier → engine), not // of CoreLoop::open. The meaningful invariant exercised below is that the - // rolled-back writes from the panicked batch do NOT appear, which holds + // rolled-back writes from the panicking flush do NOT appear, which holds // trivially under empty-replay state and confirms the panic-rollback path // never let the bad writes reach durable storage. diff --git a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs index ad6f9daa9..2446966e3 100644 --- a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs +++ b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs @@ -3,9 +3,10 @@ //! Staged Calvin apply on the Data Plane: `MetaOp::CalvinExecuteStatic` //! VALIDATES + STAGES a transaction's write plans into the commit-pending //! buffer WITHOUT mutating base, returning the local commit vote on -//! `read_set_valid`. A subsequent `MetaOp::CalvinFlush` replays the staged -//! plans to base (making the write visible), or `MetaOp::CalvinDrop` discards -//! them (leaving base unchanged). +//! `read_set_valid`. `MetaOp::CalvinResolve` then resolves the staged plans +//! into the transaction's redo record, and `MetaOp::CalvinFlush` installs it +//! at its LSN (making the write visible), or `MetaOp::CalvinDrop` discards +//! the staged state (leaving base unchanged). //! //! These drive a `CoreLoop` directly through the SPSC ring so the atomicity //! seam is observed without any scheduler timing: nothing a stage writes is @@ -86,6 +87,102 @@ fn send( rx.try_pop().unwrap().inner } +/// Resolve the transaction staged at `(epoch, 0)` on `vshard` and build the +/// flush that installs its redo record. +fn resolved_flush( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + epoch: u64, + vshard: u32, +) -> PhysicalPlan { + let resolved = send( + core, + tx, + rx, + PhysicalPlan::Meta(MetaOp::CalvinResolve { epoch, position: 0 }), + vshard, + None, + ); + assert_eq!( + resolved.status, + Status::Ok, + "resolve must succeed: {resolved:?}" + ); + PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch, + position: 0, + redo: resolved.payload.to_vec(), + collections: Vec::new(), + sum_targets: Vec::new(), + }) +} + +/// A write committed through the Calvin path to seed a write version. +struct CalvinSeed<'a> { + epoch: u64, + vshard: u32, + /// The collection `plans` write; the install records its floor at `lsn`. + collection: &'a str, + plans: Vec, + lsn: u64, +} + +/// Commit `seed.plans` as the Calvin transaction at `(seed.epoch, 0)` on +/// `seed.vshard` and install its redo record at `seed.lsn`: the path a +/// committed multi-shard write takes. The install records each written key +/// and the collection floor at that LSN. +fn commit_calvin( + core: &mut CoreLoop, + tx: &mut Producer, + rx: &mut Consumer, + seed: CalvinSeed<'_>, +) -> Response { + let staged = send( + core, + tx, + rx, + stage_static(seed.epoch, 0, seed.plans, Vec::new()), + seed.vshard, + None, + ); + assert_eq!( + staged.status, + Status::Ok, + "seed stage must succeed: {staged:?}" + ); + let resolved = send( + core, + tx, + rx, + PhysicalPlan::Meta(MetaOp::CalvinResolve { + epoch: seed.epoch, + position: 0, + }), + seed.vshard, + None, + ); + assert_eq!( + resolved.status, + Status::Ok, + "seed resolve must succeed: {resolved:?}" + ); + send( + core, + tx, + rx, + PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch: seed.epoch, + position: 0, + redo: resolved.payload.to_vec(), + collections: vec![seed.collection.to_string()], + sum_targets: Vec::new(), + }), + seed.vshard, + Some(Lsn::new(seed.lsn)), + ) +} + fn kv_put(coll: &str, key: &[u8], value: &[u8]) -> PhysicalPlan { PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, coll), @@ -183,17 +280,15 @@ fn flush_makes_staged_calvin_write_visible() { "staged write must NOT be visible before flush; got {before:?}" ); - // Flush replays the staged plans to base. + // The flush installs the resolved redo record to base. + let flush_plan = resolved_flush(&mut core, &mut tx, &mut rx, 6, 0); let flush = send( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::CalvinFlush { - epoch: 6, - position: 0, - }), + flush_plan, 0, - None, + Some(Lsn::new(60)), ); assert_eq!(flush.status, Status::Ok, "flush must succeed: {flush:?}"); @@ -225,16 +320,17 @@ fn drop_discards_invalid_staged_calvin_write() { // Seed a committed write to `dropcoll` at LSN 100 so its collection write // version floor is 100. The seed carries a WAL LSN so the version records. - let seed = send( + let seed = commit_calvin( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, + CalvinSeed { + epoch: 1, + vshard: 0, + collection: "dropcoll", plans: vec![kv_put("dropcoll", b"seed", b"v")], - }), - 0, - Some(Lsn::new(100)), + lsn: 100, + }, ); assert_eq!(seed.status, Status::Ok, "seed write must commit: {seed:?}"); @@ -328,16 +424,17 @@ fn point_read_at_write_lsn_commits_and_flush_applies() { let (mut core, mut tx, mut rx, _dir) = make_core(); // Seed a committed write to key `pk` in `pointcoll` at LSN 10. - let seed = send( + let seed = commit_calvin( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, + CalvinSeed { + epoch: 1, + vshard: 0, + collection: "pointcoll", plans: vec![kv_put("pointcoll", b"pk", b"v1")], - }), - 0, - Some(Lsn::new(10)), + lsn: 10, + }, ); assert_eq!(seed.status, Status::Ok, "seed write must commit: {seed:?}"); @@ -373,16 +470,14 @@ fn point_read_at_write_lsn_commits_and_flush_applies() { "a read at or after the last write LSN must be current -> commit vote" ); + let flush_plan = resolved_flush(&mut core, &mut tx, &mut rx, 8, point_vshard); let flush = send( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::CalvinFlush { - epoch: 8, - position: 0, - }), + flush_plan, point_vshard, - None, + Some(Lsn::new(80)), ); assert_eq!(flush.status, Status::Ok, "flush must succeed: {flush:?}"); @@ -411,16 +506,17 @@ fn stale_point_read_of_kv_key_aborts_stage_and_drop_discards() { let (mut core, mut tx, mut rx, _dir) = make_core(); // Seed a committed write to key `pk` in `stalecoll` at LSN 10. - let seed = send( + let seed = commit_calvin( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, + CalvinSeed { + epoch: 1, + vshard: 0, + collection: "stalecoll", plans: vec![kv_put("stalecoll", b"pk", b"v1")], - }), - 0, - Some(Lsn::new(10)), + lsn: 10, + }, ); assert_eq!(seed.status, Status::Ok, "seed write must commit: {seed:?}"); @@ -523,12 +619,14 @@ fn absent_kv_key_phantom_insert_causes_abort() { }; // Concurrently, the exact same key is inserted and commits at LSN 8. - let insert = send( + let insert = commit_calvin( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, + CalvinSeed { + epoch: 1, + vshard: phantom_vshard, + collection: "phantomkv", plans: vec![PhysicalPlan::Kv(KvOp::Insert { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "phantomkv"), key: b"newkey".to_vec(), @@ -538,9 +636,8 @@ fn absent_kv_key_phantom_insert_causes_abort() { returning: None, rls_filters: Vec::new(), })], - }), - phantom_vshard, - Some(Lsn::new(8)), + lsn: 8, + }, ); assert_eq!( insert.status, @@ -602,20 +699,21 @@ fn absent_document_phantom_insert_is_caught() { // at LSN 8. Its collection floor advance (phantomdocs -> 8) is what the // predicate read validates against. const NEWLY_ALLOCATED_SURROGATE: u32 = 42; - let insert = send( + let insert = commit_calvin( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, + CalvinSeed { + epoch: 1, + vshard: doc_vshard, + collection: "phantomdocs", plans: vec![doc_insert( "phantomdocs", "the-doc-id", NEWLY_ALLOCATED_SURROGATE, )], - }), - doc_vshard, - Some(Lsn::new(8)), + lsn: 8, + }, ); assert_eq!( insert.status, @@ -671,16 +769,17 @@ fn absent_document_read_without_matching_insert_still_commits() { // A concurrent insert into a DIFFERENT collection commits at LSN 8. It // advances only "othercoll"'s floor; phantomdocs is untouched. const UNRELATED_SURROGATE: u32 = 42; - let insert = send( + let insert = commit_calvin( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, + CalvinSeed { + epoch: 1, + vshard: doc_vshard, + collection: "othercoll", plans: vec![doc_insert("othercoll", "other-id", UNRELATED_SURROGATE)], - }), - doc_vshard, - Some(Lsn::new(8)), + lsn: 8, + }, ); assert_eq!( insert.status, @@ -753,14 +852,15 @@ fn already_ordered_stage_and_flush_run_past_their_deadline() { assert_eq!(staged.status, Status::Ok, "late stage must run: {staged:?}"); assert_eq!(staged.read_set_valid, Some(true)); + let flush_plan = resolved_flush(&mut core, &mut tx, &mut rx, 9, 0); let flush = send_request( &mut core, &mut tx, &mut rx, - already_ordered(PhysicalPlan::Meta(MetaOp::CalvinFlush { - epoch: 9, - position: 0, - })), + Request { + wal_lsn: Some(Lsn::new(90)), + ..already_ordered(flush_plan) + }, ); assert_eq!(flush.status, Status::Ok, "late flush must run: {flush:?}"); diff --git a/nodedb/tests/inproc/cases/core_loop.rs b/nodedb/tests/inproc/cases/core_loop.rs index 749294d68..ae5716978 100644 --- a/nodedb/tests/inproc/cases/core_loop.rs +++ b/nodedb/tests/inproc/cases/core_loop.rs @@ -102,7 +102,5 @@ mod test_transaction_matrix; mod test_transaction_matrix_helpers; #[path = "executor_tests/test_transaction_matrix_kv.rs"] mod test_transaction_matrix_kv; -#[path = "executor_tests/test_transaction_matrix_side_effects.rs"] -mod test_transaction_matrix_side_effects; #[path = "executor_tests/test_vector.rs"] mod test_vector; diff --git a/nodedb/tests/inproc/cases/executor_tests/test_conditional_update.rs b/nodedb/tests/inproc/cases/executor_tests/test_conditional_update.rs index bc81c1b8c..79a87085f 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_conditional_update.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_conditional_update.rs @@ -6,12 +6,13 @@ //! - Affected row count is correctly returned for bulk updates //! - Conditional UPDATE WHERE with predicates (stock >= N) works atomically //! - RETURNING flag returns post-update documents -//! - TransactionBatch does not auto-abort on 0-row conditional update +//! - A committed transaction does not abort on a 0-row conditional update //! - PointUpdate returns affected count use nodedb::bridge::envelope::Status; use nodedb::bridge::scan_filter::{FilterOp, ScanFilter}; -use nodedb_physical::physical_plan::{DocumentOp, MetaOp, PhysicalPlan, UpdateValue}; +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan, UpdateValue}; +use nodedb_test_support::tx_batch_helpers::commit_plans; use super::helpers::*; @@ -412,14 +413,14 @@ fn point_update_returning_returns_updated_document() { } #[test] -fn transaction_batch_does_not_abort_on_zero_row_update() { +fn a_transaction_does_not_abort_on_zero_row_update() { let (mut core, mut tx, mut rx, _dir) = make_core(); insert_product(&mut core, &mut tx, &mut rx, "t1", 1); insert_product(&mut core, &mut tx, &mut rx, "t2", 0); // Transaction: first update matches (stock >= 1), second doesn't (stock >= 100). - // Batch should NOT auto-abort on 0-row update. + // The transaction must NOT abort on the 0-row update. let filters_match = zerompk::to_msgpack_vec(&vec![filter( "stock", FilterOp::Gte, @@ -434,55 +435,53 @@ fn transaction_batch_does_not_abort_on_zero_row_update() { )]) .unwrap(); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Document(DocumentOp::BulkUpdate { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "products", + vec![ + PhysicalPlan::Document(DocumentOp::BulkUpdate { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "products", + ), + filters: filters_match, + updates: vec![( + "stock".to_string(), + UpdateValue::Literal( + nodedb_types::json_to_msgpack(&serde_json::json!(0)).unwrap(), ), - filters: filters_match, - updates: vec![( - "stock".to_string(), - UpdateValue::Literal( - nodedb_types::json_to_msgpack(&serde_json::json!(0)).unwrap(), - ), - )], - returning: None, - ollp_predicted_surrogates: None, - ollp_predicted_edges: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - resolved_sum_targets: Vec::new(), - declared_primary_key: None, - }), - PhysicalPlan::Document(DocumentOp::BulkUpdate { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "products", + )], + returning: None, + ollp_predicted_surrogates: None, + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }), + PhysicalPlan::Document(DocumentOp::BulkUpdate { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "products", + ), + filters: filters_nomatch, + updates: vec![( + "stock".to_string(), + UpdateValue::Literal( + nodedb_types::json_to_msgpack(&serde_json::json!(999)).unwrap(), ), - filters: filters_nomatch, - updates: vec![( - "stock".to_string(), - UpdateValue::Literal( - nodedb_types::json_to_msgpack(&serde_json::json!(999)).unwrap(), - ), - )], - returning: None, - ollp_predicted_surrogates: None, - ollp_predicted_edges: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - resolved_sum_targets: Vec::new(), - declared_primary_key: None, - }), - ], - }), + )], + returning: None, + ollp_predicted_surrogates: None, + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }), + ], + 10, ); assert_eq!( diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction.rs index 1e10785d2..fb5273e2c 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_transaction.rs @@ -1,55 +1,55 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Transaction batch execution over the document engine: atomic commit, -//! response identity, and rollback on a failing operation. +//! Committed transactions over the document engine: atomic commit, the +//! refusal of a plan batch, and a refused transaction that commits nothing. //! -//! Batches that span the graph and vector engines live in +//! Transactions that span the graph and vector engines live in //! `test_transaction_cross_engine`. +use nodedb::bridge::envelope::ErrorCode; use nodedb::bridge::envelope::Status; -use nodedb_physical::physical_plan::{DocumentOp, MetaOp, PhysicalPlan, VectorOp}; +use nodedb_physical::physical_plan::{DocumentOp, MetaOp, PhysicalPlan}; +use nodedb_test_support::tx_batch_helpers::{commit_plans, with_unique_refusal}; use super::helpers::*; #[test] -fn transaction_batch_commits_atomically() { +fn a_transaction_commits_atomically() { let (mut core, mut tx, mut rx, _dir) = make_core(); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "docs", - ), - document_id: "d1".into(), - value: b"{\"name\":\"alice\"}".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "docs", - ), - document_id: "d2".into(), - value: b"{\"name\":\"bob\"}".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - ], - }), + vec![ + PhysicalPlan::Document(DocumentOp::PointPut { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "docs", + ), + document_id: "d1".into(), + value: b"{\"name\":\"alice\"}".to_vec(), + surrogate: nodedb_types::Surrogate::new(1), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }), + PhysicalPlan::Document(DocumentOp::PointPut { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "docs", + ), + document_id: "d2".into(), + value: b"{\"name\":\"bob\"}".to_vec(), + surrogate: nodedb_types::Surrogate::new(2), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }), + ], + 10, ); assert_eq!(resp.status, Status::Ok); @@ -67,7 +67,7 @@ fn transaction_batch_commits_atomically() { rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), pk_bytes: Vec::new(), }), ); @@ -86,15 +86,17 @@ fn transaction_batch_commits_atomically() { rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(2), pk_bytes: Vec::new(), }), ); assert_eq!(r2.status, Status::Ok); } +/// A plan batch carries no redo record restart replay reads, so an Origin +/// core refuses it, answering under the request's own id. #[test] -fn transaction_batch_response_uses_outer_request_id() { +fn a_transaction_batch_is_refused_on_an_origin_core() { let (mut core, mut tx, mut rx, _dir) = make_core(); tx.try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored( @@ -122,12 +124,16 @@ fn transaction_batch_response_uses_outer_request_id() { core.tick(); let resp = rx.try_pop().unwrap().inner; - assert_eq!(resp.status, Status::Ok); + assert_eq!(resp.status, Status::Error); + assert!(matches!( + resp.error_code.as_deref(), + Some(ErrorCode::Unsupported { .. }) + )); assert_eq!(resp.request_id, nodedb::types::RequestId::new(42)); } #[test] -fn transaction_batch_rollback_on_failure() { +fn a_refused_transaction_commits_nothing() { let (mut core, mut tx, mut rx, _dir) = make_core(); // Pre-insert d1 via SPSC. @@ -150,86 +156,29 @@ fn transaction_batch_rollback_on_failure() { }), ); - // Create a vector index via SetVectorParams so we have a known dimension. - send_ok( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Vector(VectorOp::SetParams { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "emb", - ), - field_name: String::new(), - dim: 3, - m: 16, - ef_construction: 200, - metric: "cosine".into(), - index_type: String::new(), - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - }), - ); - // Insert one vector to create the index with dim=3. - send_ok( + // Transaction: overwrite d1, then a refused insert. + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Vector(VectorOp::Insert { + with_unique_refusal(vec![PhysicalPlan::Document(DocumentOp::PointPut { collection: nodedb_types::QualifiedCollection::new( nodedb_types::DatabaseId::DEFAULT, - "emb", + "docs", ), - vector: vec![1.0, 2.0, 3.0], - dim: 3, - field_name: String::new(), + document_id: "d1".into(), + value: b"{\"name\":\"modified\"}".to_vec(), surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ); - - // TransactionBatch: overwrite d1, then fail with wrong dimension. - let resp = send_raw( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "docs", - ), - document_id: "d1".into(), - value: b"{\"name\":\"modified\"}".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - // Dimension mismatch: index is dim=3 but vector has 2 elements. - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "emb", - ), - vector: vec![1.0, 2.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ], - }), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + })]), + 20, ); assert_eq!(resp.status, Status::Error); - // d1 should be rolled back to original value. + // d1 keeps its original value. let r = send_raw( &mut core, &mut tx, diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction_cross_engine.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction_cross_engine.rs index 5edc222e2..b6950cd9b 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction_cross_engine.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_transaction_cross_engine.rs @@ -1,13 +1,15 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Transaction batches spanning more than one engine. +//! Committed transactions spanning more than one engine. //! -//! A graph edge or a vector node written inside a batch must commit and -//! roll back with the document writes beside it — a surviving edge or -//! HNSW node after a failed batch is state no read path can account for. +//! A graph edge written inside a transaction must commit with the document +//! writes beside it, and a refused transaction must commit none of them — a +//! surviving edge after a refused transaction is state no read path can +//! account for. use nodedb::bridge::envelope::Status; -use nodedb_physical::physical_plan::{DocumentOp, GraphOp, MetaOp, PhysicalPlan, VectorOp}; +use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan}; +use nodedb_test_support::tx_batch_helpers::{commit_plans, with_unique_refusal}; use super::helpers::*; @@ -54,40 +56,38 @@ fn transaction_edge_put_committed() { ); // Transaction: insert doc + edge. - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "nodes", - ), - document_id: "carol".into(), - value: b"{\"name\":\"carol\"}".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - PhysicalPlan::Graph(GraphOp::EdgePut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "col", - ), - src_id: "alice".into(), - label: "KNOWS".into(), - dst_id: "bob".into(), - properties: Vec::new(), - src_surrogate: nodedb_types::Surrogate::ZERO, - dst_surrogate: nodedb_types::Surrogate::ZERO, - }), - ], - }), + vec![ + PhysicalPlan::Document(DocumentOp::PointPut { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "nodes", + ), + document_id: "carol".into(), + value: b"{\"name\":\"carol\"}".to_vec(), + surrogate: nodedb_types::Surrogate::ZERO, + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }), + PhysicalPlan::Graph(GraphOp::EdgePut { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "col", + ), + src_id: "alice".into(), + label: "KNOWS".into(), + dst_id: "bob".into(), + properties: Vec::new(), + src_surrogate: nodedb_types::Surrogate::ZERO, + dst_surrogate: nodedb_types::Surrogate::ZERO, + }), + ], + 10, ); assert_eq!(resp.status, Status::Ok); @@ -109,7 +109,7 @@ fn transaction_edge_put_committed() { } #[test] -fn transaction_edge_put_rolled_back_on_failure() { +fn a_refused_transaction_leaves_no_edge() { let (mut core, mut tx, mut rx, _dir) = make_core(); // Pre-insert nodes. @@ -150,84 +150,28 @@ fn transaction_edge_put_rolled_back_on_failure() { }), ); - // Set up vector index with dim=3. - send_ok( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Vector(VectorOp::SetParams { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "emb", - ), - field_name: String::new(), - dim: 3, - m: 16, - ef_construction: 200, - metric: "cosine".into(), - index_type: String::new(), - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - }), - ); - send_ok( + // Transaction: edge put, then a refused insert. + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Vector(VectorOp::Insert { + with_unique_refusal(vec![PhysicalPlan::Graph(GraphOp::EdgePut { collection: nodedb_types::QualifiedCollection::new( nodedb_types::DatabaseId::DEFAULT, - "emb", + "col", ), - vector: vec![1.0, 2.0, 3.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ); - - // Transaction: edge put + vector with wrong dimension (triggers rollback). - let resp = send_raw( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Graph(GraphOp::EdgePut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "col", - ), - src_id: "alice".into(), - label: "KNOWS".into(), - dst_id: "bob".into(), - properties: Vec::new(), - src_surrogate: nodedb_types::Surrogate::ZERO, - dst_surrogate: nodedb_types::Surrogate::ZERO, - }), - // Dimension mismatch: index is dim=3 but vector has 2 elements. - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "emb", - ), - vector: vec![1.0, 2.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ], - }), + src_id: "alice".into(), + label: "KNOWS".into(), + dst_id: "bob".into(), + properties: Vec::new(), + src_surrogate: nodedb_types::Surrogate::ZERO, + dst_surrogate: nodedb_types::Surrogate::ZERO, + })]), + 20, ); assert_eq!(resp.status, Status::Error); - // Verify edge was rolled back: neighbors should be empty. + // The edge never landed: neighbors are empty. let n = send_raw( &mut core, &mut tx, @@ -247,13 +191,13 @@ fn transaction_edge_put_rolled_back_on_failure() { // Empty result = msgpack empty array [0x90] or very short payload. assert!( payload.len() <= 3, - "edge should have been rolled back, but payload len: {}", + "the edge must not land, but payload len: {}", payload.len() ); } #[test] -fn transaction_mixed_doc_edge_vector_rollback() { +fn a_refused_transaction_leaves_neither_doc_nor_edge() { let (mut core, mut tx, mut rx, _dir) = make_core(); // Pre-insert nodes. @@ -294,97 +238,43 @@ fn transaction_mixed_doc_edge_vector_rollback() { }), ); - // Set up vector index. - send_ok( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Vector(VectorOp::SetParams { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "vec", - ), - field_name: String::new(), - dim: 3, - m: 16, - ef_construction: 200, - metric: "cosine".into(), - index_type: String::new(), - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - }), - ); - send_ok( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "vec", - ), - vector: vec![1.0, 2.0, 3.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ); - - // Transaction: doc update + edge put + vector insert (wrong dim) — all should rollback. - let resp = send_raw( + // Transaction: doc update + edge put, then a refused insert. Neither lands. + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "nodes", - ), - document_id: "n1".into(), - value: b"modified_n1".to_vec(), - surrogate: nodedb_types::Surrogate::new(1), - pk_bytes: b"n1".to_vec(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - PhysicalPlan::Graph(GraphOp::EdgePut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "col", - ), - src_id: "n1".into(), - label: "LINKED".into(), - dst_id: "n2".into(), - properties: Vec::new(), - src_surrogate: nodedb_types::Surrogate::ZERO, - dst_surrogate: nodedb_types::Surrogate::ZERO, - }), - // Fail: dim mismatch. - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "vec", - ), - vector: vec![1.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }), - ], - }), + with_unique_refusal(vec![ + PhysicalPlan::Document(DocumentOp::PointPut { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "nodes", + ), + document_id: "n1".into(), + value: b"modified_n1".to_vec(), + surrogate: nodedb_types::Surrogate::new(1), + pk_bytes: b"n1".to_vec(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }), + PhysicalPlan::Graph(GraphOp::EdgePut { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "col", + ), + src_id: "n1".into(), + label: "LINKED".into(), + dst_id: "n2".into(), + properties: Vec::new(), + src_surrogate: nodedb_types::Surrogate::ZERO, + dst_surrogate: nodedb_types::Surrogate::ZERO, + }), + ]), + 30, ); assert_eq!(resp.status, Status::Error); - // Document should be rolled back to original. + // The document keeps its original value. let r = send_raw( &mut core, &mut tx, @@ -405,7 +295,7 @@ fn transaction_mixed_doc_edge_vector_rollback() { assert_eq!(r.status, Status::Ok); assert_eq!(&*r.payload, b"original_n1"); - // Edge should be rolled back (no neighbors). + // The edge never landed (no neighbors). let n = send_raw( &mut core, &mut tx, @@ -420,5 +310,5 @@ fn transaction_mixed_doc_edge_vector_rollback() { ); assert_eq!(n.status, Status::Ok); // Empty result = msgpack empty array [0x90] or very short payload. - assert!(n.payload.len() <= 3, "edge should have been rolled back"); + assert!(n.payload.len() <= 3, "the edge must not land"); } diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix.rs index 9e39eadb3..aad15ccbd 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix.rs @@ -3,51 +3,52 @@ //! Cross-engine transaction rollback matrix: engine-pair failures. //! //! For every pair of write-trackable engines that can legally appear in one -//! `TransactionBatch` (Document, Vector, Graph, CRDT), a deterministic -//! failure on the second operation must fully roll back the first. +//! committed transaction (Document, Vector, Graph, CRDT), a deterministic +//! refusal after the first operation must leave none of it applied. //! //! Test structure (per pair): //! 1. Pre-condition: write a known state for the first engine. -//! 2. TransactionBatch: valid first op (overwrites it) + failing second op. -//! 3. Assert: the first op was rolled back; state matches the pre-condition. +//! 2. Transaction: valid first op (overwrites it) + a refused op. +//! 3. Assert: the first op is not applied; state matches the pre-condition. //! //! Adding an engine pair: add one test here following the existing pattern. //! Side-effect rollback (FTS, spatial) lives in //! `test_transaction_matrix_side_effects`. use nodedb::bridge::envelope::{ErrorCode, Status}; -use nodedb_physical::physical_plan::{CrdtOp, DocumentOp, MetaOp, PhysicalPlan, VectorOp}; +use nodedb_physical::physical_plan::{CrdtOp, DocumentOp, PhysicalPlan}; +use nodedb_test_support::tx_batch_helpers::{ + commit_plans, vector_direct_insert, with_unique_refusal, +}; use nodedb_types::{DatabaseId, QualifiedCollection}; use super::helpers::*; use super::test_transaction_matrix_helpers::*; // --------------------------------------------------------------------------- -// Pair: Document (first) × Vector (second) — vector fails +// Document, then a refused insert // --------------------------------------------------------------------------- #[test] -fn rollback_matrix_doc_then_vector_fail() { +fn rollback_matrix_doc_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); // Pre-condition: "doc1" = "original". send_ok(&mut core, &mut tx, &mut rx, doc_put("docs", b"original")); - // Seed vector index dim=3. - send_ok(&mut core, &mut tx, &mut rx, vector_set_params("vec")); - send_ok(&mut core, &mut tx, &mut rx, vector_seed("vec")); - - // TransactionBatch: overwrite doc1 + failing vector insert. - let resp = send_raw( + // Transaction: overwrite doc1, then a refused insert. + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![doc_put("docs", b"modified"), vector_fail("vec")], - }), + with_unique_refusal(vec![doc_put("docs", b"modified")]), + 10, + ); + assert_eq!( + resp.status, + Status::Error, + "the transaction must be refused" ); - assert_eq!(resp.status, Status::Error, "batch should fail"); // doc1 must be rolled back to "original". let r = send_raw(&mut core, &mut tx, &mut rx, doc_get("docs")); @@ -63,48 +64,26 @@ fn rollback_matrix_doc_then_vector_fail() { fn rollback_matrix_vector_then_doc_fail() { let (mut core, mut tx, mut rx, _dir) = make_core(); - // Seed vector index dim=3. - send_ok(&mut core, &mut tx, &mut rx, vector_set_params("vec")); - send_ok(&mut core, &mut tx, &mut rx, vector_seed("vec")); - // Pre-condition for doc conflict: doc1 already exists. send_ok(&mut core, &mut tx, &mut rx, doc_put("docs", b"preexisting")); - // Record current vector count before batch (index length is side-effect-visible - // only via a successful insert, so we verify rollback via a fresh insert). - let count_before: usize = { - // Insert a known vector and check it lands at index 1 (after the seeded one). - // We track by checking a subsequent batch that inserts and then rolls back. - // For simplicity: the batch's vector insert should be soft-deleted on rollback, - // meaning a later valid insert lands at the same logical slot. We just assert - // the batch fails. - 1 // placeholder; main assertion is batch failure + doc unchanged - }; - let _ = count_before; - - // TransactionBatch: valid vector insert + PointInsert that conflicts. - let v_plan = PhysicalPlan::Vector(VectorOp::Insert { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "vec"), - vector: vec![0.5, 0.5, 0.5], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }); - let resp = send_raw( + // Transaction: a vector-primary insert + a PointInsert that conflicts. + let v_plan = vector_direct_insert("vp", 101); + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - v_plan, - doc_insert_conflict("docs"), // doc1 already exists → constraint fail - ], - }), + vec![ + v_plan, + doc_insert_conflict("docs"), // doc1 already exists → constraint fail + ], + 20, + ); + assert_eq!( + resp.status, + Status::Error, + "the transaction must be refused" ); - assert_eq!(resp.status, Status::Error, "batch should fail"); // The error must be a constraint violation, not a rollback failure. assert!( !matches!( @@ -115,7 +94,7 @@ fn rollback_matrix_vector_then_doc_fail() { resp.error_code ); - // doc1 is still "preexisting" (transaction never committed). + // doc1 is still "preexisting" (the transaction committed nothing). let r = send_raw(&mut core, &mut tx, &mut rx, doc_get("docs")); assert_eq!(r.status, Status::Ok); assert_eq!(&*r.payload, b"preexisting"); @@ -132,24 +111,16 @@ fn rollback_matrix_doc_then_graph_fail() { // Pre-condition: doc1 = "original". send_ok(&mut core, &mut tx, &mut rx, doc_put("docs", b"original")); - // Seed vector index to trigger a failing vector insert (we'll use doc1 conflict - // instead — no easy "graph insert that always fails" exists, so we use a - // dimension-mismatch vector as the failing op and put graph second). - // - // Actually: we need a plan that fails *after* doc is written. Use PointInsert - // on an already-existing key as the failing op. - // The batch is: doc_put (overwrites) + doc_insert_conflict (same key, fails). - let resp = send_raw( + // Transaction: doc_put (overwrites) + doc_insert_conflict (same key, refused). + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - doc_put("docs", b"modified"), - doc_insert_conflict("docs"), // same key → constraint fail - ], - }), + vec![ + doc_put("docs", b"modified"), + doc_insert_conflict("docs"), // same key → constraint fail + ], + 30, ); assert_eq!(resp.status, Status::Error); @@ -160,26 +131,20 @@ fn rollback_matrix_doc_then_graph_fail() { } // --------------------------------------------------------------------------- -// Pair: Graph (first) × Vector (second, dim-mismatch fails) +// Graph, then a refused insert // --------------------------------------------------------------------------- #[test] -fn rollback_matrix_graph_then_vector_fail() { +fn rollback_matrix_graph_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - // Seed vector index dim=3 so dimension mismatch is detectable. - send_ok(&mut core, &mut tx, &mut rx, vector_set_params("vec")); - send_ok(&mut core, &mut tx, &mut rx, vector_seed("vec")); - - // TransactionBatch: edge put + failing vector insert. - let resp = send_raw( + // Transaction: edge put, then a refused insert. + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![edge_put("col", "alice", "bob"), vector_fail("vec")], - }), + with_unique_refusal(vec![edge_put("col", "alice", "bob")]), + 40, ); assert_eq!(resp.status, Status::Error); assert!( @@ -202,33 +167,20 @@ fn rollback_matrix_graph_then_vector_fail() { } // --------------------------------------------------------------------------- -// Pair: Vector (first) × Graph (second, failing via vector dim-mismatch in same batch) -// Actually: Graph × Graph — two edge puts, second targets a key that triggers -// a constraint failure by using a dimension-mismatch vector to fail. -// We use: graph edge + vector fail as the canonical "second op fails" pattern. +// Graph × Graph, then a refused insert // --------------------------------------------------------------------------- #[test] -fn rollback_matrix_graph_then_graph_and_vector_fail() { +fn rollback_matrix_two_edges_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - // Seed vector index dim=3. - send_ok(&mut core, &mut tx, &mut rx, vector_set_params("vec")); - send_ok(&mut core, &mut tx, &mut rx, vector_seed("vec")); - - // TransactionBatch: two edge puts + failing vector. Both edges must roll back. - let resp = send_raw( + // Transaction: two edge puts, then a refused insert. Neither edge lands. + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - edge_put("col", "a", "b"), - edge_put("col", "c", "d"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![edge_put("col", "a", "b"), edge_put("col", "c", "d")]), + 50, ); assert_eq!(resp.status, Status::Error); @@ -242,52 +194,44 @@ fn rollback_matrix_graph_then_graph_and_vector_fail() { } // --------------------------------------------------------------------------- -// Pair: CRDT (first, buffered) × Vector (second, fails) -// CRDT deltas are buffered and never applied to LoroDoc until commit. -// Raw CRDT Apply is forbidden in transaction batches because it bypasses -// serialized preview admission. It must reject before any sibling mutation. +// Raw CRDT Apply inside a transaction +// Raw CRDT Apply bypasses serialized preview admission, so a transaction +// cannot stage it. The refusal comes before any sibling write lands. // --------------------------------------------------------------------------- #[test] -fn rollback_matrix_crdt_buffered_then_vector_fail() { +fn rollback_matrix_raw_crdt_apply_is_refused() { let (mut core, mut tx, mut rx, _dir) = make_core(); // Pre-condition: doc1 = "original". send_ok(&mut core, &mut tx, &mut rx, doc_put("docs", b"original")); - // Seed vector index dim=3. - send_ok(&mut core, &mut tx, &mut rx, vector_set_params("vec")); - send_ok(&mut core, &mut tx, &mut rx, vector_seed("vec")); - - // TransactionBatch: forbidden CRDT Apply + writes that must never run. + // Transaction: forbidden CRDT Apply + a write that must never land. let crdt_delta: Vec = vec![0u8; 8]; // minimal placeholder delta - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Crdt(CrdtOp::Apply { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "crdt_coll"), - document_id: "crdt_doc1".into(), - delta: crdt_delta, - peer_id: 1, - mutation_id: 42, - surrogate: nodedb_types::Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }), - doc_put("docs", b"modified"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + PhysicalPlan::Crdt(CrdtOp::Apply { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "crdt_coll"), + document_id: "crdt_doc1".into(), + delta: crdt_delta, + peer_id: 1, + mutation_id: 42, + surrogate: nodedb_types::Surrogate::ZERO, + provenance: None, + constraint_version_required: 0, + expected_frontier_digest: None, + }), + doc_put("docs", b"modified"), + ]), + 60, ); assert_eq!(resp.status, Status::Error); assert!(matches!( resp.error_code.as_deref(), - Some(ErrorCode::Unsupported { detail }) if detail == "CRDT Apply is not supported inside transaction batches" + Some(ErrorCode::Internal { detail }) if detail.contains("StageWrite is only valid") )); // Rollback must succeed (not RollbackFailed). assert!( @@ -316,7 +260,7 @@ fn rollback_matrix_doc_doc_second_fails() { // Pre-condition: "doc1" already exists so PointInsert(if_absent=false) fails. send_ok(&mut core, &mut tx, &mut rx, doc_put("docs", b"preexisting")); - // Batch: write "other_doc" (new insert) + PointInsert on existing "doc1" (fails). + // Transaction: write "other_doc" (new insert) + PointInsert on existing "doc1" (refused). let other_put = PhysicalPlan::Document(DocumentOp::PointPut { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "docs"), document_id: "other_doc".into(), @@ -327,17 +271,15 @@ fn rollback_matrix_doc_doc_second_fails() { rls_filters: Vec::new(), resolved_sum_targets: Vec::new(), }); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - other_put, - doc_insert_conflict("docs"), // "doc1" already exists - ], - }), + vec![ + other_put, + doc_insert_conflict("docs"), // "doc1" already exists + ], + 70, ); assert_eq!(resp.status, Status::Error); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_helpers.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_helpers.rs index f6edcd99f..0625380ed 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_helpers.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_helpers.rs @@ -2,63 +2,12 @@ //! Plan builders shared by the cross-engine transaction rollback matrices. -use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan, VectorOp}; +use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan}; // --------------------------------------------------------------------------- // Shared helpers // --------------------------------------------------------------------------- -/// Return a `VectorOp::SetParams` plan for a named collection with dim=3. -pub fn vector_set_params(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::SetParams { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - collection, - ), - field_name: String::new(), - dim: 3, - m: 16, - ef_construction: 200, - metric: "cosine".into(), - index_type: String::new(), - pq_m: 0, - ivf_cells: 0, - ivf_nprobe: 0, - }) -} - -/// Seed a dim=3 vector index with one vector so the index exists. -pub fn vector_seed(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - collection, - ), - vector: vec![1.0, 2.0, 3.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }) -} - -/// A vector insert that will fail with dimension mismatch (index expects dim=3). -pub fn vector_fail(collection: &str) -> PhysicalPlan { - PhysicalPlan::Vector(VectorOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - collection, - ), - vector: vec![1.0, 2.0], - dim: 3, - field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: None, - provenance: None, - }) -} - /// A document PointPut for "doc1" in collection `coll`. pub fn doc_put(coll: &str, val: &[u8]) -> PhysicalPlan { PhysicalPlan::Document(DocumentOp::PointPut { diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs index 69ae94ff7..aec1998f8 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs @@ -4,15 +4,17 @@ //! //! Each test follows the same pattern as `test_transaction_matrix`: //! 1. Pre-condition: write a known state. -//! 2. TransactionBatch: valid write (first op) + deterministically failing write (second op). -//! 3. Assert: the first write was fully rolled back. +//! 2. Transaction: valid write (first op) + deterministically refused write (second op). +//! 3. Assert: the transaction committed none of its writes. use nodedb::bridge::envelope::Status; use nodedb_physical::physical_plan::{ - AggregateSpec, ColumnarInsertIntent, ColumnarOp, DocumentOp, KvOp, MetaOp, PhysicalPlan, - QueryOp, TimeseriesOp, + AggregateSpec, ColumnarInsertIntent, ColumnarOp, DocumentOp, KvOp, PhysicalPlan, QueryOp, + TimeseriesOp, }; +use nodedb_test_support::tx_batch_helpers::commit_plans; + use super::helpers::*; // --------------------------------------------------------------------------- @@ -98,17 +100,19 @@ fn rollback_matrix_kv_then_doc_fail() { // Seed the doc that will be used as a conflict trigger. send_ok(&mut core, &mut tx, &mut rx, doc_put_conflict_seed("docs")); - // TransactionBatch: overwrite KV key + failing doc insert (key exists, not if_absent). - let resp = send_raw( + // Transaction: overwrite KV key + failing doc insert (key exists, not if_absent). + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![kv_put(b"k1", b"modified"), doc_insert_conflict("docs")], - }), + vec![kv_put(b"k1", b"modified"), doc_insert_conflict("docs")], + 10, + ); + assert_eq!( + resp.status, + Status::Error, + "the transaction must be refused on conflict" ); - assert_eq!(resp.status, Status::Error, "batch must fail on conflict"); // KV key must be rolled back to "original". let r = send_raw(&mut core, &mut tx, &mut rx, kv_get(b"k1")); @@ -117,10 +121,10 @@ fn rollback_matrix_kv_then_doc_fail() { } // --------------------------------------------------------------------------- -// Pair: Document write (first) × KV DDL inside batch (rejected) +// Pair: Document write (first) × KV DDL inside a transaction (rejected) // --------------------------------------------------------------------------- // KV DDL ops (RegisterIndex, Truncate, etc.) are rejected with a typed error -// when inside a TransactionBatch. This test verifies the document write rolled +// when inside a committed transaction. This test verifies the document write rolled // back correctly when the KV write itself fails due to a prior-doc write // combined with a KV Put that fails. // @@ -136,21 +140,23 @@ fn rollback_matrix_doc_then_kv_fail() { // Seed the conflict document. send_ok(&mut core, &mut tx, &mut rx, doc_put_conflict_seed("docs")); - // TransactionBatch: write a new KV key + failing doc insert. + // Transaction: write a new KV key + failing doc insert. // On failure the new KV key must not persist. - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - kv_put(b"new_key", b"should_not_persist"), - doc_insert_conflict("docs"), - ], - }), + vec![ + kv_put(b"new_key", b"should_not_persist"), + doc_insert_conflict("docs"), + ], + 20, + ); + assert_eq!( + resp.status, + Status::Error, + "the transaction must be refused on conflict" ); - assert_eq!(resp.status, Status::Error, "batch must fail on conflict"); // "new_key" must have been rolled back — Get should return empty/NotFound. let r = send_raw(&mut core, &mut tx, &mut rx, kv_get(b"new_key")); @@ -175,29 +181,31 @@ fn rollback_matrix_kv_delete_then_doc_fail() { // Seed conflict doc. send_ok(&mut core, &mut tx, &mut rx, doc_put_conflict_seed("docs")); - // TransactionBatch: delete the KV key + failing doc insert. - let resp = send_raw( + // Transaction: delete the KV key + failing doc insert. + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Kv(KvOp::Delete { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "kv_coll", - ), - keys: vec![b"del_key".to_vec()], - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), - doc_insert_conflict("docs"), - ], - }), + vec![ + PhysicalPlan::Kv(KvOp::Delete { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "kv_coll", + ), + keys: vec![b"del_key".to_vec()], + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }), + doc_insert_conflict("docs"), + ], + 30, + ); + assert_eq!( + resp.status, + Status::Error, + "the transaction must be refused" ); - assert_eq!(resp.status, Status::Error, "batch must fail"); // "del_key" must be restored to "keep_me". let r = send_raw(&mut core, &mut tx, &mut rx, kv_get(b"del_key")); @@ -217,32 +225,34 @@ fn rollback_matrix_kv_batch_put_then_doc_fail() { send_ok(&mut core, &mut tx, &mut rx, kv_put(b"k_a", b"a_orig")); send_ok(&mut core, &mut tx, &mut rx, doc_put_conflict_seed("docs")); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Kv(KvOp::BatchPut { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "kv_coll", - ), - entries: vec![ - (b"k_a".to_vec(), b"a_new".to_vec()), - (b"k_b".to_vec(), b"b_new".to_vec()), - ], - ttl_ms: 0, - surrogates: vec![nodedb_types::Surrogate::ZERO; 2], - returning: None, - rls_filters: Vec::new(), - }), - doc_insert_conflict("docs"), - ], - }), + vec![ + PhysicalPlan::Kv(KvOp::BatchPut { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "kv_coll", + ), + entries: vec![ + (b"k_a".to_vec(), b"a_new".to_vec()), + (b"k_b".to_vec(), b"b_new".to_vec()), + ], + ttl_ms: 0, + surrogates: vec![nodedb_types::Surrogate::ZERO; 2], + returning: None, + rls_filters: Vec::new(), + }), + doc_insert_conflict("docs"), + ], + 40, + ); + assert_eq!( + resp.status, + Status::Error, + "the transaction must be refused" ); - assert_eq!(resp.status, Status::Error, "batch must fail"); // k_a must be restored to "a_orig". let r = send_raw(&mut core, &mut tx, &mut rx, kv_get(b"k_a")); @@ -256,7 +266,7 @@ fn rollback_matrix_kv_batch_put_then_doc_fail() { } // --------------------------------------------------------------------------- -// Verify: doc get after the batch returns the pre-batch state +// Verify: doc get after the transaction returns the pre-transaction state // --------------------------------------------------------------------------- #[test] @@ -273,17 +283,15 @@ fn rollback_matrix_doc_then_doc_conflict_kv_intact() { send_ok(&mut core, &mut tx, &mut rx, doc_put_conflict_seed("docs")); // Batch: KV write + doc conflict — the KV write happened, must roll back. - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - kv_put(b"anchor", b"anchor_modified"), - doc_insert_conflict("docs"), - ], - }), + vec![ + kv_put(b"anchor", b"anchor_modified"), + doc_insert_conflict("docs"), + ], + 50, ); assert_eq!(resp.status, Status::Error); @@ -292,7 +300,7 @@ fn rollback_matrix_doc_then_doc_conflict_kv_intact() { assert_eq!(r.status, Status::Ok); assert_eq!(&*r.payload, b"anchor_val"); - // The seeded conflict doc must still be readable (it was not part of the batch). + // The seeded conflict doc must still be readable (it was not part of the transaction). let r = send_raw(&mut core, &mut tx, &mut rx, doc_get("docs", "conflict_doc")); assert_eq!(r.status, Status::Ok, "conflict_doc must still exist"); } @@ -300,7 +308,7 @@ fn rollback_matrix_doc_then_doc_conflict_kv_intact() { // --------------------------------------------------------------------------- // Pair: Columnar insert (first) × Document conflict (second) — columnar rolled back // -// A row is inserted into a plain columnar collection inside a TransactionBatch. +// A row is inserted into a plain columnar collection inside a committed transaction. // The second plan is a PointInsert that conflicts (doc already exists). // After rollback, the columnar collection must be empty. // --------------------------------------------------------------------------- @@ -317,47 +325,48 @@ fn rollback_matrix_columnar_then_doc_fail() { doc_put_conflict_seed("conflict_coll"), ); - // Confirm columnar collection is empty before the batch. + // Confirm columnar collection is empty before the transaction. let before = core.scan_collection(0, 1, "metrics", 100).unwrap(); - assert!(before.is_empty(), "columnar must be empty before batch"); + assert!( + before.is_empty(), + "columnar must be empty before the transaction" + ); // Build a columnar insert payload: one row. let rows = serde_json::json!([{"id": "r1", "val": 42}]); let payload = nodedb_types::json_to_msgpack(&rows).unwrap(); - // TransactionBatch: columnar insert + failing doc insert. - let resp = send_raw( + // Transaction: columnar insert + failing doc insert. + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Columnar(ColumnarOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "metrics", - ), - payload, - format: "msgpack".into(), - intent: ColumnarInsertIntent::Insert, - on_conflict_updates: Vec::new(), - surrogates: Vec::new(), - schema_bytes: Vec::new(), - provenance: None, - wal_lsn: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), - doc_insert_conflict("conflict_coll"), - ], - }), + vec![ + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "metrics", + ), + payload, + format: "msgpack".into(), + intent: ColumnarInsertIntent::Insert, + on_conflict_updates: Vec::new(), + surrogates: Vec::new(), + schema_bytes: Vec::new(), + provenance: None, + wal_lsn: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }), + doc_insert_conflict("conflict_coll"), + ], + 60, ); assert_eq!( resp.status, Status::Error, - "batch must fail on doc conflict" + "the transaction must be refused on doc conflict" ); assert!( !matches!( @@ -380,7 +389,7 @@ fn rollback_matrix_columnar_then_doc_fail() { // --------------------------------------------------------------------------- // Pair: Columnar insert (first) × Columnar insert — verify aggregate count // -// Inserts one row outside the batch (baseline), then in a failing batch inserts +// Inserts one row outside the transaction (baseline), then in a refused transaction inserts // another row + conflicts. After rollback the aggregate count must be 1. // --------------------------------------------------------------------------- @@ -396,7 +405,7 @@ fn rollback_matrix_columnar_count_after_rollback() { doc_put_conflict_seed("conflict_coll"), ); - // Baseline: insert one row outside any batch (committed). + // Baseline: insert one row outside any transaction (committed). let baseline = serde_json::json!([{"id": "baseline", "val": 1}]); let baseline_payload = nodedb_types::json_to_msgpack(&baseline).unwrap(); send_ok( @@ -422,38 +431,40 @@ fn rollback_matrix_columnar_count_after_rollback() { }), ); - // Failed batch: inserts a second row + doc conflict. + // Refused transaction: inserts a second row + doc conflict. let extra = serde_json::json!([{"id": "rolled_back", "val": 2}]); let extra_payload = nodedb_types::json_to_msgpack(&extra).unwrap(); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Columnar(ColumnarOp::Insert { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "metrics2", - ), - payload: extra_payload, - format: "msgpack".into(), - intent: ColumnarInsertIntent::Insert, - on_conflict_updates: Vec::new(), - surrogates: Vec::new(), - schema_bytes: Vec::new(), - provenance: None, - wal_lsn: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), - doc_insert_conflict("conflict_coll"), - ], - }), + vec![ + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "metrics2", + ), + payload: extra_payload, + format: "msgpack".into(), + intent: ColumnarInsertIntent::Insert, + on_conflict_updates: Vec::new(), + surrogates: Vec::new(), + schema_bytes: Vec::new(), + provenance: None, + wal_lsn: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }), + doc_insert_conflict("conflict_coll"), + ], + 70, + ); + assert_eq!( + resp.status, + Status::Error, + "the transaction must be refused" ); - assert_eq!(resp.status, Status::Error, "batch must fail"); // Aggregate count must be 1 (only the baseline row). let agg_payload = send_ok( @@ -500,7 +511,7 @@ fn rollback_matrix_columnar_count_after_rollback() { // --------------------------------------------------------------------------- // Pair: Timeseries ingest (first) × Document conflict (second) — ts rolled back // -// A timeseries ingest inside a TransactionBatch followed by a failing doc insert +// A timeseries ingest inside a committed transaction followed by a failing doc insert // must leave the timeseries memtable empty (truncated back by apply_undo_timeseries). // --------------------------------------------------------------------------- @@ -516,39 +527,37 @@ fn rollback_matrix_timeseries_then_doc_fail() { doc_put_conflict_seed("conflict_coll"), ); - // TransactionBatch: timeseries ingest (3 rows) + doc conflict (fails). + // Transaction: timeseries ingest (3 rows) + doc conflict (fails). let ilp = "cpu,host=s1 value=0.5 1000000000\n\ cpu,host=s1 value=0.6 2000000000\n\ cpu,host=s1 value=0.7 3000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "cpu", - ), - payload: ilp.as_bytes().to_vec(), - format: "ilp".into(), - wal_lsn: None, - surrogates: Vec::new(), - provenance: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), - doc_insert_conflict("conflict_coll"), - ], - }), + vec![ + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "cpu", + ), + payload: ilp.as_bytes().to_vec(), + format: "ilp".into(), + wal_lsn: None, + surrogates: Vec::new(), + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }), + doc_insert_conflict("conflict_coll"), + ], + 80, ); assert_eq!( resp.status, Status::Error, - "batch must fail on doc conflict" + "the transaction must be refused on doc conflict" ); assert!( !matches!( @@ -598,9 +607,9 @@ fn rollback_matrix_timeseries_then_doc_fail() { } // --------------------------------------------------------------------------- -// Pair: Timeseries ingest (first) — baseline + batch — count after rollback +// Pair: Timeseries ingest (first) — baseline + transaction — count after rollback // -// Ingests rows outside any batch (committed), then a failing batch ingests more. +// Ingests rows outside any transaction (committed), then a refused transaction ingests more. // After rollback the scan must return only the committed rows. // --------------------------------------------------------------------------- @@ -639,36 +648,38 @@ fn rollback_matrix_timeseries_count_after_rollback() { }), ); - // Failed batch: ingest 3 more rows + doc conflict. + // Refused transaction: ingest 3 more rows + doc conflict. let extra_ilp = "temp,host=s1 value=3.0 3000000000\n\ temp,host=s1 value=4.0 4000000000\n\ temp,host=s1 value=5.0 5000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - "temp", - ), - payload: extra_ilp.as_bytes().to_vec(), - format: "ilp".into(), - wal_lsn: None, - surrogates: Vec::new(), - provenance: None, - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }), - doc_insert_conflict("conflict_coll"), - ], - }), + vec![ + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "temp", + ), + payload: extra_ilp.as_bytes().to_vec(), + format: "ilp".into(), + wal_lsn: None, + surrogates: Vec::new(), + provenance: None, + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }), + doc_insert_conflict("conflict_coll"), + ], + 90, + ); + assert_eq!( + resp.status, + Status::Error, + "the transaction must be refused" ); - assert_eq!(resp.status, Status::Error, "batch must fail"); // Scan must return exactly 2 rows (the baseline only). let scan_resp = send_raw( diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_side_effects.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_side_effects.rs deleted file mode 100644 index b57a7d60a..000000000 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_side_effects.rs +++ /dev/null @@ -1,208 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Cross-engine transaction rollback matrix: index side-effects. -//! -//! A point put drives FTS and spatial index maintenance as a side-effect of -//! the document write. Rolling the write back must roll those back too — -//! a posting or an R-tree entry that survives a failed batch is a row that -//! searches can still find but reads cannot. - -use nodedb::bridge::envelope::{ErrorCode, Status}; -use nodedb_physical::physical_plan::{DocumentOp, MetaOp, PhysicalPlan, TextOp}; -use nodedb_types::{DatabaseId, QualifiedCollection}; - -use super::helpers::*; -use super::test_transaction_matrix_helpers::*; - -// --------------------------------------------------------------------------- -// FTS side-effect rollback: doc with text fields, batch fails, FTS is clean -// -// When tx_point_put runs, it calls `inverted.index_document` as a side-effect. -// When the batch fails and the PutDocument undo entry is applied, `apply_undo_document` -// calls `inverted.remove_document` to revert the posting. This test proves -// that the FTS posting does NOT surface in a subsequent search after rollback. -// --------------------------------------------------------------------------- - -#[test] -fn rollback_matrix_fts_side_effect_rolled_back() { - let (mut core, mut tx, mut rx, _dir) = make_core(); - - // Seed the vector index so we have a deterministic failure trigger. - send_ok(&mut core, &mut tx, &mut rx, vector_set_params("vec")); - send_ok(&mut core, &mut tx, &mut rx, vector_seed("vec")); - - // TransactionBatch: - // plan 0: PointPut a document with a text "title" field (triggers FTS index) - // plan 1: vector insert with dim-mismatch (always fails) - let doc_value = r#"{"title":"unique_rollback_sentinel quantum database"}"#; - let resp = send_raw( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Document(DocumentOp::PointPut { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "articles"), - document_id: "fts_rollback_doc".into(), - value: doc_value.as_bytes().to_vec(), - surrogate: nodedb_types::Surrogate::new(7001), - pk_bytes: b"fts_rollback_doc".to_vec(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - vector_fail("vec"), - ], - }), - ); - assert_eq!( - resp.status, - Status::Error, - "batch must fail on dim-mismatch" - ); - assert!( - !matches!( - resp.error_code.as_deref(), - Some(ErrorCode::RollbackFailed { .. }) - ), - "rollback itself must succeed; got {:?}", - resp.error_code - ); - - // FTS search for the sentinel term must return zero results — the posting - // was removed by apply_undo_document → inverted.remove_document. - let search_resp = send_raw( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Text(TextOp::Search { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "articles"), - query: "unique_rollback_sentinel".into(), - top_k: 10, - fuzzy: false, - rls_filters: Vec::new(), - prefilter: None, - }), - ); - assert_eq!(search_resp.status, Status::Ok); - let json = super::helpers::payload_json(&search_resp.payload); - let val: serde_json::Value = - serde_json::from_str(&json).unwrap_or(serde_json::Value::Array(vec![])); - let empty = vec![]; - let arr = val.as_array().unwrap_or(&empty); - assert!( - arr.is_empty(), - "FTS posting for rolled-back doc must not appear in search results; got {json}" - ); -} - -// --------------------------------------------------------------------------- -// Spatial side-effect NOT written in tx path — confirmed by test -// -// The transactional PointPut path (tx_point_put) writes to sparse + inverted -// only. It does NOT call apply_point_put, so the spatial R-tree is never -// touched during a transaction. This test proves that after a failed batch -// containing a PointPut with a geometry field, a spatial scan returns zero -// results — confirming no stale R-tree entry was left. -// --------------------------------------------------------------------------- - -#[test] -fn rollback_matrix_spatial_not_written_in_tx_path() { - use nodedb_physical::physical_plan::{SpatialOp, SpatialPredicate}; - use nodedb_types::geometry::Geometry; - - let (mut core, mut tx, mut rx, _dir) = make_core(); - - // Seed vector index for the failing second op. - send_ok(&mut core, &mut tx, &mut rx, vector_set_params("vec")); - send_ok(&mut core, &mut tx, &mut rx, vector_seed("vec")); - - // TransactionBatch: - // plan 0: PointPut a doc with a GeoJSON geometry field - // plan 1: vector insert with dim-mismatch (always fails) - let geo_doc = r#"{"name":"poi","location":{"type":"Point","coordinates":[10.0,20.0]}}"#; - let resp = send_raw( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - PhysicalPlan::Document(DocumentOp::PointPut { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "places"), - document_id: "geo_rollback_doc".into(), - value: geo_doc.as_bytes().to_vec(), - surrogate: nodedb_types::Surrogate::new(8001), - pk_bytes: b"geo_rollback_doc".to_vec(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - vector_fail("vec"), - ], - }), - ); - assert_eq!( - resp.status, - Status::Error, - "batch must fail on dim-mismatch" - ); - assert!( - !matches!( - resp.error_code.as_deref(), - Some(ErrorCode::RollbackFailed { .. }) - ), - "rollback itself must succeed; got {:?}", - resp.error_code - ); - - // Spatial scan for a wide bounding box must return zero results. - // (The R-tree was never written since tx_point_put bypasses apply_point_put.) - let query_geometry = Geometry::Point { - coordinates: [10.0, 20.0], - }; - let scan_resp = send_raw( - &mut core, - &mut tx, - &mut rx, - PhysicalPlan::Spatial(SpatialOp::Scan { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, "places"), - field: "location".into(), - predicate: SpatialPredicate::DWithin, - query_geometry, - distance_meters: 1_000_000.0, - attribute_filters: Vec::new(), - limit: 10, - projection: Vec::new(), - rls_filters: Vec::new(), - prefilter: None, - }), - ); - assert_eq!(scan_resp.status, Status::Ok); - let json = super::helpers::payload_json(&scan_resp.payload); - let val: serde_json::Value = - serde_json::from_str(&json).unwrap_or(serde_json::Value::Array(vec![])); - let empty = vec![]; - let arr = val.as_array().unwrap_or(&empty); - assert!( - arr.is_empty(), - "spatial scan after rollback must return zero results (tx path never writes R-tree); \ - got {json}" - ); -} - -// --------------------------------------------------------------------------- - -#[test] -fn rollback_failed_error_code_is_typed() { - // Construct the error code and verify it's distinguishable. - let code = ErrorCode::RollbackFailed { - entry_index: 2, - detail: "sparse store error: disk full".into(), - }; - assert!( - matches!(code, ErrorCode::RollbackFailed { entry_index: 2, .. }), - "RollbackFailed must carry structured fields" - ); -} diff --git a/nodedb/tests/inproc/cases/mod.rs b/nodedb/tests/inproc/cases/mod.rs index 92e639697..e27b118d7 100644 --- a/nodedb/tests/inproc/cases/mod.rs +++ b/nodedb/tests/inproc/cases/mod.rs @@ -191,7 +191,6 @@ mod tenant_drop_owned_objects; mod tls_policy_enforcement; mod topic_replication_apply; mod transaction_batch_cross_engine; -mod transaction_batch_cross_engine_crash; mod transaction_batch_cross_engine_mixed; mod transaction_batch_cross_shard; mod trigger_batching; diff --git a/nodedb/tests/inproc/cases/native_handshake_e2e.rs b/nodedb/tests/inproc/cases/native_handshake_e2e.rs index d9fc2f225..51f948eec 100644 --- a/nodedb/tests/inproc/cases/native_handshake_e2e.rs +++ b/nodedb/tests/inproc/cases/native_handshake_e2e.rs @@ -40,6 +40,8 @@ impl NativeTestServer { let (event_producers, event_consumers) = create_event_bus(1); let shared = SharedState::new(dispatcher, Arc::clone(&wal)).unwrap(); + // The same gateway install production boot runs. + nodedb::bootstrap::state_wiring::install_gateway(&shared); shared .credentials .bootstrap_trust_superuser("nodedb") diff --git a/nodedb/tests/inproc/cases/transaction_batch_cross_engine.rs b/nodedb/tests/inproc/cases/transaction_batch_cross_engine.rs index 0a89a54ff..0c2099d0c 100644 --- a/nodedb/tests/inproc/cases/transaction_batch_cross_engine.rs +++ b/nodedb/tests/inproc/cases/transaction_batch_cross_engine.rs @@ -1,30 +1,20 @@ // SPDX-License-Identifier: BUSL-1.1 -//! MetaOp::TransactionBatch must atomically commit or atomically roll back +//! A committed transaction must atomically commit or atomically roll back //! across engine pairs. This file covers KV paired with every other engine. +//! Each transaction stages its plans, resolves them into one redo record, +//! and installs the record (`commit_plans`). //! //! Pairs already covered by executor_tests/test_transaction_matrix.rs and //! test_transaction_matrix_kv.rs are not duplicated here. //! Columnar/Timeseries/CRDT pairs: see transaction_batch_cross_engine_mixed.rs. -//! Crash injection tests: see transaction_batch_cross_engine_crash.rs. use nodedb_test_support::tx_batch_helpers::*; use nodedb::bridge::envelope::{ErrorCode, Status}; -use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; // ── Shared helpers ──────────────────────────────────────────────────────────── -fn seed_vec( - core: &mut nodedb::data::executor::core_loop::CoreLoop, - tx: &mut nodedb_bridge::buffer::Producer, - rx: &mut nodedb_bridge::buffer::Consumer, - coll: &str, -) { - send_ok(core, tx, rx, vector_set_params(coll)); - send_ok(core, tx, rx, vector_seed(coll)); -} - fn assert_no_rb_fail(resp: &nodedb::bridge::envelope::Response) { assert!( !matches!( @@ -41,16 +31,13 @@ fn assert_no_rb_fail(resp: &nodedb::bridge::envelope::Response) { #[test] fn commit_kv_vector() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![kv_put(b"k1", b"v1"), vector_insert_ok("vec")], - }), + vec![kv_put(b"k1", b"v1"), vector_direct_insert("vp", 101)], + 10, ); assert_eq!(resp.status, Status::Ok); let r = send_raw(&mut core, &mut tx, &mut rx, kv_get(b"k1")); @@ -58,19 +45,16 @@ fn commit_kv_vector() { } #[test] -fn rollback_kv_then_vector_fail() { +fn rollback_kv_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); send_ok(&mut core, &mut tx, &mut rx, kv_put(b"k1", b"original")); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![kv_put(b"k1", b"modified"), vector_fail("vec")], - }), + with_unique_refusal(vec![kv_put(b"k1", b"modified")]), + 20, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -84,14 +68,12 @@ fn rollback_kv_then_vector_fail() { fn commit_kv_graph() { let (mut core, mut tx, mut rx, _dir) = make_core(); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![kv_put(b"kg1", b"val"), edge_put("g", "a", "b")], - }), + vec![kv_put(b"kg1", b"val"), edge_put("g", "a", "b")], + 30, ); assert_eq!(resp.status, Status::Ok); let r = send_raw(&mut core, &mut tx, &mut rx, kv_get(b"kg1")); @@ -99,23 +81,16 @@ fn commit_kv_graph() { } #[test] -fn rollback_kv_then_graph_fail() { +fn rollback_kv_graph_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); send_ok(&mut core, &mut tx, &mut rx, kv_put(b"k2", b"original")); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - kv_put(b"k2", b"modified"), - edge_put("g", "x", "y"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![kv_put(b"k2", b"modified"), edge_put("g", "x", "y")]), + 40, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -130,14 +105,12 @@ fn rollback_kv_then_graph_fail() { fn commit_kv_columnar() { let (mut core, mut tx, mut rx, _dir) = make_core(); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![kv_put(b"kc1", b"val"), columnar_insert("metrics", "r1", 10)], - }), + vec![kv_put(b"kc1", b"val"), columnar_insert("metrics", "r1", 10)], + 50, ); assert_eq!(resp.status, Status::Ok); let r = send_raw(&mut core, &mut tx, &mut rx, kv_get(b"kc1")); @@ -145,23 +118,19 @@ fn commit_kv_columnar() { } #[test] -fn rollback_kv_then_columnar_fail() { +fn rollback_kv_columnar_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); send_ok(&mut core, &mut tx, &mut rx, kv_put(b"k3", b"original")); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - kv_put(b"k3", b"modified"), - columnar_insert("metrics", "r1", 10), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + kv_put(b"k3", b"modified"), + columnar_insert("metrics", "r1", 10), + ]), + 60, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -177,14 +146,12 @@ fn commit_kv_timeseries() { let (mut core, mut tx, mut rx, _dir) = make_core(); let ilp = "temp,host=s1 value=1.0 1000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![kv_put(b"kt1", b"val"), timeseries_ingest("temp", ilp)], - }), + vec![kv_put(b"kt1", b"val"), timeseries_ingest("temp", ilp)], + 70, ); assert_eq!(resp.status, Status::Ok); let r = send_raw(&mut core, &mut tx, &mut rx, kv_get(b"kt1")); @@ -193,24 +160,20 @@ fn commit_kv_timeseries() { } #[test] -fn rollback_kv_then_timeseries_fail() { +fn rollback_kv_timeseries_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); send_ok(&mut core, &mut tx, &mut rx, kv_put(b"k4", b"original")); let ilp = "temp,host=s1 value=1.0 1000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - kv_put(b"k4", b"modified"), - timeseries_ingest("temp", ilp), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + kv_put(b"k4", b"modified"), + timeseries_ingest("temp", ilp), + ]), + 80, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -222,23 +185,19 @@ fn rollback_kv_then_timeseries_fail() { // ── KV × CRDT ───────────────────────────────────────────────────────────────── #[test] -fn rollback_kv_then_crdt_fail() { +fn rollback_kv_crdt_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); send_ok(&mut core, &mut tx, &mut rx, kv_put(b"k5", b"original")); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - kv_put(b"k5", b"modified"), - crdt_apply("crdt_coll", "doc1"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + kv_put(b"k5", b"modified"), + crdt_upsert("crdt_coll", "doc1", 301), + ]), + 90, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); diff --git a/nodedb/tests/inproc/cases/transaction_batch_cross_engine_crash.rs b/nodedb/tests/inproc/cases/transaction_batch_cross_engine_crash.rs deleted file mode 100644 index e454efc93..000000000 --- a/nodedb/tests/inproc/cases/transaction_batch_cross_engine_crash.rs +++ /dev/null @@ -1,175 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Crash-injection tests for `MetaOp::TransactionBatch`. -//! -//! Compiled only with `--features failpoints`. Each test arms a panic on -//! the `transaction_batch::between_subapply` fail point so the second -//! sub-apply triggers a panic. The batch handler catches the unwind, -//! routes through the typed-rollback path, and returns an -//! `ErrorCode::Internal` response. Side-effects of the first (already -//! applied) sub-plan must be rolled back before the response is returned — -//! the all-or-nothing guarantee. - -#[allow(unused_imports)] -use nodedb_test_support::tx_batch_helpers::*; - -#[cfg(feature = "failpoints")] -use nodedb::bridge::dispatch::BridgeRequest; -#[cfg(feature = "failpoints")] -use nodedb::bridge::envelope::{ErrorCode, Status}; -#[cfg(feature = "failpoints")] -use nodedb::fail_point::{FailAction, FailGuard}; -#[cfg(feature = "failpoints")] -use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; - -/// Push a TransactionBatch through the bridge, tick once, return the response. -/// Panic-from-fail-point is now handled inside the handler — `tick()` never -/// unwinds for a controlled fail point installed at the sub-apply boundary. -#[cfg(feature = "failpoints")] -fn send_batch_expecting_panic_rollback( - core: &mut nodedb::data::executor::core_loop::CoreLoop, - tx: &mut nodedb_bridge::buffer::Producer, - rx: &mut nodedb_bridge::buffer::Consumer, - plans: Vec, -) -> nodedb::bridge::envelope::Response { - tx.try_push(BridgeRequest::unfloored(make_request(PhysicalPlan::Meta( - MetaOp::TransactionBatch { - plans, - txn_id: None, - }, - )))) - .unwrap(); - core.tick(); - rx.try_pop().unwrap().inner -} - -#[cfg(feature = "failpoints")] -fn assert_panic_rollback_response(resp: &nodedb::bridge::envelope::Response) { - assert_eq!( - resp.status, - Status::Error, - "expected Status::Error after panic-rollback, got {:?}", - resp.status - ); - match resp.error_code.as_deref() { - Some(ErrorCode::Internal { detail }) => { - assert!( - detail.contains("panic in sub-apply"), - "error detail did not name the panic site: {detail}" - ); - } - other => panic!("expected ErrorCode::Internal, got {other:?}"), - } -} - -#[cfg(feature = "failpoints")] -fn seed_vec( - core: &mut nodedb::data::executor::core_loop::CoreLoop, - tx: &mut nodedb_bridge::buffer::Producer, - rx: &mut nodedb_bridge::buffer::Consumer, - coll: &str, -) { - send_ok(core, tx, rx, vector_set_params(coll)); - send_ok(core, tx, rx, vector_seed(coll)); -} - -// ── Doc + Vector crash ──────────────────────────────────────────────────────── - -#[cfg(feature = "failpoints")] -#[test] -fn crash_between_subapply_doc_vector() { - let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - - let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); - - let resp = send_batch_expecting_panic_rollback( - &mut core, - &mut tx, - &mut rx, - vec![ - doc_put("docs", "crash_doc", b"should_not_persist"), - vector_insert_ok("vec"), - ], - ); - assert_panic_rollback_response(&resp); - - // Drop the guard before issuing further requests so subsequent ticks - // don't trip the fail point. - drop(_guard); - assert_doc_absent(&mut core, &mut tx, &mut rx, "docs", "crash_doc"); -} - -// ── Doc + Graph crash ───────────────────────────────────────────────────────── - -#[cfg(feature = "failpoints")] -#[test] -fn crash_between_subapply_doc_graph() { - let (mut core, mut tx, mut rx, _dir) = make_core(); - - let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); - - let resp = send_batch_expecting_panic_rollback( - &mut core, - &mut tx, - &mut rx, - vec![ - doc_put("docs", "crash_doc2", b"should_not_persist"), - edge_put("g", "crash_src", "crash_dst"), - ], - ); - assert_panic_rollback_response(&resp); - - drop(_guard); - assert_doc_absent(&mut core, &mut tx, &mut rx, "docs", "crash_doc2"); - assert_edge_absent(&mut core, &mut tx, &mut rx, "g", "crash_src"); -} - -// ── Doc + KV crash ──────────────────────────────────────────────────────────── - -#[cfg(feature = "failpoints")] -#[test] -fn crash_between_subapply_doc_kv() { - let (mut core, mut tx, mut rx, _dir) = make_core(); - - let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); - - let resp = send_batch_expecting_panic_rollback( - &mut core, - &mut tx, - &mut rx, - vec![ - doc_put("docs", "crash_doc3", b"should_not_persist"), - kv_put(b"crash_k1", b"should_not_persist"), - ], - ); - assert_panic_rollback_response(&resp); - - drop(_guard); - assert_doc_absent(&mut core, &mut tx, &mut rx, "docs", "crash_doc3"); - assert_kv_absent(&mut core, &mut tx, &mut rx, b"crash_k1"); -} - -// ── Vector + Graph crash ────────────────────────────────────────────────────── - -#[cfg(feature = "failpoints")] -#[test] -fn crash_between_subapply_vector_graph() { - let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - - let _guard = FailGuard::install("transaction_batch::between_subapply", FailAction::Panic); - - let resp = send_batch_expecting_panic_rollback( - &mut core, - &mut tx, - &mut rx, - vec![vector_insert_ok("vec"), edge_put("g", "vg_src", "vg_dst")], - ); - assert_panic_rollback_response(&resp); - - drop(_guard); - // The edge (second op) was never applied. The vector insert (first op) - // is rolled back via the undo log. - assert_edge_absent(&mut core, &mut tx, &mut rx, "g", "vg_src"); -} diff --git a/nodedb/tests/inproc/cases/transaction_batch_cross_engine_mixed.rs b/nodedb/tests/inproc/cases/transaction_batch_cross_engine_mixed.rs index 79e0fd2cc..220927fd7 100644 --- a/nodedb/tests/inproc/cases/transaction_batch_cross_engine_mixed.rs +++ b/nodedb/tests/inproc/cases/transaction_batch_cross_engine_mixed.rs @@ -1,23 +1,11 @@ // SPDX-License-Identifier: BUSL-1.1 -//! MetaOp::TransactionBatch atomicity tests for Columnar, Timeseries, and CRDT +//! Committed-transaction atomicity tests for Columnar, Timeseries, and CRDT //! engine pairs. Pairs with KV are in transaction_batch_cross_engine.rs. -//! Crash injection tests are in transaction_batch_cross_engine_crash.rs. use nodedb_test_support::tx_batch_helpers::*; use nodedb::bridge::envelope::{ErrorCode, Status}; -use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; - -fn seed_vec( - core: &mut nodedb::data::executor::core_loop::CoreLoop, - tx: &mut nodedb_bridge::buffer::Producer, - rx: &mut nodedb_bridge::buffer::Consumer, - coll: &str, -) { - send_ok(core, tx, rx, vector_set_params(coll)); - send_ok(core, tx, rx, vector_seed(coll)); -} fn seed_conflict_doc( core: &mut nodedb::data::executor::core_loop::CoreLoop, @@ -45,37 +33,31 @@ fn assert_no_rb_fail(resp: &nodedb::bridge::envelope::Response) { #[test] fn commit_columnar_vector() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - columnar_insert("metrics", "r1", 10), - vector_insert_ok("vec"), - ], - }), + vec![ + columnar_insert("metrics", "r1", 10), + vector_direct_insert("vp", 101), + ], + 10, ); assert_eq!(resp.status, Status::Ok); assert_columnar_count(&mut core, &mut tx, &mut rx, "metrics", 1); } #[test] -fn rollback_columnar_then_vector_fail() { +fn rollback_columnar_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![columnar_insert("metrics", "r1", 10), vector_fail("vec")], - }), + with_unique_refusal(vec![columnar_insert("metrics", "r1", 10)]), + 20, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -85,22 +67,18 @@ fn rollback_columnar_then_vector_fail() { // ── Columnar × Graph ────────────────────────────────────────────────────────── #[test] -fn rollback_columnar_then_graph_fail() { +fn rollback_columnar_graph_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - columnar_insert("metrics", "r1", 10), - edge_put("g", "p", "q"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + columnar_insert("metrics", "r1", 10), + edge_put("g", "p", "q"), + ]), + 30, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -111,23 +89,21 @@ fn rollback_columnar_then_graph_fail() { // ── Columnar × Timeseries ───────────────────────────────────────────────────── #[test] -fn rollback_columnar_then_timeseries_fail() { +fn rollback_columnar_timeseries_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); seed_conflict_doc(&mut core, &mut tx, &mut rx, "docs", "conflict"); let ilp = "temp,host=s1 value=1.0 1000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - columnar_insert("metrics", "r1", 10), - timeseries_ingest("temp", ilp), - doc_conflict("docs", "conflict"), - ], - }), + vec![ + columnar_insert("metrics", "r1", 10), + timeseries_ingest("temp", ilp), + doc_conflict("docs", "conflict"), + ], + 40, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -138,22 +114,18 @@ fn rollback_columnar_then_timeseries_fail() { // ── Columnar × CRDT ────────────────────────────────────────────────────────── #[test] -fn rollback_columnar_then_crdt_fail() { +fn rollback_columnar_crdt_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - columnar_insert("metrics", "r1", 10), - crdt_apply("crdt_coll", "doc1"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + columnar_insert("metrics", "r1", 10), + crdt_upsert("crdt_coll", "doc1", 301), + ]), + 50, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -165,36 +137,33 @@ fn rollback_columnar_then_crdt_fail() { #[test] fn commit_timeseries_vector() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); let ilp = "cpu,host=s1 value=0.5 1000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![timeseries_ingest("cpu", ilp), vector_insert_ok("vec")], - }), + vec![ + timeseries_ingest("cpu", ilp), + vector_direct_insert("vp", 101), + ], + 60, ); assert_eq!(resp.status, Status::Ok); assert_ts_count(&mut core, &mut tx, &mut rx, "cpu", 1); } #[test] -fn rollback_timeseries_then_vector_fail() { +fn rollback_timeseries_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); let ilp = "cpu,host=s1 value=0.5 1000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![timeseries_ingest("cpu", ilp), vector_fail("vec")], - }), + with_unique_refusal(vec![timeseries_ingest("cpu", ilp)]), + 70, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -204,23 +173,16 @@ fn rollback_timeseries_then_vector_fail() { // ── Timeseries × Graph ──────────────────────────────────────────────────────── #[test] -fn rollback_timeseries_then_graph_fail() { +fn rollback_timeseries_graph_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); let ilp = "cpu,host=s1 value=0.5 1000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - timeseries_ingest("cpu", ilp), - edge_put("g", "m", "n"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![timeseries_ingest("cpu", ilp), edge_put("g", "m", "n")]), + 80, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -231,23 +193,19 @@ fn rollback_timeseries_then_graph_fail() { // ── Timeseries × CRDT ───────────────────────────────────────────────────────── #[test] -fn rollback_timeseries_then_crdt_fail() { +fn rollback_timeseries_crdt_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); let ilp = "cpu,host=s1 value=0.5 1000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - timeseries_ingest("cpu", ilp), - crdt_apply("crdt_coll", "doc1"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + timeseries_ingest("cpu", ilp), + crdt_upsert("crdt_coll", "doc1", 301), + ]), + 90, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -257,7 +215,7 @@ fn rollback_timeseries_then_crdt_fail() { // ── CRDT × Doc ──────────────────────────────────────────────────────────────── #[test] -fn rollback_crdt_then_doc_fail() { +fn rollback_crdt_doc_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); seed_conflict_doc(&mut core, &mut tx, &mut rx, "docs", "conflict"); send_ok( @@ -267,18 +225,16 @@ fn rollback_crdt_then_doc_fail() { doc_put("docs", "sentinel", b"original"), ); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - crdt_apply("crdt_coll", "doc1"), - doc_put("docs", "sentinel", b"modified"), - doc_conflict("docs", "conflict"), - ], - }), + vec![ + crdt_upsert("crdt_coll", "doc1", 301), + doc_put("docs", "sentinel", b"modified"), + doc_conflict("docs", "conflict"), + ], + 100, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -291,22 +247,18 @@ fn rollback_crdt_then_doc_fail() { // ── CRDT × Graph ────────────────────────────────────────────────────────────── #[test] -fn rollback_crdt_then_graph_fail() { +fn rollback_crdt_graph_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - crdt_apply("crdt_coll", "doc1"), - edge_put("g", "c1", "c2"), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + crdt_upsert("crdt_coll", "doc1", 301), + edge_put("g", "c1", "c2"), + ]), + 110, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -316,22 +268,20 @@ fn rollback_crdt_then_graph_fail() { // ── CRDT × KV ───────────────────────────────────────────────────────────────── #[test] -fn rollback_crdt_then_kv_fail() { +fn rollback_crdt_kv_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); seed_conflict_doc(&mut core, &mut tx, &mut rx, "docs", "conflict"); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - crdt_apply("crdt_coll", "doc1"), - kv_put(b"ck1", b"should_not_persist"), - doc_conflict("docs", "conflict"), - ], - }), + vec![ + crdt_upsert("crdt_coll", "doc1", 301), + kv_put(b"ck1", b"should_not_persist"), + doc_conflict("docs", "conflict"), + ], + 120, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -341,22 +291,18 @@ fn rollback_crdt_then_kv_fail() { // ── CRDT × Columnar ─────────────────────────────────────────────────────────── #[test] -fn rollback_crdt_then_columnar_fail() { +fn rollback_crdt_columnar_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - crdt_apply("crdt_coll", "doc1"), - columnar_insert("metrics", "r1", 10), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + crdt_upsert("crdt_coll", "doc1", 301), + columnar_insert("metrics", "r1", 10), + ]), + 130, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); @@ -366,23 +312,19 @@ fn rollback_crdt_then_columnar_fail() { // ── CRDT × Timeseries ───────────────────────────────────────────────────────── #[test] -fn rollback_crdt_then_timeseries_fail() { +fn rollback_crdt_timeseries_on_refusal() { let (mut core, mut tx, mut rx, _dir) = make_core(); - seed_vec(&mut core, &mut tx, &mut rx, "vec"); let ilp = "cpu,host=s1 value=0.5 1000000000\n"; - let resp = send_raw( + let resp = commit_plans( &mut core, &mut tx, &mut rx, - PhysicalPlan::Meta(MetaOp::TransactionBatch { - txn_id: None, - plans: vec![ - crdt_apply("crdt_coll", "doc1"), - timeseries_ingest("cpu", ilp), - vector_fail("vec"), - ], - }), + with_unique_refusal(vec![ + crdt_upsert("crdt_coll", "doc1", 301), + timeseries_ingest("cpu", ilp), + ]), + 140, ); assert_eq!(resp.status, Status::Error); assert_no_rb_fail(&resp); diff --git a/nodedb/tests/native/cases/mod.rs b/nodedb/tests/native/cases/mod.rs index 4a5882c28..e2ccb9d9b 100644 --- a/nodedb/tests/native/cases/mod.rs +++ b/nodedb/tests/native/cases/mod.rs @@ -8,6 +8,7 @@ mod native_dml_affected_counts; mod native_dml_outcome_conformance; mod native_error_code_classification; mod native_gateway_txn_overlay; +mod native_index_ddl_opcodes; mod native_kv_counter_faults; mod native_primary_key_nullability; mod native_protocol; diff --git a/nodedb/tests/native/cases/native_index_ddl_opcodes.rs b/nodedb/tests/native/cases/native_index_ddl_opcodes.rs new file mode 100644 index 000000000..d792ab322 --- /dev/null +++ b/nodedb/tests/native/cases/native_index_ddl_opcodes.rs @@ -0,0 +1,371 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native index-DDL opcodes, end to end over the native wire. +//! +//! Each opcode runs the SQL DDL it names, so the index it makes is a catalog +//! index: `SHOW INDEXES` lists it, and inside an explicit transaction it is +//! visible to later statements and gone after ROLLBACK. Covers +//! `KvRegisterSortedIndex`, `KvDropSortedIndex`, `VectorSetParams`, +//! `DocumentDropIndex`, `KvRegisterIndex`, `KvDropIndex` and +//! `DocumentRegister`, in autocommit and inside a block. + +use nodedb_test_support::native_harness::{NativeTestServer, do_handshake, send_request, send_sql}; +use nodedb_types::protocol::opcodes::ResponseStatus; +use nodedb_types::protocol::text_fields::TextFields; +use nodedb_types::protocol::{HelloFrame, NativeResponse, OpCode}; +use nodedb_types::value::Value; +use tokio::net::TcpStream; + +/// One native session with its own request sequence. +struct Client { + stream: TcpStream, + seq: u64, +} + +impl Client { + async fn connect(server: &NativeTestServer) -> Self { + let (stream, _ack) = do_handshake(server.addr, &HelloFrame::current()) + .await + .expect("handshake"); + Self { stream, seq: 0 } + } + + fn next_seq(&mut self) -> u64 { + self.seq += 1; + self.seq + } + + /// Run `sql` and require success. + async fn sql(&mut self, sql: &str) -> NativeResponse { + let seq = self.next_seq(); + let response = send_sql(&mut self.stream, seq, sql).await; + assert_eq!( + response.status, + ResponseStatus::Ok, + "{sql} must succeed: {response:?}" + ); + response + } + + /// Send opcode `op` and require the plain opcode success reply. + async fn op(&mut self, op: OpCode, fields: TextFields) { + let seq = self.next_seq(); + let response = send_request(&mut self.stream, seq, op, fields).await; + assert_eq!( + response.status, + ResponseStatus::Ok, + "{op:?} must succeed: {response:?}" + ); + assert_eq!( + (response.rows_affected, response.command.as_deref()), + (None, None), + "{op:?} answers like an opcode, with no count and no verb" + ); + } + + /// Send read opcode `op` and require success. + async fn read(&mut self, op: OpCode, fields: TextFields) -> NativeResponse { + let seq = self.next_seq(); + let response = send_request(&mut self.stream, seq, op, fields).await; + assert_eq!( + response.status, + ResponseStatus::Ok, + "{op:?} must succeed: {response:?}" + ); + response + } + + /// Every index name `SHOW INDEXES` lists to this session. + async fn indexes(&mut self) -> Vec { + let response = self.sql("SHOW INDEXES").await; + response + .rows + .unwrap_or_default() + .into_iter() + .flatten() + .filter_map(|cell| match cell { + Value::String(text) => Some(text), + _ => None, + }) + .collect() + } + + async fn lists(&mut self, index: &str) -> bool { + self.indexes().await.iter().any(|name| name == index) + } +} + +fn sorted_index(collection: &str, name: &str) -> TextFields { + TextFields { + collection: Some(collection.to_string()), + index_name: Some(name.to_string()), + sort_columns: Some(vec![("score".to_string(), "DESC".to_string())]), + key_column: Some("id".to_string()), + ..TextFields::default() + } +} + +fn index_name(name: &str) -> TextFields { + TextFields { + index_name: Some(name.to_string()), + ..TextFields::default() + } +} + +fn on_field(collection: &str, field: &str) -> TextFields { + TextFields { + collection: Some(collection.to_string()), + field: Some(field.to_string()), + ..TextFields::default() + } +} + +async fn kv_board(client: &mut Client, name: &str) { + client + .sql(&format!( + "CREATE COLLECTION {name} (id STRING PRIMARY KEY, score INT) WITH (engine='kv')" + )) + .await; + client + .sql(&format!("INSERT INTO {name} {{ id: 'p1', score: 10 }}")) + .await; +} + +#[tokio::test] +async fn sorted_index_opcodes_create_and_drop_a_catalog_index() { + let server = NativeTestServer::start().await; + let mut client = Client::connect(&server).await; + kv_board(&mut client, "nat_sorted").await; + + client + .op( + OpCode::KvRegisterSortedIndex, + sorted_index("nat_sorted", "nat_sorted_idx"), + ) + .await; + assert!(client.lists("nat_sorted_idx").await); + client.sql("SELECT SORTED_COUNT(nat_sorted_idx)").await; + + client + .op(OpCode::KvDropSortedIndex, index_name("nat_sorted_idx")) + .await; + assert!(!client.lists("nat_sorted_idx").await); + server.shutdown().await; +} + +#[tokio::test] +async fn sorted_index_opcodes_roll_back_inside_a_block() { + let server = NativeTestServer::start().await; + let mut client = Client::connect(&server).await; + kv_board(&mut client, "nat_sorted_rb").await; + + client.sql("BEGIN").await; + client + .op( + OpCode::KvRegisterSortedIndex, + sorted_index("nat_sorted_rb", "nat_sorted_rb_idx"), + ) + .await; + assert!( + client.lists("nat_sorted_rb_idx").await, + "the block sees the index it created" + ); + client.sql("SELECT SORTED_COUNT(nat_sorted_rb_idx)").await; + client + .sql("INSERT INTO nat_sorted_rb { id: 'p2', score: 20 }") + .await; + let top = client + .read( + OpCode::KvSortedIndexTopK, + TextFields { + index_name: Some("nat_sorted_rb_idx".to_string()), + top_k_count: Some(10), + ..TextFields::default() + }, + ) + .await; + assert_eq!( + top.rows.as_ref().map(Vec::len), + Some(2), + "the TOPK opcode ranks the base row and the staged row: {top:?}" + ); + client.sql("ROLLBACK").await; + assert!(!client.lists("nat_sorted_rb_idx").await); + + // The name is free again, and a rolled-back drop keeps the index. + client + .op( + OpCode::KvRegisterSortedIndex, + sorted_index("nat_sorted_rb", "nat_sorted_rb_idx"), + ) + .await; + client.sql("BEGIN").await; + client + .op(OpCode::KvDropSortedIndex, index_name("nat_sorted_rb_idx")) + .await; + assert!(!client.lists("nat_sorted_rb_idx").await); + client.sql("ROLLBACK").await; + assert!(client.lists("nat_sorted_rb_idx").await); + client.sql("SELECT SORTED_COUNT(nat_sorted_rb_idx)").await; + server.shutdown().await; +} + +fn vector_params(collection: &str) -> TextFields { + TextFields { + collection: Some(collection.to_string()), + vector_dim: Some(3), + metric: Some("l2".to_string()), + ..TextFields::default() + } +} + +#[tokio::test] +async fn vector_set_params_creates_an_index_and_rolls_back_inside_a_block() { + let server = NativeTestServer::start().await; + let mut client = Client::connect(&server).await; + client.sql("CREATE COLLECTION nat_vec").await; + client.sql("CREATE COLLECTION nat_vec_rb").await; + + client + .op(OpCode::VectorSetParams, vector_params("nat_vec")) + .await; + assert!(client.lists("vec_nat_vec").await); + + client.sql("BEGIN").await; + client + .op(OpCode::VectorSetParams, vector_params("nat_vec_rb")) + .await; + assert!(client.lists("vec_nat_vec_rb").await); + client.sql("ROLLBACK").await; + assert!(!client.lists("vec_nat_vec_rb").await); + server.shutdown().await; +} + +#[tokio::test] +async fn document_drop_index_opcode_drops_and_rolls_back_inside_a_block() { + let server = NativeTestServer::start().await; + let mut client = Client::connect(&server).await; + client + .sql( + "CREATE COLLECTION nat_doc (id TEXT PRIMARY KEY, region TEXT) \ + WITH (engine='document_schemaless')", + ) + .await; + client + .sql("CREATE INDEX nat_doc_region ON nat_doc (region)") + .await; + + client.sql("BEGIN").await; + client + .op(OpCode::DocumentDropIndex, on_field("nat_doc", "region")) + .await; + assert!(!client.lists("nat_doc_region").await); + client.sql("ROLLBACK").await; + assert!(client.lists("nat_doc_region").await); + + client + .op(OpCode::DocumentDropIndex, on_field("nat_doc", "region")) + .await; + assert!(!client.lists("nat_doc_region").await); + server.shutdown().await; +} + +#[tokio::test] +async fn kv_index_opcodes_make_a_catalog_index() { + let server = NativeTestServer::start().await; + let mut client = Client::connect(&server).await; + client + .sql("CREATE COLLECTION nat_kv (key TEXT PRIMARY KEY) WITH (engine='kv')") + .await; + client + .sql("INSERT INTO nat_kv (key, bucket) VALUES ('s1', 'A')") + .await; + + client.sql("BEGIN").await; + client + .op(OpCode::KvRegisterIndex, on_field("nat_kv", "bucket")) + .await; + assert!(client.lists("idx_nat_kv_bucket").await); + client.sql("ROLLBACK").await; + assert!(!client.lists("idx_nat_kv_bucket").await); + + client + .op(OpCode::KvRegisterIndex, on_field("nat_kv", "bucket")) + .await; + assert!(client.lists("idx_nat_kv_bucket").await); + client + .sql("INSERT INTO nat_kv (key, bucket) VALUES ('s2', 'B')") + .await; + let rows = client + .sql("SELECT key FROM nat_kv WHERE bucket = 'A'") + .await + .rows + .unwrap_or_default(); + assert_eq!(rows, vec![vec![Value::String("s1".to_string())]]); + + client.sql("BEGIN").await; + client + .op(OpCode::KvDropIndex, on_field("nat_kv", "bucket")) + .await; + assert!(!client.lists("idx_nat_kv_bucket").await); + client.sql("ROLLBACK").await; + assert!(client.lists("idx_nat_kv_bucket").await); + + client + .op(OpCode::KvDropIndex, on_field("nat_kv", "bucket")) + .await; + assert!(!client.lists("idx_nat_kv_bucket").await); + server.shutdown().await; +} + +fn register(collection: &str, paths: &[&str]) -> TextFields { + TextFields { + collection: Some(collection.to_string()), + index_paths: Some(paths.iter().map(|p| p.to_string()).collect()), + ..TextFields::default() + } +} + +#[tokio::test] +async fn document_register_opcode_creates_the_collection_and_its_indexes() { + let server = NativeTestServer::start().await; + let mut client = Client::connect(&server).await; + + client.sql("BEGIN").await; + client + .op( + OpCode::DocumentRegister, + register("nat_reg_rb", &["region"]), + ) + .await; + assert!(client.lists("idx_nat_reg_rb_region").await); + client.sql("ROLLBACK").await; + assert!(!client.lists("idx_nat_reg_rb_region").await); + + client + .op( + OpCode::DocumentRegister, + register("nat_reg", &["region", "$.status"]), + ) + .await; + assert!(client.lists("idx_nat_reg_region").await); + assert!(client.lists("idx_nat_reg_status").await); + client + .sql("INSERT INTO nat_reg (id, region, status) VALUES ('a', 'eu', 'open')") + .await; + let rows = client + .sql("SELECT id FROM nat_reg WHERE region = 'eu'") + .await + .rows + .unwrap_or_default(); + assert_eq!(rows, vec![vec![Value::String("a".to_string())]]); + + // A repeated register is a no-op. + client + .op( + OpCode::DocumentRegister, + register("nat_reg", &["region", "$.status"]), + ) + .await; + server.shutdown().await; +} diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 3841777e5..95d642049 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -109,6 +109,7 @@ mod merge_insert_surrogate_stability; mod move_tenant_idempotent; mod move_tenant_round_trip; mod native_cluster_array; +mod native_index_ddl_restart; mod object_literal_dml_row_level_security; mod object_literal_trailing_clause; mod pg_catalog_oid_stability; @@ -295,9 +296,11 @@ mod timeseries_read_row_level_security; mod timeseries_write_row_level_security; mod transactional_ddl_atomicity; mod transactional_ddl_compensation; +mod transactional_ddl_index_families; mod transactional_ddl_visibility; mod transactional_ddl_visibility_routines; mod transactional_ddl_visibility_sequence; +mod transactional_sorted_index_reads; mod trigger_e2e; mod truncate_engine_conformance; mod truncate_engine_conformance_columnar_family; diff --git a/nodedb/tests/wire/cases/native_index_ddl_restart.rs b/nodedb/tests/wire/cases/native_index_ddl_restart.rs new file mode 100644 index 000000000..5dac075c6 --- /dev/null +++ b/nodedb/tests/wire/cases/native_index_ddl_restart.rs @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Indexes made by the native `KvRegisterIndex` and `DocumentRegister` +//! opcodes are catalog indexes: they survive a restart, SQL lists them, and +//! SQL `DROP INDEX` removes them. + +use nodedb_test_support::native_harness::{do_handshake, send_request}; +use nodedb_types::protocol::HelloFrame; +use nodedb_types::protocol::opcodes::ResponseStatus; +use nodedb_types::protocol::text_fields::TextFields; +use nodedb_types::protocol::{NativeResponse, OpCode}; +use tokio::net::TcpStream; + +use crate::harness::TestServer; + +async fn native_session(server: &TestServer) -> TcpStream { + let addr = std::net::SocketAddr::new(std::net::Ipv4Addr::LOCALHOST.into(), server.native_port); + let (stream, _ack) = do_handshake(addr, &HelloFrame::current()) + .await + .expect("native handshake"); + stream +} + +async fn run_op(stream: &mut TcpStream, seq: u64, op: OpCode, fields: TextFields) { + let response: NativeResponse = send_request(stream, seq, op, fields).await; + assert_eq!( + response.status, + ResponseStatus::Ok, + "{op:?} must succeed: {response:?}" + ); +} + +async fn lists(server: &TestServer, index: &str) -> bool { + server + .query_text("SHOW INDEXES") + .await + .unwrap() + .iter() + .any(|name| name == index) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn native_registered_indexes_survive_a_restart() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION nat_rs_kv (key TEXT PRIMARY KEY) WITH (engine='kv')") + .await + .unwrap(); + server + .exec("INSERT INTO nat_rs_kv (key, bucket) VALUES ('s1', 'A')") + .await + .unwrap(); + + let mut stream = native_session(&server).await; + run_op( + &mut stream, + 1, + OpCode::KvRegisterIndex, + TextFields { + collection: Some("nat_rs_kv".to_string()), + field: Some("bucket".to_string()), + ..TextFields::default() + }, + ) + .await; + run_op( + &mut stream, + 2, + OpCode::DocumentRegister, + TextFields { + collection: Some("nat_rs_doc".to_string()), + index_paths: Some(vec!["region".to_string()]), + ..TextFields::default() + }, + ) + .await; + drop(stream); + assert!(lists(&server, "idx_nat_rs_kv_bucket").await); + assert!(lists(&server, "idx_nat_rs_doc_region").await); + + let (server, dir) = server.take_dir(); + server.graceful_shutdown().await; + let (server, _dir) = TestServer::open_on_path(dir).await; + + assert!( + lists(&server, "idx_nat_rs_kv_bucket").await, + "the KV index is in the catalog after a restart" + ); + assert!( + lists(&server, "idx_nat_rs_doc_region").await, + "the registered document index is in the catalog after a restart" + ); + assert_eq!( + server + .query_text("SELECT key FROM nat_rs_kv WHERE bucket = 'A'") + .await + .unwrap(), + vec!["s1".to_string()] + ); + + server + .exec("DROP INDEX idx_nat_rs_kv_bucket") + .await + .expect("SQL drops the index the opcode made"); + assert!(!lists(&server, "idx_nat_rs_kv_bucket").await); +} diff --git a/nodedb/tests/wire/cases/sql_transactions_columnar_overlay.rs b/nodedb/tests/wire/cases/sql_transactions_columnar_overlay.rs index d8c8794c2..9873b9274 100644 --- a/nodedb/tests/wire/cases/sql_transactions_columnar_overlay.rs +++ b/nodedb/tests/wire/cases/sql_transactions_columnar_overlay.rs @@ -4,9 +4,8 @@ //! transaction -- staged into the per-transaction overlay with //! read-your-own-writes on columnar scans, a real affected-row count, and //! `ROLLBACK` discarding the staged rows -- mirroring the Document/KV/FTS -//! staging already in place. COMMIT's durable replay is unchanged: the -//! buffered `ColumnarOp::Insert` plan is still replayed through -//! `execute_columnar_insert` inside the COMMIT `TransactionBatch`. +//! staging already in place. COMMIT resolves the staged rows into the +//! transaction's redo record, which the redo install applies. //! //! Columnar is the first non-point-write, non-Document/KV engine wired into //! the staging overlay; row identity is the cross-engine surrogate rather diff --git a/nodedb/tests/wire/cases/sql_transactions_graph_edge_delete_overlay.rs b/nodedb/tests/wire/cases/sql_transactions_graph_edge_delete_overlay.rs index c7bdf5805..b76202919 100644 --- a/nodedb/tests/wire/cases/sql_transactions_graph_edge_delete_overlay.rs +++ b/nodedb/tests/wire/cases/sql_transactions_graph_edge_delete_overlay.rs @@ -8,7 +8,7 @@ //! //! Every edge here is a SELF-LOOP (`_from == _to == node`), keeping both //! endpoints on one home vShard so the delete is SINGLE-HOME and stages -//! through the single-shard WAL + `TransactionBatch` commit path. +//! through the single-shard redo-record commit path. use crate::harness::TestServer; @@ -118,8 +118,8 @@ async fn in_tx_edge_delete_commit_persists_removal() { .expect("in-tx GRAPH DELETE EDGE should stage at statement time"); server.exec("COMMIT").await.unwrap(); - // A single-home staged edge delete replays durably at COMMIT via the - // single-shard WAL + TransactionBatch path. + // A single-home staged edge delete lands durably at COMMIT through the + // single-shard redo-record install. let after_commit = neighbors_of(&server, "commit_node", "knows").await; assert!( !after_commit.contains(&"commit_node".to_string()), diff --git a/nodedb/tests/wire/cases/sql_transactions_timeseries_overlay.rs b/nodedb/tests/wire/cases/sql_transactions_timeseries_overlay.rs index b0ad403eb..8ae42baf8 100644 --- a/nodedb/tests/wire/cases/sql_transactions_timeseries_overlay.rs +++ b/nodedb/tests/wire/cases/sql_transactions_timeseries_overlay.rs @@ -4,9 +4,8 @@ //! transaction -- staged into the per-transaction overlay with //! read-your-own-writes on RAW timeseries scans, a real affected-row count, //! and `ROLLBACK` discarding the staged rows -- mirroring the columnar staging -//! already in place. COMMIT's durable replay is unchanged: the buffered -//! `TimeseriesOp::Ingest` plan is still replayed through -//! `execute_timeseries_ingest` inside the COMMIT `TransactionBatch`. +//! already in place. COMMIT resolves the staged ingest into the +//! transaction's redo record, which the redo install applies. //! //! A timeseries base row has no cross-engine surrogate identity in the scan //! (it is keyed internally by `series_id`), so the overlay merge is diff --git a/nodedb/tests/wire/cases/transactional_ddl_index_families.rs b/nodedb/tests/wire/cases/transactional_ddl_index_families.rs new file mode 100644 index 000000000..d84efaaf3 --- /dev/null +++ b/nodedb/tests/wire/cases/transactional_ddl_index_families.rs @@ -0,0 +1,329 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Index DDL inside an explicit transaction, one family at a time: document +//! `CREATE`/`DROP INDEX`, KV `CREATE`/`DROP SORTED INDEX`, and +//! `CREATE`/`ALTER`/`DROP VECTOR INDEX`. +//! +//! Each family is visible to later statements of its own transaction, +//! applies at COMMIT, and leaves nothing behind on ROLLBACK: neither the +//! catalog rows nor the engine state (a backfilled index, a sorted-index +//! tree, a declared vector dimension). + +use crate::harness::TestServer; + +async fn indexes(server: &TestServer) -> Vec { + server.query_text("SHOW INDEXES").await.unwrap() +} + +async fn document_collection(server: &TestServer, name: &str) { + server + .exec(&format!( + "CREATE COLLECTION {name} (id TEXT PRIMARY KEY, region TEXT) \ + WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + server + .exec(&format!( + "INSERT INTO {name} (id, region) VALUES ('a', 'eu')" + )) + .await + .unwrap(); +} + +async fn eu_rows(server: &TestServer, name: &str) -> Vec { + server + .query_text(&format!("SELECT id FROM {name} WHERE region = 'eu'")) + .await + .unwrap() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn document_index_created_in_a_transaction_is_visible_then_committed() { + let server = TestServer::start().await; + document_collection(&server, "txn_fam_doc_c").await; + + server.exec("BEGIN").await.unwrap(); + server + .exec("CREATE INDEX txn_fam_doc_idx ON txn_fam_doc_c(region)") + .await + .unwrap(); + assert!( + indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_doc_idx"), + "the transaction lists its own index" + ); + server.exec("COMMIT").await.unwrap(); + + assert!( + indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_doc_idx") + ); + assert_eq!( + eu_rows(&server, "txn_fam_doc_c").await, + vec!["a".to_string()] + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn document_index_rolled_back_leaves_nothing() { + let server = TestServer::start().await; + document_collection(&server, "txn_fam_doc_rb").await; + + server.exec("BEGIN").await.unwrap(); + server + .exec("CREATE INDEX txn_fam_doc_rb_idx ON txn_fam_doc_rb(region)") + .await + .unwrap(); + server.exec("ROLLBACK").await.unwrap(); + + assert!( + !indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_doc_rb_idx") + ); + // The name and the engine slot are free: the same index builds again. + server + .exec("CREATE INDEX txn_fam_doc_rb_idx ON txn_fam_doc_rb(region)") + .await + .expect("a rolled-back CREATE INDEX leaves the name free"); + assert_eq!( + eu_rows(&server, "txn_fam_doc_rb").await, + vec!["a".to_string()] + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn document_index_dropped_then_rolled_back_keeps_its_entries() { + let server = TestServer::start().await; + document_collection(&server, "txn_fam_doc_drop").await; + server + .exec("CREATE INDEX txn_fam_doc_drop_idx ON txn_fam_doc_drop(region)") + .await + .unwrap(); + + server.exec("BEGIN").await.unwrap(); + server + .exec("DROP INDEX txn_fam_doc_drop_idx") + .await + .unwrap(); + assert!( + !indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_doc_drop_idx"), + "the transaction no longer lists the index it dropped" + ); + server.exec("ROLLBACK").await.unwrap(); + + assert!( + indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_doc_drop_idx") + ); + assert_eq!( + eu_rows(&server, "txn_fam_doc_drop").await, + vec!["a".to_string()], + "the index keeps its entries: the purge never ran" + ); +} + +async fn kv_board(server: &TestServer, name: &str) { + server + .exec(&format!( + "CREATE COLLECTION {name} (id STRING PRIMARY KEY, score INT) WITH (engine='kv')" + )) + .await + .unwrap(); + server + .exec(&format!("INSERT INTO {name} {{ id: 'p1', score: 10 }}")) + .await + .unwrap(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sorted_index_created_in_a_transaction_is_visible_then_committed() { + let server = TestServer::start().await; + kv_board(&server, "txn_fam_sorted").await; + + server.exec("BEGIN").await.unwrap(); + server + .exec("CREATE SORTED INDEX txn_fam_sorted_idx ON txn_fam_sorted (score DESC) KEY id") + .await + .unwrap(); + assert!( + indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_sorted_idx") + ); + server.exec("COMMIT").await.unwrap(); + + server + .query_text("SELECT SORTED_COUNT(txn_fam_sorted_idx)") + .await + .expect("the committed sorted index has its tree"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sorted_index_rolled_back_leaves_nothing() { + let server = TestServer::start().await; + kv_board(&server, "txn_fam_sorted_rb").await; + + server.exec("BEGIN").await.unwrap(); + server + .exec("CREATE SORTED INDEX txn_fam_sorted_rb_idx ON txn_fam_sorted_rb (score DESC) KEY id") + .await + .unwrap(); + server.exec("ROLLBACK").await.unwrap(); + + assert!( + !indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_sorted_rb_idx") + ); + server + .expect_error( + "SELECT SORTED_COUNT(txn_fam_sorted_rb_idx)", + "does not exist", + ) + .await; + server + .exec("CREATE SORTED INDEX txn_fam_sorted_rb_idx ON txn_fam_sorted_rb (score DESC) KEY id") + .await + .expect("a rolled-back CREATE SORTED INDEX leaves no tree behind"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sorted_index_dropped_then_rolled_back_keeps_its_tree() { + let server = TestServer::start().await; + kv_board(&server, "txn_fam_sorted_drop").await; + server + .exec("CREATE SORTED INDEX txn_fam_sorted_drop_idx ON txn_fam_sorted_drop (score DESC) KEY id") + .await + .unwrap(); + + server.exec("BEGIN").await.unwrap(); + server + .exec("DROP SORTED INDEX txn_fam_sorted_drop_idx") + .await + .unwrap(); + server.exec("ROLLBACK").await.unwrap(); + + server + .query_text("SELECT SORTED_COUNT(txn_fam_sorted_drop_idx)") + .await + .expect("the tree survives a rolled-back DROP SORTED INDEX"); +} + +async fn insert_vector(server: &TestServer, coll: &str, id: &str, v: &[f32]) -> Result<(), String> { + let arr = v + .iter() + .map(|x| x.to_string()) + .collect::>() + .join(","); + server + .exec(&format!( + "INSERT INTO {coll} (id, embedding) VALUES ('{id}', ARRAY[{arr}])" + )) + .await +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn vector_index_created_and_altered_in_a_transaction_commits() { + let server = TestServer::start().await; + server.exec("CREATE COLLECTION txn_fam_vec").await.unwrap(); + + server.exec("BEGIN").await.unwrap(); + server + .exec("CREATE VECTOR INDEX txn_fam_vec_idx ON txn_fam_vec METRIC l2 DIM 3") + .await + .unwrap(); + server + .exec("ALTER VECTOR INDEX txn_fam_vec_idx ON txn_fam_vec SET (m = 32)") + .await + .expect("ALTER resolves the index this transaction created"); + server.exec("COMMIT").await.unwrap(); + + assert!( + indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_vec_idx") + ); + insert_vector(&server, "txn_fam_vec", "v1", &[1.0, 0.0, 0.0]) + .await + .expect("a vector of the committed dimension inserts"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn vector_index_rolled_back_leaves_no_declared_dimension() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION txn_fam_vec_rb") + .await + .unwrap(); + + server.exec("BEGIN").await.unwrap(); + server + .exec("CREATE VECTOR INDEX txn_fam_vec_rb_idx ON txn_fam_vec_rb METRIC l2 DIM 3") + .await + .unwrap(); + server.exec("ROLLBACK").await.unwrap(); + + assert!( + !indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_vec_rb_idx") + ); + // A 5-wide vector would be refused against a lingering DIM 3 declaration. + insert_vector(&server, "txn_fam_vec_rb", "v1", &[1.0, 0.0, 0.0, 0.0, 0.0]) + .await + .expect("the rolled-back index declared no dimension"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn vector_index_dropped_then_rolled_back_still_serves() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION txn_fam_vec_drop") + .await + .unwrap(); + server + .exec("CREATE VECTOR INDEX txn_fam_vec_drop_idx ON txn_fam_vec_drop METRIC l2 DIM 3") + .await + .unwrap(); + insert_vector(&server, "txn_fam_vec_drop", "v1", &[1.0, 0.0, 0.0]) + .await + .unwrap(); + + server.exec("BEGIN").await.unwrap(); + server + .exec("DROP VECTOR INDEX txn_fam_vec_drop_idx") + .await + .unwrap(); + server.exec("ROLLBACK").await.unwrap(); + + assert!( + indexes(&server) + .await + .iter() + .any(|n| n == "txn_fam_vec_drop_idx") + ); + let nearest = server + .query_text( + "SELECT id FROM txn_fam_vec_drop \ + ORDER BY vector_distance(embedding, ARRAY[1.0,0.0,0.0]) LIMIT 1", + ) + .await + .unwrap(); + assert_eq!(nearest, vec!["v1".to_string()]); +} diff --git a/nodedb/tests/wire/cases/transactional_sorted_index_reads.rs b/nodedb/tests/wire/cases/transactional_sorted_index_reads.rs new file mode 100644 index 000000000..e242ce6ff --- /dev/null +++ b/nodedb/tests/wire/cases/transactional_sorted_index_reads.rs @@ -0,0 +1,164 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sorted-index reads inside an explicit transaction. +//! +//! A transaction sees its own DDL and its own writes. `RANK`, `TOPK`, +//! `RANGE` and `SORTED_COUNT` answer inside the transaction that created the +//! index, and over the rows that transaction staged. Nothing of either +//! survives a ROLLBACK. + +use crate::harness::TestServer; + +async fn board(server: &TestServer, name: &str) { + server + .exec(&format!( + "CREATE COLLECTION {name} (id STRING PRIMARY KEY, score INT) WITH (engine='kv')" + )) + .await + .unwrap(); + server + .exec(&format!("INSERT INTO {name} {{ id: 'p1', score: 10 }}")) + .await + .unwrap(); +} + +/// The keys `TOPK` delivers, in rank order. Rows are `(rank, key)`. +async fn topk_keys(server: &TestServer, index: &str) -> Vec { + server + .query_rows(&format!("SELECT * FROM TOPK({index}, 10)")) + .await + .unwrap_or_else(|e| panic!("TOPK({index}): {e}")) + .into_iter() + .map(|row| row.get(1).cloned().unwrap_or_default()) + .collect() +} + +/// The keys `RANGE` delivers for an inclusive score window, sorted. +async fn range_keys(server: &TestServer, index: &str, low: i64, high: i64) -> Vec { + let mut keys: Vec = server + .query_rows(&format!("SELECT * FROM RANGE({index}, {low}, {high})")) + .await + .unwrap_or_else(|e| panic!("RANGE({index}): {e}")) + .into_iter() + .map(|row| row.get(1).cloned().unwrap_or_default()) + .collect(); + keys.sort(); + keys +} + +/// The integer a single-cell JSON reply carries under `field`. +fn json_field(text: &str, field: &str) -> Option { + let needle = format!("\"{field}\":"); + let rest = text.split_once(&needle)?.1; + let digits: String = rest + .trim_start() + .chars() + .take_while(|c| c.is_ascii_digit() || *c == '-') + .collect(); + digits.parse().ok() +} + +async fn sorted_count(server: &TestServer, index: &str) -> Option { + let rows = server + .query_text(&format!("SELECT SORTED_COUNT({index})")) + .await + .unwrap_or_else(|e| panic!("SORTED_COUNT({index}): {e}")); + json_field(rows.first()?, "count") +} + +async fn rank_of(server: &TestServer, index: &str, id: &str) -> Option { + let rows = server + .query_text(&format!("SELECT RANK({index}, '{id}')")) + .await + .unwrap_or_else(|e| panic!("RANK({index}, {id}): {e}")); + json_field(rows.first()?, "rank") +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_sorted_index_created_in_a_transaction_answers_every_read_in_it() { + let server = TestServer::start().await; + board(&server, "txn_sr_new").await; + + server.exec("BEGIN").await.unwrap(); + server + .exec("CREATE SORTED INDEX txn_sr_new_idx ON txn_sr_new (score DESC) KEY id") + .await + .unwrap(); + server + .exec("INSERT INTO txn_sr_new { id: 'p2', score: 20 }") + .await + .unwrap(); + + assert_eq!( + topk_keys(&server, "txn_sr_new_idx").await, + vec!["p2".to_string(), "p1".to_string()], + "TOPK ranks the base row and the staged row" + ); + assert_eq!(sorted_count(&server, "txn_sr_new_idx").await, Some(2)); + assert_eq!(rank_of(&server, "txn_sr_new_idx", "p1").await, Some(2)); + assert_eq!( + range_keys(&server, "txn_sr_new_idx", 15, 25).await, + vec!["p2".to_string()] + ); + server.exec("COMMIT").await.unwrap(); + + assert_eq!( + topk_keys(&server, "txn_sr_new_idx").await, + vec!["p2".to_string(), "p1".to_string()], + "the committed tree holds what the transaction read" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_committed_sorted_index_reads_the_transactions_own_writes() { + let server = TestServer::start().await; + board(&server, "txn_sr_own").await; + server + .exec("CREATE SORTED INDEX txn_sr_own_idx ON txn_sr_own (score DESC) KEY id") + .await + .unwrap(); + + server.exec("BEGIN").await.unwrap(); + server + .exec("INSERT INTO txn_sr_own { id: 'p2', score: 20 }") + .await + .unwrap(); + server + .exec("DELETE FROM txn_sr_own WHERE id = 'p1'") + .await + .unwrap(); + assert_eq!( + topk_keys(&server, "txn_sr_own_idx").await, + vec!["p2".to_string()], + "the staged insert is ranked and the staged delete is gone" + ); + assert_eq!(sorted_count(&server, "txn_sr_own_idx").await, Some(1)); + server.exec("ROLLBACK").await.unwrap(); + + assert_eq!( + topk_keys(&server, "txn_sr_own_idx").await, + vec!["p1".to_string()], + "a rolled-back write never reaches the tree" + ); + assert_eq!(sorted_count(&server, "txn_sr_own_idx").await, Some(1)); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_sorted_index_created_then_dropped_in_a_transaction_is_gone_for_it() { + let server = TestServer::start().await; + board(&server, "txn_sr_gone").await; + + server.exec("BEGIN").await.unwrap(); + server + .exec("CREATE SORTED INDEX txn_sr_gone_idx ON txn_sr_gone (score DESC) KEY id") + .await + .unwrap(); + server + .exec("DROP SORTED INDEX txn_sr_gone_idx") + .await + .unwrap(); + server + .expect_error("SELECT SORTED_COUNT(txn_sr_gone_idx)", "does not exist") + .await; + server.exec("ROLLBACK").await.unwrap(); +} From 3960fd638ab2b2ad1584858fd05693bd9850795e Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 08:39:52 +0800 Subject: [PATCH 28/64] feat(executor): stamp checkpoints with an outcome-floor and applied-above set A checkpoint used to record only the highest LSN applied. LSNs are node-global and reach a core out of mint order, so a record with a lower LSN can still be in flight when a higher one applies. A max-applied stamp then claims the lower record and restart replay skips it, losing the write. Replace the single LSN with a ReplayStamp: a `prefix` (the core's outcome floor when the artifact was written, below which every record has a final outcome) plus `applied_above`, the disjoint LSN ranges applied above that floor. Replay skips a record exactly when the stamp says so. Add LsnRanges, a BTreeMap-backed disjoint-range set that only merges touching LSNs and never bridges a gap, to track applied LSNs above the floor. Wire the stamp through every checkpoint format (columnar, KV, graph-label, sparse-vector, spatial, sync-hwm, vector) and through wal_replay_all and replay_floors so restart replay gates on it. Add an async fail-point action, WaitForFile, so a test can park one in-flight write at the funnel gate without blocking its core, plus a Control-Plane fail_gate module that calls it after a logged write's WAL append and before dispatch. Add a crash-replay test proving a sorted-index write parked at the gate while later writes are checkpointed still applies after a crash and restart. --- nodedb-types/src/fail_point.rs | 45 +++- nodedb/src/control/fail_gate.rs | 47 ++++ nodedb/src/control/mod.rs | 2 + .../submit_write/funnel/driver.rs | 7 + .../src/data/executor/applied_prefix/mod.rs | 3 + .../data/executor/applied_prefix/ranges.rs | 134 ++++++++++ .../src/data/executor/applied_prefix/stamp.rs | 198 ++++++++++++++ .../data/executor/applied_prefix/tracker.rs | 206 +++++++++++++- .../data/executor/checkpoint_decode_error.rs | 9 + .../executor/columnar_checkpoint/format.rs | 23 +- .../data/executor/columnar_checkpoint/load.rs | 36 ++- .../executor/columnar_checkpoint/manifest.rs | 7 + .../data/executor/columnar_checkpoint/mod.rs | 4 +- .../executor/columnar_checkpoint/write.rs | 64 +++-- .../core_loop/checkpoint_floors/state.rs | 23 +- nodedb/src/data/executor/core_loop/tick.rs | 10 + .../data/executor/core_loop/write_index.rs | 37 +++ .../executor/graph_label_checkpoint/write.rs | 4 +- .../handlers/control/checkpoint_crdt.rs | 4 +- .../control/checkpoint_durable_lsn.rs | 7 +- nodedb/src/data/executor/handlers/kv/ttl.rs | 7 +- .../data/executor/handlers/timeseries_wal.rs | 12 +- .../handlers/transaction/redo_apply/cover.rs | 8 + .../handlers/transaction/redo_apply/entry.rs | 33 +++ .../src/data/executor/kv_checkpoint/format.rs | 21 +- .../src/data/executor/kv_checkpoint/load.rs | 69 ++++- .../data/executor/kv_checkpoint/manifest.rs | 7 + nodedb/src/data/executor/kv_checkpoint/mod.rs | 4 +- .../src/data/executor/kv_checkpoint/write.rs | 55 ++-- nodedb/src/data/executor/replay_floors.rs | 102 +++---- .../sparse_vector_checkpoint/write.rs | 4 +- .../data/executor/spatial_checkpoint/write.rs | 4 +- .../executor/sync_hwm_checkpoint/write.rs | 4 +- .../data/executor/vector_checkpoint/write.rs | 4 +- nodedb/src/data/executor/wal_replay_all.rs | 253 ++++++++++++++++++ nodedb/tests/crash_replay_stamp.rs | 251 +++++++++++++++++ 36 files changed, 1518 insertions(+), 190 deletions(-) create mode 100644 nodedb/src/control/fail_gate.rs create mode 100644 nodedb/src/data/executor/applied_prefix/ranges.rs create mode 100644 nodedb/src/data/executor/applied_prefix/stamp.rs create mode 100644 nodedb/tests/crash_replay_stamp.rs diff --git a/nodedb-types/src/fail_point.rs b/nodedb-types/src/fail_point.rs index 56e036fc2..88948da0e 100644 --- a/nodedb-types/src/fail_point.rs +++ b/nodedb-types/src/fail_point.rs @@ -15,6 +15,10 @@ //! [`fail_point_err!`] can honour this — the call site supplies the //! mapping into its own error type, so no crate has to know about //! anyone else's. +//! - `WaitForFile(path)`: park the call site until `path` exists, so a test +//! releases it at the moment it chooses. Only an async call site can +//! honour this: it awaits the file with the crate's own timer and parks +//! only its own task. A synchronous call site refuses it. //! //! The framework is deliberately tiny — no fail-rs dep, no parsing of env //! vars, no list of probabilities. Tests install actions explicitly via @@ -47,6 +51,9 @@ mod imp { /// Return an error from the injected call site, carrying this detail. /// Ignored by bare `fail_point!` — use `fail_point_err!`. Fail(String), + /// Park the call site until this file exists. Only an async call site + /// that looks the action up with [`lookup`] can honour it. + WaitForFile(std::path::PathBuf), } /// Environment variable read once, the first time any fail point is @@ -54,7 +61,8 @@ mod imp { /// in-process `set` API cannot reach a server the test only supervises. /// /// Format: comma-separated `name=action`, where action is `panic`, - /// `sleep()`, or `fail()`. For example: + /// `abort`, `sleep()`, `fail()`, or `wait_file()`. + /// For example: /// `NODEDB_FAILPOINTS='checkpoint::after_marker_before_truncate=panic'` pub const FAILPOINTS_ENV: &str = "NODEDB_FAILPOINTS"; @@ -85,6 +93,11 @@ mod imp { rest if rest.starts_with("fail(") && rest.ends_with(')') => { FailAction::Fail(rest["fail(".len()..rest.len() - 1].to_string()) } + rest if rest.starts_with("wait_file(") && rest.ends_with(')') => { + FailAction::WaitForFile(std::path::PathBuf::from( + &rest["wait_file(".len()..rest.len() - 1], + )) + } other => panic!("{FAILPOINTS_ENV} entry {entry:?} has unknown action {other:?}"), }; actions.insert(name.trim().to_string(), action); @@ -134,6 +147,12 @@ mod imp { "fail_point {name} installed Fail({detail}) but the call site cannot return an error — use fail_point_err!" ) } + // Blocking the thread would stall every task that shares it. + FailAction::WaitForFile(path) => panic!( + "fail_point {name} installed WaitForFile({}) but the call site is \ + synchronous — only an async call site can park", + path.display() + ), } } } @@ -157,6 +176,11 @@ mod imp { std::thread::sleep(d); None } + Some(FailAction::WaitForFile(path)) => panic!( + "fail_point {name} installed WaitForFile({}) but the call site is synchronous \ + — only an async call site can park", + path.display() + ), None => None, } } @@ -279,6 +303,25 @@ mod tests { )); } + #[test] + fn env_spec_parses_a_file_gate() { + let actions = super::imp::parse_env(Some("d::gate=wait_file(/tmp/release-d)")); + assert!(matches!( + actions.get("d::gate"), + Some(FailAction::WaitForFile(path)) if path == std::path::Path::new("/tmp/release-d") + )); + } + + #[test] + #[should_panic(expected = "only an async call site can park")] + fn a_file_gate_at_a_synchronous_call_site_is_loud() { + let _g = FailGuard::install( + "nodedb::test::gate_at_sync", + FailAction::WaitForFile(std::path::PathBuf::from("/nonexistent")), + ); + eval("nodedb::test::gate_at_sync"); + } + #[test] fn empty_env_spec_arms_nothing() { assert!(super::imp::parse_env(None).is_empty()); diff --git a/nodedb/src/control/fail_gate.rs b/nodedb/src/control/fail_gate.rs new file mode 100644 index 000000000..838babaaa --- /dev/null +++ b/nodedb/src/control/fail_gate.rs @@ -0,0 +1,47 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Async fail-point gates for Control-Plane code. +//! +//! A crash test sometimes needs one request parked at a precise point while +//! every other request proceeds. A synchronous fail point cannot do that: it +//! blocks the thread, and every task that shares the thread stalls with it. +//! The gate here awaits a file with a Tokio timer, so only the task that +//! reached it waits. The test creates the file to release it. +//! +//! Compiled only with the `failpoints` feature, and called only from async +//! Control-Plane code, never from the Data Plane. + +use std::time::Duration; + +use crate::bridge::envelope::PhysicalPlan; +use crate::fail_point::{FailAction, lookup}; +use crate::types::Lsn; + +/// How often a parked task checks for its release file. +const GATE_POLL: Duration = Duration::from_millis(20); + +/// Park until the file armed for `name` with `wait_file()` exists. No-op +/// when nothing is armed for `name`. +pub(crate) async fn wait(name: &str) { + let Some(FailAction::WaitForFile(path)) = lookup(name) else { + return; + }; + while !path.exists() { + tokio::time::sleep(GATE_POLL).await; + } +} + +/// The write funnel's gate: after a logged write's WAL record is appended and +/// before any core holds the request. Named per collection, +/// `funnel::before_dispatch::`, and only a write carrying a WAL +/// LSN reaches it. +pub(crate) async fn before_dispatch(plan: &PhysicalPlan, wal_lsn: Option) { + if wal_lsn.is_none() { + return; + } + let name = format!( + "funnel::before_dispatch::{}", + plan.collection().unwrap_or_default() + ); + wait(&name).await; +} diff --git a/nodedb/src/control/mod.rs b/nodedb/src/control/mod.rs index a39ce8645..a89fe362f 100644 --- a/nodedb/src/control/mod.rs +++ b/nodedb/src/control/mod.rs @@ -22,6 +22,8 @@ pub mod distributed_applier; pub mod event_action_error; pub mod event_trigger; pub mod exec_receiver; +#[cfg(feature = "failpoints")] +pub(crate) mod fail_gate; pub mod gateway; pub mod insert_select; pub mod lease; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index b53a387e4..5c9bbf659 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -154,6 +154,13 @@ pub(crate) async fn submit_write( let wal_lsn = wal_append_outcome.wal_lsn; let resolved_now_ms = wal_append_outcome.resolved_now_ms; + // A crash test parks one collection's logged write here: its LSN is + // minted, no core holds it, and only this task waits, so a later write + // with a higher LSN applies first. The write still holds its own per-key + // admission guards, which no write to another key contends on. + #[cfg(feature = "failpoints")] + crate::control::fail_gate::before_dispatch(&plan, wal_lsn).await; + // Build the wire request and hand it to the Data-Plane dispatcher. let dispatched = dispatch_to_data_plane( shared, diff --git a/nodedb/src/data/executor/applied_prefix/mod.rs b/nodedb/src/data/executor/applied_prefix/mod.rs index 52611e21d..11a39f686 100644 --- a/nodedb/src/data/executor/applied_prefix/mod.rs +++ b/nodedb/src/data/executor/applied_prefix/mod.rs @@ -1,5 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 +mod ranges; +pub(crate) mod stamp; mod tracker; +pub(crate) use stamp::{InvalidReplayStamp, ReplayStamp}; pub(in crate::data::executor) use tracker::AppliedPrefix; diff --git a/nodedb/src/data/executor/applied_prefix/ranges.rs b/nodedb/src/data/executor/applied_prefix/ranges.rs new file mode 100644 index 000000000..cdb05505e --- /dev/null +++ b/nodedb/src/data/executor/applied_prefix/ranges.rs @@ -0,0 +1,134 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A set of LSNs kept as disjoint inclusive ranges. +//! +//! Two ranges merge only when they touch: `[5, 7]` and `[8, 9]` become +//! `[5, 9]`. A gap is never bridged. The core cannot tell a gap LSN that +//! belongs to another core from one still on its way to this core, and +//! marking the second as applied would make restart replay skip a record no +//! checkpoint holds. + +use std::collections::BTreeMap; + +use super::stamp::LsnRange; + +/// LSNs as `start -> end` ranges, both ends inclusive. +#[derive(Debug, Default, Clone)] +pub(in crate::data::executor) struct LsnRanges { + ranges: BTreeMap, +} + +impl LsnRanges { + /// Add `lsn`, merging it with the ranges it touches. + pub(in crate::data::executor) fn insert(&mut self, lsn: u64) { + if self.contains(lsn) { + return; + } + let mut start = lsn; + let mut end = lsn; + if let Some((&prev_start, &prev_end)) = self.ranges.range(..lsn).next_back() + && prev_end.checked_add(1) == Some(lsn) + { + self.ranges.remove(&prev_start); + start = prev_start; + } + if let Some(next) = lsn.checked_add(1) + && let Some(next_end) = self.ranges.remove(&next) + { + end = next_end; + } + self.ranges.insert(start, end); + } + + /// Whether `lsn` lies in a range. + pub(in crate::data::executor) fn contains(&self, lsn: u64) -> bool { + self.ranges + .range(..=lsn) + .next_back() + .is_some_and(|(_, end)| lsn <= *end) + } + + /// Drop every LSN at or below `floor`. + pub(in crate::data::executor) fn prune_through(&mut self, floor: u64) { + let Some(above) = floor.checked_add(1) else { + self.ranges.clear(); + return; + }; + let straddling = self + .ranges + .range(..above) + .next_back() + .map(|(_, end)| *end) + .filter(|end| *end >= above); + self.ranges = self.ranges.split_off(&above); + if let Some(end) = straddling { + self.ranges.insert(above, end); + } + } + + /// The highest LSN in the set. + pub(in crate::data::executor) fn max(&self) -> Option { + self.ranges.last_key_value().map(|(_, end)| *end) + } + + /// Number of stored ranges. + pub(in crate::data::executor) fn range_count(&self) -> usize { + self.ranges.len() + } + + /// Remove every LSN. + pub(in crate::data::executor) fn clear(&mut self) { + self.ranges.clear(); + } + + /// The ranges in ascending order. + pub(in crate::data::executor) fn to_ranges(&self) -> Vec { + self.ranges + .iter() + .map(|(&start, &end)| LsnRange { start, end }) + .collect() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn touching_lsns_merge_and_gaps_stay_open() { + let mut set = LsnRanges::default(); + for lsn in [5, 7, 6, 10] { + set.insert(lsn); + } + assert_eq!( + set.to_ranges(), + vec![ + LsnRange { start: 5, end: 7 }, + LsnRange { start: 10, end: 10 } + ] + ); + assert!(!set.contains(8), "a gap is never bridged"); + assert!(!set.contains(9)); + set.insert(9); + set.insert(8); + assert_eq!(set.range_count(), 1); + assert_eq!(set.max(), Some(10)); + } + + #[test] + fn pruning_keeps_only_lsns_above_the_floor() { + let mut set = LsnRanges::default(); + for lsn in [3, 4, 5, 9] { + set.insert(lsn); + } + set.prune_through(4); + assert!(!set.contains(4)); + assert!(set.contains(5)); + assert!(set.contains(9)); + set.prune_through(9); + assert_eq!(set.range_count(), 0); + set.insert(u64::MAX); + set.prune_through(u64::MAX); + assert_eq!(set.range_count(), 0); + } +} diff --git a/nodedb/src/data/executor/applied_prefix/stamp.rs b/nodedb/src/data/executor/applied_prefix/stamp.rs new file mode 100644 index 000000000..53daaf026 --- /dev/null +++ b/nodedb/src/data/executor/applied_prefix/stamp.rs @@ -0,0 +1,198 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What a published engine artifact holds, stated so restart replay can +//! decide every record exactly. +//! +//! A checkpoint used to carry one LSN, the highest one applied. LSNs are +//! node-global and records reach a core out of mint order, so a record with a +//! lower LSN can still be on its way when a higher one applies. A stamp at the +//! higher LSN then claims the lower record, restart replay skips it, and the +//! write is gone. +//! +//! A [`ReplayStamp`] states two facts instead: +//! +//! - `prefix`: the core's outcome floor when the artifact was written. Every +//! record at or below it has a final outcome. A record this core applied is +//! in the artifact. A refused record carries a durable abort marker. +//! - `applied_above`: every record above `prefix` that this core applied +//! before the artifact was written. +//! +//! Replay skips a record exactly when [`ReplayStamp::skips`] says so. Every +//! engine that publishes an artifact and gates replay on it uses this type +//! and that function, whether its artifact covers the whole engine or one +//! collection. + +use serde::{Deserialize, Serialize}; + +/// An inclusive LSN range. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub(crate) struct LsnRange { + pub start: u64, + pub end: u64, +} + +/// The records a published artifact holds. +#[derive( + Debug, + Clone, + Default, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub(crate) struct ReplayStamp { + /// The outcome floor when the artifact was written. + pub prefix: u64, + /// The LSNs above `prefix` applied before the artifact was written, as + /// ascending, disjoint, non-touching ranges. + pub applied_above: Vec, +} + +/// Why a decoded [`ReplayStamp`] cannot describe an artifact. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub(crate) enum InvalidReplayStamp { + /// A range ends before it starts. + #[error("applied range [{start}, {end}] ends before it starts")] + Inverted { start: u64, end: u64 }, + /// A range lies at or below the prefix, which already holds it. + #[error("applied range [{start}, {end}] is not above the prefix {prefix}")] + NotAbovePrefix { start: u64, end: u64, prefix: u64 }, + /// Two ranges overlap, touch, or are out of order. + #[error("applied range starting at {start} does not follow the range ending at {previous_end}")] + OutOfOrder { previous_end: u64, start: u64 }, +} + +impl ReplayStamp { + /// A stamp holding every record at or below `prefix` and nothing above. + pub(crate) fn through(prefix: u64) -> Self { + Self { + prefix, + applied_above: Vec::new(), + } + } + + /// Whether restart replay skips the record at `record_lsn`: the artifact + /// already holds it, or it has a final outcome that is not an apply. + /// + /// Every other record replays. + pub(crate) fn skips(&self, record_lsn: u64) -> bool { + if record_lsn <= self.prefix { + return true; + } + let after = self + .applied_above + .partition_point(|range| range.start <= record_lsn); + after > 0 + && self + .applied_above + .get(after - 1) + .is_some_and(|range| record_lsn <= range.end) + } + + /// Check the shape [`Self::skips`] relies on. + pub(crate) fn validate(&self) -> Result<(), InvalidReplayStamp> { + let mut previous_end: Option = None; + for range in &self.applied_above { + if range.end < range.start { + return Err(InvalidReplayStamp::Inverted { + start: range.start, + end: range.end, + }); + } + if range.start <= self.prefix { + return Err(InvalidReplayStamp::NotAbovePrefix { + start: range.start, + end: range.end, + prefix: self.prefix, + }); + } + if let Some(previous_end) = previous_end + && range.start <= previous_end.saturating_add(1) + { + return Err(InvalidReplayStamp::OutOfOrder { + previous_end, + start: range.start, + }); + } + previous_end = Some(range.end); + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn stamp(prefix: u64, ranges: &[(u64, u64)]) -> ReplayStamp { + ReplayStamp { + prefix, + applied_above: ranges + .iter() + .map(|&(start, end)| LsnRange { start, end }) + .collect(), + } + } + + #[test] + fn the_prefix_and_the_applied_ranges_skip_and_nothing_else_does() { + let stamp = stamp(10, &[(12, 13), (20, 20)]); + for lsn in [1, 10, 12, 13, 20] { + assert!(stamp.skips(lsn), "{lsn} is held by the artifact"); + } + for lsn in [11, 14, 19, 21, u64::MAX] { + assert!(!stamp.skips(lsn), "{lsn} must replay"); + } + } + + #[test] + fn a_lower_record_in_flight_at_the_stamp_replays() { + // Record 11 was still on its way when 12 applied and the checkpoint + // was written. A max-applied stamp at 12 would skip it. + let stamp = stamp(10, &[(12, 12)]); + assert!(!stamp.skips(11)); + assert!(stamp.skips(12)); + } + + #[test] + fn a_prefix_only_stamp_skips_through_its_prefix() { + let stamp = ReplayStamp::through(100); + assert!(stamp.skips(100)); + assert!(!stamp.skips(101)); + assert!(!ReplayStamp::default().skips(1)); + } + + #[test] + fn a_malformed_stamp_is_refused() { + assert!(stamp(10, &[(12, 13), (20, 20)]).validate().is_ok()); + assert_eq!( + stamp(10, &[(13, 12)]).validate(), + Err(InvalidReplayStamp::Inverted { start: 13, end: 12 }) + ); + assert!(matches!( + stamp(10, &[(9, 12)]).validate(), + Err(InvalidReplayStamp::NotAbovePrefix { .. }) + )); + assert!(matches!( + stamp(10, &[(12, 14), (15, 16)]).validate(), + Err(InvalidReplayStamp::OutOfOrder { .. }) + )); + assert!(matches!( + stamp(10, &[(20, 21), (12, 13)]).validate(), + Err(InvalidReplayStamp::OutOfOrder { .. }) + )); + } +} diff --git a/nodedb/src/data/executor/applied_prefix/tracker.rs b/nodedb/src/data/executor/applied_prefix/tracker.rs index d94d8a855..d9db7cb8c 100644 --- a/nodedb/src/data/executor/applied_prefix/tracker.rs +++ b/nodedb/src/data/executor/applied_prefix/tracker.rs @@ -2,16 +2,74 @@ //! What this core knows about the applied prefix of the WAL. //! -//! Every request from the Control Plane carries the node's outcome floor: every -//! record at or below it that any core receives has a final outcome. The core -//! keeps the highest floor it has read. +//! Every request from the Control Plane carries the node's outcome floor F: +//! every record at or below F that any core receives has a final outcome. The +//! core keeps the highest F it has read, and the set of LSNs above F that it +//! applied itself. Together they are the [`ReplayStamp`] a checkpoint takes. +//! +//! ## Why a record at or below F is in every later artifact +//! +//! A record routed here has a final outcome only once this core answered it, +//! and the core answers after the apply. The request carrying F reaches the +//! core after that answer, so every record at or below F that this core +//! applies is applied before the core reads F. +//! +//! ## Bounded size +//! +//! The set holds ranges, pruned as F advances, so it holds only the records +//! this core applied while an older record was in flight. A window held until +//! restart keeps F down for the rest of the process, so the set can still +//! grow. Past [`MAX_APPLIED_RANGES`] ranges the tracker drops the set and +//! refuses to stamp until F passes the highest LSN it dropped. Every dropped +//! LSN is then at or below F. A refused stamp fails the checkpoint, which +//! costs WAL growth and never data. +//! +//! ## After boot +//! +//! Restart replay decides every record in the WAL before the core serves a +//! request: it applies the record, or a stamp, a tombstone or an abort marker +//! says it must not. [`AppliedPrefix::seed_replayed_through`] therefore +//! raises the floor to the highest LSN replay read, before replay starts. +//! Nothing checkpoints until replay ends, and every request after boot +//! carries an LSN minted after replay, above every record replay read. +use super::ranges::LsnRanges; +use super::stamp::ReplayStamp; use crate::types::Lsn; -/// The core's view of the node's outcome floor. +/// Most ranges the applied set holds before the tracker drops it. +pub(in crate::data::executor) const MAX_APPLIED_RANGES: usize = 65_536; + +/// Why the tracker cannot state what an artifact written now holds. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +#[error( + "the applied set above outcome floor {floor} passed {MAX_APPLIED_RANGES} ranges and was \ + dropped; a stamp is exact again once the floor reaches lsn {dropped_through}" +)] +pub(crate) struct StampUnavailable { + pub floor: u64, + pub dropped_through: u64, +} + +impl From for crate::Error { + fn from(e: StampUnavailable) -> Self { + crate::Error::Storage { + engine: "checkpoint".to_string(), + detail: e.to_string(), + } + } +} + +/// The core's view of the node's outcome floor and of what it applied above +/// it. #[derive(Debug)] pub(in crate::data::executor) struct AppliedPrefix { outcome_floor: Lsn, + /// LSNs above `outcome_floor` this core applied. Empty while `dropped`. + applied_above: LsnRanges, + /// Highest LSN applied since the set was dropped. `None` while the set is + /// exact. + dropped: Option, } impl AppliedPrefix { @@ -19,16 +77,62 @@ impl AppliedPrefix { pub(in crate::data::executor) fn new() -> Self { Self { outcome_floor: Lsn::ZERO, + applied_above: LsnRanges::default(), + dropped: None, } } /// Read the floor a request carried. The kept floor never decreases. pub(in crate::data::executor) fn observe_outcome_floor(&mut self, floor: Lsn) { - if floor > self.outcome_floor { - self.outcome_floor = floor; + if floor <= self.outcome_floor { + return; + } + self.outcome_floor = floor; + self.applied_above.prune_through(floor.as_u64()); + if self + .dropped + .is_some_and(|dropped| dropped <= floor.as_u64()) + { + self.dropped = None; } } + /// Raise the floor to `lsn`, the highest LSN restart replay reads. Called + /// before replay starts. + pub(in crate::data::executor) fn seed_replayed_through(&mut self, lsn: Lsn) { + self.observe_outcome_floor(lsn); + } + + /// Record that this core applied the record at `lsn`. + pub(in crate::data::executor) fn note_applied(&mut self, lsn: Lsn) { + let lsn = lsn.as_u64(); + if lsn <= self.outcome_floor.as_u64() { + return; + } + if let Some(dropped) = self.dropped.as_mut() { + *dropped = (*dropped).max(lsn); + return; + } + self.applied_above.insert(lsn); + if self.applied_above.range_count() > MAX_APPLIED_RANGES { + self.dropped = Some(self.applied_above.max().unwrap_or(lsn).max(lsn)); + self.applied_above.clear(); + } + } + + /// What an artifact written now holds. + pub(in crate::data::executor) fn stamp(&self) -> Result { + if let Some(dropped_through) = self.dropped { + return Err(StampUnavailable { + floor: self.outcome_floor.as_u64(), + dropped_through, + }); + } + let mut stamp = ReplayStamp::through(self.outcome_floor.as_u64()); + stamp.applied_above = self.applied_above.to_ranges(); + Ok(stamp) + } + /// The highest outcome floor this core has read. pub(in crate::data::executor) fn outcome_floor(&self) -> Lsn { self.outcome_floor @@ -37,6 +141,7 @@ impl AppliedPrefix { #[cfg(test)] mod tests { + use super::super::stamp::LsnRange; use super::*; #[test] @@ -49,4 +154,93 @@ mod tests { prefix.observe_outcome_floor(Lsn::new(12)); assert_eq!(prefix.outcome_floor(), Lsn::new(12)); } + + #[test] + fn applied_lsns_above_the_floor_are_stamped_and_pruned_as_it_advances() { + let mut prefix = AppliedPrefix::new(); + prefix.observe_outcome_floor(Lsn::new(10)); + for lsn in [12, 13, 20, 8] { + prefix.note_applied(Lsn::new(lsn)); + } + assert_eq!( + prefix.stamp(), + Ok(ReplayStamp { + prefix: 10, + applied_above: vec![ + LsnRange { start: 12, end: 13 }, + LsnRange { start: 20, end: 20 } + ], + }) + ); + prefix.observe_outcome_floor(Lsn::new(15)); + assert_eq!( + prefix.stamp(), + Ok(ReplayStamp { + prefix: 15, + applied_above: vec![LsnRange { start: 20, end: 20 }], + }) + ); + prefix.observe_outcome_floor(Lsn::new(20)); + assert_eq!(prefix.stamp(), Ok(ReplayStamp::through(20))); + } + + #[test] + fn a_record_in_flight_below_an_applied_one_is_not_stamped() { + let mut prefix = AppliedPrefix::new(); + prefix.observe_outcome_floor(Lsn::new(10)); + prefix.note_applied(Lsn::new(12)); + let stamp = prefix.stamp().expect("exact"); + assert!(!stamp.skips(11), "record 11 is still on its way"); + assert!(stamp.skips(12)); + } + + #[test] + fn the_set_stays_bounded_and_refuses_to_stamp_until_the_floor_passes_it() { + let mut prefix = AppliedPrefix::new(); + // Every other LSN: no two touch, so each is its own range. + for i in 1..=(MAX_APPLIED_RANGES as u64 + 1) { + prefix.note_applied(Lsn::new(i * 2)); + } + let top = (MAX_APPLIED_RANGES as u64 + 1) * 2; + assert_eq!(prefix.applied_above.range_count(), 0, "the set is dropped"); + assert_eq!( + prefix.stamp(), + Err(StampUnavailable { + floor: 0, + dropped_through: top, + }) + ); + prefix.note_applied(Lsn::new(top + 10)); + prefix.observe_outcome_floor(Lsn::new(top)); + assert!( + prefix.stamp().is_err(), + "an LSN applied after the drop is still above the floor" + ); + prefix.observe_outcome_floor(Lsn::new(top + 10)); + assert_eq!(prefix.stamp(), Ok(ReplayStamp::through(top + 10))); + prefix.note_applied(Lsn::new(top + 12)); + assert_eq!(prefix.applied_above.range_count(), 1, "exact again"); + } + + #[test] + fn seeding_after_boot_covers_every_replayed_record() { + let mut prefix = AppliedPrefix::new(); + prefix.seed_replayed_through(Lsn::new(500)); + // Replay notes each record it applies; all are at or below the seed. + for lsn in [3, 250, 500] { + prefix.note_applied(Lsn::new(lsn)); + } + assert_eq!(prefix.applied_above.range_count(), 0); + assert_eq!(prefix.stamp(), Ok(ReplayStamp::through(500))); + // A fresh Control Plane's floor starts low; the seed is kept. + prefix.observe_outcome_floor(Lsn::new(0)); + prefix.note_applied(Lsn::new(502)); + assert_eq!( + prefix.stamp().expect("exact").applied_above, + vec![LsnRange { + start: 502, + end: 502 + }] + ); + } } diff --git a/nodedb/src/data/executor/checkpoint_decode_error.rs b/nodedb/src/data/executor/checkpoint_decode_error.rs index f18e8329e..f9538622e 100644 --- a/nodedb/src/data/executor/checkpoint_decode_error.rs +++ b/nodedb/src/data/executor/checkpoint_decode_error.rs @@ -66,6 +66,15 @@ pub(crate) enum CheckpointDecodeError { expected: u16, }, + /// A manifest's replay stamp does not have the shape replay decisions + /// rely on. Gating replay on it could skip a record no checkpoint holds. + #[error("{} carries an invalid replay stamp: {source}", path.display())] + InvalidReplayStamp { + path: PathBuf, + #[source] + source: crate::data::executor::applied_prefix::InvalidReplayStamp, + }, + /// A sparse-vector index file did not decode into an index. The decoder /// reports only success or failure, so there is no inner detail to carry. #[error("cannot decode {}", path.display())] diff --git a/nodedb/src/data/executor/columnar_checkpoint/format.rs b/nodedb/src/data/executor/columnar_checkpoint/format.rs index d03c68908..bb7216a29 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/format.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/format.rs @@ -5,12 +5,14 @@ use serde::{Deserialize, Serialize}; +use crate::data::executor::applied_prefix::ReplayStamp; + /// On-disk format version for the manifest and the collection files. /// /// A file stamped with any other version is refused rather than misparsed. /// Refusing costs a WAL replay; misparsing would install wrong rows AND a floor /// that suppresses the records which would have corrected them. -pub(crate) const COLUMNAR_CKPT_FORMAT_VERSION: u16 = 1; +pub(crate) const COLUMNAR_CKPT_FORMAT_VERSION: u16 = 2; /// Names the live generation. Writing this file is what publishes a checkpoint. #[derive( @@ -28,17 +30,16 @@ pub(crate) struct ColumnarCheckpointManifest { pub format_version: u16, /// Which `gen-{n}/` directory holds the live collection files. pub generation: u64, - /// The LSN every collection in that generation is durable THROUGH - /// (inclusive). - /// - /// This is what makes a generation self-describing: WAL replay skips - /// columnar records at or below it and replays everything above. Without it - /// a restore could not know which records it had already folded in. - /// Columnar has no safe fallback for that ignorance: `ColumnarOp::Update` is - /// delete-old-PK + insert-new-row, so re-applying one duplicates the row, - /// and on a `bitemporal=true` collection re-applying an `Insert` appends a - /// second version that `AS OF` queries can see. + /// The LSN this core reports as the columnar engine's truncation floor when + /// the generation is restored. It gates no replay: [`Self::replay`] does. pub durable_through_lsn: u64, + /// The records the generation holds. WAL replay skips exactly the columnar + /// records [`ReplayStamp::skips`] names and replays every other one. + /// + /// A single highest-applied LSN cannot state this. LSNs are node-global and + /// records reach a core out of mint order, so a record below the highest + /// applied one can still be on its way when the generation is written. + pub replay: ReplayStamp, } /// One collection's full engine state within a generation. diff --git a/nodedb/src/data/executor/columnar_checkpoint/load.rs b/nodedb/src/data/executor/columnar_checkpoint/load.rs index 04cc666ac..25620d7c5 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/load.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/load.rs @@ -107,12 +107,11 @@ impl CoreLoop { // Claimed only once every engine is in: the floor suppresses WAL // records, so claiming it over a half-restored generation would turn a // recoverable read failure into permanent data loss. - self.floors - .replay_floors - .columnar - .set(Lsn::new(manifest.durable_through_lsn)); self.floors.columnar_durable_lsn = Lsn::new(manifest.durable_through_lsn); - self.floors.columnar_published_lsn = Lsn::new(manifest.durable_through_lsn); + self.floors.columnar_published_lsn = Lsn::new(manifest.replay.prefix); + let replay_prefix = manifest.replay.prefix; + let applied_ranges = manifest.replay.applied_above.len(); + self.floors.replay_floors.columnar.set(manifest.replay); info!( core = self.core_id, @@ -121,6 +120,8 @@ impl CoreLoop { segments, geometry_rows, durable_through_lsn = manifest.durable_through_lsn, + replay_prefix, + applied_ranges, "columnar checkpoint restored" ); Ok(()) @@ -676,17 +677,22 @@ mod tests { ); } - /// The reported LSN is a deletion authority: it must be the watermark on - /// success, and it must come back as BOTH the restored durable LSN and the - /// replay floor. Getting either wrong is silent — too high gates records that - /// still needed replaying, too low replays records already folded in. + /// The reported LSN is a deletion authority: it is the watermark on success + /// and comes back as the restored durable LSN. Replay is gated by the + /// stamp instead: the outcome floor and the records applied above it, never + /// the highest applied LSN, since a record below that can still be on its + /// way. Too wide a gate drops a write; too narrow re-applies a folded one. #[test] - fn reported_lsn_becomes_the_restored_floor_and_durable_lsn() { + fn the_stamp_becomes_the_restored_floor_and_the_watermark_the_durable_lsn() { let dir = tempfile::tempdir().expect("tempdir"); let coll = "ck_lsn"; let mut core = open_core(dir.path()); seed_collection(&mut core, coll, &[(1, "a", Surrogate(601))], &[]); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(890)); + core.floors.applied_prefix.note_applied(Lsn::new(900)); core.watermark = Lsn::new(900); let reported = core @@ -709,12 +715,16 @@ mod tests { .expect("checkpoint load must succeed"); assert_eq!(restored.floors.columnar_durable_lsn, Lsn::new(900)); + assert_eq!(restored.floors.columnar_published_lsn, Lsn::new(890)); + let floor = &restored.floors.replay_floors.columnar; + assert!(floor.covers(890), "the prefix is folded in"); assert!( - restored.floors.replay_floors.columnar.covers(900), - "the stamped LSN is durable THROUGH, so its own record is folded in" + !floor.covers(891), + "a record in flight below the applied one is NOT in the restored state" ); + assert!(floor.covers(900), "the applied record is folded in"); assert!( - !restored.floors.replay_floors.columnar.covers(901), + !floor.covers(901), "a record above the stamp is NOT in the restored state and must replay" ); } diff --git a/nodedb/src/data/executor/columnar_checkpoint/manifest.rs b/nodedb/src/data/executor/columnar_checkpoint/manifest.rs index 3686adadd..c003e2036 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/manifest.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/manifest.rs @@ -44,6 +44,13 @@ impl CoreLoop { expected: COLUMNAR_CKPT_FORMAT_VERSION, }); } + manifest + .replay + .validate() + .map_err(|source| CheckpointDecodeError::InvalidReplayStamp { + path: path.clone(), + source, + })?; Ok(Some(manifest)) } } diff --git a/nodedb/src/data/executor/columnar_checkpoint/mod.rs b/nodedb/src/data/executor/columnar_checkpoint/mod.rs index 258389a1d..abed650e3 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/mod.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/mod.rs @@ -63,13 +63,13 @@ //! //! ## Why a generation + manifest, and not a stamp per file //! -//! Identical to `kv_checkpoint`: the LSN a checkpoint is durable through gates +//! Identical to `kv_checkpoint`: the replay stamp a checkpoint carries gates //! WAL replay, so it must be on disk, and recording it per file is unsound. A //! flush is per-collection and can partially fail, and a crash between two //! per-file writes leaves collection `a` stamped at LSN 900 while `b` stays at //! 400 — permanently. Writing every collection into a fresh `gen-{n}/` and //! publishing the whole set with ONE atomic manifest write removes the split by -//! construction: every live collection advances to a single LSN together, or +//! construction: every live collection advances to a single stamp together, or //! none do and the previous generation stays live. The manifest is the only //! thing that makes a generation visible, so a torn or abandoned write is inert //! garbage rather than a half-published state. diff --git a/nodedb/src/data/executor/columnar_checkpoint/write.rs b/nodedb/src/data/executor/columnar_checkpoint/write.rs index c05ebb079..4a3d518c7 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/write.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/write.rs @@ -12,46 +12,47 @@ use super::manifest::storage_err; use super::paths::{ COLUMNAR_CKPT_MANIFEST, columnar_ckpt_dir, columnar_ckpt_filename, columnar_ckpt_gen_dir, }; +use crate::data::executor::applied_prefix::ReplayStamp; use crate::data::executor::core_loop::CoreLoop; use crate::types::Lsn; impl CoreLoop { /// Flush every columnar collection on this core to disk and return the LSN - /// the columnar engine is now durable through. + /// the columnar engine reports as its truncation floor. /// - /// Returns `Ok(watermark)` only once a manifest naming a COMPLETE generation - /// has landed. Any failure returns `Err` — the caller must then clamp the + /// Returns `Ok` only once a manifest naming a COMPLETE generation has + /// landed. Any failure returns `Err` — the caller must then clamp the /// reported checkpoint LSN to the last LSN columnar was known durable /// through, so a failed flush costs WAL growth instead of data. /// - /// ## Why the core watermark is an exact stamp here + /// ## What replay skips /// - /// This runs on the core's own thread between tasks, and every columnar - /// record that mutates an engine raises the watermark AFTER applying: - /// `execute_columnar_insert` calls `note_collection_write_lsn`, and so — - /// since the fix that accompanies this checkpoint — do - /// `execute_columnar_update` and `execute_columnar_delete`. So every - /// columnar record with `lsn <= watermark` is already folded into the - /// engines exported here, and every record above it is not. + /// The manifest carries the core's [`ReplayStamp`], taken on the core's own + /// thread between tasks, so no write interleaves with the export. The + /// stamp names every record the exported engines hold: every record at or + /// below the outcome floor, and every record above it this core applied. + /// A record still on its way to the core is in neither, so restart replay + /// applies it. The tracker has no exact stamp while its applied set is + /// dropped; the checkpoint then fails. /// - /// That property is load-bearing in a way it is not for KV. KV tolerates a - /// record being replayed over a generation stamped below it, because its - /// unstamped records (index DDL) replay idempotently. Columnar has no such - /// slack: `ColumnarOp::Update` is delete-old-PK + insert-new-row, so a - /// record applied before the export and replayed again after it duplicates - /// the row. An applied-but-unstamped columnar record is therefore silent - /// corruption, which is why update/delete must note their LSN rather than - /// this stamp being defensively lowered. + /// Columnar has no slack for an applied record the stamp does not name: + /// `ColumnarOp::Update` is delete-old-PK + insert-new-row, so a record + /// applied before the export and replayed again after it duplicates the + /// row. Every columnar record that mutates an engine therefore notes its + /// LSN after applying: `execute_columnar_insert`, `execute_columnar_update` + /// and `execute_columnar_delete` call `note_collection_write_lsn`, which + /// records the LSN as applied. /// - /// A record whose live execution affected ZERO rows notes no LSN and so may - /// fall above the stamp and replay. That is safe and stays safe: it matched - /// nothing against the state that the export captured, so re-executing the - /// same predicate against that same restored state matches nothing again. + /// A record whose live execution affected ZERO rows notes no LSN and so + /// replays. That is safe and stays safe: it matched nothing against the + /// state that the export captured, so re-executing the same predicate + /// against that same restored state matches nothing again. /// - /// Every published generation raises `columnar_published_lsn`, the LSN - /// restart restores columnar from (see `redo_apply::cover`). + /// Every published generation raises `columnar_published_lsn` to the + /// stamp's prefix (see `redo_apply::cover`). pub(in crate::data::executor) fn checkpoint_columnar_engines(&mut self) -> crate::Result { let durable_through = self.watermark; + let replay = self.floors.applied_prefix.stamp()?; let ckpt_dir = columnar_ckpt_dir(&self.data_dir, self.core_id); std::fs::create_dir_all(&ckpt_dir).map_err(|e| storage_err(&ckpt_dir, "create dir", &e))?; @@ -73,9 +74,10 @@ impl CoreLoop { .map_err(|e| storage_err(&gen_dir, "create generation dir", &e))?; let written = self.write_columnar_generation(&gen_dir)?; - self.publish_columnar_generation(&ckpt_dir, generation, durable_through)?; - self.floors.columnar_published_lsn = - self.floors.columnar_published_lsn.max(durable_through); + let prefix = Lsn::new(replay.prefix); + let applied_ranges = replay.applied_above.len(); + self.publish_columnar_generation(&ckpt_dir, generation, durable_through, replay)?; + self.floors.columnar_published_lsn = self.floors.columnar_published_lsn.max(prefix); // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what @@ -101,6 +103,8 @@ impl CoreLoop { generation, collections = written, durable_through_lsn = durable_through.as_u64(), + replay_prefix = prefix.as_u64(), + applied_ranges, "columnar checkpoint published" ); Ok(durable_through) @@ -182,7 +186,7 @@ impl CoreLoop { /// Publish a written generation by atomically replacing the manifest. /// /// This single write is the commit point of the whole checkpoint: before it - /// nothing changed; after it the entire generation is live at one LSN. It + /// nothing changed; after it the entire generation is live under one stamp. It /// also fsyncs `ckpt_dir`, the same directory holding the `gen-{n}/` entry, /// so that entry cannot still be pending when the manifest naming it becomes /// visible. @@ -191,11 +195,13 @@ impl CoreLoop { ckpt_dir: &std::path::Path, generation: u64, durable_through: Lsn, + replay: ReplayStamp, ) -> crate::Result<()> { let manifest = ColumnarCheckpointManifest { format_version: COLUMNAR_CKPT_FORMAT_VERSION, generation, durable_through_lsn: durable_through.as_u64(), + replay, }; let bytes = zerompk::to_msgpack_vec(&manifest).map_err(|e| crate::Error::Serialization { diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs index b9956aea7..3341f8c7d 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs @@ -126,15 +126,16 @@ pub(in crate::data::executor) struct CheckpointFloors { /// the watermark — is what the core may report. pub(in crate::data::executor) vector_durable_lsn: Lsn, - /// LSN of the newest KV checkpoint generation on disk, restored at boot - /// from the manifest. Restart replay skips a KV record at or below it, so - /// a committed record applied at or below it must be published again - /// (`redo_apply::cover`). Never a truncation floor. + /// Replay-stamp prefix of the newest KV checkpoint generation on disk, + /// restored at boot from the manifest. Restart replay skips every KV + /// record at or below it, so a committed record applied at or below it + /// must be published again (`redo_apply::cover`). Never a truncation + /// floor. pub(in crate::data::executor) kv_published_lsn: Lsn, - /// LSN of the newest columnar checkpoint generation on disk, whichever - /// flush published it, restored at boot from the manifest. Same rule as - /// `kv_published_lsn`. + /// Replay-stamp prefix of the newest columnar checkpoint generation on + /// disk, whichever flush published it, restored at boot from the + /// manifest. Same rule as `kv_published_lsn`. pub(in crate::data::executor) columnar_published_lsn: Lsn, /// LSN of the newest vector checkpoint generation on disk, whichever @@ -178,12 +179,14 @@ pub(in crate::data::executor) struct CheckpointFloors { /// watermark — is what the core may report. pub(in crate::data::executor) spatial_durable_lsn: Lsn, - /// Per-engine "already durable through LSN X" floors recovered from on-disk - /// checkpoints during boot, before WAL replay. Consulted by the replay paths + /// Per-engine replay stamps recovered from on-disk checkpoints during + /// boot, before WAL replay. Consulted by the replay paths /// so records already folded into a restored checkpoint are not applied a /// second time. Empty outside boot, and empty means "replay everything". pub(in crate::data::executor) replay_floors: ReplayFloors, - /// The node's outcome floor as this core last read it from a request. + /// The node's outcome floor as this core last read it from a request, and + /// the records this core applied above it: the replay stamp every KV and + /// columnar checkpoint carries. pub(in crate::data::executor) applied_prefix: AppliedPrefix, } diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index 229f4474e..1cfd7f074 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -103,6 +103,16 @@ impl CoreLoop { } else { task.state = TaskState::Running; let resp = self.execute(&task); + // A crash test kills the process here: one collection's logged + // write applied, and its response never leaves the core. A task + // with no WAL record never matches. + crate::fail_point!(&match task.wal_lsn() { + Some(_) => format!( + "core::after_apply::{}", + task.plan().collection().unwrap_or_default() + ), + None => String::new(), + }); task.state = TaskState::Completed; // A failed rollback leaves this core's state unknown. self.fail_stop_on_rollback_failure(&resp); diff --git a/nodedb/src/data/executor/core_loop/write_index.rs b/nodedb/src/data/executor/core_loop/write_index.rs index 98da25d60..3c8bfe54f 100644 --- a/nodedb/src/data/executor/core_loop/write_index.rs +++ b/nodedb/src/data/executor/core_loop/write_index.rs @@ -266,6 +266,9 @@ impl CoreLoop { } /// Record a write version and advance the core watermark monotonically. + /// + /// Every applied write reaches here, so this is also where the core notes + /// the write's LSN as applied: a checkpoint's replay stamp names it. pub(in crate::data::executor) fn publish_write_version( &mut self, db: DatabaseId, @@ -274,6 +277,7 @@ impl CoreLoop { key: Option, lsn: Lsn, ) { + self.floors.applied_prefix.note_applied(lsn); self.write_index .note_write_lsn(db, tenant, collection, key, lsn); if lsn > self.watermark { @@ -935,6 +939,39 @@ pub(crate) mod tests { assert_eq!(core.watermark, Lsn::new(42)); } + #[test] + fn an_autocommit_apply_is_named_by_the_next_replay_stamp() { + let (mut core, _, _, _dir) = make_core(); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(40)); + let resp = core.execute_kv_put( + &wal_task(42), + KvWriteParams { + did: DatabaseId::DEFAULT.as_u64(), + tid: 1, + collection: "kv", + key: b"k1".as_slice(), + value: b"v1".as_slice(), + ttl_ms: 0, + surrogate: Surrogate::new(3), + returning: None, + rls_filters: &[], + }, + ); + assert_eq!(resp.status, Status::Ok); + + let stamp = core.floors.applied_prefix.stamp().expect("exact stamp"); + assert!( + stamp.skips(42), + "the applied write is in every later artifact" + ); + assert!( + !stamp.skips(41), + "a record between the floor and the applied write still replays" + ); + } + #[test] fn edge_put_records_edge_version() { let (mut core, _, _, _dir) = make_core(); diff --git a/nodedb/src/data/executor/graph_label_checkpoint/write.rs b/nodedb/src/data/executor/graph_label_checkpoint/write.rs index a4fd2f95e..c777f2fc0 100644 --- a/nodedb/src/data/executor/graph_label_checkpoint/write.rs +++ b/nodedb/src/data/executor/graph_label_checkpoint/write.rs @@ -26,8 +26,8 @@ impl CoreLoop { /// is intact and live, after it the new one is. There is no window in which /// half a core's partitions are published. /// - /// Stamping with the core watermark mirrors `checkpoint_kv_engines`: this - /// runs on the core's own thread between tasks, and a label write reaches + /// Stamping with the core watermark rests on this: the checkpoint runs + /// on the core's own thread between tasks, and a label write reaches /// `note_write_lsn` (which raises the watermark) only after /// `add_node_label` / `remove_node_label` has already mutated the bitset. So /// every label change with `lsn <= watermark` is in the export below. diff --git a/nodedb/src/data/executor/handlers/control/checkpoint_crdt.rs b/nodedb/src/data/executor/handlers/control/checkpoint_crdt.rs index e15f9278c..1c149223c 100644 --- a/nodedb/src/data/executor/handlers/control/checkpoint_crdt.rs +++ b/nodedb/src/data/executor/handlers/control/checkpoint_crdt.rs @@ -45,8 +45,8 @@ impl CoreLoop { /// caller clamps the reported checkpoint LSN to the last LSN the CRDT /// engines were known durable through. /// - /// Stamping with the core watermark mirrors `checkpoint_kv_engines`: this - /// runs on the core's own thread between tasks, and a delta apply raises the + /// Stamping with the core watermark rests on this: the checkpoint runs + /// on the core's own thread between tasks, and a delta apply raises the /// watermark only after the `LoroDoc` has already imported it. pub(in crate::data::executor) fn checkpoint_crdt_engines( &self, diff --git a/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs b/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs index 27cfa689f..e0d81e860 100644 --- a/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs +++ b/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs @@ -129,10 +129,9 @@ impl CoreLoop { /// /// Clamping matters more here than for the engines above, because columnar /// replay is not idempotent: `ColumnarOp::Update` is delete-old-PK + - /// insert-new-row. So the reported LSN both authorises truncation AND, via - /// the restored floor, decides which records replay. Overstating it would - /// not merely delete rows — it would gate the records that would have - /// rebuilt them. + /// insert-new-row, so a record whose WAL copy is gone cannot be rebuilt + /// from any other source. Which records replay is the generation's replay + /// stamp, not this LSN. pub(super) fn checkpoint_columnar_durable_lsn(&mut self) -> Lsn { match self.checkpoint_columnar_engines() { Ok(lsn) => { diff --git a/nodedb/src/data/executor/handlers/kv/ttl.rs b/nodedb/src/data/executor/handlers/kv/ttl.rs index 468c48bfc..7029bc837 100644 --- a/nodedb/src/data/executor/handlers/kv/ttl.rs +++ b/nodedb/src/data/executor/handlers/kv/ttl.rs @@ -17,10 +17,9 @@ use crate::types::TenantId; /// /// `EXPIRE` and `PERSIST` address a row identically and differ only in the /// instant they install, so they share one bundle rather than repeating the -/// five-field address list twice. The transaction wrappers in -/// `sub_plan_kv_ttl_sorted.rs` pass this straight through to these handlers, -/// so a COMMIT-time replay addresses and decides the row exactly as an -/// autocommit statement does. +/// five-field address list twice. `stage_kv_ttl.rs` and `kv/resolve/` pass +/// this straight through to these handlers, so a COMMIT-time replay +/// addresses and decides the row exactly as an autocommit statement does. /// /// `Copy` because it is a plain address: a wrapper hands the same one to the /// handler it delegates to rather than rebuilding it field by field, which is diff --git a/nodedb/src/data/executor/handlers/timeseries_wal.rs b/nodedb/src/data/executor/handlers/timeseries_wal.rs index 9d72947cd..35099b8e7 100644 --- a/nodedb/src/data/executor/handlers/timeseries_wal.rs +++ b/nodedb/src/data/executor/handlers/timeseries_wal.rs @@ -579,7 +579,9 @@ mod tests { #[test] fn columnar_records_at_or_below_the_floor_are_not_replayed() { let mut h = make_core(); - h.core.floors.replay_floors.columnar.set(Lsn::new(200)); + h.core.floors.replay_floors.columnar.set( + crate::data::executor::applied_prefix::ReplayStamp::through(200), + ); let record = columnar_wal_record("events_gated", 150, 7); h.core.replay_timeseries_wal( @@ -601,7 +603,9 @@ mod tests { #[test] fn columnar_records_above_the_floor_still_replay() { let mut h = make_core(); - h.core.floors.replay_floors.columnar.set(Lsn::new(100)); + h.core.floors.replay_floors.columnar.set( + crate::data::executor::applied_prefix::ReplayStamp::through(100), + ); let record = columnar_wal_record("events_ungated", 150, 7); h.core.replay_timeseries_wal( @@ -632,7 +636,9 @@ mod tests { let mut h = make_core(); // A floor far above the record's LSN: if the gate were applied by // record type instead of by kind, this would suppress it. - h.core.floors.replay_floors.columnar.set(Lsn::new(10_000)); + h.core.floors.replay_floors.columnar.set( + crate::data::executor::applied_prefix::ReplayStamp::through(10_000), + ); let batch = nodedb_types::timeseries::TimeseriesWalBatch { collection: "metrics_ungated".to_string(), diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs index 7c597b53f..82e3fc325 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs @@ -268,6 +268,10 @@ mod tests { surrogate: Surrogate::new(1), }); core.advance_watermark(Lsn::new(100)); + // The published stamp's prefix is what restart replay skips through. + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(100)); core.checkpoint_kv_engines().expect("publish at lsn 100"); let mut task = make_default_task(); @@ -375,6 +379,10 @@ mod tests { .expect("seed row"); core.columnar_engines.insert(key.clone(), engine); core.advance_watermark(Lsn::new(100)); + // The published stamp's prefix is what restart replay skips through. + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(100)); core.checkpoint_columnar_engines() .expect("publish at lsn 100"); diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs index 656ca544f..8bf91aaad 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs @@ -140,6 +140,8 @@ impl CoreLoop { // replay does not rebuild: the core fail-stops. The funnel keeps the // record for restart replay. let settled = self.settle_redo_install(task, &mut scope).and_then(|()| { + // Applied from here on: a checkpoint the cover writes names it. + self.floors.applied_prefix.note_applied(lsn); self.cover_applied_record(lsn, &WrittenEngines::of(&redo), &scope.arrays_written) }); if let Err(error) = settled { @@ -259,6 +261,37 @@ mod tests { task } + #[test] + fn an_installed_record_is_named_by_the_next_replay_stamp() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(10)); + let redo = redo_bytes(vec![kv_put("cache", b"k1", b"v1", 12)]); + + let response = core.execute_apply_transaction_redo( + &task_at(Some(50)), + TID, + CommittedRedo { + redo: &redo, + collections: &["cache".to_string()], + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + + let stamp = core.floors.applied_prefix.stamp().expect("exact stamp"); + assert!( + stamp.skips(50), + "the installed record is in every later artifact" + ); + assert!( + !stamp.skips(49), + "a record below it that never applied here must still replay" + ); + } + #[test] fn committed_redo_installs_its_document_and_kv_rows() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/nodedb/src/data/executor/kv_checkpoint/format.rs b/nodedb/src/data/executor/kv_checkpoint/format.rs index c718e3d34..8e4c4894f 100644 --- a/nodedb/src/data/executor/kv_checkpoint/format.rs +++ b/nodedb/src/data/executor/kv_checkpoint/format.rs @@ -5,6 +5,8 @@ use serde::{Deserialize, Serialize}; +use crate::data::executor::applied_prefix::ReplayStamp; + use super::index_format::KvCheckpointIndexes; /// On-disk format version for the manifest and the collection files. @@ -12,7 +14,7 @@ use super::index_format::KvCheckpointIndexes; /// A file stamped with any other version is refused rather than misparsed. /// Refusing costs a WAL replay; misparsing would install wrong rows AND a floor /// that suppresses the records which would have corrected them. -pub(crate) const KV_CKPT_FORMAT_VERSION: u16 = 2; +pub(crate) const KV_CKPT_FORMAT_VERSION: u16 = 3; /// Names the live generation. Writing this file is what publishes a checkpoint. #[derive( @@ -30,15 +32,16 @@ pub(crate) struct KvCheckpointManifest { pub format_version: u16, /// Which `gen-{n}/` directory holds the live collection files. pub generation: u64, - /// The LSN every collection in that generation is durable THROUGH - /// (inclusive). - /// - /// This is what makes a generation self-describing: WAL replay skips KV - /// records at or below it and replays everything above. Without it a restore - /// could not know which records it had already folded in, and would have to - /// either re-apply deltas (double-counting) or skip everything (losing every - /// write made after the flush). + /// The LSN this core reports as the KV engine's truncation floor when + /// the generation is restored. It gates no replay: [`Self::replay`] does. pub durable_through_lsn: u64, + /// The records the generation holds. WAL replay skips exactly the KV + /// records [`ReplayStamp::skips`] names and replays every other one. + /// + /// A single highest-applied LSN cannot state this. LSNs are node-global and + /// records reach a core out of mint order, so a record below the highest + /// applied one can still be on its way when the generation is written. + pub replay: ReplayStamp, } /// One checkpointed KV row. diff --git a/nodedb/src/data/executor/kv_checkpoint/load.rs b/nodedb/src/data/executor/kv_checkpoint/load.rs index b7b4196f6..d235224bc 100644 --- a/nodedb/src/data/executor/kv_checkpoint/load.rs +++ b/nodedb/src/data/executor/kv_checkpoint/load.rs @@ -25,9 +25,9 @@ impl CoreLoop { /// Rows are reinstalled by replaying them through `KvEngine::put`, and the /// index registrations are reinstalled with their exported content once the /// rows are back — see `index_restore.rs` for why that order is the only - /// sound one. The manifest's LSN then becomes the replay floor: - /// `replay_kv_wal` skips the records already folded in and applies - /// everything above. + /// sound one. The manifest's replay stamp then becomes the replay floor: + /// `replay_kv_wal` skips the records the stamp names and applies every + /// other one. /// /// # Fail-stop on corruption /// @@ -69,11 +69,10 @@ impl CoreLoop { // Claimed only once every row AND every registration is in: the floor // suppresses WAL records, so claiming it over a half-restored generation // would turn a recoverable read failure into permanent data loss. - self.floors - .replay_floors - .kv - .set(Lsn::new(manifest.durable_through_lsn)); - self.floors.kv_published_lsn = Lsn::new(manifest.durable_through_lsn); + self.floors.kv_published_lsn = Lsn::new(manifest.replay.prefix); + let replay_prefix = manifest.replay.prefix; + let applied_ranges = manifest.replay.applied_above.len(); + self.floors.replay_floors.kv.set(manifest.replay); info!( core = self.core_id, @@ -82,6 +81,8 @@ impl CoreLoop { rows, indexes, durable_through_lsn = manifest.durable_through_lsn, + replay_prefix, + applied_ranges, "KV checkpoint restored" ); Ok(()) @@ -380,15 +381,22 @@ mod tests { assert!(!alice_meta.has_ttl, "a persistent row must not gain a TTL"); } - /// The manifest is the only record of the LSN a generation is durable - /// through, and the entire replay floor rests on it: it must survive the - /// round-trip exactly. + /// The manifest is the only record of what a generation holds, and the + /// entire replay floor rests on it: it must survive the round-trip exactly. #[test] - fn manifest_roundtrips_generation_and_lsn() { + fn manifest_roundtrips_generation_lsn_and_stamp() { + let replay = crate::data::executor::applied_prefix::ReplayStamp { + prefix: 4_200, + applied_above: vec![crate::data::executor::applied_prefix::stamp::LsnRange { + start: 4_240, + end: 4_242, + }], + }; let written = KvCheckpointManifest { format_version: KV_CKPT_FORMAT_VERSION, generation: 9, durable_through_lsn: 4_242, + replay: replay.clone(), }; let tmp = tempfile::tempdir().expect("tempdir"); let path = tmp.path().join(KV_CKPT_MANIFEST); @@ -404,6 +412,7 @@ mod tests { ); assert_eq!(decoded.generation, 9); assert_eq!(decoded.format_version, KV_CKPT_FORMAT_VERSION); + assert_eq!(decoded.replay, replay, "the stamp must survive exactly"); } /// Entries still sitting in the rehash source are live rows. An export that @@ -488,4 +497,40 @@ mod tests { .load_kv_checkpoints() .expect_err("a corrupt manifest must fail the load, not silently skip it"); } + + /// A manifest whose replay stamp has an applied range at or below its + /// prefix is malformed. Gating replay on it could skip a record no + /// generation holds, so the load fails instead. + #[test] + fn a_manifest_with_a_malformed_stamp_fails_the_load() { + let dir = tempfile::tempdir().expect("tempdir"); + let core = open_core_at(dir.path()); + let ckpt_dir = kv_ckpt_dir(&core.data_dir, core.core_id); + std::fs::create_dir_all(&ckpt_dir).expect("create ckpt dir"); + let manifest = KvCheckpointManifest { + format_version: KV_CKPT_FORMAT_VERSION, + generation: 0, + durable_through_lsn: 10, + replay: crate::data::executor::applied_prefix::ReplayStamp { + prefix: 10, + applied_above: vec![crate::data::executor::applied_prefix::stamp::LsnRange { + start: 5, + end: 12, + }], + }, + }; + let bytes = zerompk::to_msgpack_vec(&manifest).expect("encode"); + nodedb_wal::segment::write_checkpoint_framed(&ckpt_dir, KV_CKPT_MANIFEST, &bytes) + .expect("write manifest"); + drop(core); + + let mut restored = open_core_at(dir.path()); + let error = restored + .load_kv_checkpoints() + .expect_err("a malformed stamp must fail the load"); + assert!( + error.to_string().contains("replay stamp"), + "the error names the stamp: {error}" + ); + } } diff --git a/nodedb/src/data/executor/kv_checkpoint/manifest.rs b/nodedb/src/data/executor/kv_checkpoint/manifest.rs index b1cd78987..005eb65a6 100644 --- a/nodedb/src/data/executor/kv_checkpoint/manifest.rs +++ b/nodedb/src/data/executor/kv_checkpoint/manifest.rs @@ -43,6 +43,13 @@ impl CoreLoop { expected: KV_CKPT_FORMAT_VERSION, }); } + manifest + .replay + .validate() + .map_err(|source| CheckpointDecodeError::InvalidReplayStamp { + path: path.clone(), + source, + })?; Ok(Some(manifest)) } } diff --git a/nodedb/src/data/executor/kv_checkpoint/mod.rs b/nodedb/src/data/executor/kv_checkpoint/mod.rs index 38139c8e7..43349a955 100644 --- a/nodedb/src/data/executor/kv_checkpoint/mod.rs +++ b/nodedb/src/data/executor/kv_checkpoint/mod.rs @@ -27,7 +27,7 @@ //! //! ## Why a generation + manifest, and not a stamp per file //! -//! The LSN a checkpoint is durable through is what lets replay skip records it +//! The replay stamp a checkpoint carries is what lets replay skip records it //! already contains, so it must be recorded on disk. Recording it *per file* //! looks simpler but is unsound: a flush is per-collection and can partially //! fail, leaving collection `a` stamped at LSN 900 while `b` stays at 400. Two @@ -41,7 +41,7 @@ //! //! Writing every collection into a fresh `gen-{n}/` and publishing the whole set //! with ONE atomic manifest write removes the split by construction — every live -//! collection advances to a single LSN together, or none do and the previous +//! collection advances to a single stamp together, or none do and the previous //! generation stays live. The manifest is the only thing that makes a generation //! visible, so a torn or abandoned write is inert garbage rather than a //! half-published state. diff --git a/nodedb/src/data/executor/kv_checkpoint/write.rs b/nodedb/src/data/executor/kv_checkpoint/write.rs index 3684809e2..2280134bb 100644 --- a/nodedb/src/data/executor/kv_checkpoint/write.rs +++ b/nodedb/src/data/executor/kv_checkpoint/write.rs @@ -11,39 +11,44 @@ use super::format::{ use super::index_export::export_collection_indexes; use super::manifest::storage_err; use super::paths::{KV_CKPT_MANIFEST, kv_ckpt_dir, kv_ckpt_filename, kv_ckpt_gen_dir}; +use crate::data::executor::applied_prefix::ReplayStamp; use crate::data::executor::core_loop::CoreLoop; use crate::types::Lsn; impl CoreLoop { /// Flush every KV collection on this core to disk and return the LSN the KV - /// engine is now durable through. + /// engine reports as its truncation floor. /// - /// Returns `Ok(watermark)` only once a manifest naming a COMPLETE generation - /// has landed. Any failure returns `Err` — the caller must then clamp the + /// Returns `Ok` only once a manifest naming a COMPLETE generation has + /// landed. Any failure returns `Err` — the caller must then clamp the /// reported checkpoint LSN to the last LSN KV was known durable through, so /// a failed flush costs WAL growth instead of data. /// - /// Stamping the generation with the core watermark is exact, not - /// approximate: this runs on the core's own thread between tasks, and a KV - /// write only reaches `note_kv_write_lsn` (which raises the watermark) after - /// it has been applied to the table. So every KV row with `lsn <= watermark` - /// is already in the tables exported here. Records above the watermark - /// belong to other engines and are gated by their own floors. + /// ## What replay skips /// - /// Index DDL is the one KV record that does NOT raise the watermark + /// The manifest carries the core's [`ReplayStamp`], + /// taken on the core's own thread between tasks, so no write interleaves + /// with the export. The stamp names every record the exported tables hold: + /// every record at or below the outcome floor, and every record above it + /// this core applied. A record still on its way to the core is in neither, + /// so restart replay applies it. The tracker has no exact stamp while its + /// applied set is dropped; the checkpoint then fails. + /// + /// Index DDL is the one KV record that notes no applied LSN /// (`execute_kv_register_index` and its siblings note no write LSN, having no - /// row to attribute one to), so a registration made after the last row write - /// is exported into a generation stamped below its own LSN, and replays again - /// on top of the restored state. That is harmless in both directions and must - /// stay so: replaying a register whose registration is already restored is a - /// no-op (`add_index` reports the field as already indexed and skips the - /// backfill), and replaying a drop whose registration the export therefore - /// never saw is a no-op too. + /// row to attribute one to), so a registration made above the outcome floor + /// can be exported into a generation whose stamp does not name it, and it + /// replays again on top of the restored state. That is harmless in both + /// directions and must stay so: replaying a register whose registration is + /// already restored is a no-op (`add_index` reports the field as already + /// indexed and skips the backfill), and replaying a drop whose registration + /// the export therefore never saw is a no-op too. /// - /// Every published generation raises `kv_published_lsn`, the LSN restart - /// restores KV from (see `redo_apply::cover`). + /// Every published generation raises `kv_published_lsn` to the stamp's + /// prefix (see `redo_apply::cover`). pub(in crate::data::executor) fn checkpoint_kv_engines(&mut self) -> crate::Result { let durable_through = self.watermark; + let replay = self.floors.applied_prefix.stamp()?; let ckpt_dir = kv_ckpt_dir(&self.data_dir, self.core_id); std::fs::create_dir_all(&ckpt_dir).map_err(|e| storage_err(&ckpt_dir, "create dir", &e))?; @@ -65,8 +70,10 @@ impl CoreLoop { .map_err(|e| storage_err(&gen_dir, "create generation dir", &e))?; let written = self.write_kv_generation(&gen_dir)?; - self.publish_kv_generation(&ckpt_dir, generation, durable_through)?; - self.floors.kv_published_lsn = self.floors.kv_published_lsn.max(durable_through); + let prefix = Lsn::new(replay.prefix); + let applied_ranges = replay.applied_above.len(); + self.publish_kv_generation(&ckpt_dir, generation, durable_through, replay)?; + self.floors.kv_published_lsn = self.floors.kv_published_lsn.max(prefix); // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what @@ -92,6 +99,8 @@ impl CoreLoop { generation, collections = written, durable_through_lsn = durable_through.as_u64(), + replay_prefix = prefix.as_u64(), + applied_ranges, "KV checkpoint published" ); Ok(durable_through) @@ -157,7 +166,7 @@ impl CoreLoop { /// Publish a written generation by atomically replacing the manifest. /// /// This single write is the commit point of the whole checkpoint: before it - /// nothing changed; after it the entire generation is live at one LSN. It + /// nothing changed; after it the entire generation is live under one stamp. It /// also fsyncs `ckpt_dir`, the same directory holding the `gen-{n}/` entry, /// so that entry cannot still be pending when the manifest naming it becomes /// visible. @@ -166,11 +175,13 @@ impl CoreLoop { ckpt_dir: &std::path::Path, generation: u64, durable_through: Lsn, + replay: ReplayStamp, ) -> crate::Result<()> { let manifest = KvCheckpointManifest { format_version: KV_CKPT_FORMAT_VERSION, generation, durable_through_lsn: durable_through.as_u64(), + replay, }; let bytes = zerompk::to_msgpack_vec(&manifest).map_err(|e| crate::Error::Serialization { diff --git a/nodedb/src/data/executor/replay_floors.rs b/nodedb/src/data/executor/replay_floors.rs index 118ecf6ac..385b76c0c 100644 --- a/nodedb/src/data/executor/replay_floors.rs +++ b/nodedb/src/data/executor/replay_floors.rs @@ -1,23 +1,28 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Per-engine "already durable through LSN X" floors recovered from on-disk -//! checkpoints at boot, consulted by WAL replay so a restored checkpoint is not -//! re-derived from records it already contains. +//! Per-engine replay stamps recovered from on-disk checkpoints at boot, +//! consulted by WAL replay so a restored checkpoint is not re-derived from +//! records it already contains. //! -//! ## Why replaying ABOVE the floor is safe +//! ## Why replaying the records a stamp does not hold is safe //! +//! A checkpoint's [`ReplayStamp`] names exactly the records its state holds: +//! every record at or below its prefix, and the records above the prefix this +//! core applied before the checkpoint. Every other record replays, through the +//! same paths, in LSN order, on top of the restored state. +//! +//! A replayed record can have a lower LSN than a record the state already +//! holds: it was still on its way when the higher one applied. The live core +//! applied the two in that same order, so replay reproduces the live result. //! Every KV WAL record replays either as an absolute overwrite (`kv_put`, -//! `kv_batch_put`, `kv_delete`, `kv_truncate`) or as a delta re-executed against -//! the engine's current state (`kv_incr`, `kv_cas`, `kv_field_set`, -//! `kv_transfer`, ...). Restoring a checkpoint durable through LSN F reproduces -//! exactly the engine state that existed after record F was applied, so feeding -//! the records above F back through the same replay paths — in LSN order, on top -//! of that state — reaches the state a full from-zero replay would. -//! -//! It is records at or below F that MUST be skipped. For the absolute-overwrite +//! `kv_batch_put`, `kv_delete`, `kv_truncate`) or as a delta re-executed +//! against the engine's current state (`kv_incr`, `kv_cas`, `kv_field_set`, +//! `kv_transfer`, ...), and both land where the live apply landed. +//! +//! Records the stamp holds MUST be skipped. For the absolute-overwrite //! records re-applying is merely redundant, but for the delta records it is -//! corruption: an increment already folded into the checkpoint would be counted -//! twice. +//! corruption: an increment already folded into the checkpoint would be +//! counted twice. //! //! ## Why a floor is engine-wide rather than per-collection //! @@ -102,7 +107,7 @@ //! a floor here, or with a per-collection watermark it carries itself. It must //! not assume a cursor will cover it. -use crate::types::Lsn; +use crate::data::executor::applied_prefix::ReplayStamp; /// Checkpoint-restored replay floors for every engine on one core. /// @@ -140,40 +145,40 @@ pub(in crate::data::executor) struct ReplayFloors { pub(in crate::data::executor) columnar: ReplayFloor, } -/// The LSN an engine's restored checkpoint is durable through. +/// What an engine's restored checkpoint holds. /// /// `None` means no checkpoint was restored, so nothing is gated and the full WAL -/// replays. Shared by every engine in [`ReplayFloors`]: the gating rule (an -/// inclusive `record_lsn <= durable_through` check) is identical across -/// engines — what differs between them is WHY they need one at all, which is -/// documented on each field above rather than on this type. +/// replays. Shared by every engine in [`ReplayFloors`]: the gating rule is +/// [`ReplayStamp::skips`] for every engine. What differs between them is WHY +/// they need one at all, which is documented on each field above rather than +/// on this type. #[derive(Debug, Default)] pub(in crate::data::executor) struct ReplayFloor { - durable_through: Option, + stamp: Option, } impl ReplayFloor { - /// Record that the restored checkpoint is durable through `lsn`. + /// Record what the restored checkpoint holds. /// /// Set once per boot, from the manifest that named the restored generation. - pub(in crate::data::executor) fn set(&mut self, lsn: Lsn) { - self.durable_through = Some(lsn); + pub(in crate::data::executor) fn set(&mut self, stamp: ReplayStamp) { + self.stamp = Some(stamp); } /// Whether a record at `record_lsn` is already folded into the restored - /// checkpoint and must therefore NOT be replayed. - /// - /// Inclusive: the manifest's LSN is the one the generation is durable - /// THROUGH, so that record's effect is already present. + /// checkpoint, or has a final outcome that is not an apply, and must + /// therefore NOT be replayed. Every KV and columnar skip site asks here. pub(in crate::data::executor) fn covers(&self, record_lsn: u64) -> bool { - self.durable_through - .is_some_and(|floor| record_lsn <= floor.as_u64()) + self.stamp + .as_ref() + .is_some_and(|stamp| stamp.skips(record_lsn)) } } #[cfg(test)] mod tests { use super::*; + use crate::data::executor::applied_prefix::stamp::LsnRange; #[test] fn unset_floor_covers_nothing() { @@ -186,33 +191,30 @@ mod tests { } #[test] - fn covers_is_inclusive_of_the_stamped_lsn() { + fn covers_is_inclusive_of_the_stamped_prefix() { let mut floor = ReplayFloor::default(); - floor.set(Lsn::new(100)); - assert!(floor.covers(99), "below the floor is already durable"); - assert!(floor.covers(100), "the floor itself is already durable"); - assert!(!floor.covers(101), "above the floor must replay"); - } - - #[test] - fn unset_columnar_floor_covers_nothing() { - let floor = ReplayFloor::default(); - assert!(!floor.covers(1)); - assert!( - !floor.covers(u64::MAX), - "no checkpoint restored must never gate a record" - ); + floor.set(ReplayStamp::through(100)); + assert!(floor.covers(99), "below the prefix is already durable"); + assert!(floor.covers(100), "the prefix itself is already durable"); + assert!(!floor.covers(101), "above the prefix must replay"); } #[test] - fn columnar_covers_is_inclusive_of_the_stamped_lsn() { + fn covers_the_applied_set_and_replays_the_gaps_in_it() { let mut floor = ReplayFloor::default(); - floor.set(Lsn::new(100)); - assert!(floor.covers(99), "below the floor is already durable"); - assert!(floor.covers(100), "the floor itself is already durable"); + floor.set(ReplayStamp { + prefix: 100, + applied_above: vec![LsnRange { + start: 103, + end: 104, + }], + }); + assert!(floor.covers(103)); + assert!(floor.covers(104)); assert!( !floor.covers(101), - "above the floor must replay — gating it would drop the write" + "a record in flight when the checkpoint was written must replay" ); + assert!(!floor.covers(105)); } } diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs index e83bb96f0..638014203 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs @@ -23,8 +23,8 @@ impl CoreLoop { /// reported checkpoint LSN to the last LSN this engine was known durable /// through, so a failed flush costs WAL growth instead of data. /// - /// Stamping the generation with the core watermark mirrors - /// `checkpoint_kv_engines`: this runs on the core's own thread between + /// Stamping the generation with the core watermark rests on this: it runs + /// on the core's own thread between /// tasks, so every sparse-vector write the core has admitted is already /// folded into the in-memory indexes exported here. Where a sparse-vector /// write did not itself raise the watermark, the stamp merely UNDERSTATES diff --git a/nodedb/src/data/executor/spatial_checkpoint/write.rs b/nodedb/src/data/executor/spatial_checkpoint/write.rs index 8a1a097f7..b861068d8 100644 --- a/nodedb/src/data/executor/spatial_checkpoint/write.rs +++ b/nodedb/src/data/executor/spatial_checkpoint/write.rs @@ -46,8 +46,8 @@ impl CoreLoop { /// drop geometry entries while the rows they point at survive — a spatial /// predicate silently stops matching rows a full scan still returns. /// - /// Stamping with the core watermark mirrors `checkpoint_kv_engines`: this - /// runs on the core's own thread between tasks, and a geometry write raises + /// Stamping with the core watermark rests on this: the checkpoint runs + /// on the core's own thread between tasks, and a geometry write raises /// the watermark only after the R-tree has already been mutated. pub(crate) fn checkpoint_spatial_indexes(&self) -> crate::Result { let durable_lsn = self.watermark; diff --git a/nodedb/src/data/executor/sync_hwm_checkpoint/write.rs b/nodedb/src/data/executor/sync_hwm_checkpoint/write.rs index b376b684a..aabe52f2e 100644 --- a/nodedb/src/data/executor/sync_hwm_checkpoint/write.rs +++ b/nodedb/src/data/executor/sync_hwm_checkpoint/write.rs @@ -24,8 +24,8 @@ impl CoreLoop { /// is intact and live, after it the new one is. There is no window in which /// half a gate is published. /// - /// Stamping with the core watermark mirrors `checkpoint_kv_engines`: this - /// runs on the core's own thread between tasks, and `sync_commit` advances + /// Stamping with the core watermark rests on this: the checkpoint runs + /// on the core's own thread between tasks, and `sync_commit` advances /// the HWM only after the frame's `SyncSeqAdvance` record is durable, so /// every advance the core has admitted is already in the maps exported here. pub(in crate::data::executor) fn checkpoint_sync_hwm(&self) -> crate::Result { diff --git a/nodedb/src/data/executor/vector_checkpoint/write.rs b/nodedb/src/data/executor/vector_checkpoint/write.rs index b804bdef3..8eaaac560 100644 --- a/nodedb/src/data/executor/vector_checkpoint/write.rs +++ b/nodedb/src/data/executor/vector_checkpoint/write.rs @@ -36,8 +36,8 @@ impl CoreLoop { /// partial success cannot be expressed, because the LSN it would justify /// does not exist. /// - /// Stamping with the core watermark mirrors `checkpoint_kv_engines`: this - /// runs on the core's own thread between tasks, and a vector write raises + /// Stamping with the core watermark rests on this: the checkpoint runs + /// on the core's own thread between tasks, and a vector write raises /// the watermark only after the collection has already been mutated, so /// every write with `lsn <= watermark` is in the bytes written below. /// diff --git a/nodedb/src/data/executor/wal_replay_all.rs b/nodedb/src/data/executor/wal_replay_all.rs index db459c67b..7a99db993 100644 --- a/nodedb/src/data/executor/wal_replay_all.rs +++ b/nodedb/src/data/executor/wal_replay_all.rs @@ -41,6 +41,18 @@ impl CoreLoop { } let core_id = self.core_id; + // Replay decides every record handed to it before this core serves a + // request or writes a checkpoint: it applies the record, or a stamp, + // a tombstone or an abort marker says it must not. Every one of them + // is therefore at or below the outcome floor once replay ends. Raised + // before the passes so the records replay applies are not collected + // one by one into the applied set. + if let Some(wal_end) = records.iter().map(|record| record.header.lsn).max() { + self.floors + .applied_prefix + .seed_replayed_through(crate::types::Lsn::new(wal_end)); + } + crate::fail_point!("replay::before_engine_passes"); // Every engine-bearing record class — standalone (autocommit) records @@ -91,3 +103,244 @@ impl CoreLoop { } } } + +#[cfg(test)] +mod tests { + use nodedb_wal::record::{RecordType, WalRecordArgs}; + use nodedb_wal::{TombstoneSet, WalRecord}; + + use nodedb_types::Value; + use nodedb_types::columnar::{ + COLUMNAR_IMAGE_KIND, ColumnDef, ColumnType, ColumnarImageWalRecord, ColumnarImageWalRow, + ColumnarSchema, + }; + + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::CoreLoop; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; + use crate::types::{DatabaseId, Lsn, TenantId}; + use crate::wal::{RedoRecord, RedoSubRecord}; + + const TID: u64 = 1; + + fn kv_put_record(lsn: u64, key: &[u8]) -> WalRecord { + WalRecord::new(WalRecordArgs { + record_type: RecordType::Put as u32, + lsn, + tenant_id: 1, + vshard_id: 0, + database_id: 0, + payload: zerompk::to_msgpack_vec(&( + "kv_put", + "cache", + key.to_vec(), + b"v".to_vec(), + 0u64, + None::, + 1u32, + )) + .expect("encode kv put"), + encryption_key: None, + preamble_bytes: None, + }) + .expect("build record") + } + + /// Every record replay reads is decided when replay ends, so a checkpoint + /// written afterwards names all of them through its prefix, and the + /// applied set holds none of them. + #[test] + fn replay_seeds_the_applied_prefix_through_the_wal_end() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let records = vec![kv_put_record(40, b"a"), kv_put_record(77, b"b")]; + + core.replay_all_wal(&records, 1, &TombstoneSet::new()); + + assert_eq!(core.floors.applied_prefix.outcome_floor(), Lsn::new(77)); + let stamp = core.floors.applied_prefix.stamp().expect("exact stamp"); + assert_eq!(stamp.prefix, 77); + assert!(stamp.applied_above.is_empty()); + } + + fn redo_bytes(op: RedoSubRecord) -> Vec { + RedoRecord { + version: 1, + ops: vec![op], + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo") + } + + fn redo_record(lsn: u64, redo: &[u8]) -> WalRecord { + WalRecord::new(WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn, + tenant_id: TID, + vshard_id: 0, + database_id: 0, + payload: redo.to_vec(), + encryption_key: None, + preamble_bytes: None, + }) + .expect("build redo record") + } + + /// Install `redo` live at `lsn`, the way a committed record reaches a core. + fn install(core: &mut CoreLoop, lsn: u64, redo: &[u8], collection: &str) { + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(lsn)); + let response = core.execute_apply_transaction_redo( + &task, + TID, + CommittedRedo { + redo, + collections: &[collection.to_string()], + sum_targets: &[], + }, + ); + assert_eq!( + response.status, + Status::Ok, + "install at {lsn}: {response:?}" + ); + } + + fn kv_put(key: &[u8], value: &[u8], surrogate: u32) -> RedoSubRecord { + RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(&( + "kv_put", + "cache", + key.to_vec(), + value.to_vec(), + 0u64, + None::, + surrogate, + )) + .expect("encode kv put"), + } + } + + fn columnar_row(id: i64, value: i64, surrogate: u32) -> RedoSubRecord { + let mut image = std::collections::HashMap::new(); + image.insert("id".to_string(), Value::Integer(id)); + image.insert("v".to_string(), Value::Integer(value)); + let record = ColumnarImageWalRecord { + kind: COLUMNAR_IMAGE_KIND.to_string(), + collection: "m".to_string(), + schema_bytes: Vec::new(), + rows: vec![ColumnarImageWalRow { + surrogate, + prior_pk_msgpack: Vec::new(), + image_msgpack: nodedb_types::value_to_msgpack(&Value::Object(image)) + .expect("encode image"), + }], + }; + RedoSubRecord { + record_type: RecordType::TimeseriesBatch as u32, + payload: zerompk::to_msgpack_vec(&record).expect("encode image record"), + } + } + + fn columnar_ids(core: &CoreLoop) -> Vec { + let key = (DatabaseId::DEFAULT, TenantId::new(TID), "m".to_string()); + let mut ids: Vec = core + .columnar_engines + .get(&key) + .expect("engine") + .scan_memtable_rows() + .filter_map(|row| match row.first() { + Some(Value::Integer(id)) => Some(*id), + _ => None, + }) + .collect(); + ids.sort_unstable(); + ids + } + + /// Record B, at LSN 30, applies and a checkpoint is written while record A, + /// at LSN 20, is still on its way. A applies after the checkpoint and the + /// core stops before another one. Restart replay must apply A and skip B. + #[test] + fn a_kv_record_in_flight_at_a_checkpoint_replays_after_a_crash() { + let dir = tempfile::tempdir().expect("tempdir"); + let a = redo_bytes(kv_put(b"a", b"held", 1)); + let b = redo_bytes(kv_put(b"b", b"applied", 2)); + { + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(10)); + install(&mut core, 30, &b, "cache"); + core.checkpoint_kv_engines().expect("checkpoint"); + install(&mut core, 20, &a, "cache"); + } + + let (mut restored, _req, _resp) = make_core_with_dir(dir.path()); + restored.load_kv_checkpoints().expect("load"); + restored.replay_all_wal( + &[redo_record(20, &a), redo_record(30, &b)], + 1, + &TombstoneSet::new(), + ); + let now = crate::engine::kv::current_ms(); + let expected: [(&[u8], &[u8]); 2] = [(b"a", b"held"), (b"b", b"applied")]; + for (key, value) in expected { + assert_eq!( + restored.kv_engine.get(0, TID, "cache", key, now).as_deref(), + Some(value), + "the replayed state equals the live one" + ); + } + } + + #[test] + fn a_columnar_record_in_flight_at_a_checkpoint_replays_after_a_crash() { + let dir = tempfile::tempdir().expect("tempdir"); + let a = redo_bytes(columnar_row(1, 10, 1)); + let b = redo_bytes(columnar_row(2, 20, 2)); + let live = { + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let schema = ColumnarSchema { + columns: vec![ + ColumnDef::required("id", ColumnType::Int64).with_primary_key(), + ColumnDef::required("v", ColumnType::Int64), + ], + version: 1, + }; + core.columnar_engines.insert( + (DatabaseId::DEFAULT, TenantId::new(TID), "m".to_string()), + nodedb_columnar::MutationEngine::new("m".to_string(), schema), + ); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(10)); + install(&mut core, 30, &b, "m"); + core.checkpoint_columnar_engines().expect("checkpoint"); + install(&mut core, 20, &a, "m"); + columnar_ids(&core) + }; + assert_eq!(live, vec![1, 2]); + + let (mut restored, _req, _resp) = make_core_with_dir(dir.path()); + restored.load_columnar_checkpoints().expect("load"); + assert_eq!( + columnar_ids(&restored), + vec![2], + "the checkpoint holds B and not A" + ); + restored.replay_all_wal( + &[redo_record(20, &a), redo_record(30, &b)], + 1, + &TombstoneSet::new(), + ); + assert_eq!( + columnar_ids(&restored), + live, + "replay applies A once and never applies B again" + ); + } +} diff --git a/nodedb/tests/crash_replay_stamp.rs b/nodedb/tests/crash_replay_stamp.rs new file mode 100644 index 000000000..487ea347c --- /dev/null +++ b/nodedb/tests/crash_replay_stamp.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A KV write still in flight when a checkpoint is written survives a crash. +//! +//! LSNs are node-global and a write reaches its core out of mint order. Client +//! row writes apply through the one Raft apply loop, which finishes each entry +//! before it starts the next, so two of them never overtake each other. A +//! write outside that loop can: the engine step of an index DDL committed in +//! a transaction runs on its own session. The test uses that step as write A: +//! +//! 1. `COMMIT` of a block holding `CREATE SORTED INDEX` lands the catalog +//! record, then appends A, the sorted-index registration, and parks it at +//! the funnel gate. Its LSN is minted and no core holds it. +//! 2. Client writes B apply through the Raft loop with higher LSNs, until a +//! KV checkpoint's replay stamp names one of them above its prefix. That +//! proves the checkpoint was written while A was minted and not applied. +//! 3. The test releases A. A applies, and the process aborts before A's +//! response leaves, so no later checkpoint holds A. +//! 4. After restart, replay must apply A: the stamp does not name it. A stamp +//! holding only the highest applied LSN would skip A, and the index would +//! have a catalog record and no tree. +//! +//! The columnar engine has no write outside the Raft loop, so its in-flight +//! case runs in-process in `wal_replay_all.rs`. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +mod crash_harness; + +use std::time::{Duration, Instant}; + +use crash_harness::{CrashHarness, diagnostics}; + +/// One checkpoint per second, so several are written while A is parked. +const CHECKPOINT_INTERVAL_SECS: &str = "1"; + +/// How long the test waits for a checkpoint whose stamp names a B: thirty +/// checkpoint cycles at one per second. A is parked until the test releases +/// it, so this bounds only the wait for the checkpoint manager. +const STAMP_DEADLINE: Duration = Duration::from_secs(30); + +/// How long the process may take to abort once A is released. +const CRASH_TIMEOUT: Duration = Duration::from_secs(60); + +const HELD: &str = "stamp_kv_lo"; +const INDEX: &str = "stamp_kv_idx"; +const SEEDED_ROWS: u64 = 3; + +#[tokio::test(flavor = "multi_thread")] +async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { + let mut h = CrashHarness::new() + .with_env("NODEDB_CHECKPOINT_INTERVAL_SECS", CHECKPOINT_INTERVAL_SECS) + .with_env( + "RUST_LOG", + "warn,nodedb::data::executor::kv_checkpoint=info", + ); + h.spawn(); + h.wait_ready(); + h.exec(&format!( + "CREATE COLLECTION {HELD} (k STRING PRIMARY KEY, score INT) WITH (engine='kv')" + )) + .await; + for i in 0..SEEDED_ROWS { + h.exec(&format!( + "INSERT INTO {HELD} (k, score) VALUES ('p{i}', {})", + i * 10 + )) + .await; + } + h.exec("CREATE COLLECTION stamp_kv_hi (k STRING PRIMARY KEY, v STRING) WITH (engine='kv')") + .await; + + // Boot 2 arms the gate and the abort, keyed to the held collection. Both + // match only a request carrying a WAL LSN, so boot itself passes them. + h.kill_9(); + let release = h.data_dir().join("release-held-write"); + h.set_env( + "NODEDB_FAILPOINTS", + &format!( + "funnel::before_dispatch::{HELD}=wait_file({}),core::after_apply::{HELD}=abort", + release.display() + ), + ); + h.reopen(); + + let conn_str = h.pgwire_conn_str(); + let held_task = tokio::spawn(async move { + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .map_err(|e| e.to_string())?; + tokio::spawn(connection); + for sql in [ + "BEGIN".to_string(), + format!("CREATE SORTED INDEX {INDEX} ON {HELD} (score DESC) KEY k"), + "COMMIT".to_string(), + ] { + client + .simple_query(&sql) + .await + .map_err(|e| format!("{sql}: {e}"))?; + } + Ok::<(), String>(()) + }); + + // Apply B writes until a checkpoint names one of them above its prefix. + let deadline = Instant::now() + STAMP_DEADLINE; + let mut applied = 0usize; + loop { + h.exec(&format!( + "INSERT INTO stamp_kv_hi (k, v) VALUES ('k{applied:03}', 'v{applied}')" + )) + .await; + applied += 1; + tokio::time::sleep(Duration::from_millis(200)).await; + let log = boot_section(&h.server_log(), 2); + if applied_ranges(&log, "KV checkpoint published") + .iter() + .any(|n| *n > 0) + { + break; + } + assert!( + Instant::now() < deadline, + "no KV checkpoint named an applied LSN above its prefix within \ + {STAMP_DEADLINE:?}: write A never parked, or no checkpoint ran while it was.{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + } + assert!( + !held_task.is_finished(), + "write A finished before its release: the gate never parked it" + ); + let mut live = h.query_col_idx("SELECT v FROM stamp_kv_hi", 0).await; + live.sort(); + + // Release A. It applies, and the process aborts before its response + // leaves, so no checkpoint written after it can hold it. + std::fs::write(&release, b"release").expect("create the release file"); + h.await_self_crash(CRASH_TIMEOUT); + let marker = format!("fail_point aborting process: core::after_apply::{HELD}"); + assert!( + h.server_log().contains(&marker), + "the process exited, but not after write A applied.{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + // A's client lost its connection with the process; its result says nothing. + let _ = held_task.await; + + h.clear_env("NODEDB_FAILPOINTS"); + h.reopen(); + + let ranges = applied_ranges(&boot_section(&h.server_log(), 3), "KV checkpoint restored"); + assert!( + ranges.iter().any(|n| *n > 0), + "the restored generation must name an applied LSN above its prefix, or this run did \ + not reproduce the in-flight write (restored stamps: {ranges:?}).{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + + let count = h + .query_col_idx(&format!("SELECT SORTED_COUNT({INDEX})"), 0) + .await; + assert_eq!( + count.first().and_then(|text| json_field(text, "count")), + Some(SEEDED_ROWS), + "write A applied after the checkpoint and before the crash; replay must rebuild \ + the index tree from it, never skip it as covered by a higher applied LSN \ + (got {count:?})" + ); + let mut replayed = h.query_col_idx("SELECT v FROM stamp_kv_hi", 0).await; + replayed.sort(); + assert_eq!( + replayed, live, + "the replayed state must equal the live state: every B write once" + ); +} + +/// The server output of boot `n`, from its harness marker to the next one. +fn boot_section(log: &str, n: u32) -> String { + let marker = format!("=== crash harness boot {n} (pid"); + let Some(start) = log.find(&marker) else { + return String::new(); + }; + let rest = &log[start..]; + let next = format!("=== crash harness boot {} (pid", n + 1); + match rest.find(&next) { + Some(end) => rest[..end].to_string(), + None => rest.to_string(), + } +} + +/// The `applied_ranges` field of every log line carrying `message`. +fn applied_ranges(log: &str, message: &str) -> Vec { + strip_ansi(log) + .lines() + .filter(|line| line.contains(message)) + .filter_map(|line| { + let rest = line.split_once("applied_ranges=")?.1; + let digits: String = rest.chars().take_while(char::is_ascii_digit).collect(); + digits.parse().ok() + }) + .collect() +} + +/// The unsigned integer a single-cell JSON reply carries under `field`. +fn json_field(text: &str, field: &str) -> Option { + let rest = text.split_once(&format!("\"{field}\":"))?.1; + let digits: String = rest + .trim_start() + .chars() + .take_while(char::is_ascii_digit) + .collect(); + digits.parse().ok() +} + +/// `text` without terminal colour escape sequences. +fn strip_ansi(text: &str) -> String { + let mut out = String::with_capacity(text.len()); + let mut chars = text.chars(); + while let Some(c) = chars.next() { + if c == '\u{1b}' { + for next in chars.by_ref() { + if next == 'm' { + break; + } + } + } else { + out.push(c); + } + } + out +} + +#[test] +fn log_fields_are_read_through_colour_codes() { + let log = "INFO KV checkpoint published \u{1b}[3mapplied_ranges\u{1b}[0m\u{1b}[2m=\u{1b}[0m2\n\ + INFO KV checkpoint published applied_ranges=0\n"; + assert_eq!(applied_ranges(log, "KV checkpoint published"), vec![2, 0]); + let booted = "=== crash harness boot 1 (pid 1) ===\nfirst-line\n\ + === crash harness boot 2 (pid 2) ===\nsecond-line\n"; + assert!(boot_section(booted, 2).contains("second-line")); + assert!(!boot_section(booted, 2).contains("first-line")); + assert!(boot_section(booted, 1).contains("first-line")); + assert!(!boot_section(booted, 1).contains("second-line")); + assert_eq!(json_field("{\"count\":3}", "count"), Some(3)); +} From 8bb83f04e3d3352a7c7b583750caa33b61d178b0 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 11:00:08 +0800 Subject: [PATCH 29/64] feat(control): route every autocommit write through a durable path KV counter atomics, sorted-index registration, rate-gate counters, and weighted-pick's audit write used to dispatch straight to the Data Plane, bypassing the write-admission funnel and Raft. That write has no WAL record and no replica ever sees it: a crash loses it, and a follower's KV_INCR answers with a stale value. Add a durable-write module (`dispatch_utils::durable_write`) that proposes a write through Raft when this node runs a proposer and the plan is replicable, and otherwise dispatches it through the funnel's `AppendHere` route under the write-admission guard. Route every planned and hand-built autocommit write through it: `dispatch_authorized_durable_write` for tasks that still need the clone-write and authorization gates, `dispatch_durable_autocommit_write` for one already built as an `AutocommitWrite`, and `dispatch_authorized_task_by_class` for a transport that dispatches reads and writes through one call site. Add `refuse_unlogged_write` at the read-dispatch boundary so a write that reaches a route with no WAL append is rejected before it applies, rather than silently losing durability. Extend the in-transaction staging gate's `InTxnRoute` with an `Autocommit` variant so a write outside a transaction block (or one a transaction cannot buffer) is distinguished from a read at the gate, and every caller that matched `InTxnRoute::Read` for both cases now routes the write side through the durable path. Publish the change-feed event for a replicated write from the proposing node in both the gateway and the executor's replicated-entry path, since a replica applies with `ChangeFeedOwner::Unowned`. Harden `weighted_pick`'s row scan and audit write to surface a refusal or a decode failure as an error instead of silently returning an empty result or logging and continuing. Add a `CrashHarness::standalone` boot mode (no Raft proposer, so autocommit writes take the local funnel route) and cover the new path with crash, in-process, native, and cluster-replication tests for KV counter atomics. --- .config/nextest.toml | 2 +- .../cases/kv_atomic_autocommit_replicates.rs | 207 ++++++++++ .../tests/common_suite/cases/mod.rs | 1 + nodedb/src/control/exec_receiver/executor.rs | 16 +- nodedb/src/control/gateway/dispatcher.rs | 84 +++- .../control/server/dispatch_utils/dispatch.rs | 8 + .../dispatch_utils/durability_barrier.rs | 36 ++ .../server/dispatch_utils/durable_write.rs | 190 +++++++++ .../src/control/server/dispatch_utils/mod.rs | 5 + .../server/http/routes/promql/remote.rs | 2 +- .../http/routes/query/materialized/shape.rs | 2 +- .../server/http/routes/query/ndjson.rs | 3 +- .../server/http/routes/ws_rpc/execute_sql.rs | 5 +- .../server/native/dispatch/raw_dispatch.rs | 18 +- .../server/native/dispatch/single_task.rs | 14 +- .../server/native/dispatch/sql_gateway.rs | 3 +- .../server/native/dispatch/sql_loop.rs | 12 +- .../handler/routing/execute_dml_hooks.rs | 6 +- .../control/server/resp/gateway_dispatch.rs | 2 +- .../neutral/collection/dml/parse/dispatch.rs | 39 +- .../ddl/neutral/graph_ops/edge_stage.rs | 12 +- .../shared/ddl/neutral/kv_atomic/dispatch.rs | 177 ++++++--- .../ddl/neutral/kv_sorted_index/dispatch.rs | 46 ++- .../server/shared/ddl/neutral/rate_gate.rs | 58 ++- .../shared/ddl/neutral/weighted_pick.rs | 158 +++++--- .../server/shared/session/staging_gate.rs | 67 ++-- nodedb/src/control/system_txn/run.rs | 7 + nodedb/tests/crash_harness/mod.rs | 12 + nodedb/tests/crash_kv_atomic_autocommit.rs | 111 ++++++ nodedb/tests/crash_replay_stamp.rs | 194 +++++---- nodedb/tests/crash_resp_kv_write.rs | 12 +- .../inproc/cases/kv_atomic_autocommit_wal.rs | 376 ++++++++++++++++++ nodedb/tests/inproc/cases/mod.rs | 1 + nodedb/tests/native/cases/mod.rs | 1 + .../cases/native_kv_atomic_autocommit_wal.rs | 182 +++++++++ 35 files changed, 1756 insertions(+), 313 deletions(-) create mode 100644 nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs create mode 100644 nodedb/src/control/server/dispatch_utils/durable_write.rs create mode 100644 nodedb/tests/crash_kv_atomic_autocommit.rs create mode 100644 nodedb/tests/inproc/cases/kv_atomic_autocommit_wal.rs create mode 100644 nodedb/tests/native/cases/native_kv_atomic_autocommit_wal.rs diff --git a/.config/nextest.toml b/.config/nextest.toml index 5efbb0c7b..cc1394558 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -147,7 +147,7 @@ slow-timeout = { period = "30s", terminate-after = 8 } # hide the cause rather than fix it. The tail this costs is bounded — roughly a # dozen crash/shutdown tests at about ten seconds each. [[profile.default.overrides]] -filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' +filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | binary(crash_replay_stamp) | binary(crash_kv_atomic_autocommit) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' test-group = 'server-process' threads-required = 'num-test-threads' diff --git a/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs b/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs new file mode 100644 index 000000000..dfac58743 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs @@ -0,0 +1,207 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! An autocommit SQL-function write reaches every replica. +//! +//! `KV_INCR` and `CREATE SORTED INDEX` build their `KvOp` by hand instead of +//! planning a statement. In cluster mode each one must be proposed through +//! the data group's Raft log like a planned write, so every replica applies +//! it. A write applied on the receiving node alone exists nowhere else. +//! +//! - `KV_INCR` runs on the counter's data-group leader, and the test kills +//! that leader. A survivor must read the incremented counter: had the write +//! applied on the leader alone, it died with it. +//! - `CREATE SORTED INDEX` builds a tree on the core that owns the rows. A +//! sorted-index read runs on the node that receives it, so a count read on +//! each follower reads that follower's own tree. + +use crate::common; +use common::cluster_harness::TestCluster; + +use std::time::{Duration, Instant}; + +use nodedb::types::{DatabaseId, VShardId}; + +const COUNTERS: &str = "repl_kv_ctr"; +const BOARD: &str = "repl_kv_board"; +const INDEX: &str = "repl_kv_board_idx"; + +fn pg_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +/// The first column of the first row `sql` returns, or the error it raised. +async fn first_cell(client: &tokio_postgres::Client, sql: &str) -> Result, String> { + let rows = client.simple_query(sql).await.map_err(|e| pg_detail(&e))?; + Ok(rows.into_iter().find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get(0).map(str::to_string), + _ => None, + })) +} + +/// Whether a `SORTED_COUNT` read returned a count of three. +fn counts_three(read: &Result, String>) -> bool { + matches!(read, Ok(Some(doc)) if doc.replace(' ', "").contains("\"count\":3")) +} + +/// The leader node id of the data group that owns `collection`. +fn group_leader(cluster: &TestCluster, collection: &str) -> u64 { + let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection); + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let group = routing + .group_for_vshard(vshard.as_u32()) + .expect("the collection's vShard maps to a data group"); + routing + .group_info(group) + .map(|info| info.leader) + .unwrap_or(0) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn an_autocommit_kv_incr_survives_its_leader() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COUNTERS} (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')" + )) + .await + .expect("create the counter collection"); + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {COUNTERS} (key, n) VALUES ('ctr', 5)" + )) + .await + .unwrap_or_else(|e| panic!("seed the counter: {}", pg_detail(&e))); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let leader_id = group_leader(&cluster, COUNTERS); + assert_ne!(leader_id, 0, "the counter's data group has no leader"); + let leader = cluster + .nodes + .iter() + .find(|n| n.node_id == leader_id) + .expect("the leader node is in the cluster"); + let incremented = first_cell( + &leader.client, + &format!("SELECT KV_INCR('{COUNTERS}', 'ctr', 3)"), + ) + .await + .unwrap_or_else(|e| panic!("KV_INCR on the group leader: {e}")); + assert!( + incremented.as_deref().is_some_and(|doc| doc.contains('8')), + "the live KV_INCR returns 5 + 3: {incremented:?}" + ); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let mut nodes = cluster.nodes; + let leader_idx = nodes + .iter() + .position(|n| n.node_id == leader_id) + .expect("leader node present"); + nodes.remove(leader_idx).shutdown().await; + + let read = format!("SELECT n FROM {COUNTERS} WHERE key = 'ctr'"); + for node in &nodes { + let deadline = Instant::now() + Duration::from_secs(30); + let mut last = Err(String::from("never read")); + while Instant::now() < deadline { + last = first_cell(&node.client, &read).await; + if matches!(&last, Ok(Some(n)) if n == "8") { + break; + } + tokio::time::sleep(Duration::from_millis(200)).await; + } + assert_eq!( + last, + Ok(Some("8".to_string())), + "survivor node {} must read the counter the autocommit KV_INCR moved; 5 means \ + the increment applied on the killed leader alone", + node.node_id + ); + } + + for node in nodes { + node.shutdown().await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_sorted_index_is_built_on_every_replica() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {BOARD} (k TEXT PRIMARY KEY, score INT) WITH (engine='kv')" + )) + .await + .expect("create the board collection"); + for (key, score) in [("p0", 10), ("p1", 20), ("p2", 30)] { + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {BOARD} (k, score) VALUES ('{key}', {score})" + )) + .await + .unwrap_or_else(|e| panic!("insert {key}: {}", pg_detail(&e))); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let leader_id = group_leader(&cluster, BOARD); + assert_ne!(leader_id, 0, "the board's data group has no leader"); + let leader = cluster + .nodes + .iter() + .find(|n| n.node_id == leader_id) + .expect("the leader node is in the cluster"); + leader + .client + .simple_query(&format!( + "CREATE SORTED INDEX {INDEX} ON {BOARD} (score DESC) KEY k" + )) + .await + .unwrap_or_else(|e| panic!("CREATE SORTED INDEX: {}", pg_detail(&e))); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let read = format!("SELECT SORTED_COUNT({INDEX})"); + for node in cluster.nodes.iter().filter(|n| n.node_id != leader_id) { + let deadline = Instant::now() + Duration::from_secs(30); + let mut last = Err(String::from("never read")); + while Instant::now() < deadline { + last = first_cell(&node.client, &read).await; + if counts_three(&last) { + break; + } + tokio::time::sleep(Duration::from_millis(200)).await; + } + assert!( + counts_three(&last), + "follower node {} must count every row in its own replica of the index tree; \ + a missing tree means the registration applied on the receiving node alone \ + (last read: {last:?})", + node.node_id + ); + } + + for node in cluster.nodes { + node.shutdown().await; + } +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs index 7842c819a..8f71c9950 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs @@ -55,6 +55,7 @@ mod http_gateway_migration; mod ilp_gateway_migration; mod install_snapshot_crdt_constraints_cluster; mod install_snapshot_e2e_cluster; +mod kv_atomic_autocommit_replicates; mod learner_cleanup; mod linearizable_read_leadership; mod listeners_gateway_smoke; diff --git a/nodedb/src/control/exec_receiver/executor.rs b/nodedb/src/control/exec_receiver/executor.rs index c7d32f5ea..d10258b14 100644 --- a/nodedb/src/control/exec_receiver/executor.rs +++ b/nodedb/src/control/exec_receiver/executor.rs @@ -307,7 +307,21 @@ impl LocalPlanExecutor { { // Replicated writes carry no read watermark → 0: it floors a // session's later reads, and this RPC seam has no session. - Ok((payload, _write_version)) => ExecuteResponse::ok(vec![payload], 0, 0), + Ok((payload, write_version)) => { + // Replicas apply with `ChangeFeedOwner::Unowned`. This + // node proposed the write once, so it publishes the + // change event. + crate::control::server::dispatch_utils::publish_change_set_with_lsn( + &self.state, + tenant_id, + database_id, + crate::control::server::dispatch_utils::extract_write_change_set( + &plan, tenant_id, + ), + write_version, + ); + ExecuteResponse::ok(vec![payload], 0, 0) + } // A replicated write's apply verdict is a Data-Plane // verdict: carry its code, never flatten to internal. Err(e) => ExecuteResponse::err(execution_error_to_typed(e)), diff --git a/nodedb/src/control/gateway/dispatcher.rs b/nodedb/src/control/gateway/dispatcher.rs index e0f3b7507..fbc989a24 100644 --- a/nodedb/src/control/gateway/dispatcher.rs +++ b/nodedb/src/control/gateway/dispatcher.rs @@ -14,14 +14,17 @@ use nodedb_cluster::rpc_codec::TypedClusterError; use crate::Error; use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; use crate::control::server::dispatch_utils::{ - dispatch_to_data_plane_with_txn, reject_data_plane_error, + AutocommitWrite, dispatch_autocommit_write, dispatch_to_data_plane_with_txn, + extract_write_change_set, publish_change_set_with_lsn, reject_data_plane_error, }; use crate::control::server::result_stream::ResultStream; +use crate::control::server::shared::write_admission::plan_is_write; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; use super::dispatch_remote::{RemoteDispatchArgs, dispatch_remote, dispatch_remote_stream}; use super::route::{RouteDecision, TaskRoute}; +use super::router::is_task_vshard_scoped; use super::version_check::check_descriptor_versions; use super::version_set::GatewayVersionSet; @@ -275,15 +278,15 @@ async fn dispatch_local( let resp = shared .vshard_admission_sequencer .run(vshard_id, || async { - dispatch_to_data_plane_with_txn( + dispatch_local_plan(LocalPlan { shared, tenant_id, database_id, vshard_id, - route.plan, + plan: route.plan, trace_id, - None, - ) + txn_id: None, + }) .await }) .await?; @@ -308,6 +311,15 @@ async fn dispatch_local( let (payload, write_version) = crate::control::wal_replication::propose_replicated_entry(shared, proposer, entry) .await?; + // Replicas apply with `ChangeFeedOwner::Unowned`. This node proposed + // the write once, so it publishes the change event. + publish_change_set_with_lsn( + shared, + tenant_id, + database_id, + extract_write_change_set(&route.plan, tenant_id), + write_version, + ); return Ok(DispatchOutcome { payloads: vec![payload], // A write carries no read watermark (Lsn::ZERO); its post-write @@ -318,15 +330,15 @@ async fn dispatch_local( }); } - let resp = dispatch_to_data_plane_with_txn( + let resp = dispatch_local_plan(LocalPlan { shared, tenant_id, database_id, vshard_id, - route.plan, + plan: route.plan, trace_id, txn_id, - ) + }) .await?; // The remote sibling turns `ExecuteResponse.error` into `Err`; the local // route must reject its own error status the same way. Keeping only the @@ -341,6 +353,62 @@ async fn dispatch_local( }) } +/// One plan this node applies on its own cores. +struct LocalPlan<'a> { + shared: &'a Arc, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: PhysicalPlan, + trace_id: TraceId, + txn_id: Option, +} + +/// Dispatch a plan to this node's cores on the route its class needs. +/// +/// A base-state write enters the funnel with `AppendHere`, which appends its +/// redo record under the write-admission guard. It reaches here when no Raft +/// proposal carries it: a standalone node, a plan with no replicated +/// encoding, or a write a transaction cannot buffer. The transaction meta-ops +/// own their durability, and a staged write is logged at COMMIT, so both take +/// the read route with everything else. +async fn dispatch_local_plan(local: LocalPlan<'_>) -> Result { + let LocalPlan { + shared, + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + } = local; + if plan_is_write(&plan) && !is_task_vshard_scoped(&plan) { + return dispatch_autocommit_write( + shared, + AutocommitWrite { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + event_source: crate::event::EventSource::User, + txn_id, + }, + ) + .await; + } + dispatch_to_data_plane_with_txn( + shared, + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + ) + .await +} + /// Whether the core refused the task with `ErrorCode::NotFound`. /// /// `reject_data_plane_error` passes this refusal as an empty success. diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index bd91df676..1563539c1 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -16,6 +16,10 @@ use super::submit_write::{ use super::types::{AutocommitWrite, DataPlaneDispatch, WriteDispatch}; /// Dispatch a clone-checked, capability-bearing external task to the Data Plane. +/// +/// The read route: it appends no WAL record. A write whose caller owns no +/// record for it goes through `dispatch_authorized_durable_write`, or +/// `dispatch_authorized_task_by_class` where one call site carries both. pub async fn dispatch_authorized_to_data_plane( shared: &SharedState, checked: CloneCheckedTask, @@ -289,6 +293,9 @@ pub(crate) async fn dispatch_autocommit_write( /// id so the Data Plane can resolve this transaction's staging overlay /// (read-your-own-writes) and route `StageWrite`. Used by the native endpoint, /// whose in-transaction tasks flow through this shared path. +/// +/// It appends no WAL record, so it refuses a write that only the funnel's +/// `AppendHere` route logs. A staged write is not such a write: COMMIT logs it. pub(crate) async fn dispatch_to_data_plane_with_txn( shared: &SharedState, tenant_id: TenantId, @@ -298,6 +305,7 @@ pub(crate) async fn dispatch_to_data_plane_with_txn( trace_id: TraceId, txn_id: Option, ) -> crate::Result { + super::durability_barrier::refuse_unlogged_write(&plan)?; dispatch_to_data_plane_inner( shared, DataPlaneDispatch { diff --git a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs index 2f3af9557..f92ec8bb0 100644 --- a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs +++ b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs @@ -98,6 +98,24 @@ pub(super) fn funnel_minted_redo_engine(plan: &PhysicalPlan) -> Option<&'static } } +/// Refuse a write whose redo record only the funnel's `AppendHere` route +/// mints, on a dispatch route that appends nothing. +/// +/// The read route and the staged-write route supply no LSN and no minted +/// records. A write that reaches one of them applies with no WAL record, so +/// the refusal fires before the write is enqueued, never after it applied. +pub(super) fn refuse_unlogged_write(plan: &PhysicalPlan) -> crate::Result<()> { + match funnel_minted_redo_engine(plan) { + None => Ok(()), + Some(engine) => Err(crate::Error::Internal { + detail: format!( + "a {engine} write reached a dispatch route that appends no WAL record; an \ + autocommit write must dispatch through the durable write route" + ), + }), + } +} + /// Called at the durable-at-ack barrier when there is no LSN to wait on. /// /// `missing_redo_engine` is [`funnel_minted_redo_engine`]'s verdict for the @@ -188,6 +206,24 @@ mod tests { assert_eq!(funnel_minted_redo_engine(&plan), None); } + /// The read route refuses a KV write before it is enqueued, and passes a + /// read and a staged write. + #[test] + fn the_read_route_refuses_a_write_it_cannot_log() { + assert!(refuse_unlogged_write(&kv_put()).is_err()); + let staged = PhysicalPlan::Meta(MetaOp::StageWrite { + plan: Box::new(kv_put()), + }); + assert!(refuse_unlogged_write(&staged).is_ok()); + let read = PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + rls_filters: Vec::new(), + surrogate_ceiling: None, + }); + assert!(refuse_unlogged_write(&read).is_ok()); + } + /// A zero-edge batch appends nothing on purpose, so it must not be held to /// an invariant its non-empty sibling satisfies. #[test] diff --git a/nodedb/src/control/server/dispatch_utils/durable_write.rs b/nodedb/src/control/server/dispatch_utils/durable_write.rs new file mode 100644 index 000000000..6797151fd --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/durable_write.rs @@ -0,0 +1,190 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The durable route for an autocommit write. +//! +//! A planned autocommit write reaches its engine one of two ways: +//! +//! - Cluster mode, replicable write: the write is proposed through Raft. Every +//! replica applies the committed entry through the funnel, which appends the +//! redo record. The proposing node publishes the change event. +//! - Otherwise: the write enters the funnel with `WalDurability::AppendHere`. +//! The funnel appends the redo record under the write-admission guard, inside +//! the write's outcome-floor window. +//! +//! A write dispatched any other way applies with no WAL record. A crash loses +//! it, and no replica sees it. Every caller that holds an autocommit write +//! dispatches it through this module. + +use std::sync::atomic::Ordering; + +use crate::bridge::envelope::{PhysicalPlan, Response, Status}; +use crate::control::server::shared::clone_write::CloneCheckedTask; +use crate::control::server::shared::write_admission::plan_is_write; +use crate::control::state::SharedState; +use crate::control::wal_replication::{ + ReplicableWrite, propose_replicated_entry, to_replicated_entry, +}; +use crate::types::{DatabaseId, Lsn, RequestId, TenantId, TraceId, VShardId}; + +use super::change_events::{extract_write_change_set, publish_change_set_with_lsn}; +use super::dispatch::{ + dispatch_authorized_autocommit_write, dispatch_authorized_to_data_plane, + dispatch_autocommit_write, +}; +use super::types::AutocommitWrite; + +/// Dispatch a trusted autocommit write on the durable route. +/// +/// A write that carries a transaction id applies at the statement inside an +/// open block: a write the transaction cannot buffer. It is never proposed +/// on its own, so it takes the local `AppendHere` route. +pub(crate) async fn dispatch_durable_autocommit_write( + shared: &SharedState, + write: AutocommitWrite, +) -> crate::Result { + if write.txn_id.is_none() + && let Some(response) = propose_if_replicable( + shared, + WriteTarget { + tenant_id: write.tenant_id, + database_id: write.database_id, + vshard_id: write.vshard_id, + }, + &write.plan, + ) + .await? + { + return Ok(response); + } + dispatch_autocommit_write(shared, write).await +} + +/// Dispatch an authorized autocommit write on the durable route. +/// +/// Same routing as [`dispatch_durable_autocommit_write`], for a task that +/// passed the clone-write gate and authorization. +/// +/// In cluster mode a write whose RLS write policy is decided per row cannot +/// be proposed bare: a follower has no writing identity to decide the policy +/// against. It resolves to a concrete row set first, on this node, and the +/// resolved write is proposed (`control::write_resolve`), as the planned +/// pgwire and native writes do. +pub(crate) async fn dispatch_authorized_durable_write( + shared: &SharedState, + checked: CloneCheckedTask, + trace_id: TraceId, +) -> crate::Result { + if checked.txn_id().is_none() + && shared.async_raft_proposer().is_some() + && let Some(resolver) = crate::control::write_resolve::resolver_for_plan(checked.plan()) + { + return crate::control::write_resolve::run_authorized_write_resolve( + shared, + checked.into_authorized(), + resolver, + ) + .await; + } + if checked.txn_id().is_none() + && let Some(response) = propose_if_replicable( + shared, + WriteTarget { + tenant_id: checked.tenant_id(), + database_id: checked.database_id(), + vshard_id: checked.vshard_id(), + }, + checked.plan(), + ) + .await? + { + return Ok(response); + } + dispatch_authorized_autocommit_write(shared, checked, trace_id).await +} + +/// Dispatch one authorized task by its class: a write on the durable route, +/// anything else on the read route. +/// +/// For a transport fallback that dispatches reads and writes through one call +/// site with no gateway installed. +pub(crate) async fn dispatch_authorized_task_by_class( + shared: &SharedState, + checked: CloneCheckedTask, + trace_id: TraceId, +) -> crate::Result { + if plan_is_write(checked.plan()) { + dispatch_authorized_durable_write(shared, checked, trace_id).await + } else { + dispatch_authorized_to_data_plane(shared, checked, trace_id).await + } +} + +/// The coordinates a write is proposed under. +struct WriteTarget { + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, +} + +/// Propose `plan` through Raft when this node runs a proposer and the plan +/// encodes to a replicated entry. `None` means the write takes the local route. +/// +/// A Data-Plane verdict comes back as `Err(Error::DataPlane(code))`, the shape +/// the pgwire replicated path returns. +async fn propose_if_replicable( + shared: &SharedState, + target: WriteTarget, + plan: &PhysicalPlan, +) -> crate::Result> { + let Some(proposer) = shared.async_raft_proposer() else { + return Ok(None); + }; + let WriteTarget { + tenant_id, + database_id, + vshard_id, + } = target; + let replicable = ReplicableWrite::decide_for_replication(plan)?; + let Some(entry) = to_replicated_entry(tenant_id, database_id, vshard_id, &replicable)? else { + return Ok(None); + }; + let (payload, write_version) = propose_replicated_entry(shared, proposer, entry).await?; + // Replicas apply with `ChangeFeedOwner::Unowned`. The proposing node + // handled the write once, so it publishes the change event. + publish_change_set_with_lsn( + shared, + tenant_id, + database_id, + extract_write_change_set(plan, tenant_id), + write_version, + ); + Ok(Some(replicated_write_response( + shared, + payload, + write_version, + ))) +} + +/// The response a committed and applied replicated write answers with. +/// +/// `write_version` is the written collection's `coll_write_lsn` after the +/// write. It is the watermark and the read version, so a session can floor a +/// later read at it. +fn replicated_write_response( + shared: &SharedState, + payload: Vec, + write_version: Lsn, +) -> Response { + Response { + request_id: RequestId::new(shared.request_id_counter.fetch_add(1, Ordering::Relaxed)), + status: Status::Ok, + attempt: 1, + partial: false, + payload: payload.into(), + watermark_lsn: write_version, + error_code: None, + read_set_valid: None, + read_version_lsn: write_version, + write_set: Vec::new(), + } +} diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index e1b77515a..6260f3edd 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -6,6 +6,7 @@ mod change_events; mod collect; mod dispatch; mod durability_barrier; +mod durable_write; mod error_status; mod minted; mod submit_write; @@ -26,6 +27,10 @@ pub(crate) use dispatch::{ dispatch_trusted_internal_write_to_data_plane, }; pub use durability_barrier::writes_acked_without_durability; +pub(crate) use durable_write::{ + dispatch_authorized_durable_write, dispatch_authorized_task_by_class, + dispatch_durable_autocommit_write, +}; pub(crate) use error_status::reject_data_plane_error; pub(crate) use minted::{ Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, diff --git a/nodedb/src/control/server/http/routes/promql/remote.rs b/nodedb/src/control/server/http/routes/promql/remote.rs index f694c788c..e0a70a3d8 100644 --- a/nodedb/src/control/server/http/routes/promql/remote.rs +++ b/nodedb/src/control/server/http/routes/promql/remote.rs @@ -182,7 +182,7 @@ pub async fn remote_write( }; gw.execute(&gw_ctx, checked).await } - None => crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( + None => crate::control::server::dispatch_utils::dispatch_authorized_durable_write( &state.shared, checked, TraceId::generate(), diff --git a/nodedb/src/control/server/http/routes/query/materialized/shape.rs b/nodedb/src/control/server/http/routes/query/materialized/shape.rs index ad9148e54..073777685 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/shape.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/shape.rs @@ -323,7 +323,7 @@ pub(super) async fn run_task_loop( None => { // Single-node boot: gateway not yet initialised — dispatch locally. let response = - crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( + crate::control::server::dispatch_utils::dispatch_authorized_durable_write( &state.shared, checked, trace_id, diff --git a/nodedb/src/control/server/http/routes/query/ndjson.rs b/nodedb/src/control/server/http/routes/query/ndjson.rs index 097f16619..9d99551af 100644 --- a/nodedb/src/control/server/http/routes/query/ndjson.rs +++ b/nodedb/src/control/server/http/routes/query/ndjson.rs @@ -307,8 +307,9 @@ pub async fn query_ndjson( }; gw.execute(&gw_ctx, checked).await } + // A write takes the durable route, a read the read route. None => { - crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( + crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( &state.shared, checked, trace_id, diff --git a/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs b/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs index 12806adc2..4fd83ccea 100644 --- a/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs +++ b/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs @@ -295,8 +295,9 @@ pub async fn execute_sql( gw.execute(&gw_ctx, checked).await } None => { - // Single-node boot: gateway not yet initialised — dispatch locally. - crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( + // Single-node boot: gateway not yet initialised — dispatch + // locally. A write takes the durable route, a read the read route. + crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( shared, checked, trace_id, ) .await diff --git a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs index dd4900a1d..fa521357c 100644 --- a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs @@ -57,6 +57,21 @@ pub(super) async fn dispatch_authorized_single_task( CloneCheckedOutcome::Handled(resp) => return Ok(resp), CloneCheckedOutcome::Proceed(checked) => checked, }; + // A write whose RLS write policy is decided per row cannot be proposed + // bare: a follower has no writing identity to decide it against. It + // resolves to a concrete row set here, while the identity is live, the + // way the planned native and pgwire writes resolve. + if checked.txn_id().is_none() + && ctx.state.async_raft_proposer().is_some() + && let Some(resolver) = crate::control::write_resolve::resolver_for_plan(checked.plan()) + { + return crate::control::write_resolve::run_authorized_write_resolve( + ctx.state, + checked.into_authorized(), + resolver, + ) + .await; + } // A staged write and the other transaction meta-ops run on the core of // the task's own vShard. The gateway would route them to vShard 0. let gateway = ctx @@ -182,8 +197,7 @@ pub(super) async fn dispatch_without_gateway( if crate::control::crdt_admission::changes_crdt_frontier(op) ); let write = || async move { - dispatch_utils::dispatch_authorized_autocommit_write(ctx.state, checked, TraceId::ZERO) - .await + dispatch_utils::dispatch_authorized_durable_write(ctx.state, checked, TraceId::ZERO).await }; if frontier_mutation { ctx.state diff --git a/nodedb/src/control/server/native/dispatch/single_task.rs b/nodedb/src/control/server/native/dispatch/single_task.rs index b417f29d3..d141355ee 100644 --- a/nodedb/src/control/server/native/dispatch/single_task.rs +++ b/nodedb/src/control/server/native/dispatch/single_task.rs @@ -32,9 +32,9 @@ use super::{ /// Routes through the same protocol-neutral in-transaction staging gate /// (`route_in_tx_write`) the SQL-planned dispatch loops (`sql_loop.rs`, /// pgwire's `execute_dml_hooks.rs`) already use. Outside a transaction block -/// this is a no-op passthrough (`InTxnRoute::Read` with the task unchanged), -/// so autocommit direct ops (including `KvBatchPut`) dispatch exactly as -/// before. Inside a transaction block, a stageable write (e.g. `KvBatchPut`) +/// the task comes back unchanged (`InTxnRoute::Read`, or `Autocommit` for a +/// write), and the gateway gives a write its durable route. Inside a +/// transaction block, a stageable write (e.g. `KvBatchPut`) /// is applied to the per-transaction overlay at statement time instead of /// hitting durable storage directly. Otherwise a native direct-op write /// inside `BEGIN...COMMIT` would commit immediately and survive `ROLLBACK`, @@ -60,8 +60,8 @@ pub(super) async fn dispatch_single_task( // Only when metering is enabled — the default is disabled, so this is a // no-op on the hot path for every deployment that hasn't turned it on. - // Covers the plain `Read` dispatch below (autocommit writes/reads, and - // in-transaction reads). `Staged` meters itself inside + // Covers the `Read` / `Autocommit` dispatch below (reads, and writes that + // apply now). `Staged` meters itself inside // `staging_gate::stage_write` — the single choke-point every `Staged` // route (this file, `sql_loop.rs`, the expander's per-op staging, and // pgwire's `execute_dml_hooks.rs`) dispatches through, so it is metered @@ -100,7 +100,9 @@ pub(super) async fn dispatch_single_task( ) .await { - Ok(InTxnRoute::Read(routed_task)) => *routed_task, + // A write here reaches the gateway, which proposes it through Raft or + // appends its redo record in the funnel. + Ok(InTxnRoute::Read(routed_task) | InTxnRoute::Autocommit(routed_task)) => *routed_task, // A buffered write applies at COMMIT: no count and no verb yet. Ok(InTxnRoute::Buffered) => return NativeResponse::ok(seq), Ok(InTxnRoute::Staged(outcome)) => { diff --git a/nodedb/src/control/server/native/dispatch/sql_gateway.rs b/nodedb/src/control/server/native/dispatch/sql_gateway.rs index 2e350dffe..7731841a8 100644 --- a/nodedb/src/control/server/native/dispatch/sql_gateway.rs +++ b/nodedb/src/control/server/native/dispatch/sql_gateway.rs @@ -98,8 +98,9 @@ pub(super) async fn dispatch_task_via_gateway( // renders its SQLSTATE and numeric code from it. gw.execute_response(&gw_ctx, checked).await } + // A write takes the durable route, a read the read route. None => { - crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( + crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( ctx.state, checked, TraceId::generate(), diff --git a/nodedb/src/control/server/native/dispatch/sql_loop.rs b/nodedb/src/control/server/native/dispatch/sql_loop.rs index 45797f8c4..0c8e63be7 100644 --- a/nodedb/src/control/server/native/dispatch/sql_loop.rs +++ b/nodedb/src/control/server/native/dispatch/sql_loop.rs @@ -120,8 +120,8 @@ pub(super) async fn run_dispatch_loop( // Extracted from the same clone above, before `task` is moved into // the routing call below — metering needs the collection/engine // shape after this task's dispatch succeeds. Only covers the direct - // dispatch below (`InTxnRoute::Read`, i.e. autocommit writes/reads - // and in-transaction reads); `Buffered`/`Staged` tasks `continue` + // dispatch below (`InTxnRoute::Read` / `Autocommit`: reads, and + // writes that apply now); `Buffered`/`Staged` tasks `continue` // before reaching the metering call and are not billed here — a // `Buffered` task performs no dispatch yet (replayed at COMMIT), and // a `Staged` task's dispatch happens inside `route_in_tx_write`'s @@ -145,8 +145,8 @@ pub(super) async fn run_dispatch_loop( // buffered for COMMIT-time replay; stageable writes are applied to // the per-transaction overlay immediately for a real affected count // and statement-time constraint errors. Outside a transaction block, - // `route_in_tx_write` always returns `Read(task)` unchanged, so the - // autocommit path is untouched. + // `route_in_tx_write` returns the task unchanged, as `Read` or as + // `Autocommit` for a write. // In-transaction `MERGE` and `UPDATE ... FROM` are resolved + staged at // STATEMENT time by the expander (read-your-own-writes for later // statements in the same txn); every other task falls through to the @@ -205,7 +205,9 @@ pub(super) async fn run_dispatch_loop( PlanKind::ReturningRows ); let task = match routed { - Ok(InTxnRoute::Read(routed_task)) => *routed_task, + // A write here reaches the gateway, which proposes it through + // Raft or appends its redo record in the funnel. + Ok(InTxnRoute::Read(routed_task) | InTxnRoute::Autocommit(routed_task)) => *routed_task, Ok(InTxnRoute::Buffered) => { if returns_rows { return resp(error_to_native( diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs index 0b57d5855..36f7ae56a 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs @@ -136,7 +136,11 @@ impl NodeDbPgHandler { } match routed { - Ok(InTxnRoute::Read(routed_task)) => Ok(TxnRouteOutcome::Proceed(routed_task)), + // The pgwire dispatch proposes a write through Raft or appends its + // redo record in the funnel. + Ok(InTxnRoute::Read(routed_task) | InTxnRoute::Autocommit(routed_task)) => { + Ok(TxnRouteOutcome::Proceed(routed_task)) + } Ok(InTxnRoute::Buffered) => Ok(TxnRouteOutcome::Handled(HandledWrite::Opaque)), Ok(InTxnRoute::Staged(outcome)) => Ok(TxnRouteOutcome::Handled(HandledWrite::Dml( staged_dml_outcome(outcome.kind, outcome.affected), diff --git a/nodedb/src/control/server/resp/gateway_dispatch.rs b/nodedb/src/control/server/resp/gateway_dispatch.rs index b27844de9..3f3b50f97 100644 --- a/nodedb/src/control/server/resp/gateway_dispatch.rs +++ b/nodedb/src/control/server/resp/gateway_dispatch.rs @@ -141,7 +141,7 @@ pub(super) async fn dispatch_kv_write( detail: GatewayErrorMap::to_resp(&e), }) } - None => dispatch_utils::dispatch_authorized_autocommit_write(state, checked, TraceId::ZERO) + None => dispatch_utils::dispatch_authorized_durable_write(state, checked, TraceId::ZERO) .await .map_err(map_busy_error), }; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index 0e87463ab..68c717ee9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -27,7 +27,8 @@ use crate::types::TraceId; use super::types::ddl_err; -/// Dispatch a plan to WAL + Data Plane, returning an error response on failure. +/// Dispatch a write plan on the durable route, returning an error response on +/// failure. `None` means the write applied. pub(in crate::control::server::shared::ddl::neutral::collection) async fn dispatch_plan( state: &SharedState, identity: &AuthenticatedIdentity, @@ -69,18 +70,28 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn dispat } }; - if let Err(error) = - crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( - state, - checked, - TraceId::ZERO, - ) - .await + // The durable route: Raft in cluster mode, else the funnel's `AppendHere`. + match crate::control::server::dispatch_utils::dispatch_authorized_durable_write( + state, + checked, + TraceId::ZERO, + ) + .await { - let (_, sqlstate, message) = error_to_sqlstate(&error); - return Some(Err(ddl_err(sqlstate, message))); + Err(error) => { + let (_, sqlstate, message) = error_to_sqlstate(&error); + Some(Err(ddl_err(sqlstate, message))) + } + // A refusal arrives as an error status inside an `Ok` response. + Ok(response) if response.status == crate::bridge::envelope::Status::Error => { + let (_, sqlstate, message) = match response.error_code.as_deref() { + Some(code) => error_code_to_sqlstate(code), + None => ("ERROR", "XX000", "unknown data plane error".to_owned()), + }; + Some(Err(ddl_err(sqlstate, message))) + } + Ok(_) => None, } - None } /// Authorize a write target before triggers, sequences, or catalog reads run. @@ -343,7 +354,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a } let task = match routed { - Ok(InTxnRoute::Read(task)) => *task, + Ok(InTxnRoute::Read(task) | InTxnRoute::Autocommit(task)) => *task, Ok(InTxnRoute::Buffered) | Ok(InTxnRoute::Staged(_)) => { drop(initial_authorized); // A buffered/staged write produces its rows at COMMIT, not @@ -389,8 +400,10 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a ddl_err(sqlstate, message) })? { crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => resp, + // A write takes the durable route: Raft in cluster mode, else the + // funnel's `AppendHere`. A read takes the read route. crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(checked) => { - crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write( + crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( state, checked, TraceId::ZERO, diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs index bf29d2128..1b5bb30d3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs @@ -151,11 +151,13 @@ pub(super) async fn stage_edge_write_in_txn( match routed { Ok(InTxnRoute::Staged(outcome)) => Ok(outcome.affected as u64), // Edge writes are stageable (`is_stageable_write`), so inside a - // transaction block the gate always returns `Staged`. `Read` (not in a - // block) and `Buffered` (non-stageable write) cannot occur for a - // caller that already checked `InBlock`; there is no affected count to - // report for either, so treat them as a no-op tag rather than panicking. - Ok(InTxnRoute::Read(_)) | Ok(InTxnRoute::Buffered) => Ok(0), + // transaction block the gate returns `Staged`. Any other route means + // the caller's `InBlock` check and the gate disagree. A report of zero + // rows drops the write silently, so the statement fails instead. + Ok(InTxnRoute::Read(_) | InTxnRoute::Autocommit(_) | InTxnRoute::Buffered) => Err(ddl_err( + "XX000", + "a graph edge write reached the transaction staging gate and was not staged", + )), Err(StagingGateError::Dispatch(e)) => { let (_, sqlstate, message) = error_to_sqlstate(&e); Err(ddl_err(sqlstate, message)) diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs index 790b6b5a9..063177913 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs @@ -5,9 +5,12 @@ //! directory (`handlers` here, and the sibling `kv_sorted_index`, //! `weighted_pick`, `rate_gate`, `transfer` modules). +use std::sync::Arc; + use serde_json::{Map, Value as JsonValue}; use crate::bridge::envelope::Status; +use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::server::response_shape::types::ShapedRows; use crate::control::server::shared::session::DmlTxnCtx; @@ -23,27 +26,25 @@ use super::super::read_gate::CollectionReadGate; /// and return the JSON response as a single text-column row keyed by the /// lower-cased function name. /// -/// Outside a transaction block (or for the read half of the gate, which -/// never applies here since every `KvOp` this module builds is a write), -/// `route_in_tx_write` dispatches immediately -- byte-identical to the -/// pre-staging behavior. Inside a transaction, `KvOp::Incr` / `IncrFloat` / -/// `Cas` / `GetSet` are staged into the per-transaction overlay -/// (`is_stageable_write`) and this function reads the computed value back -/// from `StagedWriteOutcome::payload` (forwarded verbatim by the staging -/// gate for `StagedTagKind::RawPayload`), so a `SELECT KV_INCR(...)` inside -/// `BEGIN..COMMIT` returns the same value the staged overlay now holds, and -/// a following `SELECT KV_INCR(...)` on the same key in the same -/// transaction chains off it. +/// Outside a transaction block the gate answers `Autocommit`, and the op takes +/// the durable route every planned autocommit write takes: proposed through +/// Raft in cluster mode, otherwise the write funnel with `AppendHere`. Either +/// way a WAL record reproduces the value the op computed, and a replica +/// applies it. +/// +/// Inside a transaction, `KvOp::Incr` / `IncrFloat` / `Cas` / `GetSet` / +/// `Transfer` / `TransferItem` are staged into the per-transaction overlay +/// (`is_stageable_write`). This function reads the computed value back from +/// `StagedWriteOutcome::payload`, so a `SELECT KV_INCR(...)` inside +/// `BEGIN..COMMIT` returns the value the staged overlay now holds, and a +/// following `SELECT KV_INCR(...)` on the same key chains off it. /// /// `collections` names every collection the op touches, in the caller's own -/// words rather than read back out of the plan: these `KvOp`s carry no -/// collection the plan-classification helpers report, and `TRANSFER_ITEM` -/// touches two. Each is authorized here before the op is routed anywhere. +/// words rather than read back out of the plan: `TRANSFER_ITEM` touches two. +/// Each is authorized here before the op is routed anywhere. /// -/// Reused by the sibling `transfer.rs` module for the identical -/// in-transaction routing for `TRANSFER` / `TRANSFER_ITEM` instead of the -/// direct `dispatch_to_data_plane` call it used before those two `KvOp`s -/// became stageable. +/// Reused by the sibling `transfer.rs` module for `TRANSFER` / +/// `TRANSFER_ITEM`. pub(crate) async fn dispatch_and_respond( state: &SharedState, identity: &AuthenticatedIdentity, @@ -61,10 +62,10 @@ pub(crate) async fn dispatch_and_respond( let database_id = DatabaseId::DEFAULT; // Every caller here names its collections in the SQL text and reaches the - // Data Plane through a hand-built `KvOp`, which carries no identity and is - // never authorized downstream. The op reports the value it replaced or - // computed, so it is a read as much as a write and needs both grants — and - // a cross-collection move needs them on each side, hence the slice. + // Data Plane through a hand-built `KvOp`. The op reports the value it + // replaced or computed, so it is a read as much as a write and needs both + // grants — and a cross-collection move needs them on each side, hence the + // slice. let gate = CollectionReadGate::for_request(state, identity, database_id); for collection in collections { gate.authorize(collection)?; @@ -72,15 +73,13 @@ pub(crate) async fn dispatch_and_respond( } // Row-level security is resolved by the same injection pass the - // planner-driven path runs, against the very same op. That is the point of - // routing it through here rather than restating a verdict locally: these - // functions build the identical `KvOp`s a planned statement builds, so a - // local refusal here while the planner enforced — or the reverse — would - // give the same operation two different answers depending on which syntax - // reached it. The pass compiles the write policy into the op's own gate + // planner-driven path runs, against the very same op. These functions + // build the identical `KvOp`s a planned statement builds. A local refusal + // here while the planner enforced, or the reverse, gives the same + // operation two answers that depend on the syntax that reached it. The pass compiles the write policy into the op's own gate // slot for the Data Plane to decide the image against, injects the read - // filter where the reply is a row body, and still refuses outright where - // the reply is a value computed from a row the policy hides. + // filter where the reply is a row body, and refuses outright where the + // reply is a value computed from a row the policy hides. gate.inject_rls(&mut plan)?; let task = PhysicalTask { @@ -112,41 +111,27 @@ pub(crate) async fn dispatch_and_respond( .await; let payload = match routed { - Ok(InTxnRoute::Read(task)) => { - let task = *task; - match crate::control::server::dispatch_utils::dispatch_to_data_plane_with_txn( - state, - task.tenant_id, - task.database_id, - task.vshard_id, - task.plan, - TraceId::ZERO, - task.txn_id, - ) - .await - { - // A refused write comes back as `Ok(Response)` carrying - // `Status::Error` — `submit_write` reports the dispatch itself - // as having succeeded and puts the verdict inside the response. - // Its payload is empty, so forwarding it unchecked would answer - // `SELECT KV_INCR(...)` with one blank column and let the caller - // read a refusal as a completed write. Every terminal outcome - // these functions can produce arrives this way — a policy - // refusal, a type mismatch, an overflow, an insufficient - // balance, a missing key — so the status is what decides, - // never the payload's emptiness. - Ok(resp) if resp.status == Status::Error => { - return Err(data_plane_error(resp.error_code.map(|code| *code))); - } - Ok(resp) => resp.payload.as_ref().to_vec(), - Err(e) => return Err(ddl_err("XX000", e.to_string())), - } - } - // Every `KvOp` this module builds is stageable once in a - // transaction (`is_stageable_write`), so `Buffered` never occurs; - // handled defensively with an empty payload rather than a panic. - Ok(InTxnRoute::Buffered) => Vec::new(), + Ok(InTxnRoute::Autocommit(task)) => dispatch_autocommit(state, identity, *task).await?, Ok(InTxnRoute::Staged(outcome)) => outcome.payload, + // Every `KvOp` this module builds is a write, and a stageable one once + // in a transaction (`is_stageable_write`). A read route has no + // durable apply, and a buffered route has no value to answer with, so + // either one is a classification break, never an answer. + Ok(InTxnRoute::Read(_)) => { + return Err(ddl_err( + "XX000", + format!("{func_name}: the staging gate classified this write as a read"), + )); + } + Ok(InTxnRoute::Buffered) => { + return Err(ddl_err( + "XX000", + format!( + "{func_name}: the staging gate buffered this write, so it has no value to \ + return at the statement" + ), + )); + } Err(StagingGateError::Dispatch(e)) => return Err(ddl_err("XX000", e.to_string())), Err(StagingGateError::Rejected { code }) => return Err(data_plane_error(code)), }; @@ -156,6 +141,70 @@ pub(crate) async fn dispatch_and_respond( Ok(vec![single_text_col(&col_name, payload_text)]) } +/// Apply one autocommit op on the durable route and return its payload. +/// +/// The task passes the same clone-write gate and authorization every +/// transport runs before a write dispatches. +async fn dispatch_autocommit( + state: &SharedState, + identity: &AuthenticatedIdentity, + task: PhysicalTask, +) -> Result, DdlError> { + use crate::control::server::shared::clone_write::{ + CloneCheckedOutcome, InterceptAndAuthorizeParams, intercept_and_authorize, + }; + + let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); + let outcome = intercept_and_authorize(InterceptAndAuthorizeParams { + state, + task, + identity, + tenant_id: identity.tenant_id, + permissions: &state.permissions, + roles: &state.roles, + emitter: &emitter, + }) + .await + .map_err(|e| error_to_ddl(&e))?; + let response = match outcome { + CloneCheckedOutcome::Handled(resp) => resp, + CloneCheckedOutcome::Proceed(checked) => { + crate::control::server::dispatch_utils::dispatch_authorized_durable_write( + state, + checked, + TraceId::ZERO, + ) + .await + .map_err(|e| error_to_ddl(&e))? + } + }; + // A refused write comes back as `Ok(Response)` carrying `Status::Error`: + // the funnel reports the dispatch as successful and puts the verdict in the + // response. Its payload is empty, so an unchecked forward answers + // `SELECT KV_INCR(...)` with one blank column. Every terminal outcome these + // functions can produce arrives this way on the local route — a policy + // refusal, a type mismatch, an overflow, an insufficient balance, a + // missing key — so the status decides, never the payload's emptiness. + if response.status == Status::Error { + return Err(data_plane_error(response.error_code.map(|code| *code))); + } + Ok(response.payload.as_ref().to_vec()) +} + +/// Map a dispatch error to the client-facing error. A Data-Plane verdict +/// arrives here as `Error::DataPlane` on the replicated route, and it renders +/// the same way the local route's error status does. +fn error_to_ddl(error: &crate::Error) -> DdlError { + match error { + crate::Error::DataPlane(code) => data_plane_error(Some(code.clone())), + other => { + let (_, sqlstate, message) = + crate::control::server::pgwire::types::error_to_sqlstate(other); + ddl_err(sqlstate, message) + } + } +} + /// Translate a terminal Data-Plane verdict into the client-facing error. /// /// Shared by both routes a refusal can arrive on — the staging gate's diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs index a2ed87aa8..be8cb49dc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs @@ -106,20 +106,26 @@ pub(super) async fn dispatch_read( } } -/// Dispatch a sorted-index registration or teardown. +/// Dispatch a sorted-index registration or teardown on the durable route. /// -/// These go through the autocommit write funnel rather than the read path so -/// the funnel appends their WAL record (`kv_register_sorted_index` / -/// `kv_drop_sorted_index`) under the write-admission guard. The manager holds -/// the tree only in memory, so that record plus the KV checkpoint is all that -/// carries a registration across a restart: dispatched as a read, the catalog -/// would keep listing an index whose tree no longer exists anywhere. +/// In cluster mode the op is proposed through Raft, so every replica of the +/// collection's vShard builds or drops the tree from the committed entry. +/// Otherwise the write funnel appends its WAL record +/// (`kv_register_sorted_index` / `kv_drop_sorted_index`) under the +/// write-admission guard. The manager holds the tree only in memory, so that +/// record plus the KV checkpoint is all that carries a registration across a +/// restart. Dispatched as a read, the catalog keeps listing an index whose +/// tree no longer exists anywhere. +/// +/// A Data-Plane verdict from the replicated route arrives as +/// `Error::DataPlane`. It comes back here as the error-status response the +/// local route gives, so [`refusal`] reads both the one way. async fn dispatch_durable( state: &SharedState, target: &SortedIndexTarget<'_>, plan: PhysicalPlan, ) -> Result { - crate::control::server::dispatch_utils::dispatch_autocommit_write( + let dispatched = crate::control::server::dispatch_utils::dispatch_durable_autocommit_write( state, crate::control::server::dispatch_utils::AutocommitWrite { tenant_id: target.tenant_id, @@ -131,8 +137,28 @@ async fn dispatch_durable( txn_id: None, }, ) - .await - .map_err(|e| ddl_err("XX000", e.to_string())) + .await; + match dispatched { + Ok(resp) => Ok(resp), + Err(crate::Error::DataPlane(code)) => Ok(verdict_response(code)), + Err(e) => Err(ddl_err("XX000", e.to_string())), + } +} + +/// The error-status response the local route gives for a Data-Plane verdict. +fn verdict_response(code: ErrorCode) -> Response { + Response { + request_id: crate::types::RequestId::new(0), + status: Status::Error, + attempt: 0, + partial: false, + payload: crate::bridge::envelope::Payload::empty(), + watermark_lsn: crate::types::Lsn::ZERO, + error_code: Some(Box::new(code)), + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + } } /// Decode a row-shaped sorted-index reply. diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index 6b8ec1e05..eaf8962fd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -116,16 +116,7 @@ pub async fn rate_check( shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); - match crate::control::server::dispatch_utils::dispatch_to_data_plane( - state, - tenant_id, - crate::types::DatabaseId::DEFAULT, - vshard, - plan, - TraceId::ZERO, - ) - .await - { + match dispatch_counter_write(state, tenant_id, vshard, plan).await { Ok(resp) if resp.status == Status::Ok => { let payload_text = crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); @@ -265,17 +256,8 @@ pub async fn rate_reset( rls_filters: Vec::new(), }); - match crate::control::server::dispatch_utils::dispatch_to_data_plane( - state, - tenant_id, - crate::types::DatabaseId::DEFAULT, - vshard, - plan, - TraceId::ZERO, - ) - .await - { - Ok(_) => { + match dispatch_counter_write(state, tenant_id, vshard, plan).await { + Ok(resp) if resp.status == Status::Ok => { let result = serde_json::json!({ "gate": gate_name, "key": key, @@ -283,12 +265,46 @@ pub async fn rate_reset( }); Ok(vec![single_text_col("rate_reset", result.to_string())]) } + // A refusal arrives as an error status inside an `Ok` response. The + // counter is still there, so the reset did not happen. + Ok(resp) => Err(ddl_err( + "XX000", + format!( + "RATE_RESET: the counter delete was refused: {:?}", + resp.error_code + ), + )), Err(e) => Err(ddl_err("XX000", e.to_string())), } } // ── Helpers ──────────────────────────────────────────────────────────── +/// Apply a write to the rate-gate counter collection on the durable route: +/// Raft in cluster mode, else the write funnel's `AppendHere`. A counter +/// written any other way has no WAL record, so a crash resets the gate and a +/// replica never counts the call. +async fn dispatch_counter_write( + state: &SharedState, + tenant_id: crate::types::TenantId, + vshard: VShardId, + plan: PhysicalPlan, +) -> crate::Result { + crate::control::server::dispatch_utils::dispatch_durable_autocommit_write( + state, + crate::control::server::dispatch_utils::AutocommitWrite { + tenant_id, + database_id: DatabaseId::DEFAULT, + vshard_id: vshard, + plan, + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + }, + ) + .await +} + /// Read TTL remaining for a KV key (in milliseconds). async fn read_ttl_ms( state: &SharedState, diff --git a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs index dd713cc8b..5afe571cf 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs @@ -108,9 +108,8 @@ pub async fn weighted_pick( let mut weights: Vec = Vec::with_capacity(entries.len()); let mut keys: Vec = Vec::with_capacity(entries.len()); - for (key_bytes, value_bytes) in &entries { - let key_str = String::from_utf8_lossy(key_bytes).to_string(); - let weight = extract_weight(value_bytes, &weight_col).unwrap_or(0.0); + for (key_str, row) in entries { + let weight = row_weight(&row, &weight_col).unwrap_or(0.0); if weight < 0.0 { return Err(ddl_err( "42601", @@ -160,47 +159,59 @@ pub async fn weighted_pick( .map(|d| d.as_nanos()) .unwrap_or(0) ); - let audit_value = nodedb_types::json_to_msgpack(&audit_entry).unwrap_or_default(); + let audit_value = nodedb_types::json_to_msgpack(&audit_entry) + .map_err(|e| ddl_err("XX000", format!("WEIGHTED_PICK: audit entry encode: {e}")))?; let audit_key_bytes = audit_key.into_bytes(); - match state.surrogate_assigner.assign( - crate::types::DatabaseId::DEFAULT, - tenant_id, - "_system_random_audit", - &audit_key_bytes, - ) { - Ok(audit_surrogate) => { - let audit_plan = PhysicalPlan::Kv(KvOp::Put { - collection: nodedb_types::QualifiedCollection::new( - DatabaseId::DEFAULT, - "_system_random_audit", - ), - key: audit_key_bytes, - value: audit_value, - ttl_ms: 0, - surrogate: audit_surrogate, - returning: None, - rls_filters: Vec::new(), - }); - // Audit write failure doesn't block the pick result, but log the error. - if let Err(e) = crate::control::server::dispatch_utils::dispatch_to_data_plane( - state, - tenant_id, + let audit_surrogate = state + .surrogate_assigner + .assign( + crate::types::DatabaseId::DEFAULT, + tenant_id, + "_system_random_audit", + &audit_key_bytes, + ) + .map_err(|e| ddl_err("XX000", format!("WEIGHTED_PICK: audit surrogate bind: {e}")))?; + let audit_plan = PhysicalPlan::Kv(KvOp::Put { + collection: nodedb_types::QualifiedCollection::new( + DatabaseId::DEFAULT, + "_system_random_audit", + ), + key: audit_key_bytes, + value: audit_value, + ttl_ms: 0, + surrogate: audit_surrogate, + returning: None, + rls_filters: Vec::new(), + }); + // The caller asked for an audited pick, so the pick is answered only + // once its audit record is durable: Raft in cluster mode, else the + // write funnel's `AppendHere`. A pick returned without its record is + // an unaudited pick reported as an audited one. + let resp = crate::control::server::dispatch_utils::dispatch_durable_autocommit_write( + state, + crate::control::server::dispatch_utils::AutocommitWrite { + tenant_id, + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::from_collection_in_database( DatabaseId::DEFAULT, - VShardId::from_collection_in_database( - DatabaseId::DEFAULT, - "_system_random_audit", - ), - audit_plan, - TraceId::ZERO, - ) - .await - { - tracing::warn!(error = %e, "WEIGHTED_PICK: audit write failed"); - } - } - Err(e) => { - tracing::warn!(error = %e, "WEIGHTED_PICK: audit surrogate bind failed"); - } + "_system_random_audit", + ), + plan: audit_plan, + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + }, + ) + .await + .map_err(|e| ddl_err("XX000", format!("WEIGHTED_PICK: audit write: {e}")))?; + if resp.status != crate::bridge::envelope::Status::Ok { + return Err(ddl_err( + "XX000", + format!( + "WEIGHTED_PICK: the audit write was refused: {:?}", + resp.error_code + ), + )); } } @@ -228,14 +239,15 @@ pub async fn weighted_pick( // ── Helpers ──────────────────────────────────────────────────────────── -/// Scan all entries from a KV collection. +/// Scan every row of a KV collection as `(key, row)`. A KV scan row is a +/// document map carrying the row's `key` beside its value fields. async fn scan_all_entries( state: &SharedState, gate: &CollectionReadGate<'_>, tenant_id: crate::types::TenantId, vshard: VShardId, collection: &str, -) -> Result, Vec)>, DdlError> { +) -> Result, DdlError> { let mut plan = PhysicalPlan::Kv(KvOp::Scan { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, collection), cursor: Vec::new(), @@ -260,39 +272,63 @@ async fn scan_all_entries( .await .map_err(|e| ddl_err("XX000", e.to_string()))?; + // A refused scan is not an empty collection: a pick from it draws from + // rows the scan never returned. if resp.status != Status::Ok { - return Ok(Vec::new()); + return Err(ddl_err( + "XX000", + format!( + "WEIGHTED_PICK: the scan of '{collection}' was refused: {:?}", + resp.error_code + ), + )); } // KV scan returns a flat msgpack array of entry maps. let payload_text = crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - let json: serde_json::Value = sonic_rs::from_str(&payload_text).unwrap_or_default(); + let json: serde_json::Value = sonic_rs::from_str(&payload_text).map_err(|e| { + ddl_err( + "XX000", + format!("WEIGHTED_PICK: the scan of '{collection}' returned undecodable rows: {e}"), + ) + })?; let entries = match json { serde_json::Value::Array(arr) => arr, - _ => return Ok(Vec::new()), + other => { + return Err(ddl_err( + "XX000", + format!("WEIGHTED_PICK: the scan of '{collection}' returned {other}, not rows"), + )); + } }; let mut all_entries = Vec::with_capacity(entries.len()); - for entry in &entries { - let key_b64 = entry.get("key").and_then(|k| k.as_str()).unwrap_or(""); - let val_b64 = entry.get("value").and_then(|v| v.as_str()).unwrap_or(""); - - let key_bytes = base64::Engine::decode(&base64::engine::general_purpose::STANDARD, key_b64) - .unwrap_or_default(); - let val_bytes = base64::Engine::decode(&base64::engine::general_purpose::STANDARD, val_b64) - .unwrap_or_default(); - - all_entries.push((key_bytes, val_bytes)); + for row in entries { + let key = scan_row_key(&row, collection)?; + all_entries.push((key, row)); } Ok(all_entries) } -/// Extract a numeric weight from a MessagePack-encoded value. -fn extract_weight(value_bytes: &[u8], weight_col: &str) -> Option { - let doc: serde_json::Value = nodedb_types::json_from_msgpack(value_bytes).ok()?; - let v = doc.get(weight_col)?; +/// The key of one KV scan row. A row without a text `key` fails the pick: a +/// skipped row changes the row set the pick draws from. +fn scan_row_key(row: &serde_json::Value, collection: &str) -> Result { + row.get("key") + .and_then(|v| v.as_str()) + .map(str::to_owned) + .ok_or_else(|| { + ddl_err( + "XX000", + format!("WEIGHTED_PICK: a scan row of '{collection}' has no text 'key'"), + ) + }) +} + +/// The numeric weight column of one KV scan row. +fn row_weight(row: &serde_json::Value, weight_col: &str) -> Option { + let v = row.get(weight_col)?; v.as_f64().or_else(|| v.as_i64().map(|i| i as f64)) } diff --git a/nodedb/src/control/server/shared/session/staging_gate.rs b/nodedb/src/control/server/shared/session/staging_gate.rs index 1065d146a..40ecbd555 100644 --- a/nodedb/src/control/server/shared/session/staging_gate.rs +++ b/nodedb/src/control/server/shared/session/staging_gate.rs @@ -2,13 +2,18 @@ //! Protocol-neutral in-transaction write-routing gate. //! -//! Decides, for a single physical task submitted while a connection is -//! inside an explicit transaction block, whether the task is a plain read -//! (falls through to normal dispatch), a write that gets buffered for -//! COMMIT-time replay ("OK" now, durable apply later), or a stageable write -//! that must be applied to the per-transaction overlay immediately (real -//! command tag + statement-time constraint errors now, still buffered for -//! COMMIT's durable replay). +//! Decides, for a single physical task, whether it is: +//! +//! - a read, handed back for the caller's read dispatch; +//! - a write that applies now on the durable autocommit route: outside a +//! transaction block, or a write a transaction cannot buffer; +//! - a write buffered for COMMIT-time replay ("OK" now, durable apply later); +//! - a stageable write applied to the per-transaction overlay now (real +//! command tag and statement-time constraint errors), still buffered for +//! COMMIT's durable replay. +//! +//! A write never comes back as a read, so no caller can send one down the +//! read route, which appends no WAL record. //! //! This is the shared seam every protocol's dispatch loop routes through //! (pgwire SQL today; native and the DSL/UPSERT path in later units), so the @@ -26,7 +31,7 @@ use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; use crate::control::server::shared::sql::staging_predicates::{ is_stageable_write, require_affected_count, stageable_write_shape, }; -use crate::control::server::shared::write_admission::plan_requires_txn_buffering; +use crate::control::server::shared::write_admission::{plan_is_write, plan_requires_txn_buffering}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TxnId, VShardId}; use nodedb_physical::physical_plan::{ClusterArrayOp, CrdtOp, MetaOp}; @@ -42,10 +47,21 @@ pub use crate::control::server::shared::sql::staging_predicates::StagedTagKind; /// Outcome of routing a single task through the in-transaction staging gate. pub enum InTxnRoute { - /// Not a write (or not in a transaction block at all): the task is - /// handed back, possibly with `txn_id` stamped for read-your-own-writes, - /// for the caller's normal dispatch path. + /// Not a write. The task is handed back for the caller's read dispatch. + /// Inside a transaction block it carries the transaction id, so the Data + /// Plane reads this transaction's overlay (read-your-own-writes). Read(Box), + /// A write that applies now, on the caller's durable autocommit route: + /// the session is outside a transaction block, or the write is one a + /// transaction cannot buffer (index DDL, a Calvin-routed bulk write). + /// + /// The caller must dispatch it through the write funnel with + /// `AppendHere`, or propose it through Raft in cluster mode + /// (`dispatch_utils::dispatch_authorized_durable_write`). The read route + /// appends no WAL record, and it refuses such a write. + /// + /// Inside a block the task carries the transaction id. + Autocommit(Box), /// A non-stageable write: buffered for COMMIT-time replay. The caller /// pushes an immediate "OK" tag. Buffered, @@ -81,12 +97,11 @@ pub struct DmlTxnCtx<'a> { /// An owned, session-less scope for callers with no BEGIN/COMMIT transaction /// concept over their transport (stateless HTTP, autocommit test helpers). /// -/// It owns a fresh [`SessionStore`] and a private legacy session identity; -/// because a fresh store reports [`TransactionState::Idle`] for that identity, -/// [`route_in_tx_write`] always takes the `Read` (immediate autocommit -/// dispatch) branch through a [`DmlTxnCtx`] borrowed from here — byte-identical -/// to the pre-gate behavior. Keep the scope alive for the duration of the -/// dispatch call that borrows its [`ctx`](Self::ctx). +/// It owns a fresh [`SessionStore`] and a private legacy session identity. +/// A fresh store reports [`TransactionState::Idle`] for that identity, so +/// [`route_in_tx_write`] answers `Read` for a read and `Autocommit` for a +/// write through a [`DmlTxnCtx`] borrowed from here. Keep the scope alive for +/// the duration of the dispatch call that borrows its [`ctx`](Self::ctx). pub struct DetachedTxnScope { sessions: SessionStore, session_id: SessionId, @@ -156,7 +171,11 @@ where Fut: Future>, { if sessions.transaction_state(session_id) != TransactionState::InBlock { - return Ok(InTxnRoute::Read(Box::new(task))); + return Ok(if plan_is_write(&task.plan) { + InTxnRoute::Autocommit(Box::new(task)) + } else { + InTxnRoute::Read(Box::new(task)) + }); } if matches!( @@ -168,14 +187,16 @@ where )); } - let is_write = plan_requires_txn_buffering(&task.plan); - - if !is_write { - // Not a write: an in-transaction read. Stamp the active transaction + if !plan_requires_txn_buffering(&task.plan) { + // Not buffered: it runs at the statement. Stamp the active transaction // id onto the task so the Data Plane can check this transaction's // staging overlay for read-your-own-writes on point lookups. task.txn_id = sessions.tx_id(session_id); - return Ok(InTxnRoute::Read(Box::new(task))); + return Ok(if plan_is_write(&task.plan) { + InTxnRoute::Autocommit(Box::new(task)) + } else { + InTxnRoute::Read(Box::new(task)) + }); } // A distributed array write is a routing wrapper with no Data-Plane diff --git a/nodedb/src/control/system_txn/run.rs b/nodedb/src/control/system_txn/run.rs index 188cba6a1..534bcd9b9 100644 --- a/nodedb/src/control/system_txn/run.rs +++ b/nodedb/src/control/system_txn/run.rs @@ -155,6 +155,13 @@ pub async fn run_statements_atomically( let read = match routed { Ok(InTxnRoute::Read(task)) => dispatch_read(state, *task).await.err(), + // Refused before BEGIN by `runs_in_a_system_transaction`: a write + // the transaction cannot buffer applies at once and survives a + // rollback. + Ok(InTxnRoute::Autocommit(_)) => Some(crate::Error::Internal { + detail: "a write a system transaction cannot buffer reached its staging gate" + .into(), + }), Ok(InTxnRoute::Buffered | InTxnRoute::Staged(_)) => None, Err(error) => Some(staging_error(error)), }; diff --git a/nodedb/tests/crash_harness/mod.rs b/nodedb/tests/crash_harness/mod.rs index c356cb731..e0344b820 100644 --- a/nodedb/tests/crash_harness/mod.rs +++ b/nodedb/tests/crash_harness/mod.rs @@ -200,6 +200,18 @@ impl CrashHarness { self } + /// Boot without the single-node Calvin stack. The node runs no Raft + /// proposer, so every autocommit write takes the local funnel route, and + /// two writes can apply out of LSN order. Writes a config file into the + /// data directory and points `NODEDB_CONFIG` at it. Call before `spawn`. + pub fn standalone(self) -> CrashHarness { + let config = self.data_dir_path.join("standalone.toml"); + std::fs::write(&config, "[server]\nsingle_node_calvin = false\n") + .expect("write the standalone config file"); + let path = config.to_string_lossy().into_owned(); + self.with_env("NODEDB_CONFIG", &path) + } + /// Set (or replace) a server env override in place, between spawns. /// Unlike [`CrashHarness::with_env`], this lets a crash-during-recovery /// test arm `NODEDB_FAILPOINTS` for exactly one boot — left armed, every diff --git a/nodedb/tests/crash_kv_atomic_autocommit.rs b/nodedb/tests/crash_kv_atomic_autocommit.rs new file mode 100644 index 000000000..9820aee03 --- /dev/null +++ b/nodedb/tests/crash_kv_atomic_autocommit.rs @@ -0,0 +1,111 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! An autocommit `KV_INCR` survives `kill -9`. +//! +//! `KV_INCR` builds its `KvOp` by hand instead of planning a statement. +//! Outside a transaction block it must take the durable route a planned write +//! takes. The route depends on the node: +//! +//! - The default single-node Calvin stack proposes it through Raft, and the +//! applying funnel appends its WAL record. +//! - A standalone node has no proposer, and the funnel appends the record on +//! the local route. +//! +//! The checkpoint interval is pushed beyond the test's runtime, so the values +//! read after the crash come from WAL replay alone. + +mod crash_harness; + +use std::time::{Duration, Instant}; + +use crash_harness::CrashHarness; + +/// A checkpoint between the writes and the kill holds the values independent +/// of the WAL, and a lost record then passes as a survived one. +const NO_CHECKPOINT_SECS: &str = "3600"; + +/// A second guard against the harness running long enough to reach a +/// checkpoint cycle anyway. +const MAX_TEST_WALL_CLOCK: Duration = Duration::from_secs(120); + +#[tokio::test(flavor = "multi_thread")] +async fn an_autocommit_kv_incr_survives_kill_9_on_the_raft_route() { + survives_kill_9(CrashHarness::new()).await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn an_autocommit_kv_incr_survives_kill_9_on_the_local_route() { + survives_kill_9(CrashHarness::new().standalone()).await; +} + +async fn survives_kill_9(h: CrashHarness) { + let mut h = h.with_env("NODEDB_CHECKPOINT_INTERVAL_SECS", NO_CHECKPOINT_SECS); + let spawned_at = Instant::now(); + h.spawn(); + h.wait_ready(); + + h.exec("CREATE COLLECTION crash_ctr (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')") + .await; + h.exec("CREATE COLLECTION crash_ctr_raw (key TEXT PRIMARY KEY, value TEXT) WITH (engine='kv')") + .await; + h.exec("CREATE COLLECTION crash_score (key TEXT PRIMARY KEY, score FLOAT) WITH (engine='kv')") + .await; + + for delta in [5, 2] { + h.query_col_idx(&format!("SELECT KV_INCR('crash_ctr', 'k', {delta})"), 0) + .await; + h.query_col_idx(&format!("SELECT KV_INCR('crash_ctr_raw', 'k', {delta})"), 0) + .await; + } + for delta in ["0.1", "0.2"] { + h.query_col_idx( + &format!("SELECT KV_INCR_FLOAT('crash_score', 's', {delta})"), + 0, + ) + .await; + } + + let counter = h + .query_col_idx("SELECT n FROM crash_ctr WHERE key = 'k'", 0) + .await; + assert_eq!(counter, vec!["7".to_string()], "live typed counter"); + let raw = counter_value(&h, "SELECT KV_INCR('crash_ctr_raw', 'k', 0)").await; + assert_eq!(raw, serde_json::json!(7), "live raw counter"); + let score = h + .query_col_idx("SELECT score FROM crash_score WHERE key = 's'", 0) + .await; + + assert!( + spawned_at.elapsed() < MAX_TEST_WALL_CLOCK, + "the test ran long enough to reach a checkpoint cycle; tighten the test or the bound" + ); + h.kill_9(); + h.reopen(); + + assert_eq!( + h.query_col_idx("SELECT n FROM crash_ctr WHERE key = 'k'", 0) + .await, + counter, + "an acknowledged autocommit KV_INCR did not survive kill -9 and WAL replay" + ); + assert_eq!( + counter_value(&h, "SELECT KV_INCR('crash_ctr_raw', 'k', 0)").await, + raw, + "an acknowledged autocommit KV_INCR on a raw counter did not survive kill -9" + ); + assert_eq!( + h.query_col_idx("SELECT score FROM crash_score WHERE key = 's'", 0) + .await, + score, + "WAL replay must add the same decimal digits the live KV_INCR_FLOAT added" + ); +} + +/// The `value` field of the JSON document a counter function returns. +async fn counter_value(h: &CrashHarness, sql: &str) -> serde_json::Value { + let rows = h.query_col_idx(sql, 0).await; + assert_eq!(rows.len(), 1, "{sql} returns one row: {rows:?}"); + let doc: serde_json::Value = + serde_json::from_str(&rows[0]).unwrap_or_else(|e| panic!("{sql} returns JSON: {e}")); + doc["value"].clone() +} diff --git a/nodedb/tests/crash_replay_stamp.rs b/nodedb/tests/crash_replay_stamp.rs index 487ea347c..e18f0d8fb 100644 --- a/nodedb/tests/crash_replay_stamp.rs +++ b/nodedb/tests/crash_replay_stamp.rs @@ -1,27 +1,22 @@ // SPDX-License-Identifier: BUSL-1.1 -//! A KV write still in flight when a checkpoint is written survives a crash. +//! A write still in flight when a checkpoint is written survives a crash. //! -//! LSNs are node-global and a write reaches its core out of mint order. Client -//! row writes apply through the one Raft apply loop, which finishes each entry -//! before it starts the next, so two of them never overtake each other. A -//! write outside that loop can: the engine step of an index DDL committed in -//! a transaction runs on its own session. The test uses that step as write A: +//! LSNs are node-global, and a write reaches its core out of mint order. The +//! server runs standalone: with no Raft proposer, every autocommit write takes +//! the local funnel route, so two writes to different keys apply in any order. +//! Each engine case runs the same sequence: //! -//! 1. `COMMIT` of a block holding `CREATE SORTED INDEX` lands the catalog -//! record, then appends A, the sorted-index registration, and parks it at -//! the funnel gate. Its LSN is minted and no core holds it. -//! 2. Client writes B apply through the Raft loop with higher LSNs, until a -//! KV checkpoint's replay stamp names one of them above its prefix. That -//! proves the checkpoint was written while A was minted and not applied. +//! 1. Write A, an `INSERT` into the held collection, mints its LSN and parks at +//! the funnel gate. No core holds it. +//! 2. Writes B, `INSERT`s into a second collection, apply with higher LSNs +//! until a checkpoint's replay stamp names one of them above its prefix. +//! That proves the checkpoint was written while A was minted and not +//! applied. //! 3. The test releases A. A applies, and the process aborts before A's //! response leaves, so no later checkpoint holds A. //! 4. After restart, replay must apply A: the stamp does not name it. A stamp -//! holding only the highest applied LSN would skip A, and the index would -//! have a catalog record and no tree. -//! -//! The columnar engine has no write outside the Raft loop, so its in-flight -//! case runs in-process in `wal_replay_all.rs`. +//! that holds only the highest applied LSN skips A, and A's row is lost. //! //! Requires `--features failpoints`. @@ -44,33 +39,83 @@ const STAMP_DEADLINE: Duration = Duration::from_secs(30); /// How long the process may take to abort once A is released. const CRASH_TIMEOUT: Duration = Duration::from_secs(60); -const HELD: &str = "stamp_kv_lo"; -const INDEX: &str = "stamp_kv_idx"; -const SEEDED_ROWS: u64 = 3; +/// One engine's run of the in-flight sequence. +struct Case { + /// The collection write A goes to. + held: &'static str, + /// The collection the B writes go to. + applied: &'static str, + create_held: &'static str, + create_applied: &'static str, + /// Write A. + insert_held: &'static str, + /// Reads A's row back as one column. + read_held: &'static str, + /// The value `read_held` returns once A applied. + held_value: &'static str, + /// Write B number `n`. + insert_applied: fn(usize) -> String, + /// Reads every B row as one column. + read_applied: &'static str, + /// The engine's checkpoint module, as a `RUST_LOG` target. + log_target: &'static str, + /// The log message of a published checkpoint. + published: &'static str, + /// The log message of a checkpoint restored at boot. + restored: &'static str, +} #[tokio::test(flavor = "multi_thread")] async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { + run(Case { + held: "stamp_kv_lo", + applied: "stamp_kv_hi", + create_held: "CREATE COLLECTION stamp_kv_lo (k STRING PRIMARY KEY, v STRING) \ + WITH (engine='kv')", + create_applied: "CREATE COLLECTION stamp_kv_hi (k STRING PRIMARY KEY, v STRING) \ + WITH (engine='kv')", + insert_held: "INSERT INTO stamp_kv_lo (k, v) VALUES ('held', 'a')", + read_held: "SELECT v FROM stamp_kv_lo WHERE k = 'held'", + held_value: "a", + insert_applied: |n| format!("INSERT INTO stamp_kv_hi (k, v) VALUES ('k{n:03}', 'v{n}')"), + read_applied: "SELECT v FROM stamp_kv_hi", + log_target: "nodedb::data::executor::kv_checkpoint", + published: "KV checkpoint published", + restored: "KV checkpoint restored", + }) + .await; +} + +#[tokio::test(flavor = "multi_thread")] +async fn a_columnar_write_in_flight_at_a_checkpoint_survives_kill_9() { + run(Case { + held: "stamp_col_lo", + applied: "stamp_col_hi", + create_held: "CREATE COLLECTION stamp_col_lo COLUMNS (id TEXT, v TEXT) \ + WITH (engine='columnar')", + create_applied: "CREATE COLLECTION stamp_col_hi COLUMNS (id TEXT, v TEXT) \ + WITH (engine='columnar')", + insert_held: "INSERT INTO stamp_col_lo (id, v) VALUES ('held', 'a')", + read_held: "SELECT v FROM stamp_col_lo WHERE id = 'held'", + held_value: "a", + insert_applied: |n| format!("INSERT INTO stamp_col_hi (id, v) VALUES ('r{n:03}', 'v{n}')"), + read_applied: "SELECT v FROM stamp_col_hi", + log_target: "nodedb::data::executor::columnar_checkpoint", + published: "columnar checkpoint published", + restored: "columnar checkpoint restored", + }) + .await; +} + +async fn run(case: Case) { let mut h = CrashHarness::new() + .standalone() .with_env("NODEDB_CHECKPOINT_INTERVAL_SECS", CHECKPOINT_INTERVAL_SECS) - .with_env( - "RUST_LOG", - "warn,nodedb::data::executor::kv_checkpoint=info", - ); + .with_env("RUST_LOG", &format!("warn,{}=info", case.log_target)); h.spawn(); h.wait_ready(); - h.exec(&format!( - "CREATE COLLECTION {HELD} (k STRING PRIMARY KEY, score INT) WITH (engine='kv')" - )) - .await; - for i in 0..SEEDED_ROWS { - h.exec(&format!( - "INSERT INTO {HELD} (k, score) VALUES ('p{i}', {})", - i * 10 - )) - .await; - } - h.exec("CREATE COLLECTION stamp_kv_hi (k STRING PRIMARY KEY, v STRING) WITH (engine='kv')") - .await; + h.exec(case.create_held).await; + h.exec(case.create_applied).await; // Boot 2 arms the gate and the abort, keyed to the held collection. Both // match only a request carrying a WAL LSN, so boot itself passes them. @@ -79,28 +124,24 @@ async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { h.set_env( "NODEDB_FAILPOINTS", &format!( - "funnel::before_dispatch::{HELD}=wait_file({}),core::after_apply::{HELD}=abort", - release.display() + "funnel::before_dispatch::{held}=wait_file({}),core::after_apply::{held}=abort", + release.display(), + held = case.held, ), ); h.reopen(); let conn_str = h.pgwire_conn_str(); + let insert_held = case.insert_held; let held_task = tokio::spawn(async move { let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) .await .map_err(|e| e.to_string())?; tokio::spawn(connection); - for sql in [ - "BEGIN".to_string(), - format!("CREATE SORTED INDEX {INDEX} ON {HELD} (score DESC) KEY k"), - "COMMIT".to_string(), - ] { - client - .simple_query(&sql) - .await - .map_err(|e| format!("{sql}: {e}"))?; - } + client + .simple_query(insert_held) + .await + .map_err(|e| format!("{insert_held}: {e}"))?; Ok::<(), String>(()) }); @@ -108,23 +149,18 @@ async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { let deadline = Instant::now() + STAMP_DEADLINE; let mut applied = 0usize; loop { - h.exec(&format!( - "INSERT INTO stamp_kv_hi (k, v) VALUES ('k{applied:03}', 'v{applied}')" - )) - .await; + h.exec(&(case.insert_applied)(applied)).await; applied += 1; tokio::time::sleep(Duration::from_millis(200)).await; let log = boot_section(&h.server_log(), 2); - if applied_ranges(&log, "KV checkpoint published") - .iter() - .any(|n| *n > 0) - { + if applied_ranges(&log, case.published).iter().any(|n| *n > 0) { break; } assert!( Instant::now() < deadline, - "no KV checkpoint named an applied LSN above its prefix within \ - {STAMP_DEADLINE:?}: write A never parked, or no checkpoint ran while it was.{}\n{}", + "no {} named an applied LSN above its prefix within {STAMP_DEADLINE:?}: write A \ + never parked, or no checkpoint ran while it was.{}\n{}", + case.published, h.keep_data_dir_note(), diagnostics::log_tail_section(&h.server_log()) ); @@ -133,14 +169,17 @@ async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { !held_task.is_finished(), "write A finished before its release: the gate never parked it" ); - let mut live = h.query_col_idx("SELECT v FROM stamp_kv_hi", 0).await; + let mut live = h.query_col_idx(case.read_applied, 0).await; live.sort(); // Release A. It applies, and the process aborts before its response // leaves, so no checkpoint written after it can hold it. std::fs::write(&release, b"release").expect("create the release file"); h.await_self_crash(CRASH_TIMEOUT); - let marker = format!("fail_point aborting process: core::after_apply::{HELD}"); + let marker = format!( + "fail_point aborting process: core::after_apply::{}", + case.held + ); assert!( h.server_log().contains(&marker), "the process exited, but not after write A applied.{}\n{}", @@ -153,7 +192,7 @@ async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { h.clear_env("NODEDB_FAILPOINTS"); h.reopen(); - let ranges = applied_ranges(&boot_section(&h.server_log(), 3), "KV checkpoint restored"); + let ranges = applied_ranges(&boot_section(&h.server_log(), 3), case.restored); assert!( ranges.iter().any(|n| *n > 0), "the restored generation must name an applied LSN above its prefix, or this run did \ @@ -162,21 +201,20 @@ async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { diagnostics::log_tail_section(&h.server_log()) ); - let count = h - .query_col_idx(&format!("SELECT SORTED_COUNT({INDEX})"), 0) - .await; + let held = h.query_col_idx(case.read_held, 0).await; assert_eq!( - count.first().and_then(|text| json_field(text, "count")), - Some(SEEDED_ROWS), - "write A applied after the checkpoint and before the crash; replay must rebuild \ - the index tree from it, never skip it as covered by a higher applied LSN \ - (got {count:?})" + held, + vec![case.held_value.to_string()], + "write A to {} applied after the checkpoint and before the crash; replay must \ + apply it, never skip it as covered by a higher applied LSN", + case.held ); - let mut replayed = h.query_col_idx("SELECT v FROM stamp_kv_hi", 0).await; + let mut replayed = h.query_col_idx(case.read_applied, 0).await; replayed.sort(); assert_eq!( replayed, live, - "the replayed state must equal the live state: every B write once" + "the replayed state of {} must equal the live state: every B write once", + case.applied ); } @@ -207,17 +245,6 @@ fn applied_ranges(log: &str, message: &str) -> Vec { .collect() } -/// The unsigned integer a single-cell JSON reply carries under `field`. -fn json_field(text: &str, field: &str) -> Option { - let rest = text.split_once(&format!("\"{field}\":"))?.1; - let digits: String = rest - .trim_start() - .chars() - .take_while(char::is_ascii_digit) - .collect(); - digits.parse().ok() -} - /// `text` without terminal colour escape sequences. fn strip_ansi(text: &str) -> String { let mut out = String::with_capacity(text.len()); @@ -247,5 +274,4 @@ fn log_fields_are_read_through_colour_codes() { assert!(!boot_section(booted, 2).contains("first-line")); assert!(boot_section(booted, 1).contains("first-line")); assert!(!boot_section(booted, 1).contains("second-line")); - assert_eq!(json_field("{\"count\":3}", "count"), Some(3)); } diff --git a/nodedb/tests/crash_resp_kv_write.rs b/nodedb/tests/crash_resp_kv_write.rs index ef68b549a..5a51574f8 100644 --- a/nodedb/tests/crash_resp_kv_write.rs +++ b/nodedb/tests/crash_resp_kv_write.rs @@ -108,7 +108,17 @@ async fn resp_kv_set_survives_kill_9() { /// replay must read and write decimal text the way the live write did. #[tokio::test(flavor = "multi_thread")] async fn resp_kv_counters_replay_to_the_acknowledged_values_after_kill_9() { - let mut h = no_incidental_checkpoint(); + counters_replay_after_kill_9(no_incidental_checkpoint()).await; +} + +/// A standalone node runs no Raft proposer, so the gateway applies each RESP +/// write on its own cores. It must still append the write's WAL record. +#[tokio::test(flavor = "multi_thread")] +async fn resp_kv_counters_replay_after_kill_9_on_a_standalone_node() { + counters_replay_after_kill_9(no_incidental_checkpoint().standalone()).await; +} + +async fn counters_replay_after_kill_9(mut h: CrashHarness) { let spawned_at = Instant::now(); h.spawn(); h.wait_ready(); diff --git a/nodedb/tests/inproc/cases/kv_atomic_autocommit_wal.rs b/nodedb/tests/inproc/cases/kv_atomic_autocommit_wal.rs new file mode 100644 index 000000000..e882b26a8 --- /dev/null +++ b/nodedb/tests/inproc/cases/kv_atomic_autocommit_wal.rs @@ -0,0 +1,376 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! An autocommit SQL-function write is WAL-durable, and replay reproduces it. +//! +//! `KV_INCR`, `KV_INCR_FLOAT`, `KV_CAS`, `KV_GETSET`, `TRANSFER`, +//! `TRANSFER_ITEM`, `CREATE SORTED INDEX`, `RATE_CHECK`, `RATE_RESET` and an +//! audited `WEIGHTED_PICK` build their `KvOp` by hand instead of planning a +//! statement. Outside a transaction block each one must take the durable +//! route a planned write takes: the write funnel appends a WAL record for it. +//! +//! Each case runs the function, checks the WAL holds the record for it, then +//! restarts on the same data directory. The harness restores a core from WAL +//! replay alone, so the value read after the restart is the value replay +//! computed. A write dispatched on the read route has no record, and the +//! restart loses it. + +use nodedb_test_support::pgwire_harness::{TestDataDir, TestServer}; + +/// The MessagePack encoding of the short string `s`. +fn fixstr(s: &str) -> Vec { + assert!(s.len() < 32, "a fixstr holds at most 31 bytes: {s}"); + let mut encoded = vec![0xa0 | s.len() as u8]; + encoded.extend_from_slice(s.as_bytes()); + encoded +} + +fn contains(haystack: &[u8], needle: &[u8]) -> bool { + haystack + .windows(needle.len()) + .any(|window| window == needle) +} + +/// Whether the WAL holds a record of the KV op `op` on `collection`. +fn wal_holds(server: &TestServer, op: &str, collection: &str) -> bool { + server.shared.wal.sync().expect("sync the WAL"); + let records = server.shared.wal.replay().expect("read the WAL"); + let (op, collection) = (fixstr(op), fixstr(collection)); + records + .iter() + .any(|record| contains(&record.payload, &op) && contains(&record.payload, &collection)) +} + +/// Stop the server and start a new one on its data directory. +async fn restart(server: TestServer) -> (TestServer, TestDataDir) { + let (server, dir) = server.take_dir(); + server.graceful_shutdown().await; + TestServer::open_on_path(dir).await +} + +async fn exec(server: &TestServer, sql: &str) { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); +} + +/// The first column of every row `sql` returns. +async fn column(server: &TestServer, sql: &str) -> Vec { + server + .query_text(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")) +} + +/// The JSON document a single-row KV function returns. +async fn json(server: &TestServer, sql: &str) -> serde_json::Value { + let rows = column(server, sql).await; + assert_eq!(rows.len(), 1, "{sql} returns one row: {rows:?}"); + serde_json::from_str(&rows[0]).unwrap_or_else(|e| panic!("{sql} returns JSON: {e}")) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_incr_replays_to_the_typed_and_the_raw_value() { + let server = TestServer::start().await; + exec( + &server, + "CREATE COLLECTION wal_incr (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')", + ) + .await; + exec( + &server, + "CREATE COLLECTION wal_incr_raw (key TEXT PRIMARY KEY, value TEXT) WITH (engine='kv')", + ) + .await; + + json(&server, "SELECT KV_INCR('wal_incr', 'k', 5)").await; + let typed = json(&server, "SELECT KV_INCR('wal_incr', 'k', 2)").await; + assert_eq!(typed["value"], 7, "{typed}"); + json(&server, "SELECT KV_INCR('wal_incr_raw', 'k', 4)").await; + let raw = json(&server, "SELECT KV_INCR('wal_incr_raw', 'k', 3)").await; + assert_eq!(raw["value"], 7, "{raw}"); + assert!(wal_holds(&server, "kv_incr", "wal_incr")); + assert!(wal_holds(&server, "kv_incr", "wal_incr_raw")); + let typed_live = column(&server, "SELECT n FROM wal_incr WHERE key = 'k'").await; + + let (server, _dir) = restart(server).await; + assert_eq!( + column(&server, "SELECT n FROM wal_incr WHERE key = 'k'").await, + typed_live, + "replay must rebuild the typed counter row the live increments wrote" + ); + assert_eq!( + json(&server, "SELECT KV_INCR('wal_incr_raw', 'k', 0)").await["value"], + raw["value"], + "replay must rebuild the raw counter the live increments wrote" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_incr_float_replays_to_the_exact_decimal_value() { + let server = TestServer::start().await; + exec( + &server, + "CREATE COLLECTION wal_score (key TEXT PRIMARY KEY, score FLOAT) WITH (engine='kv')", + ) + .await; + exec( + &server, + "CREATE COLLECTION wal_score_raw (key TEXT PRIMARY KEY, value TEXT) WITH (engine='kv')", + ) + .await; + + json(&server, "SELECT KV_INCR_FLOAT('wal_score', 's', 0.1)").await; + json(&server, "SELECT KV_INCR_FLOAT('wal_score', 's', 0.2)").await; + json(&server, "SELECT KV_INCR_FLOAT('wal_score_raw', 's', 0.1)").await; + json(&server, "SELECT KV_INCR_FLOAT('wal_score_raw', 's', 0.2)").await; + assert!(wal_holds(&server, "kv_incr_float", "wal_score")); + assert!(wal_holds(&server, "kv_incr_float", "wal_score_raw")); + let typed_live = column(&server, "SELECT score FROM wal_score WHERE key = 's'").await; + let raw_live = json(&server, "SELECT KV_INCR_FLOAT('wal_score_raw', 's', 0)").await; + + let (server, _dir) = restart(server).await; + assert_eq!( + column(&server, "SELECT score FROM wal_score WHERE key = 's'").await, + typed_live, + "replay must add the same decimal digits the live increments added" + ); + assert_eq!( + json(&server, "SELECT KV_INCR_FLOAT('wal_score_raw', 's', 0)").await["value"], + raw_live["value"], + "replay must store the same decimal text the live increments stored" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_cas_and_kv_getset_replay_to_the_value_they_set() { + let server = TestServer::start().await; + exec( + &server, + "CREATE COLLECTION wal_swap (key TEXT PRIMARY KEY, value TEXT) WITH (engine='kv')", + ) + .await; + + let cas = json(&server, "SELECT KV_CAS('wal_swap', 'state', '', 'idle')").await; + assert_eq!(cas["success"], true, "{cas}"); + json( + &server, + "SELECT KV_GETSET('wal_swap', 'tok', 'first-token')", + ) + .await; + assert!(wal_holds(&server, "kv_cas", "wal_swap")); + assert!(wal_holds(&server, "kv_getset", "wal_swap")); + + let (server, _dir) = restart(server).await; + let cas = json( + &server, + "SELECT KV_CAS('wal_swap', 'state', 'idle', 'ended')", + ) + .await; + assert_eq!( + cas["success"], true, + "replay must restore the value the live KV_CAS set: {cas}" + ); + let getset = json( + &server, + "SELECT KV_GETSET('wal_swap', 'tok', 'second-token')", + ) + .await; + let old = getset["old_value"] + .as_str() + .unwrap_or_else(|| panic!("replay must restore the KV_GETSET value: {getset}")); + let old = base64::Engine::decode(&base64::engine::general_purpose::STANDARD, old) + .expect("old_value is base64"); + assert_eq!(old, b"first-token"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn transfer_and_transfer_item_replay_to_the_moved_state() { + let server = TestServer::start().await; + exec( + &server, + "CREATE COLLECTION wal_acct (key TEXT PRIMARY KEY, balance INT) WITH (engine='kv')", + ) + .await; + exec( + &server, + "CREATE COLLECTION wal_items (key TEXT PRIMARY KEY, name TEXT) WITH (engine='kv')", + ) + .await; + exec( + &server, + "INSERT INTO wal_acct (key, balance) VALUES ('a', 100)", + ) + .await; + exec( + &server, + "INSERT INTO wal_acct (key, balance) VALUES ('b', 10)", + ) + .await; + exec( + &server, + "INSERT INTO wal_items (key, name) VALUES ('ownerA:sword', 'Sword')", + ) + .await; + + json( + &server, + "SELECT TRANSFER('wal_acct', 'a', 'b', 'balance', 30)", + ) + .await; + json( + &server, + "SELECT TRANSFER_ITEM('wal_items', 'wal_items', 'sword', 'ownerA', 'ownerB')", + ) + .await; + assert!(wal_holds(&server, "kv_transfer", "wal_acct")); + assert!(wal_holds(&server, "kv_transfer_item", "wal_items")); + let source = column(&server, "SELECT balance FROM wal_acct WHERE key = 'a'").await; + let dest = column(&server, "SELECT balance FROM wal_acct WHERE key = 'b'").await; + + let (server, _dir) = restart(server).await; + assert_eq!( + column(&server, "SELECT balance FROM wal_acct WHERE key = 'a'").await, + source + ); + assert_eq!( + column(&server, "SELECT balance FROM wal_acct WHERE key = 'b'").await, + dest + ); + assert_eq!( + column( + &server, + "SELECT name FROM wal_items WHERE key = 'ownerB:sword'" + ) + .await, + vec!["Sword".to_string()], + "replay must move the item to its destination key" + ); + assert!( + column( + &server, + "SELECT name FROM wal_items WHERE key = 'ownerA:sword'" + ) + .await + .is_empty(), + "replay must remove the item from its source key" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_sorted_index_registration_replays_to_its_tree() { + let server = TestServer::start().await; + exec( + &server, + "CREATE COLLECTION wal_board (k TEXT PRIMARY KEY, score INT) WITH (engine='kv')", + ) + .await; + for (key, score) in [("p0", 10), ("p1", 20), ("p2", 30)] { + exec( + &server, + &format!("INSERT INTO wal_board (k, score) VALUES ('{key}', {score})"), + ) + .await; + } + exec( + &server, + "CREATE SORTED INDEX wal_board_idx ON wal_board (score DESC) KEY k", + ) + .await; + assert!(wal_holds(&server, "kv_register_sorted_index", "wal_board")); + + let (server, _dir) = restart(server).await; + let count = json(&server, "SELECT SORTED_COUNT(wal_board_idx)").await; + assert_eq!( + count["count"], 3, + "replay must rebuild the index tree over every row: {count}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_rate_gate_counter_replays_through_check_and_reset() { + let server = TestServer::start().await; + json(&server, "SELECT RATE_CHECK('wal_gate', 'u1', 5, 600)").await; + json(&server, "SELECT RATE_CHECK('wal_gate', 'u1', 5, 600)").await; + assert!(wal_holds(&server, "kv_incr", "_system_rate_gates")); + + let (server, dir) = restart(server).await; + let remaining = json(&server, "SELECT RATE_REMAINING('wal_gate', 'u1', 5, 600)").await; + assert_eq!( + remaining["current"], 2, + "replay must restore both counted calls: {remaining}" + ); + + json(&server, "SELECT RATE_RESET('wal_gate', 'u1')").await; + assert!(wal_holds(&server, "kv_delete", "_system_rate_gates")); + let (server, _dir) = restart_again(server, dir).await; + let remaining = json(&server, "SELECT RATE_REMAINING('wal_gate', 'u1', 5, 600)").await; + assert_eq!( + remaining["current"], 0, + "replay must apply the reset after the counted calls: {remaining}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn an_audited_weighted_pick_logs_its_audit_record() { + let server = TestServer::start().await; + exec( + &server, + "CREATE COLLECTION wal_wp (key TEXT PRIMARY KEY, w FLOAT) WITH (engine='kv')", + ) + .await; + exec(&server, "INSERT INTO wal_wp (key, w) VALUES ('x', 1.0)").await; + + let picked = column( + &server, + "SELECT * FROM WEIGHTED_PICK('wal_wp', weight => 'w', count => 1, AUDIT => TRUE)", + ) + .await; + assert_eq!(picked.len(), 1, "one pick: {picked:?}"); + assert!( + wal_holds(&server, "kv_put", "_system_random_audit"), + "an audited pick answers only once its audit record is in the WAL" + ); +} + +/// A pick draws the stored key by its stored weight: a zero-weight row is +/// never drawn, and the reported key and weight are the row's own. +#[tokio::test] +async fn a_weighted_pick_draws_the_stored_key_by_its_weight() { + let server = TestServer::start().await; + exec( + &server, + "CREATE COLLECTION wp_draw (key TEXT PRIMARY KEY, w FLOAT) WITH (engine='kv')", + ) + .await; + exec( + &server, + "INSERT INTO wp_draw (key, w) VALUES ('never', 0.0)", + ) + .await; + exec( + &server, + "INSERT INTO wp_draw (key, w) VALUES ('always', 2.5)", + ) + .await; + + let sql = "SELECT * FROM WEIGHTED_PICK('wp_draw', weight => 'w', count => 5, \ + WITH REPLACEMENT)"; + let rows = server + .query_rows(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + assert_eq!(rows.len(), 5, "five draws: {rows:?}"); + for row in &rows { + assert_eq!(row[1], "always", "only the weighted row is drawn: {rows:?}"); + assert_eq!(row[2], "2.5", "the stored weight is reported: {rows:?}"); + } +} + +/// Restart a server that itself came from a restart. Such a server holds a +/// placeholder directory, so the data directory the earlier restart returned +/// is the one reopened. +async fn restart_again(server: TestServer, dir: TestDataDir) -> (TestServer, TestDataDir) { + server.graceful_shutdown().await; + TestServer::open_on_path(dir).await +} diff --git a/nodedb/tests/inproc/cases/mod.rs b/nodedb/tests/inproc/cases/mod.rs index e27b118d7..23dd214d4 100644 --- a/nodedb/tests/inproc/cases/mod.rs +++ b/nodedb/tests/inproc/cases/mod.rs @@ -115,6 +115,7 @@ mod http_streams; mod http_ws; mod http_ws_authorization; mod intake_throttle; +mod kv_atomic_autocommit_wal; mod kv_atomic_surrogate_identity; mod kv_field_transfer_surrogate_identity; mod maintenance_does_not_starve_interactive; diff --git a/nodedb/tests/native/cases/mod.rs b/nodedb/tests/native/cases/mod.rs index e2ccb9d9b..d494b9a13 100644 --- a/nodedb/tests/native/cases/mod.rs +++ b/nodedb/tests/native/cases/mod.rs @@ -9,6 +9,7 @@ mod native_dml_outcome_conformance; mod native_error_code_classification; mod native_gateway_txn_overlay; mod native_index_ddl_opcodes; +mod native_kv_atomic_autocommit_wal; mod native_kv_counter_faults; mod native_primary_key_nullability; mod native_protocol; diff --git a/nodedb/tests/native/cases/native_kv_atomic_autocommit_wal.rs b/nodedb/tests/native/cases/native_kv_atomic_autocommit_wal.rs new file mode 100644 index 000000000..e63d37f30 --- /dev/null +++ b/nodedb/tests/native/cases/native_kv_atomic_autocommit_wal.rs @@ -0,0 +1,182 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! An autocommit native KV atomic opcode is WAL-durable, and replay +//! reproduces it. +//! +//! `KvIncr`, `KvIncrFloat`, `KvCas` and `KvGetSet` reach the gateway as direct +//! ops. On a node with no Raft proposer the gateway applies them on its own +//! cores, and it must give each one the durable route: the write funnel +//! appends a WAL record for it. The harness restores a core from WAL replay +//! alone, so the value read after the restart is the value replay computed. +//! The native protocol has no transfer opcode, so `TRANSFER` is covered by +//! the SQL cases alone. + +use nodedb_test_support::native_harness::{do_handshake, send_request, send_sql}; +use nodedb_test_support::pgwire_harness::TestServer; + +use nodedb_types::protocol::opcodes::ResponseStatus; +use nodedb_types::protocol::text_fields::TextFields; +use nodedb_types::protocol::{HelloFrame, NativeResponse, OpCode}; +use tokio::net::TcpStream; + +/// The MessagePack encoding of the short string `s`. +fn fixstr(s: &str) -> Vec { + assert!(s.len() < 32, "a fixstr holds at most 31 bytes: {s}"); + let mut encoded = vec![0xa0 | s.len() as u8]; + encoded.extend_from_slice(s.as_bytes()); + encoded +} + +fn contains(haystack: &[u8], needle: &[u8]) -> bool { + haystack + .windows(needle.len()) + .any(|window| window == needle) +} + +/// Whether the WAL holds a record of the KV op `op` on `collection`. +fn wal_holds(server: &TestServer, op: &str, collection: &str) -> bool { + server.shared.wal.sync().expect("sync the WAL"); + let records = server.shared.wal.replay().expect("read the WAL"); + let (op, collection) = (fixstr(op), fixstr(collection)); + records + .iter() + .any(|record| contains(&record.payload, &op) && contains(&record.payload, &collection)) +} + +async fn native_session(server: &TestServer) -> TcpStream { + let addr = format!("127.0.0.1:{}", server.native_port) + .parse() + .expect("native addr"); + let (stream, _ack) = do_handshake(addr, &HelloFrame::current()) + .await + .expect("native handshake"); + stream +} + +fn ok(resp: NativeResponse, what: &str) -> NativeResponse { + assert_ne!(resp.status, ResponseStatus::Error, "{what}: {resp:?}"); + resp +} + +fn fields(collection: &str, key: &str) -> TextFields { + TextFields { + collection: Some(collection.to_string()), + key: Some(key.to_string()), + ..Default::default() + } +} + +/// The first column of every row `sql` returns over pgwire. +async fn column(server: &TestServer, sql: &str) -> Vec { + server + .query_text(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")) +} + +/// The JSON document a single-row KV function returns over pgwire. +async fn json(server: &TestServer, sql: &str) -> serde_json::Value { + let rows = column(server, sql).await; + assert_eq!(rows.len(), 1, "{sql} returns one row: {rows:?}"); + serde_json::from_str(&rows[0]).unwrap_or_else(|e| panic!("{sql} returns JSON: {e}")) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn native_kv_atomic_opcodes_replay_to_the_values_they_wrote() { + let server = TestServer::start().await; + let mut stream = native_session(&server).await; + for (seq, sql) in [ + "CREATE COLLECTION nat_ctr (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')", + "CREATE COLLECTION nat_score (key TEXT PRIMARY KEY, score FLOAT) WITH (engine='kv')", + "CREATE COLLECTION nat_swap (key TEXT PRIMARY KEY, value TEXT) WITH (engine='kv')", + ] + .into_iter() + .enumerate() + { + ok(send_sql(&mut stream, seq as u64 + 1, sql).await, sql); + } + + let incr = TextFields { + incr_delta: Some(5), + ..fields("nat_ctr", "k") + }; + ok( + send_request(&mut stream, 10, OpCode::KvIncr, incr.clone()).await, + "KvIncr", + ); + ok( + send_request(&mut stream, 11, OpCode::KvIncr, incr).await, + "KvIncr", + ); + for (seq, delta) in [(12, "0.1"), (13, "0.2")] { + let incr_float = TextFields { + incr_float_delta: Some(delta.to_string()), + ..fields("nat_score", "s") + }; + ok( + send_request(&mut stream, seq, OpCode::KvIncrFloat, incr_float).await, + "KvIncrFloat", + ); + } + let cas = TextFields { + expected: Some(Vec::new()), + new_value: Some(b"idle".to_vec()), + ..fields("nat_swap", "state") + }; + ok( + send_request(&mut stream, 14, OpCode::KvCas, cas).await, + "KvCas", + ); + let getset = TextFields { + new_value: Some(b"first-token".to_vec()), + ..fields("nat_swap", "tok") + }; + ok( + send_request(&mut stream, 15, OpCode::KvGetSet, getset).await, + "KvGetSet", + ); + + assert!(wal_holds(&server, "kv_incr", "nat_ctr")); + assert!(wal_holds(&server, "kv_incr_float", "nat_score")); + assert!(wal_holds(&server, "kv_cas", "nat_swap")); + assert!(wal_holds(&server, "kv_getset", "nat_swap")); + let counter = column(&server, "SELECT n FROM nat_ctr WHERE key = 'k'").await; + assert_eq!(counter, vec!["10".to_string()]); + let score = column(&server, "SELECT score FROM nat_score WHERE key = 's'").await; + drop(stream); + + let (server, dir) = server.take_dir(); + server.graceful_shutdown().await; + let (server, _dir) = TestServer::open_on_path(dir).await; + + assert_eq!( + column(&server, "SELECT n FROM nat_ctr WHERE key = 'k'").await, + counter, + "replay must rebuild the counter both KvIncr ops moved" + ); + assert_eq!( + column(&server, "SELECT score FROM nat_score WHERE key = 's'").await, + score, + "replay must add the same decimal digits both KvIncrFloat ops added" + ); + let cas = json( + &server, + "SELECT KV_CAS('nat_swap', 'state', 'idle', 'ended')", + ) + .await; + assert_eq!( + cas["success"], true, + "replay must restore the value KvCas set: {cas}" + ); + let getset = json( + &server, + "SELECT KV_GETSET('nat_swap', 'tok', 'second-token')", + ) + .await; + let old = getset["old_value"] + .as_str() + .unwrap_or_else(|| panic!("replay must restore the KvGetSet value: {getset}")); + let old = base64::Engine::decode(&base64::engine::general_purpose::STANDARD, old) + .expect("old_value is base64"); + assert_eq!(old, b"first-token"); +} From 96d31d45433e62e8179e0a72de198d1264daee32 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 13:41:35 +0800 Subject: [PATCH 30/64] feat(executor): stamp checkpoints with LSN-range replay stamps Move the applied-prefix stamp type to types::replay_stamp so vector, array, and timeseries checkpoints can all publish an exact record of what they hold, instead of a single high-water LSN. A single LSN wrongly claims out-of-order records that mint below it but arrive after, so restart replay skips them and the write is lost. Vector and array checkpoint manifests now carry a ReplayStamp and gate recovery on it. Timeseries collections get an analogous TsReplayStamp that names applied rows and effective truncates per partition and per collection directory, written before each flush and before retention removes a partition, so replay never re-applies or drops a record. Timeseries ingest also groups a committed record's sub-records so a memtable flush never lands mid-record, and retention enforcement moves into its own dispatch handler. --- .../src/data/executor/applied_prefix/mod.rs | 3 +- nodedb/src/data/executor/array_checkpoint.rs | 410 +++++++++++++-- .../core_loop/checkpoint_floors/init.rs | 10 +- .../core_loop/checkpoint_floors/state.rs | 29 +- nodedb/src/data/executor/core_loop/open.rs | 4 +- nodedb/src/data/executor/core_loop/state.rs | 24 +- nodedb/src/data/executor/core_loop/tick.rs | 9 + .../src/data/executor/dispatch/array/entry.rs | 50 +- .../data/executor/dispatch/array/mutate.rs | 69 ++- .../dispatch/meta_retention/handlers.rs | 105 +--- .../executor/dispatch/meta_retention/mod.rs | 4 +- .../dispatch/meta_retention/ts_retention.rs | 119 +++++ .../src/data/executor/dispatch/timeseries.rs | 44 +- .../handlers/point/apply_put/vector/put.rs | 81 ++- .../handlers/point/apply_put/vector/types.rs | 1 - nodedb/src/data/executor/handlers/purge.rs | 4 +- .../data/executor/handlers/snapshot/create.rs | 7 + .../handlers/snapshot/restore_segments.rs | 28 +- .../executor/handlers/timeseries/admission.rs | 13 +- .../executor/handlers/timeseries/flush.rs | 61 ++- .../handlers/timeseries/group_flush.rs | 494 ++++++++++++++++++ .../executor/handlers/timeseries/ingest.rs | 72 +-- .../handlers/timeseries/ingest_dispatch.rs | 54 +- .../data/executor/handlers/timeseries/mod.rs | 1 + .../handlers/timeseries/redo_ingest.rs | 27 +- .../executor/handlers/timeseries/truncate.rs | 161 +++++- .../data/executor/handlers/timeseries_wal.rs | 118 +++-- .../handlers/timeseries_wal_payload.rs | 2 +- .../handlers/transaction/redo_apply/cover.rs | 51 +- .../handlers/transaction/redo_apply/entry.rs | 21 +- .../handlers/transaction/redo_apply/settle.rs | 22 +- .../handlers/transaction/resolve/array.rs | 2 +- .../handlers/transaction/undo/apply.rs | 22 +- .../handlers/transaction/undo/entry.rs | 13 +- .../handlers/transaction/undo/timeseries.rs | 77 ++- .../handlers/unregister_collection.rs | 3 +- nodedb/src/data/executor/handlers/vector.rs | 8 - .../handlers/vector_direct_resolve/apply.rs | 2 - .../executor/handlers/vector_direct_row.rs | 8 - .../executor/handlers/vector_direct_update.rs | 9 +- .../data/executor/handlers/vector_multi.rs | 13 - .../data/executor/handlers/vector_upsert.rs | 1 - .../data/executor/handlers/vector_write.rs | 6 - nodedb/src/data/executor/replay_floors.rs | 43 +- nodedb/src/data/executor/replay_policy.rs | 22 + .../sparse_vector_checkpoint/format.rs | 8 +- .../executor/sparse_vector_checkpoint/load.rs | 14 +- .../sparse_vector_checkpoint/manifest.rs | 7 + .../executor/sparse_vector_checkpoint/mod.rs | 16 +- .../sparse_vector_checkpoint/write.rs | 29 +- .../executor/timeseries_checkpoint/flush.rs | 216 +++++++- .../executor/timeseries_checkpoint/load.rs | 39 +- .../executor/timeseries_checkpoint/mod.rs | 59 +-- .../executor/timeseries_checkpoint/stamp.rs | 324 ++++++++++++ .../data/executor/vector_checkpoint/format.rs | 8 +- .../data/executor/vector_checkpoint/load.rs | 74 ++- .../executor/vector_checkpoint/manifest.rs | 7 + .../data/executor/vector_checkpoint/mod.rs | 14 +- .../executor/vector_checkpoint/publish.rs | 6 +- .../data/executor/vector_checkpoint/write.rs | 23 +- nodedb/src/data/executor/wal_replay/array.rs | 101 ++-- .../data/executor/wal_replay_columnar_dml.rs | 8 +- .../executor/wal_replay_columnar_truncate.rs | 174 ++++-- .../data/executor/wal_replay_redo_document.rs | 44 +- nodedb/src/data/executor/wal_replay_vector.rs | 181 +++---- .../data/executor/wal_replay_vector_direct.rs | 27 +- .../executor/wal_replay_vector_extended.rs | 110 ++-- .../executor/wal_replay_vector_resolved.rs | 9 +- .../data/executor/wal_replay_vector_sparse.rs | 22 +- nodedb/src/data/runtime/boot_restore.rs | 23 +- nodedb/src/engine/array/compact.rs | 8 +- nodedb/src/engine/array/compaction/merger.rs | 22 +- nodedb/src/engine/array/engine.rs | 15 +- nodedb/src/engine/array/flush.rs | 84 ++- nodedb/src/engine/array/mod.rs | 6 +- nodedb/src/engine/array/purge/execute.rs | 18 +- nodedb/src/engine/array/purge/plan.rs | 9 + nodedb/src/engine/array/recovery.rs | 23 +- nodedb/src/engine/array/rollback.rs | 39 +- nodedb/src/engine/array/store/catalog/scan.rs | 5 +- nodedb/src/engine/array/store/manifest.rs | 61 ++- nodedb/src/engine/array/test_support.rs | 9 + nodedb/src/engine/array/wal.rs | 7 +- nodedb/src/engine/array/write.rs | 44 +- .../columnar_memtable/memtable/ingest.rs | 8 +- nodedb/src/storage/snapshot_executor.rs | 2 + nodedb/src/types/mod.rs | 1 + .../stamp.rs => types/replay_stamp.rs} | 110 +++- nodedb/tests/crash_replay_stamp.rs | 173 +++++- .../inproc/cases/bitemporal_array_basic.rs | 3 +- .../cases/bitemporal_array_compaction.rs | 18 +- .../inproc/cases/bitemporal_array_gdpr.rs | 26 +- .../inproc/cases/bitemporal_array_recovery.rs | 12 +- .../cases/bitemporal_array_retention.rs | 15 + .../cases/bitemporal_array_temporal_purge.rs | 21 +- .../cases/bitemporal_array_variant_cube.rs | 3 +- 96 files changed, 3332 insertions(+), 1280 deletions(-) create mode 100644 nodedb/src/data/executor/dispatch/meta_retention/ts_retention.rs create mode 100644 nodedb/src/data/executor/handlers/timeseries/group_flush.rs create mode 100644 nodedb/src/data/executor/timeseries_checkpoint/stamp.rs rename nodedb/src/{data/executor/applied_prefix/stamp.rs => types/replay_stamp.rs} (63%) diff --git a/nodedb/src/data/executor/applied_prefix/mod.rs b/nodedb/src/data/executor/applied_prefix/mod.rs index 11a39f686..3c9a8a0aa 100644 --- a/nodedb/src/data/executor/applied_prefix/mod.rs +++ b/nodedb/src/data/executor/applied_prefix/mod.rs @@ -1,8 +1,9 @@ // SPDX-License-Identifier: BUSL-1.1 mod ranges; -pub(crate) mod stamp; mod tracker; +pub(crate) use crate::types::replay_stamp as stamp; + pub(crate) use stamp::{InvalidReplayStamp, ReplayStamp}; pub(in crate::data::executor) use tracker::AppliedPrefix; diff --git a/nodedb/src/data/executor/array_checkpoint.rs b/nodedb/src/data/executor/array_checkpoint.rs index f689029f0..4937841a9 100644 --- a/nodedb/src/data/executor/array_checkpoint.rs +++ b/nodedb/src/data/executor/array_checkpoint.rs @@ -6,15 +6,13 @@ //! //! `ArrayStore::memtable` is a plain in-memory `Memtable` (see //! `engine::array::memtable`). Every `INSERT INTO ARRAY` / `DELETE FROM ARRAY` -//! lands there and advances the core watermark -//! (`dispatch::array::mutate` calls `note_write_lsn` with the Control Plane's -//! `wal_lsn`), so the periodic checkpoint reported those writes as durable and -//! the manager truncated the `ArrayPut` / `ArrayDelete` records that were their -//! only copy. The memtable is drained to a segment only by an explicit -//! `NDARRAY_FLUSH` or by the `flush_cell_threshold` auto-flush — neither of -//! which is ordered against the truncation the checkpoint authorises. A restart -//! after truncation returned an array missing every cell written since the last -//! time a user happened to run `NDARRAY_FLUSH`. +//! lands there and advances the core watermark (`dispatch::array::mutate` +//! calls `note_write_lsn` with the Control Plane's `wal_lsn`). The periodic +//! checkpoint reports those writes as durable, and the manager then removes +//! the `ArrayPut` / `ArrayDelete` records below that LSN. The checkpoint must +//! therefore flush every memtable first. An explicit `NDARRAY_FLUSH` and the +//! `flush_cell_threshold` flush in `dispatch::array::mutate` are not ordered +//! against that removal, so they cannot stand in for it. //! //! ## Why this reuses `ArrayEngine::flush` rather than writing a checkpoint blob //! @@ -36,38 +34,36 @@ //! the engine already knows how to write and read back — and a worse one: it //! would bypass the tile compression, the per-tile MBR statistics the query //! planner prunes on, and the compaction path that later merges those segments. -//! The bug was never a missing format; it was that the existing flush was -//! reachable only by an explicit user command. Calling it here is the fix. +//! The checkpoint calls the existing flush instead. //! -//! ## What LSN is durable after a flush +//! ## What a flush stamps //! -//! Segments carry a `flush_lsn` and the manifest a `durable_lsn = max(flush_lsn)` -//! — "every WAL record at or below this is already in a segment". Stamping the -//! flush with the core watermark makes that claim true: this runs on the core's -//! own thread between tasks, and an array write reaches `note_write_lsn` (which -//! raises the watermark) only after `put_cells` / `delete_cells` has already -//! stamped the memtable, so every cell with `lsn <= watermark` is in the drain. +//! Every flush publishes the core's replay stamp in the array's manifest: the +//! records this core applied, by prefix and ranges. This runs on the core's +//! own thread between tasks, and an applied write is noted before the next +//! task runs, so every record whose cells the memtable holds is named. A +//! record the stamp does not name was not applied yet, and replay applies it. +//! +//! The flush also reports the core watermark as this engine's durable point, +//! the LSN the checkpoint manager may truncate below. //! //! An array whose memtable is empty flushes nothing and leaves its manifest's -//! `durable_lsn` where it stands. That is not a gap: an empty memtable means -//! every cell ever applied to this array is already in a segment, so the -//! watermark is durable for it either way. The stamp is left alone rather than -//! advanced with an empty write because doing so would rewrite every array's -//! manifest on every cycle to record a claim nothing reads back. +//! stamp where it stands. That is not a gap: an empty memtable means every +//! cell ever applied to this array is already in a segment, and the stamp of +//! the flush that wrote it named the record. //! -//! ## Why the floor is the manifest, not `replay_floors.rs` +//! ## Why the stamp is the manifest's, not `replay_floors.rs` //! -//! Arrays flush independently of one another, so the "already durable through" -//! watermark is per-array and already recorded: it is the `durable_lsn` this -//! flush stamps into each array's manifest. `replay_array_wal` skips every -//! record at or below it. +//! Arrays flush independently of one another, so each array's manifest +//! carries the stamp of its own last flush. `replay_array_wal` skips a record +//! exactly when that stamp names it (`array_replay_skips`). //! -//! Re-applying such a record would usually be merely redundant — the writes are -//! absolute and keyed by `(tile, coord)` with `system_from_ms` taken from the -//! WAL payload, so it writes the identical version back. It is not redundant -//! after a bitemporal audit purge: the purge physically removes a superseded -//! tile-version from the flushed segment, and a still-retained `ArrayPut` below -//! the watermark would re-materialise exactly the version the purge erased. +//! Re-applying a named record is not merely redundant. Its cells would land in +//! the memtable while a segment already holds the same tile version, and the +//! next flush would write that version into a second segment. After a +//! bitemporal audit purge it is worse: the purge physically removes a +//! superseded tile-version from the flushed segment, and a still-retained +//! `ArrayPut` would re-materialise exactly the version the purge erased. use nodedb_array::types::ArrayId; use tracing::{info, warn}; @@ -97,6 +93,7 @@ impl CoreLoop { /// the first read (`ensure_array_open`) or by replay. pub(in crate::data::executor) fn checkpoint_array_engines(&mut self) -> crate::Result { let durable_through = self.watermark; + let stamp = self.floors.applied_prefix.stamp()?; // Collected first: `flush` takes `&mut self.array_engine`, so the id // iterator cannot stay borrowed across the loop. @@ -110,7 +107,7 @@ impl CoreLoop { for id in &ids { // `Ok(None)` = empty memtable, nothing to write; see the module docs // for why the watermark is still durable for that array. - match self.array_engine.flush(id, durable_through.as_u64()) { + match self.array_engine.flush(id, stamp.clone()) { Ok(Some(_)) => flushed += 1, Ok(None) => {} Err(e) => { @@ -146,10 +143,53 @@ impl CoreLoop { arrays = ids.len(), flushed, durable_through_lsn = durable_through.as_u64(), + replay_prefix = stamp.prefix, + applied_ranges = stamp.applied_above.len(), "array checkpoint flushed" ); Ok(durable_through) } + + /// Flush `array_id` once its memtable reached the engine's cell + /// threshold. [`Self::flush_array`] states what the flush stamps. + pub(in crate::data::executor) fn flush_array_if_full( + &mut self, + array_id: &ArrayId, + ) -> crate::Result<()> { + let full = self + .array_engine + .needs_flush(array_id) + .map_err(|e| array_flush_error(array_id, e))?; + if full { + self.flush_array(array_id)?; + } + Ok(()) + } + + /// Flush `array_id`, stamped with what this core applied. + /// + /// The caller noted every record it applied to the array before calling, + /// so the stamp names each record whose cells the flush writes. + pub(in crate::data::executor) fn flush_array( + &mut self, + array_id: &ArrayId, + ) -> crate::Result<()> { + let stamp = self.floors.applied_prefix.stamp()?; + self.array_engine + .flush(array_id, stamp) + .map_err(|e| array_flush_error(array_id, e))?; + Ok(()) + } +} + +fn array_flush_error( + array_id: &ArrayId, + e: crate::engine::array::engine::ArrayEngineError, +) -> crate::Error { + crate::Error::Storage { + engine: "array".to_string(), + detail: format!("flush of array '{}': {e}", array_id.name), + } } #[cfg(test)] @@ -172,7 +212,10 @@ mod tests { use crate::bridge::dispatch::{BridgeRequest, BridgeResponse}; use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Response, Status}; use crate::engine::array::wal::ArrayPutCell; + use crate::types::replay_stamp::{LsnRange, ReplayStamp}; use crate::types::*; + use nodedb_wal::record::{RecordType, WalRecordArgs}; + use nodedb_wal::{TombstoneSet, WalRecord}; const TID: u64 = 1; @@ -252,14 +295,12 @@ mod tests { } fn put(&mut self, id: &ArrayId, x: i64, y: i64, v: i64, wal_lsn: u64) { - let cells = vec![ArrayPutCell { - coord: vec![CoordValue::Int64(x), CoordValue::Int64(y)], - attrs: vec![CellValue::Int64(v)], - surrogate: nodedb_types::Surrogate::ZERO, - system_from_ms: 1, - valid_from_ms: 0, - valid_until_ms: i64::MAX, - }]; + self.put_at(id, x, y, v, 1, wal_lsn); + } + + /// `put` with an explicit system time, for bitemporal versions. + fn put_at(&mut self, id: &ArrayId, x: i64, y: i64, v: i64, sys_ms: i64, wal_lsn: u64) { + let cells = vec![cell(x, y, v, sys_ms)]; let r = self.send(ArrayOp::Put { array_id: id.clone(), cells_msgpack: zerompk::to_msgpack_vec(&cells).expect("encode cells"), @@ -384,8 +425,7 @@ mod tests { assert_eq!( after.slice_all(&id), vec![(1, 2, 30), (9, 9, 40)], - "every checkpointed cell must come back from its on-disk segment — \ - pre-fix the checkpoint flushed nothing and both cells were gone" + "every checkpointed cell must come back from its on-disk segment" ); } @@ -448,11 +488,11 @@ mod tests { ); } - /// A committed record applied at an LSN the array's durable watermark - /// already covers is flushed into a segment: restart replay skips it, so + /// A committed record applied at an LSN the array's manifest stamp + /// already names is flushed into a segment: restart replay skips it, so /// the segment is its only copy. #[test] - fn a_committed_record_applied_below_the_durable_lsn_is_flushed() { + fn a_committed_record_the_manifest_stamp_names_is_flushed() { use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; use crate::engine::array::wal::{ArrayPutPayload, encode_put_with_version}; use crate::wal::{RedoRecord, RedoSubRecord}; @@ -465,10 +505,17 @@ mod tests { before.open_array(&id); before.put(&id, 1, 2, 30, 100); before.core.advance_watermark(Lsn::new(100)); + // The outcome floor passed lsn 50 while that record was still on its + // way to this core, so the flush's stamp prefix names it. + before + .core + .floors + .applied_prefix + .observe_outcome_floor(Lsn::new(100)); before .core .checkpoint_array_engines() - .expect("flush at lsn 100"); + .expect("flush stamped through lsn 100"); let payload = encode_put_with_version(&ArrayPutPayload { array_id: id.clone(), @@ -512,8 +559,269 @@ mod tests { assert_eq!( after.slice_all(&id), vec![(1, 2, 30), (9, 9, 40)], - "the cell committed at lsn 50 survives a restart that replays nothing \ - at or below lsn 100" + "the cell committed at lsn 50 survives a restart that skips every \ + record the stamp names" + ); + } + + fn cell(x: i64, y: i64, v: i64, sys_ms: i64) -> ArrayPutCell { + ArrayPutCell { + coord: vec![CoordValue::Int64(x), CoordValue::Int64(y)], + attrs: vec![CellValue::Int64(v)], + surrogate: nodedb_types::Surrogate::ZERO, + system_from_ms: sys_ms, + valid_from_ms: 0, + valid_until_ms: i64::MAX, + } + } + + /// The `ArrayPut` WAL record a live put of one cell appends. + fn put_record(id: &ArrayId, x: i64, y: i64, v: i64, sys_ms: i64, lsn: u64) -> WalRecord { + use crate::engine::array::wal::{ArrayPutPayload, encode_put_with_version}; + let payload = encode_put_with_version(&ArrayPutPayload { + array_id: id.clone(), + cells: vec![cell(x, y, v, sys_ms)], + provenance: None, + }) + .expect("encode put"); + WalRecord::new(WalRecordArgs { + record_type: RecordType::ArrayPut as u32, + lsn, + tenant_id: TID, + vshard_id: 0, + database_id: DatabaseId::DEFAULT.as_u64(), + payload, + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record") + } + + /// Restart the way `replay_all_wal` does: raise the floor to the WAL end, + /// then replay `records` into the array. + fn restart_and_replay(dir: &std::path::Path, id: &ArrayId, records: &[WalRecord]) -> Core { + let mut core = Core::open_at(dir); + core.open_array(id); + let wal_end = records.iter().map(|r| r.header.lsn).max().unwrap_or(0); + core.core + .floors + .applied_prefix + .seed_replayed_through(Lsn::new(wal_end)); + core.core.replay_array_wal(records, 1, &TombstoneSet::new()); + core + } + + /// Tiles over every segment the array's manifest names. + fn segment_tiles(core: &Core, id: &ArrayId) -> u32 { + core.core + .array_engine + .store(id) + .expect("array open") + .manifest() + .segments + .iter() + .map(|s| s.tile_count) + .sum() + } + + /// The replay gate asks the stamp in the array's manifest, record by + /// record. A record below the highest named LSN that the stamp does not + /// name replays. + #[test] + fn array_replay_skips_exactly_the_records_the_stamp_names() { + let dir = tempfile::tempdir().expect("tempdir"); + let id = aid(); + let mut core = Core::open_at(dir.path()); + core.open_array(&id); + core.core + .floors + .applied_prefix + .observe_outcome_floor(Lsn::new(5)); + core.put(&id, 1, 1, 10, 10); + core.core.checkpoint_array_engines().expect("flush"); + + assert_eq!( + core.core + .array_engine + .store(&id) + .expect("open") + .manifest() + .replay, + ReplayStamp { + prefix: 5, + applied_above: vec![LsnRange { start: 10, end: 10 }], + } + ); + for (lsn, skips) in [(3, true), (5, true), (7, false), (10, true), (11, false)] { + assert_eq!( + core.core.array_replay_skips(&id, lsn), + skips, + "record at lsn {lsn}" + ); + } + } + + /// A threshold flush stamps the record whose write filled the memtable. + #[test] + fn a_threshold_flush_stamps_the_write_that_filled_the_memtable() { + let dir = tempfile::tempdir().expect("tempdir"); + let id = aid(); + let mut core = Core::open_at(dir.path()); + core.core.array_engine.set_flush_cell_threshold(1); + core.open_array(&id); + core.put(&id, 1, 1, 10, 40); + + let manifest = core.core.array_engine.store(&id).expect("open").manifest(); + assert_eq!(manifest.segments.len(), 1, "the put filled the memtable"); + assert!( + manifest.replay.skips(40), + "the threshold flush names the record it wrote: {:?}", + manifest.replay + ); + } + + /// Floor 10. B at lsn 30 applies, a checkpoint flushes it, then A at lsn + /// 20 applies. A restart replays A once and never re-applies B, and the + /// array reads as it did live. + #[test] + fn an_array_write_in_flight_at_a_checkpoint_replays_once() { + let dir = tempfile::tempdir().expect("tempdir"); + let id = aid(); + let records = [ + put_record(&id, 1, 1, 20, 1, 20), + put_record(&id, 9, 9, 30, 1, 30), + ]; + + let mut before = Core::open_at(dir.path()); + before.open_array(&id); + before + .core + .floors + .applied_prefix + .observe_outcome_floor(Lsn::new(10)); + before.put(&id, 9, 9, 30, 30); + before.core.checkpoint_array_engines().expect("flush B"); + before.put(&id, 1, 1, 20, 20); + let live = before.slice_all(&id); + assert_eq!(live, vec![(1, 1, 20), (9, 9, 30)]); + drop(before); + + let mut after = restart_and_replay(dir.path(), &id, &records); + assert_eq!(after.slice_all(&id), live, "replay equals live"); + after.core.checkpoint_array_engines().expect("flush A"); + assert_eq!( + segment_tiles(&after, &id), + 2, + "B's tile is in the first segment only; the second holds A alone" + ); + drop(after); + + // Every record is named now: a second restart applies nothing. + let mut again = restart_and_replay(dir.path(), &id, &records); + assert_eq!(again.slice_all(&id), live); + assert_eq!(segment_tiles(&again, &id), 2); + again.core.checkpoint_array_engines().expect("empty flush"); + assert_eq!( + segment_tiles(&again, &id), + 2, + "nothing replayed into memory" + ); + } + + /// An audit purge removes a superseded tile version from a segment. The + /// records that wrote it are named by the stamp, so replay does not bring + /// the version back. + #[test] + fn replay_does_not_rematerialise_a_purged_version() { + use nodedb_array::query::ceiling::CeilingResult; + + let dir = tempfile::tempdir().expect("tempdir"); + let id = aid(); + let records = [ + put_record(&id, 0, 0, 100, 100, 1), + put_record(&id, 0, 0, 200, 200, 2), + put_record(&id, 0, 0, 300, 300, 3), + ]; + let ceiling = |core: &Core, sys: i64| { + core.core + .array_engine + .store(&id) + .expect("open") + .ceiling_for_coord(&[CoordValue::Int64(0), CoordValue::Int64(0)], sys, None) + .expect("ceiling") + }; + + let mut before = Core::open_at(dir.path()); + before.open_array(&id); + for (v, lsn) in [(100, 1), (200, 2), (300, 3)] { + before.put_at(&id, 0, 0, v, v, lsn); + before + .core + .checkpoint_array_engines() + .expect("flush one version"); + } + let dropped = before + .core + .array_engine + .temporal_purge(TenantId::new(TID), DatabaseId::DEFAULT, &id.name, 250) + .expect("purge"); + assert!(dropped >= 1, "the purge dropped the superseded versions"); + assert!(matches!(ceiling(&before, 249), CeilingResult::NotFound)); + drop(before); + + let after = restart_and_replay(dir.path(), &id, &records); + assert!( + matches!(ceiling(&after, 249), CeilingResult::NotFound), + "a purged version came back from replay: {:?}", + ceiling(&after, 249) + ); + assert!(matches!(ceiling(&after, i64::MAX), CeilingResult::Live(_))); + } + + /// A restart opens only the arrays its WAL tail writes. The catalog still + /// holds every array, as the Control Plane seeds it from `_system.arrays` + /// at boot, so the first write to an array nothing reopened opens it from + /// the catalog instead of failing as unknown. + #[test] + fn the_first_write_after_a_restart_opens_its_array_from_the_catalog() { + use crate::control::array_catalog::ArrayCatalogEntry; + + let dir = tempfile::tempdir().expect("tempdir"); + let id = aid(); + + let mut before = Core::open_at(dir.path()); + before.open_array(&id); + before.put(&id, 1, 2, 30, 10); + before.core.checkpoint_array_engines().expect("flush"); + drop(before); + + let mut after = Core::open_at(dir.path()); + after + .core + .array_catalog + .write() + .expect("catalog lock") + .register(ArrayCatalogEntry { + array_id: id.clone(), + name: id.name.clone(), + schema_msgpack: zerompk::to_msgpack_vec(&schema()).expect("encode schema"), + schema_hash: SCHEMA_HASH, + created_at_ms: 0, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + }) + .expect("seed the catalog"); + assert!( + after.core.array_engine.store(&id).is_err(), + "nothing opened the array on this core yet" + ); + + after.put(&id, 3, 3, 50, 20); + assert_eq!( + after.slice_all(&id), + vec![(1, 2, 30), (3, 3, 50)], + "the write lands beside the cells the array held before the restart" ); } } diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs index 97186db1e..ce7457459 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs @@ -37,14 +37,14 @@ impl CheckpointFloors { // through, so this stays at zero until this process's own flush // succeeds. ts_durable_lsn: Lsn::ZERO, - // Vector, CRDT and spatial checkpoint files carry no core-level LSN - // to restore — a vector file holds its collection's replay gate, a - // Loro snapshot holds CRDT versions, and an R-tree file holds - // neither — so all three stay at zero until this process's own flush - // succeeds, and clamp truncation to zero until it does. + // CRDT and spatial checkpoint files carry no core-level LSN to + // restore — a Loro snapshot holds CRDT versions, and an R-tree file + // holds none — so both stay at zero until this process's own flush + // succeeds. Vector restores its LSN from its manifest at load. vector_durable_lsn: Lsn::ZERO, // No generation is published until one is loaded or written. vector_published_lsn: Lsn::ZERO, + sparse_vector_published_lsn: Lsn::ZERO, kv_published_lsn: Lsn::ZERO, columnar_published_lsn: Lsn::ZERO, crdt_durable_lsn: Lsn::ZERO, diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs index 3341f8c7d..b8a244e51 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs @@ -109,12 +109,9 @@ pub(in crate::data::executor) struct CheckpointFloors { /// the WAL (i.e. in `{data_dir}/vector-ckpt/`), advanced only by a fully /// successful `checkpoint_vector_indexes`. /// - /// Not restored at boot: `load_vector_checkpoints` restores the INDEXES, but - /// each file carries only its own collection's `checkpoint_wal_lsn`, which - /// is a per-collection replay gate and says nothing about what this CORE is - /// durable through. So this starts at zero and is first advanced by this - /// process's own successful flush; clamping to zero until then costs WAL - /// growth, never data. + /// Restored at boot from the live generation's manifest + /// (`load_vector_checkpoints`), and otherwise advanced by this process's + /// own successful flush. /// /// Scoped to what the rebuild cannot reach. /// `rebuild_vector_indexes_from_store` re-indexes every document of a @@ -138,14 +135,17 @@ pub(in crate::data::executor) struct CheckpointFloors { /// manifest. Same rule as `kv_published_lsn`. pub(in crate::data::executor) columnar_published_lsn: Lsn, - /// LSN of the newest vector checkpoint generation on disk, whichever - /// flush published it, restored at boot from the manifest. Restart - /// replay skips a vector record at or below the LSN its collection was - /// published with, so a committed record applied at or below this one - /// must be published again (`redo_apply::cover`). Never a truncation - /// floor: that is `vector_durable_lsn`. + /// Replay-stamp prefix of the newest vector checkpoint generation on + /// disk, whichever flush published it, restored at boot from the + /// manifest. Same rule as `kv_published_lsn`. Never a truncation floor: + /// that is `vector_durable_lsn`. pub(in crate::data::executor) vector_published_lsn: Lsn, + /// Replay-stamp prefix of the newest sparse-vector checkpoint generation + /// on disk, restored at boot from the manifest. Same rule as + /// `kv_published_lsn`. + pub(in crate::data::executor) sparse_vector_published_lsn: Lsn, + /// Highest LSN the CRDT engines are known to be durable through OUTSIDE the /// WAL (i.e. in `{data_dir}/crdt-ckpt/`), advanced only by a fully /// successful `checkpoint_crdt_engines`. @@ -186,7 +186,8 @@ pub(in crate::data::executor) struct CheckpointFloors { pub(in crate::data::executor) replay_floors: ReplayFloors, /// The node's outcome floor as this core last read it from a request, and - /// the records this core applied above it: the replay stamp every KV and - /// columnar checkpoint carries. + /// the records this core applied above it: the replay stamp every KV, + /// columnar, vector and sparse-vector checkpoint, array manifest and + /// timeseries partition carries. pub(in crate::data::executor) applied_prefix: AppliedPrefix, } diff --git a/nodedb/src/data/executor/core_loop/open.rs b/nodedb/src/data/executor/core_loop/open.rs index c816ceaee..be5e98c69 100644 --- a/nodedb/src/data/executor/core_loop/open.rs +++ b/nodedb/src/data/executor/core_loop/open.rs @@ -140,13 +140,13 @@ impl CoreLoop { columnar_engines: HashMap::new(), columnar_flushed_segments: HashMap::new(), columnar_flushed_surrogates: HashMap::new(), - ts_max_ingested_lsn: HashMap::new(), + ts_replay_stamps: HashMap::new(), + ts_replay_cursor: None, last_ts_ingest: None, ts_last_value_caches: HashMap::new(), ts_series_catalogs: HashMap::new(), ts_registries: HashMap::new(), ts_truncate_backlog: Vec::new(), - ts_truncate_floors: HashMap::new(), continuous_agg_mgr: crate::engine::timeseries::continuous_agg::ContinuousAggregateManager::new(), checkpoint_coordinator: crate::storage::checkpoint::CheckpointCoordinator::new( diff --git a/nodedb/src/data/executor/core_loop/state.rs b/nodedb/src/data/executor/core_loop/state.rs index 950151a83..fb4c10a6c 100644 --- a/nodedb/src/data/executor/core_loop/state.rs +++ b/nodedb/src/data/executor/core_loop/state.rs @@ -270,11 +270,18 @@ pub struct CoreLoop { pub(in crate::data::executor) columnar_flushed_surrogates: HashMap<(DatabaseId, TenantId, String), FlushedSurrogateTable>, - /// Per-collection max WAL LSN that has been ingested into the memtable. - /// Used by the WAL catch-up deduplication: if a catch-up record's LSN - /// is <= this value, the Data Plane skips it (already ingested). - /// Key: (DatabaseId, TenantId, collection). - pub(in crate::data::executor) ts_max_ingested_lsn: HashMap<(DatabaseId, TenantId, String), u64>, + /// Per-collection replay stamps: the records whose rows a partition + /// holds or a truncate removed, and the truncates that took effect. See + /// `timeseries_checkpoint::stamp`. Key: (DatabaseId, TenantId, collection). + pub(in crate::data::executor) ts_replay_stamps: HashMap< + (DatabaseId, TenantId, String), + crate::data::executor::timeseries_checkpoint::stamp::TsReplayStamp, + >, + + /// While restart replay runs the timeseries pass: the LSN through which it + /// has passed every record. A flush or truncate then stamps through it + /// instead of the core stamp. `None` outside that pass. + pub(in crate::data::executor) ts_replay_cursor: Option, /// Last time any timeseries ingest was processed on this core. /// Used by idle flush: if no ingest for 5 seconds, `maybe_run_maintenance` @@ -310,13 +317,6 @@ pub struct CoreLoop { /// removal failed at batch finalize; the maintenance tick retries them. pub(in crate::data::executor) ts_truncate_backlog: Vec, - /// WAL LSN of the last truncate applied to each timeseries collection. - /// An ingest carrying a `wal_lsn` at or below it was written before the - /// truncate (a WAL catch-up redelivery) and is refused, so a row the - /// truncate removed can never come back through the catch-up path. - /// Key: (DatabaseId, TenantId, collection). - pub(in crate::data::executor) ts_truncate_floors: HashMap<(DatabaseId, TenantId, String), u64>, - /// Continuous aggregate manager for this core. Fires on memtable flush. pub(in crate::data::executor) continuous_agg_mgr: crate::engine::timeseries::continuous_agg::ContinuousAggregateManager, diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index 1cfd7f074..70c64323e 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -103,6 +103,15 @@ impl CoreLoop { } else { task.state = TaskState::Running; let resp = self.execute(&task); + // Every engine's next checkpoint, flush or manifest stamps what + // this core applied, so a record applied here is noted whichever + // engine applied it. A refused record carries a durable abort + // marker instead, and a stamp never names it. + if resp.status == Status::Ok + && let Some(lsn) = task.wal_lsn() + { + self.floors.applied_prefix.note_applied(lsn); + } // A crash test kills the process here: one collection's logged // write applied, and its response never leaves the core. A task // with no WAL record never matches. diff --git a/nodedb/src/data/executor/dispatch/array/entry.rs b/nodedb/src/data/executor/dispatch/array/entry.rs index 702687a5f..0cb95f313 100644 --- a/nodedb/src/data/executor/dispatch/array/entry.rs +++ b/nodedb/src/data/executor/dispatch/array/entry.rs @@ -27,6 +27,19 @@ impl CoreLoop { if is_write && let Some(r) = self.check_engine_pressure(task, nodedb_mem::EngineId::Array) { return r; } + // A write, flush or compaction addresses an array this core may not + // have opened since it booted: restart replay opens only the arrays + // its WAL tail writes, and a read opens its own array. Opened from the + // catalog here, like a read, or the engine refuses the array as + // unknown. + if let ArrayOp::Put { array_id, .. } + | ArrayOp::Delete { array_id, .. } + | ArrayOp::Flush { array_id, .. } + | ArrayOp::Compact { array_id, .. } = op + && let Err(resp) = self.ensure_array_open(task, array_id) + { + return resp; + } match op { ArrayOp::OpenArray { array_id, @@ -138,12 +151,39 @@ impl CoreLoop { } } + /// [`Self::ensure_array_open`] when the catalog holds `array_id`. An array + /// with no catalog entry was dropped, and there is nothing to open. + pub(in crate::data::executor) fn open_array_if_cataloged( + &mut self, + task: &ExecutionTask, + array_id: &ArrayId, + ) -> Result<(), Response> { + let cataloged = self + .array_catalog + .read() + .map_err(|_| { + self.response_error( + task, + ErrorCode::Internal { + detail: "array catalog lock poisoned".to_string(), + }, + ) + })? + .lookup_by_id(array_id) + .is_some(); + if cataloged { + self.ensure_array_open(task, array_id)?; + } + Ok(()) + } + /// Idempotently open the array on this core, looking the schema up - /// from the shared `ArrayCatalogHandle`. Read handlers (Slice / - /// Project / Aggregate / Elementwise) call this at entry so that a - /// SQL read against a per-core engine that has not yet seen an - /// explicit `OpenArray` dispatch (e.g. the very first read after a - /// restart) auto-opens via the catalog instead of erroring. + /// from the shared `ArrayCatalogHandle`. Every read handler (Slice / + /// Project / Aggregate / Elementwise), and `dispatch_array` for every + /// write, flush and compaction, calls this at entry, so an op against a + /// per-core engine that has not yet seen an explicit `OpenArray` dispatch + /// (the first op after a restart) auto-opens via the catalog instead of + /// erroring. pub(in crate::data::executor) fn ensure_array_open( &mut self, task: &ExecutionTask, diff --git a/nodedb/src/data/executor/dispatch/array/mutate.rs b/nodedb/src/data/executor/dispatch/array/mutate.rs index 22eb04304..ec394d303 100644 --- a/nodedb/src/data/executor/dispatch/array/mutate.rs +++ b/nodedb/src/data/executor/dispatch/array/mutate.rs @@ -54,17 +54,8 @@ impl CoreLoop { }, ); } - // Advance the collection floor for this committed array write. The - // Control Plane allocated `wal_lsn` from the central WAL writer, so it - // is the committed write LSN in the same space as the core watermark. - if wal_lsn > 0 { - self.note_write_lsn( - task.request.database_id, - task.request.tenant_id, - &array_id.name, - None, - crate::types::Lsn::new(wal_lsn), - ); + if let Err(e) = self.settle_array_write(task, array_id, wal_lsn) { + return self.response_error(task, e); } // Advance HWM after durable write so the producer's array stream // frontier is tracked and reconstructable on replay. @@ -132,16 +123,8 @@ impl CoreLoop { }, ); } - // Advance the collection floor for this committed array delete (see - // `handle_array_put` for why `wal_lsn` is the committed write LSN). - if wal_lsn > 0 { - self.note_write_lsn( - task.request.database_id, - task.request.tenant_id, - &array_id.name, - None, - crate::types::Lsn::new(wal_lsn), - ); + if let Err(e) = self.settle_array_write(task, array_id, wal_lsn) { + return self.response_error(task, e); } if let Some(prov) = provenance { self.sync_commit(prov); @@ -155,10 +138,14 @@ impl CoreLoop { array_id: &ArrayId, wal_lsn: u64, ) -> Response { - // The Control Plane allocated `wal_lsn` from the central WAL - // writer; the engine just stamps it as the segment's flush - // watermark. - if let Err(e) = self.array_engine.flush(array_id, wal_lsn) { + // The flush record is applied once the segment it asks for is written, + // so the stamp that segment's manifest carries names it too. + if wal_lsn > 0 { + self.floors + .applied_prefix + .note_applied(crate::types::Lsn::new(wal_lsn)); + } + if let Err(e) = self.flush_array(array_id) { return self.response_error( task, ErrorCode::Internal { @@ -169,6 +156,38 @@ impl CoreLoop { encode_count_response(self, task, "flushed", 1) } + /// Account for a live cell write or delete that landed in the memtable. + /// + /// The Control Plane allocated `wal_lsn` from the central WAL writer, so + /// it is the committed write LSN in the same space as the core watermark. + /// It advances the collection floor and is noted as applied. The note + /// comes before the threshold flush, so the stamp that flush publishes + /// names this record. A write with no WAL record (`wal_lsn == 0`) notes + /// nothing. + /// + /// A failed threshold flush fails the response. The cells stay in the + /// memtable and the WAL record stays, so no write is lost; the next + /// checkpoint flush retries and clamps its reported LSN if it fails too. + fn settle_array_write( + &mut self, + task: &ExecutionTask, + array_id: &ArrayId, + wal_lsn: u64, + ) -> crate::Result<()> { + if wal_lsn > 0 { + let lsn = crate::types::Lsn::new(wal_lsn); + self.note_write_lsn( + task.request.database_id, + task.request.tenant_id, + &array_id.name, + None, + lsn, + ); + self.floors.applied_prefix.note_applied(lsn); + } + self.flush_array_if_full(array_id) + } + /// Stage, restore, and purge form the reversible physical side of DROP. pub(in crate::data::executor) fn handle_array_drop( &mut self, diff --git a/nodedb/src/data/executor/dispatch/meta_retention/handlers.rs b/nodedb/src/data/executor/dispatch/meta_retention/handlers.rs index 69b9bb5db..595ac9508 100644 --- a/nodedb/src/data/executor/dispatch/meta_retention/handlers.rs +++ b/nodedb/src/data/executor/dispatch/meta_retention/handlers.rs @@ -86,100 +86,6 @@ impl CoreLoop { } } - /// `MetaOp::EnforceTimeseriesRetention`: drop partitions older than - /// `max_age_ms`. Bitemporal collections use `max_system_ts` as the - /// retention axis; non-bitemporal partitions fall through to `max_ts`. - pub(in crate::data::executor::dispatch) fn meta_enforce_timeseries_retention( - &mut self, - task: &ExecutionTask, - collection: &str, - max_age_ms: i64, - ) -> Response { - let now_ms = now_ms(); - let cutoff = now_ms - max_age_ms; - let mut deleted = 0usize; - let ts_base = crate::data::executor::handlers::timeseries::paths::ts_collection_dir( - &self.data_dir, - task.request.database_id.as_u64(), - task.request.tenant_id.as_u64(), - collection, - ); - - let bitemporal = self.is_bitemporal( - task.request.database_id.as_u64(), - task.request.tenant_id.as_u64(), - collection, - ); - - let ts_key = ( - task.request.database_id, - task.request.tenant_id, - collection.to_string(), - ); - if let Some(registry) = self.ts_registries.get_mut(&ts_key) { - let expired: Vec<(i64, String)> = registry - .iter() - .filter(|(_, e)| { - let axis_ts = if bitemporal && e.meta.max_system_ts > 0 { - e.meta.max_system_ts - } else { - e.meta.max_ts - }; - axis_ts < cutoff - && e.meta.state != nodedb_types::timeseries::PartitionState::Deleted - }) - .map(|(&start, e)| (start, e.dir_name.clone())) - .collect(); - - for (start_ts, dir_name) in expired { - let partition_path = ts_base.join(&dir_name); - if partition_path.exists() - && let Err(e) = std::fs::remove_dir_all(&partition_path) - { - tracing::warn!( - path = %partition_path.display(), - error = %e, - "failed to delete expired partition" - ); - continue; - } - registry.mark_deleted(start_ts); - deleted += 1; - } - - if deleted > 0 { - tracing::info!( - collection, - deleted, - max_age_ms, - "retention enforcement complete" - ); - } - } - - if let Some(lvc) = self.ts_last_value_caches.get_mut(&ts_key) { - let evicted = lvc.evict_older_than(cutoff); - if !evicted.is_empty() { - tracing::debug!( - collection, - evicted = evicted.len(), - "evicted stale LVC entries" - ); - // Drop the same series from the catalog. Leaving them behind - // would let it accumulate every series the collection ever saw - // while the cache it exists to serve has already released them. - if let Some(catalog) = self.ts_series_catalogs.get_mut(&ts_key) { - for id in evicted { - catalog.forget(id); - } - } - } - } - - let payload = (deleted as u64).to_le_bytes().to_vec(); - self.response_with_payload(task, payload) - } - /// `MetaOp::ApplyContinuousAggRetention`. pub(in crate::data::executor::dispatch) fn meta_apply_continuous_agg_retention( &mut self, @@ -400,6 +306,15 @@ impl CoreLoop { cutoff_system_ms: i64, ) -> Response { let tenant = TenantId::new(tenant_id); + // An array no op has touched since boot is not open on this core, and + // the engine purges nothing from an array it has not opened. Opened + // from the catalog first; a dropped array has no entry and purges + // nothing. + let id = + nodedb_array::types::ArrayId::in_database(tenant, task.request.database_id, array_id); + if let Err(resp) = self.open_array_if_cataloged(task, &id) { + return resp; + } match self.array_engine.temporal_purge( tenant, task.request.database_id, @@ -480,7 +395,7 @@ impl CoreLoop { } } -fn now_ms() -> i64 { +pub(super) fn now_ms() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .unwrap_or_else(|e| { diff --git a/nodedb/src/data/executor/dispatch/meta_retention/mod.rs b/nodedb/src/data/executor/dispatch/meta_retention/mod.rs index 08b43d1e9..3fa60d641 100644 --- a/nodedb/src/data/executor/dispatch/meta_retention/mod.rs +++ b/nodedb/src/data/executor/dispatch/meta_retention/mod.rs @@ -5,11 +5,13 @@ //! Split off from `dispatch/other.rs` to respect the per-file size budget. //! Handlers are grouped by concern: //! -//! - `handlers` — retention + continuous-agg + last-value + edge-store / +//! - `handlers` — continuous-agg + last-value + edge-store / //! document-strict / timeseries-columnar temporal-purge handlers and the //! dispatch entry point [`CoreLoop::dispatch_meta_retention`]. //! - `columnar_plain` — plain columnar temporal-purge (segment-scanning, //! delete-bitmap marking). +//! - `ts_retention` — timeseries partition retention. pub mod columnar_plain; pub mod handlers; +pub mod ts_retention; diff --git a/nodedb/src/data/executor/dispatch/meta_retention/ts_retention.rs b/nodedb/src/data/executor/dispatch/meta_retention/ts_retention.rs new file mode 100644 index 000000000..429fa2b9a --- /dev/null +++ b/nodedb/src/data/executor/dispatch/meta_retention/ts_retention.rs @@ -0,0 +1,119 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `MetaOp::EnforceTimeseriesRetention`: drop a timeseries collection's +//! expired partitions. + +use crate::bridge::envelope::Response; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; + +use super::handlers::now_ms; + +impl CoreLoop { + /// `MetaOp::EnforceTimeseriesRetention`: drop partitions older than + /// `max_age_ms`. Bitemporal collections use `max_system_ts` as the + /// retention axis; non-bitemporal partitions fall through to `max_ts`. + pub(in crate::data::executor::dispatch) fn meta_enforce_timeseries_retention( + &mut self, + task: &ExecutionTask, + collection: &str, + max_age_ms: i64, + ) -> Response { + let now_ms = now_ms(); + let cutoff = now_ms - max_age_ms; + let mut deleted = 0usize; + let ts_base = crate::data::executor::handlers::timeseries::paths::ts_collection_dir( + &self.data_dir, + task.request.database_id.as_u64(), + task.request.tenant_id.as_u64(), + collection, + ); + + let bitemporal = self.is_bitemporal( + task.request.database_id.as_u64(), + task.request.tenant_id.as_u64(), + collection, + ); + + let ts_key = ( + task.request.database_id, + task.request.tenant_id, + collection.to_string(), + ); + let expired: Vec<(i64, String)> = self + .ts_registries + .get(&ts_key) + .map(|registry| { + registry + .iter() + .filter(|(_, e)| { + let axis_ts = if bitemporal && e.meta.max_system_ts > 0 { + e.meta.max_system_ts + } else { + e.meta.max_ts + }; + axis_ts < cutoff + && e.meta.state != nodedb_types::timeseries::PartitionState::Deleted + }) + .map(|(&start, e)| (start, e.dir_name.clone())) + .collect() + }) + .unwrap_or_default(); + // The collection directory's stamp must name every record of the + // partitions removed below, or replay re-appends the ones the WAL + // still holds. Nothing is removed until it does. + if !expired.is_empty() + && let Err(e) = self.persist_ts_collection_stamp(&ts_key) + { + return self.response_error(task, e); + } + if let Some(registry) = self.ts_registries.get_mut(&ts_key) { + for (start_ts, dir_name) in expired { + let partition_path = ts_base.join(&dir_name); + if partition_path.exists() + && let Err(e) = std::fs::remove_dir_all(&partition_path) + { + tracing::warn!( + path = %partition_path.display(), + error = %e, + "failed to delete expired partition" + ); + continue; + } + registry.mark_deleted(start_ts); + deleted += 1; + } + + if deleted > 0 { + tracing::info!( + collection, + deleted, + max_age_ms, + "retention enforcement complete" + ); + } + } + + if let Some(lvc) = self.ts_last_value_caches.get_mut(&ts_key) { + let evicted = lvc.evict_older_than(cutoff); + if !evicted.is_empty() { + tracing::debug!( + collection, + evicted = evicted.len(), + "evicted stale LVC entries" + ); + // Drop the same series from the catalog. Leaving them behind + // would let it accumulate every series the collection ever saw + // while the cache it exists to serve has already released them. + if let Some(catalog) = self.ts_series_catalogs.get_mut(&ts_key) { + for id in evicted { + catalog.forget(id); + } + } + } + } + + let payload = (deleted as u64).to_le_bytes().to_vec(); + self.response_with_payload(task, payload) + } +} diff --git a/nodedb/src/data/executor/dispatch/timeseries.rs b/nodedb/src/data/executor/dispatch/timeseries.rs index 62ce17c16..8b6b798d3 100644 --- a/nodedb/src/data/executor/dispatch/timeseries.rs +++ b/nodedb/src/data/executor/dispatch/timeseries.rs @@ -247,8 +247,9 @@ mod tests { }) } - /// A flushed partition must carry the last record's LSN, or boot replay's - /// dedup gate never fires and records replay on top of rows already on disk. + /// A flushed partition's stamp must name the record it holds, or boot + /// replay's skip gate never fires and the record replays on top of rows + /// already on disk. #[test] fn autocommit_ingest_stamps_the_envelope_lsn_on_the_partition_it_flushes() { let mut h = make_core(); @@ -269,17 +270,26 @@ mod tests { TenantId::new(TENANT), COLLECTION.to_string(), ); - let registry = h.core.ts_registries.get(&key).expect("registry"); - let stamps: Vec = registry - .iter() - .map(|(_, entry)| entry.meta.last_flushed_wal_lsn) - .collect(); - assert_eq!( - stamps, - vec![42], - "the flushed partition must carry the record's WAL LSN, or replay's \ - dedup gate never fires for it" + let stamp = &h.core.ts_replay_stamps.get(&key).expect("stamp").rows; + assert!( + stamp.skips(42) && !stamp.skips(41), + "the flushed partition's stamp must name the record's WAL LSN and \ + nothing else: {stamp:?}" ); + let registry = h.core.ts_registries.get(&key).expect("registry"); + let dirs: Vec = registry.iter().map(|(_, e)| e.dir_name.clone()).collect(); + assert_eq!(dirs.len(), 1); + let dir = crate::data::executor::handlers::timeseries::paths::ts_collection_dir( + &h.core.data_dir, + DatabaseId::DEFAULT.as_u64(), + TENANT, + COLLECTION, + ) + .join(&dirs[0]); + let on_disk = crate::data::executor::timeseries_checkpoint::stamp::read_ts_stamp(&dir) + .expect("read") + .expect("the partition carries its stamp"); + assert_eq!(&on_disk.rows, stamp); } /// A read policy governs a raw timeseries scan: only the rows it admits @@ -390,7 +400,7 @@ mod tests { } /// Nothing minted an LSN, so nothing may be claimed as flushed: a stamp - /// invented here would gate away records that are genuinely un-flushed. + /// naming a record here would gate away records that are un-flushed. #[test] fn an_ingest_with_no_lsn_anywhere_stamps_nothing() { let mut h = make_core(); @@ -407,7 +417,13 @@ mod tests { TenantId::new(TENANT), COLLECTION.to_string(), ); - assert_eq!(h.core.ts_max_ingested_lsn.get(&key), None); + h.core + .flush_ts_collection(TenantId::new(TENANT), DatabaseId::DEFAULT, COLLECTION, 0) + .expect("flush"); + assert_eq!( + h.core.ts_replay_stamps.get(&key).map(|s| s.rows.clone()), + Some(crate::types::replay_stamp::ReplayStamp::default()) + ); } /// A `RETURNING` ingest whose tags overflow the cardinality limit is diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs index d6e320003..900f77912 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs @@ -76,19 +76,16 @@ impl CoreLoop { .get(&index_key) .cloned() .unwrap_or_default(); - let skip = { - let coll = self - .vector_collections - .entry(index_key.clone()) - .or_insert_with(|| { - nodedb_vector::VectorCollection::new(*dim as usize, params) - }); - // Skip a straddling-segment record the restored - // checkpoint already absorbed (replay only; a - // live write always carries a higher, unseen - // LSN). - wal_lsn != 0 && wal_lsn <= coll.checkpoint_wal_lsn() - }; + // Skip a record the restored vector checkpoint holds. Its + // stamp names only records applied before the checkpoint, + // all of them replayed before the core serves a request, + // so a live write is never named. + let skip = wal_lsn != 0 && self.vector_replay_skips(wal_lsn); + self.vector_collections + .entry(index_key.clone()) + .or_insert_with(|| { + nodedb_vector::VectorCollection::new(*dim as usize, params) + }); if skip { continue; } @@ -100,7 +97,6 @@ impl CoreLoop { field_name, storage_key, floats, - wal_lsn, }) { inserts.push(delta); } @@ -160,17 +156,11 @@ impl CoreLoop { Self::vector_index_key(database_id, tid, collection, field_name); self.check_vector_width(&store_key, field_name, floats.len())?; let dim = floats.len(); - let skip = { - let coll = self - .vector_collections - .entry(store_key.clone()) - .or_insert_with(|| nodedb_vector::VectorCollection::new(dim, params)); - // Skip a straddling-segment record the restored - // checkpoint already absorbed (replay only; a - // live write always carries a higher, unseen - // LSN). - wal_lsn != 0 && wal_lsn <= coll.checkpoint_wal_lsn() - }; + // Same stamp gate as the strict arm above. + let skip = wal_lsn != 0 && self.vector_replay_skips(wal_lsn); + self.vector_collections + .entry(store_key.clone()) + .or_insert_with(|| nodedb_vector::VectorCollection::new(dim, params)); if skip { continue; } @@ -182,7 +172,6 @@ impl CoreLoop { field_name, storage_key, floats, - wal_lsn, }) { inserts.push(delta); } @@ -254,7 +243,6 @@ impl CoreLoop { field_name, storage_key, floats, - wal_lsn, } = params; let _ = self.remove_document_vector_index_field( database_id, @@ -265,7 +253,6 @@ impl CoreLoop { ); let coll = self.vector_collections.get_mut(&index_key)?; let vector_id = coll.insert_with_surrogate(floats, storage_key.surrogate()); - coll.note_checkpoint_lsn(wal_lsn); self.vector_doc_map.insert( ( index_key.0, @@ -572,4 +559,42 @@ mod tests { ); } } + + /// A document put the restored vector checkpoint's stamp names is not + /// indexed again on replay; one it does not name is indexed. + #[test] + fn a_put_the_vector_stamp_names_is_not_indexed_again() { + let mut harness = make_core(); + let core = &mut harness.core; + let (db_id, tid, collection) = (0u64, 1u64, "docs"); + register_bare_field(core, db_id, tid, collection); + core.floors + .replay_floors + .vector + .set(crate::types::replay_stamp::ReplayStamp { + prefix: 5, + applied_above: vec![crate::types::replay_stamp::LsnRange { start: 10, end: 10 }], + }); + + for (surrogate, wal_lsn) in [(1, 10), (2, 7)] { + let storage_key = crate::engine::document::store::StorageKey::for_surrogate( + Surrogate::new(surrogate), + ); + let doc = doc_with_vectors(&[("embedding", &[1.0, 0.0, 0.0])]); + core.apply_point_put_vector_indexes(VectorIndexPutParams { + database_id: db_id, + tid, + collection, + storage_key, + value: &doc, + wal_lsn, + }) + .expect("vector indexing must accept this fixture"); + } + assert_eq!( + physical_len(core, db_id, tid, collection, "embedding"), + 1, + "the put at lsn 10 is held by the checkpoint; the in-flight put at lsn 7 is indexed" + ); + } } diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs index 42577ee8d..1ab9f9db0 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/types.rs @@ -44,5 +44,4 @@ pub(super) struct VectorFieldInsert<'a> { pub(super) field_name: &'a str, pub(super) storage_key: crate::engine::document::store::StorageKey, pub(super) floats: Vec, - pub(super) wal_lsn: u64, } diff --git a/nodedb/src/data/executor/handlers/purge.rs b/nodedb/src/data/executor/handlers/purge.rs index aaa168499..82dfed25a 100644 --- a/nodedb/src/data/executor/handlers/purge.rs +++ b/nodedb/src/data/executor/handlers/purge.rs @@ -125,12 +125,10 @@ impl CoreLoop { self.columnar_memtable_mem .retain(|(_, t, _), _| *t != tid_key); self.ts_registries.retain(|(_, t, _), _| *t != tid_key); - self.ts_max_ingested_lsn - .retain(|(_, t, _), _| *t != tid_key); + self.ts_replay_stamps.retain(|(_, t, _), _| *t != tid_key); self.ts_last_value_caches .retain(|(_, t, _), _| *t != tid_key); self.ts_series_catalogs.retain(|(_, t, _), _| *t != tid_key); - self.ts_truncate_floors.retain(|(_, t, _), _| *t != tid_key); before - self.columnar_memtables.len() }; diff --git a/nodedb/src/data/executor/handlers/snapshot/create.rs b/nodedb/src/data/executor/handlers/snapshot/create.rs index 5ef5eb029..cd21be88a 100644 --- a/nodedb/src/data/executor/handlers/snapshot/create.rs +++ b/nodedb/src/data/executor/handlers/snapshot/create.rs @@ -327,6 +327,13 @@ impl CoreLoop { if !dir_entry.file_type()?.is_file() { continue; } + // A replay stamp names this core's WAL records; a restore + // elsewhere writes its own. + if name_str + == crate::data::executor::timeseries_checkpoint::stamp::TS_STAMP_FILE + { + continue; + } let bytes = std::fs::read(dir_entry.path())?; files.push((name_str.to_string(), bytes)); } diff --git a/nodedb/src/data/executor/handlers/snapshot/restore_segments.rs b/nodedb/src/data/executor/handlers/snapshot/restore_segments.rs index 14ea00ac5..db7293d68 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore_segments.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore_segments.rs @@ -9,6 +9,7 @@ use std::path::Path; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::timeseries_checkpoint::stamp::{TS_STAMP_FILE, write_ts_stamp}; use crate::types::TsFlushedCollectionBlob; /// The partition commit marker. A partition directory without it is treated as @@ -210,11 +211,21 @@ impl CoreLoop { // Segment files first; `partition.meta` is the commit point and // must not become visible before the columns it describes. for (filename, bytes) in &part_blob.files { - if filename.as_str() == PARTITION_META { + if filename.as_str() == PARTITION_META || filename.as_str() == TS_STAMP_FILE { continue; } durable_write_into(&staging_dir, filename, bytes)?; } + // A stamp names records of the core that wrote it, so the + // snapshot's never applies here. The restored rows come from + // no local record: the partition carries the collection's + // local stamp, which names nothing new. + let local_stamp = self + .ts_replay_stamps + .get(®_key) + .cloned() + .unwrap_or_default(); + write_ts_stamp(&staging_dir, &local_stamp)?; if let Some((_, bytes)) = part_blob .files .iter() @@ -472,8 +483,13 @@ mod tests { assert_eq!( names(&partition_dir), - vec![PARTITION_META.to_string(), "value.col".to_string()], - "the swapped-in partition must hold exactly the snapshot's files" + vec![ + PARTITION_META.to_string(), + TS_STAMP_FILE.to_string(), + "value.col".to_string() + ], + "the swapped-in partition must hold exactly the snapshot's files and \ + the local stamp" ); assert_eq!( std::fs::read(partition_dir.join("value.col")).expect("read col"), @@ -504,7 +520,11 @@ mod tests { let partition_dir = ts_dir(root.path()).join("ts-5_6"); assert_eq!( names(&partition_dir), - vec![PARTITION_META.to_string(), "value.col".to_string()] + vec![ + PARTITION_META.to_string(), + TS_STAMP_FILE.to_string(), + "value.col".to_string() + ] ); // No staging file may be left inside the published dir. assert!( diff --git a/nodedb/src/data/executor/handlers/timeseries/admission.rs b/nodedb/src/data/executor/handlers/timeseries/admission.rs index 27a8300c5..71a0c6869 100644 --- a/nodedb/src/data/executor/handlers/timeseries/admission.rs +++ b/nodedb/src/data/executor/handlers/timeseries/admission.rs @@ -7,13 +7,12 @@ //! caller must flush first — never partway through. //! //! That ordering is what makes the partition stamp honest. -//! `flush_ts_collection` labels the partition it writes with the collection's -//! max ingested WAL LSN, and boot replay skips every record at or below the -//! highest stamp it finds. A flush that fires between two rows of record L -//! writes a partition holding SOME of L but stamped L-1 (L is recorded only -//! once the record is fully ingested), so replay does not skip L and appends -//! every one of its rows a second time — on an append-only engine nothing -//! masks that. +//! `flush_ts_collection` labels the partition it writes with the records it +//! holds (`timeseries_checkpoint::stamp`), and boot replay skips exactly those. +//! A record is noted only once all of its rows landed. A flush that fires +//! between two rows of record L writes a partition holding SOME of L whose +//! stamp does not name L, so replay appends every one of L's rows a second +//! time — on an append-only engine nothing masks that. use std::collections::HashSet; diff --git a/nodedb/src/data/executor/handlers/timeseries/flush.rs b/nodedb/src/data/executor/handlers/timeseries/flush.rs index 0e5b13923..e3428aa06 100644 --- a/nodedb/src/data/executor/handlers/timeseries/flush.rs +++ b/nodedb/src/data/executor/handlers/timeseries/flush.rs @@ -8,6 +8,7 @@ use std::path::Path; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::timeseries_checkpoint::stamp::write_ts_stamp; use crate::engine::timeseries::columnar_segment::ColumnarSegmentWriter; use crate::engine::timeseries::partition_registry::PartitionRegistry; use crate::types::{DatabaseId, TenantId}; @@ -27,17 +28,28 @@ impl CoreLoop { /// /// These rows have no durable copy but the WAL, and the coordinated /// checkpoint calls this flush and then reports the LSN that authorises - /// deleting it. Draining first — as this did while its only callers were the - /// ingest-path thresholds and the idle timer — meant an encode or write - /// failure took the rows out of memory without putting them anywhere: the - /// scan stopped returning them for the rest of the process's life, and only - /// a restart's WAL replay brought them back. The partition is therefore + /// deleting it. An encode or write failure must not take the rows out of + /// memory without putting them anywhere. The partition is therefore /// written from a BORROW (`ColumnarMemtable::flush_view`) and the drain /// happens only once `write_partition` has returned `Ok` — its - /// `partition.meta` write being the commit point. Every failure path now + /// `partition.meta` write being the commit point. Every failure path /// leaves the memtable exactly as it was, so a failed flush costs a retry /// while the caller's clamped checkpoint LSN keeps the WAL records behind /// it. + /// + /// ## What the partition's stamp names + /// + /// The partition carries the collection's replay stamp as of this flush + /// (`timeseries_checkpoint::stamp`): the collection stamp plus every + /// record this core applied. Every row in the view belongs to one of + /// those records, and every one of those records has its rows here or in + /// an earlier partition. Restart replay skips exactly the named records. + /// + /// The ingest path resolves everything that could stop it mid-record + /// before the first row of a record goes in (its record-boundary admission + /// gate), and notes a record applied only once its rows all landed. A + /// flush from between two rows of a record would break the stamp in + /// either direction, so no caller may introduce one. pub(in crate::data::executor) fn flush_ts_collection( &mut self, tid: TenantId, @@ -63,28 +75,20 @@ impl CoreLoop { let writer = ColumnarSegmentWriter::new(&segment_dir); let view = mt.flush_view(); - // Use the max ingested WAL LSN for this collection so the partition - // records which WAL records have been flushed. Read before the write and - // never advanced by it. - // - // This is a collection-wide SCALAR, so the only state it can express is - // "every record at or below N is WHOLLY on disk" — and boot replay reads - // it exactly that way, skipping every record at or below the highest - // stamp it finds. "All of <= L-1 plus part of L" has no representation - // here, which is why the ingest path resolves everything that could stop - // it mid-record BEFORE the first row of a record goes in (its - // record-boundary admission gate) and stamps a record's LSN only once - // the record is fully ingested. Those two together are what make the - // claim this stamp rests on true by construction: every row in the view - // belongs to a record at or below it. - // - // A flush fired from between two rows of a record would break it in - // whichever direction it stamped — the predecessor's LSN duplicates the - // record on replay, the record's own LSN loses the rows not yet - // flushed — so no caller may introduce one. - let flush_wal_lsn = self.ts_max_ingested_lsn.get(&key).copied().unwrap_or(0); + let stamp = self.ts_flush_stamp(&key)?; + // Informational: `PartitionMeta` is shared with Lite, and replay reads + // the stamp file, never this LSN. + let flush_wal_lsn = stamp.rows.highest(); let partition_name = unique_partition_name(&segment_dir, view.min_ts, view.max_ts, flush_wal_lsn)?; + // The stamp lands before `partition.meta`, the partition's commit + // point, so a committed partition always carries it. + let partition_dir = segment_dir.join(&partition_name); + std::fs::create_dir_all(&partition_dir).map_err(|e| crate::Error::Storage { + engine: "timeseries".into(), + detail: format!("create partition dir {}: {e}", partition_dir.display()), + })?; + write_ts_stamp(&partition_dir, &stamp)?; let ts_kek = self.segment_keks.ts_segment_kek.as_ref(); let meta = writer .write_partition(&partition_name, &view, 0, flush_wal_lsn, ts_kek) @@ -104,6 +108,9 @@ impl CoreLoop { }); }; let drain = mt.drain(); + let replay_prefix = stamp.rows.prefix; + let applied_ranges = stamp.rows.applied_above.len(); + self.ts_replay_stamps.insert(key.clone(), stamp); // The memtable is empty, so drop its memory reservation. The // reservation tracks the full resident footprint, kept current by @@ -114,6 +121,8 @@ impl CoreLoop { tracing::info!( collection, rows = meta.row_count, + replay_prefix, + applied_ranges, "timeseries columnar flush complete" ); diff --git a/nodedb/src/data/executor/handlers/timeseries/group_flush.rs b/nodedb/src/data/executor/handlers/timeseries/group_flush.rs new file mode 100644 index 000000000..c4d324602 --- /dev/null +++ b/nodedb/src/data/executor/handlers/timeseries/group_flush.rs @@ -0,0 +1,494 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One committed record installs into a timeseries collection as one unit. +//! +//! A committed transaction's redo record can carry several timeseries +//! sub-records into one collection, all at the record's LSN. A partition +//! stamp names whole records only, so no flush may run between two of those +//! sub-records: the partition would hold part of the record, and restart +//! replay would either append that part again or lose the rest. +//! +//! So the replay arm measures the whole record before it writes any of it: +//! the rows and bytes each collection receives from every sub-record at that +//! LSN (`ts_record_loads`). A memtable that cannot take the record whole +//! flushes once, before the record's first row lands +//! (`CoreLoop::flush_ts_for_record`). No sub-record flushes. A record that +//! alone passes the memtable budget installs over it: it is committed, so it +//! must install, and the governor charge of the resident bytes reports the +//! overshoot as memory pressure. One flush after the record settles brings +//! the memtable back under budget, with a stamp that names the whole record +//! (`settle_redo_timeseries` live, `CoreLoop::settle_ts_replayed_record` in +//! restart replay). + +use std::collections::HashMap; + +use nodedb_wal::WalRecord; +use nodedb_wal::record::RecordType; + +use crate::data::executor::core_loop::CoreLoop; +use crate::engine::timeseries::ilp; +use crate::types::{DatabaseId, TenantId}; +use crate::wal::decode_batch_record; + +/// `(database, tenant, collection)`. +pub(in crate::data::executor) type TsKey = (DatabaseId, TenantId, String); + +/// What one record writes into one timeseries collection. +#[derive(Debug, Default, Clone, PartialEq, Eq)] +pub(in crate::data::executor) struct TsRecordLoad { + /// Rows the record writes. + pub rows: usize, + /// Payload bytes the rows arrive as: the measure the budget test adds to + /// the memtable's resident bytes. + pub bytes: usize, + /// Every ILP line the record writes, one per line, for the tag-headroom + /// test. Empty for a sample batch, which carries no tags. + pub ilp: String, +} + +/// Every timeseries collection the record at `records[start]` writes on +/// `core_id`, and the load it writes into each, summed over every +/// `TimeseriesBatch` at that record's LSN. `records` is in LSN order, so the +/// record's sub-records follow `start` directly. +pub(in crate::data::executor) fn ts_record_loads( + records: &[WalRecord], + start: usize, + num_cores: usize, + core_id: usize, +) -> HashMap { + let mut loads: HashMap = HashMap::new(); + let Some(first) = records.get(start) else { + return loads; + }; + let lsn = first.header.lsn; + for record in records[start..] + .iter() + .take_while(|record| record.header.lsn == lsn) + { + if RecordType::from_raw(record.logical_record_type()) != Some(RecordType::TimeseriesBatch) { + continue; + } + let target_core = if num_cores > 0 { + record.header.vshard_id as usize % num_cores + } else { + 0 + }; + if target_core != core_id { + continue; + } + let Ok(decoded) = decode_batch_record(&record.payload) else { + continue; + }; + if decoded.kind.as_deref() == Some("columnar") { + continue; + } + let key = ( + DatabaseId::new(record.header.database_id), + TenantId::new(record.header.tenant_id), + decoded.collection, + ); + let load = loads.entry(key).or_default(); + add_payload(load, &decoded.payload, decoded.format.as_deref()); + } + loads +} + +/// Add one sub-record's rows to `load`, decoded the way the replay arm +/// decodes them (`replay_timeseries_payload`). +fn add_payload(load: &mut TsRecordLoad, payload: &[u8], format: Option<&str>) { + load.bytes = load.bytes.saturating_add(payload.len()); + let lines: Vec = match format { + Some("ilp-msgpack") => zerompk::from_msgpack::>(payload).unwrap_or_default(), + Some("ilp") => utf8_lines(payload), + _ => { + if let Ok(batch) = + zerompk::from_msgpack::(payload) + { + load.rows = load.rows.saturating_add(batch.samples.len()); + return; + } + match format { + None => utf8_lines(payload), + // A JSON or msgpack row payload: its rows are counted by the + // ingest. Its bytes still count toward the budget test. + Some(_) => Vec::new(), + } + } + }; + for line in lines { + if line.trim().is_empty() { + continue; + } + load.rows += 1; + load.ilp.push_str(&line); + load.ilp.push('\n'); + } +} + +fn utf8_lines(payload: &[u8]) -> Vec { + std::str::from_utf8(payload) + .map(|text| text.lines().map(str::to_string).collect()) + .unwrap_or_default() +} + +/// The record the timeseries replay arm is applying: its LSN and the +/// collections it writes on this core. +#[derive(Debug)] +pub(in crate::data::executor) struct TsRecordInProgress { + lsn: u64, + keys: Vec, +} + +impl CoreLoop { + /// Called by the timeseries replay arm at `records[index]`, a record + /// routed to this core. When it starts a new record, restart replay + /// settles the previous one, and every writing pass flushes the + /// memtables that cannot take the new record whole. The validate pass of + /// a committed-redo apply writes nothing and passes through. + pub(in crate::data::executor) fn begin_ts_record( + &mut self, + records: &[WalRecord], + index: usize, + num_cores: usize, + current: &mut Option, + ) { + let Some(record) = records.get(index) else { + return; + }; + let lsn = record.header.lsn; + if self.validating_committed_redo() || current.as_ref().is_some_and(|c| c.lsn == lsn) { + return; + } + if let Some(previous) = current.take() { + self.end_ts_record(previous); + } + let loads = ts_record_loads(records, index, num_cores, self.core_id); + if let Err(e) = self.flush_ts_for_record(&loads) { + self.report_ts_record_flush_error(lsn, e); + } + *current = Some(TsRecordInProgress { + lsn, + keys: loads.into_keys().collect(), + }); + } + + /// The timeseries replay arm applied every sub-record of `record`. In + /// restart replay, flush each memtable it left at or over its budget. A + /// committed-redo install settles in `settle_redo_timeseries` instead. + pub(in crate::data::executor) fn end_ts_record(&mut self, record: TsRecordInProgress) { + if self.applying_committed_redo() { + return; + } + if let Err(e) = self.settle_ts_replayed_record(&record.keys) { + self.report_ts_record_flush_error(record.lsn, e); + } + } + + /// A flush around one record failed. A committed-redo install fails and + /// rolls back: it can retry. Restart replay must apply the committed + /// record all the same; its rows stay in the memtable over budget, and + /// the next checkpoint flush retries and clamps its reported LSN. + fn report_ts_record_flush_error(&mut self, lsn: u64, error: crate::Error) { + if let Some(scope) = self.redo_apply.scope.as_mut() { + scope.record_error(error); + return; + } + tracing::error!( + core = self.core_id, + lsn, + error = %error, + "timeseries flush around a replayed record failed; its rows stay in the \ + memtable until the next checkpoint flush" + ); + } + + /// Flush every memtable in `loads` that cannot take its record whole: + /// its resident bytes plus the record's reach the memtable budget, its + /// tag dictionaries lack room for the record's values, or the engine + /// budget is under pressure. An empty or absent memtable flushes nothing: + /// a flush cannot make room there. + /// + /// Runs before the record's first row lands, and before a committed-redo + /// install records any pre-image, so the undo never restores rows the + /// flush moved to disk. + fn flush_ts_for_record(&mut self, loads: &HashMap) -> crate::Result<()> { + let soft_limit = self.ts_tuning.memtable_budget_bytes; + for (key, load) in loads { + let Some(resident) = self + .columnar_memtables + .get(key) + .filter(|mt| !mt.is_empty()) + .map(|mt| mt.memory_bytes()) + else { + continue; + }; + let lines = ilp::parse_batch(&load.ilp) + .map(|batch| batch.into_lines()) + .unwrap_or_default(); + let fits = resident.saturating_add(load.bytes) < soft_limit + && !self.ts_ingest_needs_flush(key, &lines); + if fits { + continue; + } + let now_ms = self.ingest_now_ms(); + self.flush_ts_collection(key.1, key.0, &key.2, now_ms)?; + } + Ok(()) + } + + /// After restart replay applied a whole record: flush each collection it + /// wrote whose memtable is at or over its budget. The replay cursor has + /// passed the record, so the flush's stamp names all of it. + fn settle_ts_replayed_record(&mut self, keys: &[TsKey]) -> crate::Result<()> { + let soft_limit = self.ts_tuning.memtable_budget_bytes; + for key in keys { + let over = self + .columnar_memtables + .get(key) + .is_some_and(|mt| mt.memory_bytes() >= soft_limit); + if over { + let now_ms = self.ingest_now_ms(); + self.flush_ts_collection(key.1, key.0, &key.2, now_ms)?; + } + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_wal::record::WalRecordArgs; + + fn record(lsn: u64, collection: &str, lines: &[&str]) -> WalRecord { + let lines: Vec = lines.iter().map(|l| l.to_string()).collect(); + let resolved = zerompk::to_msgpack_vec(&lines).expect("encode lines"); + let payload = + crate::control::server::wal_dispatch::encode_timeseries_batch_payload_with_format( + collection, + &resolved, + None, + "ilp-msgpack", + ) + .expect("encode sub-record"); + WalRecord::new(WalRecordArgs { + record_type: RecordType::TimeseriesBatch as u32, + lsn, + tenant_id: 1, + vshard_id: 0, + database_id: 0, + payload, + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record") + } + + #[test] + fn a_record_load_sums_every_sub_record_at_its_lsn() { + let records = [ + record(7, "m", &["m,host=a value=1i 1"]), + record(7, "m", &["m,host=b value=2i 2", "m,host=c value=3i 3"]), + record(7, "n", &["n,host=a value=1i 1"]), + record(8, "m", &["m,host=d value=4i 4"]), + ]; + let loads = ts_record_loads(&records, 0, 1, 0); + let m = (DatabaseId::new(0), TenantId::new(1), "m".to_string()); + let n = (DatabaseId::new(0), TenantId::new(1), "n".to_string()); + assert_eq!(loads.len(), 2); + assert_eq!(loads[&m].rows, 3, "the lsn-8 record is not part of it"); + assert_eq!(loads[&n].rows, 1); + assert_eq!(loads[&m].ilp.lines().count(), 3); + assert!(loads[&m].bytes > 0); + } + + // ── One record installs as one unit ────────────────────────────────────── + + use crate::bridge::envelope::Status; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; + use crate::types::Lsn; + use crate::wal::{RedoRecord, RedoSubRecord}; + + const COLL: &str = "metrics"; + + fn key() -> TsKey { + (DatabaseId::DEFAULT, TenantId::new(1), COLL.to_string()) + } + + /// A committed redo record whose sub-records each ingest one ILP line + /// into `metrics`. + fn redo(lines: &[&str]) -> Vec { + let ops = lines + .iter() + .map(|line| { + let resolved = + zerompk::to_msgpack_vec(&vec![line.to_string()]).expect("encode lines"); + RedoSubRecord { + record_type: RecordType::TimeseriesBatch as u32, + payload: + crate::control::server::wal_dispatch::encode_timeseries_batch_payload_with_format( + COLL, + &resolved, + None, + "ilp-msgpack", + ) + .expect("encode sub-record"), + } + }) + .collect(); + RedoRecord { + version: 1, + ops, + calvin_stamp: None, + } + .to_bytes() + .expect("encode redo") + } + + fn redo_wal_record(lsn: u64, redo: &[u8]) -> WalRecord { + WalRecord::new(WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn, + tenant_id: 1, + vshard_id: 0, + database_id: DatabaseId::DEFAULT.as_u64(), + payload: redo.to_vec(), + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record") + } + + fn install(core: &mut CoreLoop, lsn: u64, redo: &[u8]) { + let mut task = make_default_task(); + task.wal_lsn = Some(Lsn::new(lsn)); + let response = core.install_committed_redo( + &task, + 1, + CommittedRedo { + redo, + collections: &[COLL.to_string()], + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + } + + /// Row counts of the collection's partitions, sorted. + fn partition_rows(core: &CoreLoop) -> Vec { + let mut rows: Vec = core + .ts_registries + .get(&key()) + .map(|registry| registry.iter().map(|(_, e)| e.meta.row_count).collect()) + .unwrap_or_default(); + rows.sort_unstable(); + rows + } + + fn memtable_rows(core: &CoreLoop) -> u64 { + core.columnar_memtables + .get(&key()) + .map_or(0, |mt| mt.row_count()) + } + + /// A core and the bridge ends that must outlive it. + type Booted = ( + CoreLoop, + nodedb_bridge::buffer::Producer, + nodedb_bridge::buffer::Consumer, + ); + + /// Boot over `dir` with a one-byte memtable budget, and replay + /// `records` as restart replay does. + fn restart_tiny_budget(dir: &std::path::Path, records: &[WalRecord]) -> Booted { + let (mut core, req, resp) = make_core_with_dir(dir); + core.ts_tuning.memtable_budget_bytes = 1; + core.load_ts_registries().expect("load"); + let wal_end = records.iter().map(|r| r.header.lsn).max().unwrap_or(0); + core.floors + .applied_prefix + .seed_replayed_through(Lsn::new(wal_end)); + core.replay_transaction_redo_wal(records, 1, &nodedb_wal::TombstoneSet::new()) + .expect("replay"); + (core, req, resp) + } + + const SEED: &str = "metrics,host=s value=1 1000000000"; + const FIRST: &str = "metrics,host=a value=2 2000000000"; + const SECOND: &str = "metrics,host=b value=3 3000000000"; + + /// Live: the memtable holds a row and is over its budget once the first + /// of two sub-records lands. The record flushes once before it and once + /// after it, never between: one partition holds the seed, one holds the + /// whole record. A restart applies neither again. + #[test] + fn a_live_install_never_flushes_between_its_sub_records() { + let dir = tempfile::tempdir().expect("tempdir"); + let seed = redo(&[SEED]); + let pair = redo(&[FIRST, SECOND]); + { + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + install(&mut core, 5, &seed); + assert_eq!(memtable_rows(&core), 1); + core.ts_tuning.memtable_budget_bytes = 1; + install(&mut core, 20, &pair); + assert_eq!( + partition_rows(&core), + vec![1, 2], + "one flush before the record and one after it" + ); + assert_eq!(memtable_rows(&core), 0); + let stamp = &core.ts_replay_stamps.get(&key()).expect("stamp").rows; + assert!(stamp.skips(20), "the settle flush names the whole record"); + } + + let records = [redo_wal_record(5, &seed), redo_wal_record(20, &pair)]; + let (core, _req, _resp) = restart_tiny_budget(dir.path(), &records); + assert_eq!(partition_rows(&core), vec![1, 2]); + assert_eq!( + memtable_rows(&core), + 0, + "every record is named: none replays" + ); + } + + /// Restart replay: the record's sub-records are only in the WAL. The + /// memtable is over its budget once the first one lands, yet the second + /// lands in the same generation, and the record flushes whole. A second + /// restart applies nothing again. + #[test] + fn restart_replay_never_flushes_between_a_records_sub_records() { + let dir = tempfile::tempdir().expect("tempdir"); + let seed = redo(&[SEED]); + let pair = redo(&[FIRST, SECOND]); + { + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + install(&mut core, 5, &seed); + install(&mut core, 20, &pair); + assert_eq!(memtable_rows(&core), 3); + assert!( + partition_rows(&core).is_empty(), + "nothing flushed before the crash" + ); + } + + let records = [redo_wal_record(5, &seed), redo_wal_record(20, &pair)]; + let (core, req, resp) = restart_tiny_budget(dir.path(), &records); + assert_eq!( + partition_rows(&core), + vec![1, 2], + "the seed record and the pair each flush whole" + ); + assert_eq!(memtable_rows(&core), 0); + drop((core, req, resp)); + + let (again, _req, _resp) = restart_tiny_budget(dir.path(), &records); + assert_eq!(partition_rows(&again), vec![1, 2]); + assert_eq!( + memtable_rows(&again), + 0, + "both records applied exactly once" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/timeseries/ingest.rs b/nodedb/src/data/executor/handlers/timeseries/ingest.rs index 63a0a408b..acc3c4faa 100644 --- a/nodedb/src/data/executor/handlers/timeseries/ingest.rs +++ b/nodedb/src/data/executor/handlers/timeseries/ingest.rs @@ -206,16 +206,8 @@ impl CoreLoop { ); } - if mode == TimeseriesApplyMode::RedoInstall - && let Err(error) = self.prepare_redo_ts_ingest( - task.request.database_id, - tid, - collection, - &lines, - now_ms, - ) - { - return self.response_error(task, error); + if mode == TimeseriesApplyMode::RedoInstall { + self.record_redo_ts_pre_image(task.request.database_id, tid, collection); } let bitemporal = @@ -250,32 +242,21 @@ impl CoreLoop { // The WAL has already committed this record, so the admission gate // resolves every possible mid-record stop before the first row lands. + // A replayed or installed sub-record never flushes here: the replay + // arm flushed before the record's first sub-record (`group_flush`), + // and a flush between two sub-records would split the record. let soft_limit = self.ts_tuning.memtable_budget_bytes; - if self.ts_ingest_needs_flush(&key, &lines) { - // A redo install flushed before it took its pre-image. A flush - // now would drain rows that pre-image holds, so the install - // fails and rolls back instead. - if mode == TimeseriesApplyMode::RedoInstall { - return self.response_error( - task, - ErrorCode::Internal { - detail: format!( - "'{collection}' still needs a flush after the flush that preceded \ - its committed-redo install" - ), - }, - ); - } - if let Err(e) = + if mode == TimeseriesApplyMode::Immediate + && self.ts_ingest_needs_flush(&key, &lines) + && let Err(e) = self.flush_ts_collection(tid, task.request.database_id, collection, now_ms) - { - return self.response_error( - task, - ErrorCode::Internal { - detail: format!("pre-ingest ts flush failed: {e}"), - }, - ); - } + { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("pre-ingest ts flush failed: {e}"), + }, + ); } let Some(mt) = self.columnar_memtables.get_mut(&key) else { @@ -305,6 +286,15 @@ impl CoreLoop { }); let accepted = outcome.accepted; let rejected = outcome.rejected; + // The record's rows landed, whatever the response below says. A + // flush from here on names the record; the pre-ingest flush above ran + // before it and does not. A committed-redo install is noted by the + // apply once the whole record installed. + if mode != TimeseriesApplyMode::RedoInstall + && let Some(lsn) = wal_lsn + { + self.note_ts_record_applied(lsn); + } if rejected > 0 { tracing::warn!( @@ -358,13 +348,6 @@ impl CoreLoop { None => Vec::new(), }; - if accepted > 0 - && let Some(lsn) = wal_lsn - { - let entry = self.ts_max_ingested_lsn.entry(key.clone()).or_insert(0); - *entry = (*entry).max(lsn); - } - let Some(mt) = self.columnar_memtables.get(&key) else { return self.response_error( task, @@ -374,8 +357,9 @@ impl CoreLoop { ); }; let needs_flush = mt.memory_bytes() >= soft_limit; - if mode == TimeseriesApplyMode::Immediate { - if needs_flush + if mode != TimeseriesApplyMode::RedoInstall { + if mode == TimeseriesApplyMode::Immediate + && needs_flush && let Err(e) = self.flush_ts_collection(tid, task.request.database_id, collection, now_ms) { @@ -388,7 +372,7 @@ impl CoreLoop { } if accepted > 0 { - // no-determinism: Instant::now runs only for the operational idle/checkpoint timer in Immediate mode and is skipped in Calvin staged apply. + // no-determinism: Instant::now runs only for the operational idle/checkpoint timer, outside a committed-redo install. self.last_ts_ingest = Some(std::time::Instant::now()); } diff --git a/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs b/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs index 6a65d9a41..1afac0965 100644 --- a/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs +++ b/nodedb/src/data/executor/handlers/timeseries/ingest_dispatch.rs @@ -14,12 +14,20 @@ use crate::data::executor::task::ExecutionTask; /// Side-effect policy for a timeseries ingest. #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub(in crate::data::executor) enum TimeseriesApplyMode { + /// A live write: one record, one sub-record. The memtable flushes before + /// the rows land when it cannot take them whole, and after they land + /// when it is over its budget. Immediate, - /// The install pass of a committed redo record. The memtable flushes the - /// rows it already holds when it has no room, then the ingest records - /// the pre-image its undo restores. No flush, budget recharge or timer - /// update runs after the rows land: the apply settles those once the - /// whole record installed (`settle_redo_timeseries`). + /// Restart replay of one sub-record. The replay arm flushed before the + /// record's first sub-record when the memtable could not take the whole + /// record, and settles it once the last sub-record landed + /// (`group_flush`). No flush runs here: it would split the record. + Replay, + /// The install pass of a committed redo record. The replay arm flushed + /// before the record's first sub-record, as in `Replay`; the ingest + /// records the pre-image its undo restores. No flush, budget recharge or + /// timer update runs after the rows land: the apply settles those once + /// the whole record installed (`settle_redo_timeseries`). RedoInstall, } @@ -113,37 +121,17 @@ impl CoreLoop { } let key = (task.request.database_id, tid, collection.to_string()); - let already_flushed = if let Some(lsn) = wal_lsn - && let Some(registry) = self.ts_registries.get(&key) - { - let max_flushed = registry - .iter() - .map(|(_, e)| e.meta.last_flushed_wal_lsn) - .max() - .unwrap_or(0); - max_flushed > 0 && lsn <= max_flushed - } else { - false - }; - // A record written before a later truncate describes rows the - // truncate removed: the same "nothing to write" answer as a record - // already on disk. Strictly before: a record at the truncate's own LSN - // is a sibling sub-record of the same transaction redo group, applied - // in the order the transaction wrote it. - let already_flushed = already_flushed - || wal_lsn.is_some_and(|lsn| { - self.ts_truncate_floors - .get(&key) - .is_some_and(|floor| lsn < *floor) - }); - // Both tests are restart-replay watermarks. A committed-redo apply - // installs its record once, whatever a live flush or truncate stamped - // since the record's LSN was minted. - let already_flushed = self.replay_watermark_skips(already_flushed); + // A record the collection's replay stamp names is already in a + // partition, or a truncate removed its rows: nothing to write. Every + // other record applies, including one below the highest named LSN, + // which was still on its way when that flush or truncate ran. A + // committed-redo apply installs its record once, whatever the stamp + // names (`replay_watermark_skips`). + let already_flushed = wal_lsn.is_some_and(|lsn| self.ts_replay_skips(&key, lsn)); if already_flushed { if let Some(prov) = provenance - && mode == TimeseriesApplyMode::Immediate + && mode != TimeseriesApplyMode::RedoInstall { self.sync_commit(prov); let applied_seq = self.sync_hwm_value(prov.producer_id, prov.stream_id); diff --git a/nodedb/src/data/executor/handlers/timeseries/mod.rs b/nodedb/src/data/executor/handlers/timeseries/mod.rs index 1b87c7f27..920e2bf9c 100644 --- a/nodedb/src/data/executor/handlers/timeseries/mod.rs +++ b/nodedb/src/data/executor/handlers/timeseries/mod.rs @@ -6,6 +6,7 @@ mod admission; pub mod aggregate; pub mod encode; pub mod flush; +mod group_flush; pub mod ingest; mod ingest_dispatch; pub mod ingest_formats; diff --git a/nodedb/src/data/executor/handlers/timeseries/redo_ingest.rs b/nodedb/src/data/executor/handlers/timeseries/redo_ingest.rs index 8d7ef7d1a..e2396bf19 100644 --- a/nodedb/src/data/executor/handlers/timeseries/redo_ingest.rs +++ b/nodedb/src/data/executor/handlers/timeseries/redo_ingest.rs @@ -1,13 +1,13 @@ // SPDX-License-Identifier: BUSL-1.1 //! The admission test every ILP ingest runs before its rows land, and the -//! preparation a committed-redo install adds to it. +//! pre-image a committed-redo install records. //! -//! An install flushes the rows the memtable already holds when it has no -//! room for the new ones, and only then records its pre-image. The undo -//! therefore restores a memtable that holds nothing the flush moved to disk. +//! An install never flushes per sub-record: the replay arm flushed the +//! memtable before the record's first sub-record when it could not take the +//! whole record (`group_flush`). The pre-image is recorded after that flush, +//! so the undo restores a memtable that holds nothing the flush moved to disk. -use crate::bridge::envelope::ErrorCode; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::undo::UndoEntry; use crate::engine::timeseries::ilp; @@ -40,25 +40,16 @@ impl CoreLoop { }) } - /// Prepare the install of `lines` into `collection`: flush the memtable - /// when it has no room, then record the pre-image the undo restores. - pub(super) fn prepare_redo_ts_ingest( + /// Record the pre-image the undo of one installed sub-record into + /// `collection` restores. + pub(super) fn record_redo_ts_pre_image( &mut self, database_id: DatabaseId, tid: TenantId, collection: &str, - lines: &[ilp::IlpLine<'_>], - now_ms: i64, - ) -> Result<(), ErrorCode> { + ) { let key = (database_id, tid, collection.to_string()); - if self.ts_ingest_needs_flush(&key, lines) { - self.flush_ts_collection(tid, database_id, collection, now_ms) - .map_err(|e| ErrorCode::Internal { - detail: format!("pre-install ts flush of '{collection}' failed: {e}"), - })?; - } let undo = self.capture_timeseries_ingest_undo(&key); self.record_redo_undo([UndoEntry::TimeseriesIngest(undo)]); - Ok(()) } } diff --git a/nodedb/src/data/executor/handlers/timeseries/truncate.rs b/nodedb/src/data/executor/handlers/timeseries/truncate.rs index d9830fd07..b8741913f 100644 --- a/nodedb/src/data/executor/handlers/timeseries/truncate.rs +++ b/nodedb/src/data/executor/handlers/timeseries/truncate.rs @@ -2,16 +2,29 @@ //! `TimeseriesOp::Truncate`: remove every row of a timeseries collection. //! -//! In-memory state (memtable, its reservation, partition registry, ingest -//! watermark, last-value cache, series catalog) is moved out whole. The -//! partition directory is renamed to an aside name in one atomic step +//! In-memory state (memtable, its reservation, partition registry, +//! last-value cache, series catalog) is moved out whole. The partition +//! directory is renamed to an aside name in one atomic step //! (`truncating_dir_name`), so a scan can never observe half a directory. //! Autocommit removes the aside directory at once; inside a transaction //! batch it stays until the batch finalizes (`finalize_timeseries_truncates`) //! or the undo renames it back (`apply_undo_timeseries_truncate`). A crash //! between rename and removal leaves an aside directory that boot removes -//! (`remove_truncating_leftovers`): the truncate's WAL record precedes the -//! rename, so replay re-applies it either way. +//! (`remove_truncating_leftovers`). +//! +//! ## The stamp a truncate leaves +//! +//! After the rename the truncate creates a fresh collection directory holding +//! only the collection's replay stamp (`timeseries_checkpoint::stamp`). Its +//! rows name every record the truncate removed: the collection stamp plus +//! every record this core applied. Its truncates name this truncate. Restart +//! replay then never re-applies the truncate, and a record below its LSN that +//! applied after it replays and survives. +//! +//! The stamp is written after the rename. A crash between the two leaves no +//! stamp naming the truncate, so replay re-applies it; no later write applied +//! in between, because this core ran nothing else. A stamp that cannot be +//! written puts the directory back and fails the truncate. use std::path::{Path, PathBuf}; @@ -21,6 +34,10 @@ use crate::bridge::envelope::Response; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::undo::{TimeseriesTruncateUndo, UndoEntry}; use crate::data::executor::task::ExecutionTask; +use crate::data::executor::timeseries_checkpoint::stamp::{ + TsCollectionKey, TsReplayStamp, write_ts_stamp, +}; +use crate::types::replay_stamp::ReplayStamp; /// Suffix marking a partition directory renamed aside by a truncate. pub(in crate::data::executor) const TRUNCATING_MARKER: &str = ".truncating-"; @@ -85,23 +102,28 @@ impl CoreLoop { if let Err(e) = rename_aside(&original, &moved) { return self.response_error(task, e); } - Some((original, moved)) + Some(moved) } else { None }; + let stamp = match self.write_truncate_stamp(&key, &original, task.wal_lsn()) { + Ok(stamp) => stamp, + Err(e) => { + if let Err(restore) = restore_after_failed_stamp(&original, moved_dir.as_ref()) { + return self.response_error(task, restore); + } + return self.response_error(task, e); + } + }; let memtable = self.columnar_memtables.remove(&key); let memtable_mem = self.columnar_memtable_mem.remove(&key); let registry = self.ts_registries.remove(&key); - let max_ingested_lsn = self.ts_max_ingested_lsn.remove(&key); let last_value_cache = self.ts_last_value_caches.remove(&key); let series_catalog = self.ts_series_catalogs.remove(&key); let truncated = memtable.as_ref().map(|m| m.row_count()).unwrap_or(0) + registry.as_ref().map(|r| r.total_row_count()).unwrap_or(0); - let truncate_floor_before = match task.wal_lsn() { - Some(lsn) => self.ts_truncate_floors.insert(key.clone(), lsn.as_u64()), - None => self.ts_truncate_floors.get(&key).copied(), - }; + let replay_stamp = self.ts_replay_stamps.insert(key.clone(), stamp); self.continuous_agg_mgr .reset_for_source(db.as_u64(), collection); @@ -110,18 +132,18 @@ impl CoreLoop { Some(log) => log.push(UndoEntry::TimeseriesTruncate(Box::new( TimeseriesTruncateUndo { collection_key: key, + original_dir: original, moved_dir, memtable, memtable_mem, registry, - max_ingested_lsn, last_value_cache, series_catalog, - truncate_floor: truncate_floor_before, + replay_stamp, }, ))), None => { - if let Some((_, moved)) = &moved_dir + if let Some(moved) = &moved_dir && let Err(e) = remove_dir_tree(moved) { return self.response_error(task, e); @@ -150,7 +172,7 @@ impl CoreLoop { let UndoEntry::TimeseriesTruncate(undo) = entry else { continue; }; - let Some((_, moved)) = &undo.moved_dir else { + let Some(moved) = &undo.moved_dir else { continue; }; if let Err(e) = remove_dir_tree(moved) { @@ -165,6 +187,27 @@ impl CoreLoop { } } + /// Create `original` afresh and write into it the replay stamp this + /// truncate leaves: every record this core applied, and the truncate at + /// `lsn`. Returns the stamp. + fn write_truncate_stamp( + &self, + key: &TsCollectionKey, + original: &Path, + lsn: Option, + ) -> crate::Result { + let mut stamp = self.ts_flush_stamp(key)?; + if let Some(lsn) = lsn { + stamp.truncates = stamp.truncates.union(&ReplayStamp::naming(lsn.as_u64())); + } + std::fs::create_dir_all(original).map_err(|e| crate::Error::Storage { + engine: "timeseries".into(), + detail: format!("create collection directory {}: {e}", original.display()), + })?; + write_ts_stamp(original, &stamp)?; + Ok(stamp) + } + /// Retry every aside directory removal a batch finalize deferred. /// Returns how many are still pending. pub(in crate::data::executor) fn retry_ts_truncate_backlog(&mut self) -> usize { @@ -184,6 +227,23 @@ impl CoreLoop { } } +/// Put the collection back as it was before a truncate whose stamp could not +/// be written: remove the fresh directory, then rename the aside one back. +fn restore_after_failed_stamp(original: &Path, moved: Option<&PathBuf>) -> crate::Result<()> { + remove_dir_tree(original)?; + if let Some(moved) = moved { + std::fs::rename(moved, original).map_err(|e| crate::Error::Storage { + engine: "timeseries".into(), + detail: format!( + "rename partition directory {} back to {}: {e}", + moved.display(), + original.display() + ), + })?; + } + Ok(()) +} + /// Rename `original` to `moved`. An aside directory already at `moved` is a /// leftover of an earlier truncate at the same LSN (a replay) and is removed /// first; the truncate is committed either way. @@ -235,6 +295,7 @@ mod tests { use super::*; use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::data::executor::timeseries_checkpoint::stamp::TS_STAMP_FILE; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use nodedb_physical::physical_plan::TimeseriesOp; @@ -298,6 +359,16 @@ mod tests { .unwrap_or(0) } + /// The entries of `dir`, by name, sorted. + fn entries(dir: &Path) -> Vec { + let mut names: Vec = std::fs::read_dir(dir) + .expect("read dir") + .map(|e| e.expect("entry").file_name().to_string_lossy().into_owned()) + .collect(); + names.sort(); + names + } + fn decode_truncated(resp: &crate::bridge::envelope::Response) -> u64 { let v = nodedb_types::value_from_msgpack(resp.payload.as_bytes()).expect("payload"); v.as_object() @@ -307,7 +378,8 @@ mod tests { } /// Autocommit: memtable rows and a flushed partition both go, the - /// directory is removed, and the count covers both. + /// directory is replaced by one holding only the truncate's stamp, and + /// the count covers both. #[test] fn autocommit_truncate_removes_memtable_partitions_and_directory() { let dir = tempfile::tempdir().expect("tempdir"); @@ -336,7 +408,20 @@ mod tests { assert_eq!(decode_truncated(&resp), 3); assert_eq!(memtable_rows(&core), 0); assert_eq!(partition_rows(&core), 0); - assert!(!ts_dir.exists(), "the partition directory is gone"); + assert_eq!( + entries(&ts_dir), + vec![TS_STAMP_FILE.to_string()], + "the partitions are gone; the fresh directory holds the stamp" + ); + let stamp = core.ts_replay_stamps.get(&key()).expect("stamp"); + assert!(stamp.rows.skips(1) && stamp.rows.skips(2), "{stamp:?}"); + assert!(stamp.truncates.skips(3) && !stamp.truncates.skips(2)); + assert_eq!( + crate::data::executor::timeseries_checkpoint::stamp::read_ts_stamp(&ts_dir) + .expect("read") + .as_ref(), + Some(stamp) + ); assert!( !core.columnar_memtable_mem.contains_key(&key()), "the memtable reservation is released" @@ -351,11 +436,11 @@ mod tests { assert_eq!(memtable_rows(&core), 1); } - /// A catch-up redelivery of a record written before the truncate is - /// refused: the truncate's LSN floors every later ingest carrying an - /// older LSN. + /// A catch-up redelivery of a record the truncate removed is refused: the + /// truncate's stamp names it. A record below the truncate's LSN that was + /// still on its way when the truncate applied is not named, and applies. #[test] - fn ingest_below_the_truncate_lsn_is_refused_after_the_truncate() { + fn only_a_record_the_truncate_removed_is_refused_after_it() { let dir = tempfile::tempdir().expect("tempdir"); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); assert_eq!( @@ -366,19 +451,28 @@ mod tests { assert_eq!(resp.status, Status::Ok); assert_eq!( - ingest(&mut core, "metrics,host=a value=1i\n", 4), + ingest(&mut core, "metrics,host=a value=1i\n", 1), Status::Ok ); assert_eq!( memtable_rows(&core), 0, - "a pre-truncate record writes nothing" + "the redelivered record the truncate removed writes nothing" + ); + assert_eq!( + ingest(&mut core, "metrics,host=a value=1i\n", 4), + Status::Ok + ); + assert_eq!( + memtable_rows(&core), + 1, + "a record in flight at the truncate applies after it" ); assert_eq!( ingest(&mut core, "metrics,host=a value=1i\n", 6), Status::Ok ); - assert_eq!(memtable_rows(&core), 1, "a post-truncate record is stored"); + assert_eq!(memtable_rows(&core), 2, "a post-truncate record is stored"); } /// Inside a batch the directory is renamed aside and every in-memory @@ -398,25 +492,34 @@ mod tests { Status::Ok ); let ts_dir = super::super::paths::ts_collection_dir(dir.path(), 0, TENANT, COLLECTION); + let stamp_before = core.ts_replay_stamps.get(&key()).cloned(); let mut undo = Vec::new(); let resp = core.execute_timeseries_truncate(&task_at(3), COLLECTION, Some(&mut undo)); assert_eq!(resp.status, Status::Ok); assert_eq!(decode_truncated(&resp), 2); - assert!(!ts_dir.exists(), "the live directory is renamed aside"); + assert_eq!( + entries(&ts_dir), + vec![TS_STAMP_FILE.to_string()], + "the live directory is renamed aside and replaced by the stamp" + ); let aside = ts_dir.with_file_name(truncating_dir_name(COLLECTION, 3)); assert!(aside.exists()); assert_eq!(memtable_rows(&core), 0); assert_eq!(partition_rows(&core), 0); core.rollback_undo_log(0, TENANT, undo).expect("rollback"); - assert!(ts_dir.exists(), "the directory is renamed back"); + assert!( + entries(&ts_dir).iter().any(|name| name.starts_with("ts-")), + "the directory is renamed back" + ); assert!(!aside.exists()); assert_eq!(memtable_rows(&core), 1); assert_eq!(partition_rows(&core), 1); - assert!( - !core.ts_truncate_floors.contains_key(&key()), - "the floor is rewound" + assert_eq!( + core.ts_replay_stamps.get(&key()).cloned(), + stamp_before, + "the stamp is rewound" ); assert!(core.columnar_memtable_mem.contains_key(&key())); } diff --git a/nodedb/src/data/executor/handlers/timeseries_wal.rs b/nodedb/src/data/executor/handlers/timeseries_wal.rs index 35099b8e7..8e8c7447d 100644 --- a/nodedb/src/data/executor/handlers/timeseries_wal.rs +++ b/nodedb/src/data/executor/handlers/timeseries_wal.rs @@ -3,10 +3,14 @@ //! WAL replay for timeseries records. //! //! On startup, replays `TimeseriesBatch` records into the per-core -//! columnar memtable. Only replays records with LSN > `last_flushed_wal_lsn` -//! per partition (not max_ts — safe with out-of-order data). A -//! committed-redo apply runs the same arm without that skip -//! (`replay_policy`). +//! columnar memtable. A record the collection's replay stamp names is in a +//! partition already, or a truncate removed it, and is skipped +//! (`timeseries_checkpoint::stamp`). A committed-redo apply runs the same arm +//! without that skip (`replay_policy`). +//! +//! Restart replay walks the records in LSN order and sets the replay cursor +//! to the last LSN it passed. A flush or truncate during the pass stamps +//! through the cursor, never past a record replay has not reached. use crate::data::executor::core_loop::CoreLoop; use crate::types::DatabaseId; @@ -19,9 +23,9 @@ impl CoreLoop { /// /// Called once during startup, after `open()` but before the event loop. /// Processes `TimeseriesBatch` records and the columnar-family truncate - /// records, ignoring records for other vShards. Uses LSN-based skip: only - /// replays records with LSN > last flushed LSN, and never a record a - /// later truncate of its collection already removed. + /// records, ignoring records for other vShards. Skips every record its + /// collection's replay stamp names, and never replays a record a later + /// truncate of its collection already removed. pub fn replay_timeseries_wal( &mut self, records: &[nodedb_wal::WalRecord], @@ -31,11 +35,19 @@ impl CoreLoop { use crate::data::executor::wal_replay_columnar_truncate::TruncateFloors; use nodedb_wal::record::RecordType; - let truncate_floors = TruncateFloors::collect(records, num_cores, self.core_id); + let mut truncate_floors = + TruncateFloors::collect(records, num_cores, self.core_id, &self.ts_replay_stamps); let mut replayed = 0usize; let mut skipped = 0usize; - - for record in records { + // Records replayed below the highest LSN their collection's stamp + // names: in flight when that stamp was written. + let mut in_flight = 0usize; + let restart = !self.applying_committed_redo(); + // The record whose sub-records the arm is applying. A record installs + // into a collection as one unit (`group_flush`). + let mut current = None; + + for (index, record) in records.iter().enumerate() { let logical_type = record.logical_record_type(); let record_type = RecordType::from_raw(logical_type); @@ -59,8 +71,14 @@ impl CoreLoop { skipped += 1; continue; } + // Every record below this one is passed. + if restart { + self.ts_replay_cursor = Some(record.header.lsn.saturating_sub(1)); + } + self.begin_ts_record(records, index, num_cores, &mut current); if is_truncate { + truncate_floors.pass(record); if self.replay_truncate_record(record, tombstones) { replayed += 1; } else { @@ -174,19 +192,11 @@ impl CoreLoop { continue; } - // Check if this record was already flushed (LSN-based skip). A - // restart watermark only: see `replay_policy`. - if let Some(registry) = self.ts_registries.get(&key) { - // Find the max flushed LSN across all partitions. - let max_flushed_lsn = registry - .iter() - .map(|(_, e)| e.meta.last_flushed_wal_lsn) - .max() - .unwrap_or(0); - if self.replay_watermark_skips(record_lsn <= max_flushed_lsn) { - skipped += 1; - continue; - } + // A partition holds this record, or a truncate removed it. A + // restart gate only: see `replay_policy`. + if self.ts_replay_skips(&key, record_lsn) { + skipped += 1; + continue; } if redo_apply && kind.as_deref() == Some("columnar") { @@ -196,6 +206,10 @@ impl CoreLoop { if self.claim_for_validation() { continue; } + let passed = self + .ts_replay_stamps + .get(&key) + .is_some_and(|stamp| stamp.rows.highest() > record_lsn); let accepted = match kind.as_deref() { // The columnar floor is consulted HERE and not above the `kind` @@ -256,24 +270,21 @@ impl CoreLoop { continue; } - // Track the max WAL LSN ingested per collection for flush metadata, - // AFTER the record has been applied — never before. - // - // `flush_ts_collection` stamps the partition it writes with this - // scalar, and the stamp claims "every record at or below N is - // WHOLLY on disk". Replaying a record can itself fire the - // record-boundary flush in the ingest handler (a full tag - // dictionary is resolved by flushing first, then taking the record - // whole). Advancing the scalar to the in-flight record before that - // dispatch stamped the partition with a record it holds NONE of, so - // a crash there lost the record outright: the next replay skipped it - // against a stamp no partition had earned. Advancing after the - // apply keeps the stamp at the last record the memtable fully - // absorbed, which is exactly what the flush can honestly claim. - let entry = self.ts_max_ingested_lsn.entry(key).or_insert(0); - *entry = (*entry).max(record_lsn); + // Every row of the record landed; a flush from here on names it. + // A committed-redo apply notes its record once the whole record + // installed. + if restart { + self.note_ts_record_applied(record_lsn); + } replayed += accepted; + in_flight += usize::from(passed); + } + if let Some(record) = current { + self.end_ts_record(record); + } + if restart { + self.ts_replay_cursor = None; } if replayed > 0 { @@ -281,6 +292,7 @@ impl CoreLoop { core = self.core_id, replayed, skipped, + in_flight, collections = self.columnar_memtables.len(), "WAL timeseries replay complete" ); @@ -384,8 +396,7 @@ mod tests { /// must name the previous record. Stamping it with the in-flight record /// makes boot replay skip a record no partition holds: the rows are gone. /// - /// This fails if the `ts_max_ingested_lsn` advance moves back ahead of the - /// apply. + /// This fails if the replay cursor passes a record before its rows land. #[test] fn a_replay_flush_is_stamped_with_the_last_fully_applied_record() { let mut h = make_core(); @@ -421,12 +432,12 @@ mod tests { "the partition holds record 10 and none of record 11, so it may \ claim only record 10" ); - assert_eq!( - h.core.ts_max_ingested_lsn.get(&key).copied(), - Some(11), - "record 11 is applied by the end of replay, so the collection's \ - max ingested LSN must have reached it" + let stamp = &h.core.ts_replay_stamps.get(&key).expect("stamp").rows; + assert!( + stamp.skips(10) && !stamp.skips(11), + "the stamp names record 10 only: {stamp:?}" ); + assert_eq!(h.core.ts_replay_cursor, None, "the pass clears its cursor"); } /// A `TimeseriesBatch`-typed WAL record carrying a map-shaped @@ -733,15 +744,22 @@ mod tests { TenantId::new(7), "metrics_big".to_string(), ); - let mt = h + let memtable_rows = h .core .columnar_memtables .get(&key) - .expect("memtable created by replay"); + .expect("memtable created by replay") + .row_count(); + let partition_rows = h + .core + .ts_registries + .get(&key) + .map_or(0, |registry| registry.total_row_count()); assert_eq!( - mt.row_count(), + memtable_rows + partition_rows, N as u64, - "every sample of an over-limit replayed batch must be retained" + "every sample of an over-limit replayed batch must be retained; the \ + record lands whole, then flushes whole once it settled" ); } diff --git a/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs b/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs index 0756ea8b8..ac551f807 100644 --- a/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs +++ b/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs @@ -120,7 +120,7 @@ impl CoreLoop { let mode = if installing { TimeseriesApplyMode::RedoInstall } else { - TimeseriesApplyMode::Immediate + TimeseriesApplyMode::Replay }; let response = self.execute_timeseries_ingest(TimeseriesIngestExec { task: &task, diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs index 82e3fc325..e58b3d183 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs @@ -1,12 +1,12 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Keep every published engine watermark true after a committed record -//! applies below it. +//! Keep every published replay stamp true after a committed record applies +//! below its prefix. //! -//! Restart replay skips a record at or below an engine's published -//! watermark: the vector checkpoint's per-collection LSN, the KV and -//! columnar checkpoint floors, an array manifest's durable LSN. The watermark -//! claims that the published artifact holds every record at or below it. +//! Restart replay skips a record at or below the prefix of an engine's +//! published replay stamp: the vector, KV and columnar checkpoint manifests, +//! an array manifest. The prefix claims that the published artifact holds +//! every record at or below it. //! Records apply in the order Raft //! commits them, not in LSN order, so a record can apply after an artifact //! was published at a higher LSN. That record lives only in memory, and the @@ -28,6 +28,8 @@ use super::sub_ops::kv_ops; pub(super) struct WrittenEngines { /// A vector index the vector checkpoint publishes. pub vectors: bool, + /// A sparse-vector index the sparse-vector checkpoint publishes. + pub sparse_vectors: bool, /// A KV collection the KV checkpoint publishes. pub kv: bool, /// A columnar collection the columnar checkpoint publishes. @@ -38,6 +40,7 @@ impl WrittenEngines { pub(super) fn of(redo: &RedoRecord) -> Self { Self { vectors: redo.ops.iter().any(writes_vector_index), + sparse_vectors: redo.ops.iter().any(writes_sparse_vector_index), kv: !kv_ops(&redo.ops).is_empty() || redo .ops @@ -65,6 +68,13 @@ fn writes_vector_index(op: &RedoSubRecord) -> bool { ) } +fn writes_sparse_vector_index(op: &RedoSubRecord) -> bool { + matches!( + RecordType::from_raw(op.record_type), + Some(RecordType::SparseVectorPut | RecordType::SparseVectorDelete) + ) +} + fn is_kv_truncate(op: &RedoSubRecord) -> bool { RecordType::from_raw(op.record_type) == Some(RecordType::Delete) && zerompk::from_msgpack::<(String, String)>(&op.payload) @@ -105,8 +115,8 @@ impl CoreLoop { } } - /// Publish again every artifact whose watermark covers `lsn` but not the - /// record just applied at it. + /// Publish again every artifact whose stamp names `lsn` but does not hold + /// the record just applied at it. pub(super) fn cover_applied_record( &mut self, lsn: Lsn, @@ -118,6 +128,12 @@ impl CoreLoop { self.checkpoint_vector_indexes() .map_err(|error| republish_error("vector checkpoint", lsn, published, &error))?; } + if engines.sparse_vectors && lsn <= self.floors.sparse_vector_published_lsn { + let published = self.floors.sparse_vector_published_lsn; + self.checkpoint_sparse_vector_indexes().map_err(|error| { + republish_error("sparse-vector checkpoint", lsn, published, &error) + })?; + } if engines.kv && lsn <= self.floors.kv_published_lsn { let published = self.floors.kv_published_lsn; self.checkpoint_kv_engines() @@ -129,16 +145,14 @@ impl CoreLoop { .map_err(|error| republish_error("columnar checkpoint", lsn, published, &error))?; } for array_id in arrays { - let durable = self.array_durable_lsn(array_id); - if lsn.as_u64() > durable { + if !self.array_stamp_names(array_id, lsn.as_u64()) { continue; } - self.array_engine - .flush(array_id, durable) + self.flush_array(array_id) .map_err(|error| ErrorCode::Internal { detail: format!( - "a record applied at lsn {} below array '{}' durable lsn {durable} \ - could not be flushed: {error}", + "a record applied at lsn {} that array '{}' manifest stamp names \ + could not be flushed: {error}", lsn.as_u64(), array_id.name ), @@ -197,9 +211,12 @@ mod tests { let key = CoreLoop::vector_index_key(DatabaseId::DEFAULT.as_u64(), TID, "docs", ""); let mut collection = VectorCollection::new(2, HnswParams::default()); collection.insert_with_surrogate(vec![0.1, 0.9], Surrogate::new(1)); - collection.note_checkpoint_lsn(100); core.vector_collections.insert(key.clone(), collection); core.advance_watermark(Lsn::new(100)); + // The published stamp's prefix is what restart replay skips through. + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(100)); core.checkpoint_vector_indexes() .expect("publish at lsn 100"); @@ -225,8 +242,8 @@ mod tests { assert_eq!(response.status, Status::Ok, "apply: {response:?}"); drop(core); - // Restart replay skips lsn 50 against the restored collection, so the - // published generation is the only copy of the second vector. + // Restart replay skips lsn 50, which the restored stamp's prefix names, + // so the published generation is the only copy of the second vector. let dir_path = dir.path().to_path_buf(); let (mut restored, _tx2, _rx2) = make_core_with_dir(&dir_path); restored.load_vector_checkpoints().expect("load"); diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs index 8bf91aaad..b75fe007a 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs @@ -134,14 +134,15 @@ impl CoreLoop { Ok(scope) => scope, Err(refusal) => return self.response_error(task, refusal.into_code()), }; - // Settle, then publish again every artifact whose watermark covers + // The install applied the record: a flush the settle runs, and a + // checkpoint the cover writes, name it. + self.floors.applied_prefix.note_applied(lsn); + // Settle, then publish again every artifact whose stamp prefix covers // the record, so restart replay does not skip it. Neither step can be // rolled back once it started, so a failure leaves live state restart // replay does not rebuild: the core fail-stops. The funnel keeps the // record for restart replay. let settled = self.settle_redo_install(task, &mut scope).and_then(|()| { - // Applied from here on: a checkpoint the cover writes names it. - self.floors.applied_prefix.note_applied(lsn); self.cover_applied_record(lsn, &WrittenEngines::of(&redo), &scope.arrays_written) }); if let Err(error) = settled { @@ -378,10 +379,10 @@ mod tests { } } - /// Restart replay skips a timeseries record at or below the highest - /// flushed partition stamp. A committed redo applied online after a live - /// flush stamped past its LSN must still install, and must flush so the - /// stamp's claim holds for it on the next restart. + /// Restart replay skips a timeseries record the collection stamp names. + /// A committed redo applied online after a flush whose stamp prefix + /// passed its LSN must still install, and must flush so the stamp's + /// claim holds for it on the next restart. #[test] fn an_online_redo_apply_below_a_flushed_partition_stamp_still_installs() { let dir = tempfile::tempdir().expect("tempdir"); @@ -393,7 +394,8 @@ mod tests { "metrics".to_string(), ); - // A write at LSN 100 lands and flushes: the partition stamps LSN 100. + // A write at LSN 100 lands, the outcome floor passes it, and a flush + // stamps through LSN 100. let earlier = RedoRecord { version: 1, ops: vec![metric_samples_sub("metrics", 1_700_000_000_000, 1.0)], @@ -416,6 +418,9 @@ mod tests { &nodedb_wal::TombstoneSet::new(), ) .expect("seed replay"); + core.floors + .applied_prefix + .observe_outcome_floor(crate::types::Lsn::new(100)); core.flush_ts_collection(tenant, crate::types::DatabaseId::DEFAULT, "metrics", 0) .expect("flush the seeded partition"); diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs index 98754086d..6f304c403 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs @@ -73,8 +73,7 @@ impl CoreLoop { self.settle_redo_timeseries(key, lsn)?; } for array_id in &scope.arrays_written { - self.array_engine - .flush_if_full(array_id) + self.flush_array_if_full(array_id) .map_err(|e| ErrorCode::Internal { detail: format!("array '{}' threshold flush failed: {e}", array_id.name), })?; @@ -109,12 +108,12 @@ impl CoreLoop { /// Settle one timeseries collection the install ingested into: charge /// the memory budget for its memtable, and flush it when it is over its - /// soft limit or when a partition already claims the record's LSN. + /// soft limit or when the collection stamp already names the record. /// - /// Restart replay skips every record at or below the highest partition - /// stamp. A record applied after a flush stamped past its LSN sits only - /// in the memtable, so the claim is false for it until the memtable - /// flushes too. + /// Restart replay skips every record the collection's replay stamp + /// names. A record applied after a stamp's prefix passed its LSN sits + /// only in the memtable, so the claim is false for it until the memtable + /// flushes too, with a stamp that names it. fn settle_redo_timeseries(&mut self, key: CollectionKey, lsn: u64) -> Result<(), ErrorCode> { let (database_id, tid, collection) = key; self.recharge_ts_memtable_budget(tid, database_id, &collection); @@ -126,14 +125,7 @@ impl CoreLoop { .columnar_memtables .get(&memtable_key) .is_some_and(|mt| mt.memory_bytes() >= self.ts_tuning.memtable_budget_bytes); - let below_stamp = self - .ts_registries - .get(&memtable_key) - .is_some_and(|registry| { - registry - .iter() - .any(|(_, entry)| entry.meta.last_flushed_wal_lsn >= lsn) - }); + let below_stamp = self.ts_rows_named(&memtable_key, lsn); if !over_budget && !below_stamp { return Ok(()); } diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/array.rs b/nodedb/src/data/executor/handlers/transaction/resolve/array.rs index f35ba8781..3064f5ecb 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/array.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/array.rs @@ -6,7 +6,7 @@ //! ride the buffered-plan path rather than a per-surrogate overlay post-image, //! and their redo replay re-runs the engine's native cell batch //! (`replay_array_wal`, dispatched from the redo reconstitute path, which -//! respects the array's `ArrayFlush` watermark). This module reads the +//! respects the array manifest's replay stamp). This module reads the //! [`ArrayOp`] plan node directly and emits the SAME `RecordType::ArrayPut` / //! `RecordType::ArrayDelete` sub-record the autocommit array path produces, //! reusing its version-tagged encoders (`engine::array::wal`): diff --git a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs index a741f3481..a3a71eda4 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs @@ -202,7 +202,6 @@ impl CoreLoop { memtable_memory_bytes_before, last_value_cache_before, series_catalog_before, - max_ingested_lsn_before, last_ts_ingest_before, reservation_bytes_before, } = token; @@ -273,14 +272,6 @@ impl CoreLoop { self.ts_series_catalogs.remove(&collection_key); } } - match max_ingested_lsn_before { - Some(lsn) => { - self.ts_max_ingested_lsn.insert(collection_key, lsn); - } - None => { - self.ts_max_ingested_lsn.remove(&collection_key); - } - } self.last_ts_ingest = last_ts_ingest_before; Ok(()) } @@ -350,7 +341,6 @@ mod tests { let mut cache = LastValueCache::new(); cache.update(1, 10, 1.0); core.ts_last_value_caches.insert(key.clone(), cache.clone()); - core.ts_max_ingested_lsn.insert(key.clone(), 7); let mut catalog = nodedb_types::timeseries::SeriesCatalog::new(); catalog.resolve(&nodedb_types::timeseries::SeriesKey::new( "cpu", @@ -367,7 +357,6 @@ mod tests { memtable_memory_bytes_before: Some(memory_bytes), last_value_cache_before: Some(cache), series_catalog_before: Some(catalog.clone()), - max_ingested_lsn_before: Some(7), last_ts_ingest_before: Some(prior_timer), reservation_bytes_before: None, }; @@ -388,7 +377,6 @@ mod tests { .get_mut(&key) .expect("cache") .update(1, 20, 2.0); - core.ts_max_ingested_lsn.insert(key.clone(), 99); core.ts_series_catalogs .get_mut(&key) .expect("catalog") @@ -423,7 +411,6 @@ mod tests { .map(|entry| (entry.ts, entry.value)), Some((10, 1.0)) ); - assert_eq!(core.ts_max_ingested_lsn.get(&key), Some(&7)); assert_eq!(core.last_ts_ingest, Some(prior_timer)); } @@ -443,7 +430,6 @@ mod tests { memtable_memory_bytes_before: None, last_value_cache_before: None, series_catalog_before: None, - max_ingested_lsn_before: None, last_ts_ingest_before: None, reservation_bytes_before: None, }; @@ -453,14 +439,12 @@ mod tests { .insert(key.clone(), nodedb_types::timeseries::SeriesCatalog::new()); core.ts_last_value_caches .insert(key.clone(), LastValueCache::new()); - core.ts_max_ingested_lsn.insert(key.clone(), 1); core.last_ts_ingest = Some(std::time::Instant::now()); core.apply_undo_timeseries(0, UndoEntry::TimeseriesIngest(token)) .expect("undo"); assert!(!core.columnar_memtables.contains_key(&key)); assert!(!core.ts_last_value_caches.contains_key(&key)); - assert!(!core.ts_max_ingested_lsn.contains_key(&key)); assert!( !core.ts_series_catalogs.contains_key(&key), "the catalog the ingest created is gone" @@ -597,7 +581,6 @@ mod tests { TenantId::new(TID), "metrics".to_string(), ); - assert_eq!(core.ts_max_ingested_lsn.get(&key), Some(&lsn)); core.flush_ts_collection( TenantId::new(TID), @@ -606,6 +589,11 @@ mod tests { 0, ) .expect("flush committed transaction rows"); + let stamp = &core.ts_replay_stamps.get(&key).expect("stamp").rows; + assert!( + stamp.skips(lsn), + "the partition's stamp names the transaction record: {stamp:?}" + ); let max_flushed_lsn = core .ts_registries .get(&key) diff --git a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs index 18fecc1cc..321103e76 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/entry.rs @@ -27,7 +27,6 @@ pub(in crate::data::executor) struct TimeseriesIngestUndo { /// The collection's series catalog. Ingest registers each new series in /// it. pub series_catalog_before: Option, - pub max_ingested_lsn_before: Option, pub last_ts_ingest_before: Option, pub reservation_bytes_before: Option, } @@ -64,16 +63,18 @@ pub(in crate::data::executor) type SpatialDocMapEntry = ( /// renames it back and a commit removes it (`finalize_timeseries_truncates`). pub(in crate::data::executor) struct TimeseriesTruncateUndo { pub collection_key: (nodedb_types::DatabaseId, TenantId, String), - /// `(original, moved)` when the collection had a partition directory. - pub moved_dir: Option<(std::path::PathBuf, std::path::PathBuf)>, + /// The collection's live directory. The truncate leaves a fresh one + /// holding only its replay stamp; the rollback removes it. + pub original_dir: std::path::PathBuf, + /// The aside name of the directory the collection had, if it had one. + pub moved_dir: Option, pub memtable: Option, pub memtable_mem: Option, pub registry: Option, - pub max_ingested_lsn: Option, pub last_value_cache: Option, pub series_catalog: Option, - /// `ts_truncate_floors[key]` before this truncate raised it. - pub truncate_floor: Option, + /// The collection's replay stamp before this truncate raised it. + pub replay_stamp: Option, } /// Tracks a write operation for rollback purposes. diff --git a/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs b/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs index 0c9c8d74e..23a646441 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/timeseries.rs @@ -5,10 +5,11 @@ //! state `execute_timeseries_truncate` moved out and renames the partition //! directory back to its live name. //! -//! The rename is the only durable step. A rename that fails is fatal to the -//! rollback (`Err` → `RollbackFailed`): the memory state would say the rows -//! exist while the partitions sit under the aside name, and the next scan -//! would answer from half a collection. +//! Removing the truncate's fresh directory and the rename back are the +//! durable steps. Either failing is fatal to the rollback (`Err` → +//! `RollbackFailed`): the memory state would say the rows exist while the +//! partitions sit under the aside name, and the next scan would answer from +//! half a collection. use crate::data::executor::core_loop::CoreLoop; use crate::types::{DatabaseId, TenantId}; @@ -18,7 +19,7 @@ use super::{TimeseriesIngestUndo, TimeseriesTruncateUndo}; impl CoreLoop { /// The complete in-memory pre-image of a timeseries collection before an /// ingest mutates it: the memtable, its config and resident footprint, - /// the last-value cache, the ingest watermark, the ingest timer, and the + /// the last-value cache, the series catalog, the ingest timer, and the /// memtable's reservation. pub(in crate::data::executor) fn capture_timeseries_ingest_undo( &self, @@ -32,7 +33,6 @@ impl CoreLoop { memtable_memory_bytes_before: memtable.map(|memtable| memtable.memory_bytes()), last_value_cache_before: self.ts_last_value_caches.get(collection_key).cloned(), series_catalog_before: self.ts_series_catalogs.get(collection_key).cloned(), - max_ingested_lsn_before: self.ts_max_ingested_lsn.get(collection_key).copied(), last_ts_ingest_before: self.last_ts_ingest, reservation_bytes_before: self .columnar_memtable_mem @@ -48,41 +48,41 @@ impl CoreLoop { ) -> Result<(), (usize, String)> { let TimeseriesTruncateUndo { collection_key, + original_dir: original, moved_dir, memtable, memtable_mem, registry, - max_ingested_lsn, last_value_cache, series_catalog, - truncate_floor, + replay_stamp, } = undo; - if let Some((original, moved)) = moved_dir { - // A directory a later sub-plan created under the live name holds - // rows the reverse-order rollback has already withdrawn. - if original.exists() - && let Err(e) = std::fs::remove_dir_all(&original) - { - return Err(( - entry_index, - format!( - "timeseries truncate undo: remove {} before restoring {}: {e}", - original.display(), - moved.display() - ), - )); - } - if let Err(e) = std::fs::rename(&moved, &original) { - return Err(( - entry_index, - format!( - "timeseries truncate undo: rename {} back to {}: {e}", - moved.display(), - original.display() - ), - )); - } + // The live name holds the truncate's fresh directory, and any rows a + // later sub-plan wrote there, which the reverse-order rollback has + // already withdrawn. + if original.exists() + && let Err(e) = std::fs::remove_dir_all(&original) + { + return Err(( + entry_index, + format!( + "timeseries truncate undo: remove {}: {e}", + original.display() + ), + )); + } + if let Some(moved) = moved_dir + && let Err(e) = std::fs::rename(&moved, &original) + { + return Err(( + entry_index, + format!( + "timeseries truncate undo: rename {} back to {}: {e}", + moved.display(), + original.display() + ), + )); } reinstall(&mut self.columnar_memtables, &collection_key, memtable); @@ -92,11 +92,6 @@ impl CoreLoop { memtable_mem, ); reinstall(&mut self.ts_registries, &collection_key, registry); - reinstall( - &mut self.ts_max_ingested_lsn, - &collection_key, - max_ingested_lsn, - ); reinstall( &mut self.ts_last_value_caches, &collection_key, @@ -107,11 +102,7 @@ impl CoreLoop { &collection_key, series_catalog, ); - reinstall( - &mut self.ts_truncate_floors, - &collection_key, - truncate_floor, - ); + reinstall(&mut self.ts_replay_stamps, &collection_key, replay_stamp); Ok(()) } } diff --git a/nodedb/src/data/executor/handlers/unregister_collection.rs b/nodedb/src/data/executor/handlers/unregister_collection.rs index 39d53dac6..16b68840e 100644 --- a/nodedb/src/data/executor/handlers/unregister_collection.rs +++ b/nodedb/src/data/executor/handlers/unregister_collection.rs @@ -275,10 +275,9 @@ impl CoreLoop { } self.columnar_memtable_mem.remove(&key); self.ts_registries.remove(&key); - self.ts_max_ingested_lsn.remove(&key); + self.ts_replay_stamps.remove(&key); self.ts_last_value_caches.remove(&key); self.ts_series_catalogs.remove(&key); - self.ts_truncate_floors.remove(&key); r }; diff --git a/nodedb/src/data/executor/handlers/vector.rs b/nodedb/src/data/executor/handlers/vector.rs index 3428939df..528cca889 100644 --- a/nodedb/src/data/executor/handlers/vector.rs +++ b/nodedb/src/data/executor/handlers/vector.rs @@ -225,14 +225,6 @@ impl CoreLoop { match self.get_or_create_vector_index(database_id, tid, collection, dim, field_name) { Ok(collection_ref) => { collection_ref.insert_with_surrogate(vector.to_vec(), surrogate); - // Advance this collection's checkpoint watermark to the write's - // WAL LSN so a later checkpoint records that this insert is - // already absorbed; startup replay then skips the straddling - // WAL record instead of appending a duplicate HNSW node. `None` - // (unassigned LSN) leaves the watermark untouched. - if let Some(lsn) = task.wal_lsn() { - collection_ref.note_checkpoint_lsn(lsn.as_u64()); - } let seal_key = CoreLoop::vector_build_key(&index_key); if !defer_seal && collection_ref.needs_seal() diff --git a/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs b/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs index 006c2f210..0334166d6 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs @@ -248,7 +248,6 @@ impl CoreLoop { old_sidecar: old_payload.clone(), }; self.apply_vector_direct_update_row( - task, &index_key, tid, collection, @@ -272,7 +271,6 @@ impl CoreLoop { self.remove_vector_direct_row(&index_key, tid, collection, *surrogate)?; } self.write_vector_direct_row(VectorDirectRowWrite { - task, index_key: &index_key, tid, collection, diff --git a/nodedb/src/data/executor/handlers/vector_direct_row.rs b/nodedb/src/data/executor/handlers/vector_direct_row.rs index 19578db48..a1421222c 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_row.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_row.rs @@ -35,7 +35,6 @@ pub(in crate::data::executor) struct VectorDirectIndexSpec<'a> { /// One row to store, for [`CoreLoop::write_vector_direct_row`]. pub(in crate::data::executor) struct VectorDirectRowWrite<'a> { - pub task: &'a ExecutionTask, pub index_key: &'a VectorIndexKey, pub tid: u64, pub collection: &'a str, @@ -239,7 +238,6 @@ impl CoreLoop { row: VectorDirectRowWrite<'_>, ) -> Result<(), ErrorCode> { let VectorDirectRowWrite { - task, index_key, tid, collection, @@ -255,12 +253,6 @@ impl CoreLoop { }); }; let node_id = coll.insert_with_surrogate(vector.to_vec(), surrogate); - // Advance the checkpoint watermark so a later vector checkpoint records - // this write as absorbed; startup replay then skips the straddling WAL - // record instead of appending a duplicate node. - if let Some(lsn) = task.wal_lsn() { - coll.note_checkpoint_lsn(lsn.as_u64()); - } coll.payload.insert_row(node_id, fields); let key = StorageKey::for_surrogate(surrogate); diff --git a/nodedb/src/data/executor/handlers/vector_direct_update.rs b/nodedb/src/data/executor/handlers/vector_direct_update.rs index bd2cf68a0..55954d124 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_update.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_update.rs @@ -200,8 +200,8 @@ impl CoreLoop { let mut written: Vec<(StorageKey, Vec)> = Vec::with_capacity(planned.len()); for row in planned { - if let Err(e) = self - .apply_vector_direct_update_row(task, &index_key, tid, collection, &row, new_vector) + if let Err(e) = + self.apply_vector_direct_update_row(&index_key, tid, collection, &row, new_vector) { return self.response_error(task, e); } @@ -241,7 +241,6 @@ impl CoreLoop { /// only the bitmap entries and sidecar move. pub(in crate::data::executor) fn apply_vector_direct_update_row( &mut self, - task: &ExecutionTask, index_key: &VectorIndexKey, tid: u64, collection: &str, @@ -252,7 +251,6 @@ impl CoreLoop { if let Some(vector) = new_vector { self.remove_vector_direct_row(index_key, tid, collection, row.surrogate)?; return self.write_vector_direct_row(VectorDirectRowWrite { - task, index_key, tid, collection, @@ -276,9 +274,6 @@ impl CoreLoop { }; coll.payload.delete_row(node_id, &old); coll.payload.insert_row(node_id, &row.fields); - if let Some(lsn) = task.wal_lsn() { - coll.note_checkpoint_lsn(lsn.as_u64()); - } let key = StorageKey::for_surrogate(row.surrogate); if let Err(e) = self .sparse diff --git a/nodedb/src/data/executor/handlers/vector_multi.rs b/nodedb/src/data/executor/handlers/vector_multi.rs index 9125c2887..8c9f9cfa6 100644 --- a/nodedb/src/data/executor/handlers/vector_multi.rs +++ b/nodedb/src/data/executor/handlers/vector_multi.rs @@ -135,13 +135,6 @@ impl CoreLoop { // Insert all vectors with shared surrogate. let ids = coll.insert_multi_vector(&vector_slices, document_surrogate); - // Advance the checkpoint watermark to this write's WAL LSN so a later - // checkpoint records these nodes as absorbed; startup replay then skips - // the straddling WAL record instead of appending duplicate HNSW nodes. - // `None` (unassigned LSN) leaves the watermark untouched. - if let Some(lsn) = task.wal_lsn() { - coll.note_checkpoint_lsn(lsn.as_u64()); - } // Auto-seal if needed. let seal_key = CoreLoop::vector_build_key(&index_key); @@ -193,12 +186,6 @@ impl CoreLoop { }; let deleted = coll.delete_multi_vector(document_surrogate); - // Advance the watermark to this delete's WAL LSN (single per-collection - // value covering inserts and deletes), so a checkpoint records the - // removal as absorbed and replay does not re-run it below the mark. - if let Some(lsn) = task.wal_lsn() { - coll.note_checkpoint_lsn(lsn.as_u64()); - } if deleted > 0 { self.checkpoint_coordinator.mark_dirty("vector", deleted); // Record this write's version keyed by the shared document diff --git a/nodedb/src/data/executor/handlers/vector_upsert.rs b/nodedb/src/data/executor/handlers/vector_upsert.rs index 38f9972de..395fc1ecc 100644 --- a/nodedb/src/data/executor/handlers/vector_upsert.rs +++ b/nodedb/src/data/executor/handlers/vector_upsert.rs @@ -289,7 +289,6 @@ impl CoreLoop { return self.response_error(task, e); } if let Err(e) = self.write_vector_direct_row(VectorDirectRowWrite { - task, index_key: &index_key, tid, collection, diff --git a/nodedb/src/data/executor/handlers/vector_write.rs b/nodedb/src/data/executor/handlers/vector_write.rs index 1d4abaf71..8dd929008 100644 --- a/nodedb/src/data/executor/handlers/vector_write.rs +++ b/nodedb/src/data/executor/handlers/vector_write.rs @@ -50,12 +50,6 @@ impl CoreLoop { let s = surrogates.get(i).copied().unwrap_or(Surrogate::ZERO); collection_ref.insert_with_surrogate(vector.clone(), s); } - // Advance the checkpoint watermark so a later vector checkpoint - // records these writes as absorbed; startup replay then skips the - // straddling WAL records instead of appending duplicate nodes. - if let Some(lsn) = task.wal_lsn() { - collection_ref.note_checkpoint_lsn(lsn.as_u64()); - } let seal_key = CoreLoop::vector_build_key(&index_key); if !defer_seal && collection_ref.needs_seal() diff --git a/nodedb/src/data/executor/replay_floors.rs b/nodedb/src/data/executor/replay_floors.rs index 385b76c0c..a877f0c4b 100644 --- a/nodedb/src/data/executor/replay_floors.rs +++ b/nodedb/src/data/executor/replay_floors.rs @@ -42,15 +42,13 @@ //! //! ## Which engines need one //! -//! Only those whose WAL records are DELTAS against current state. A floor is not -//! a general "I restored a checkpoint" marker, and adding one where replay is -//! already idempotent gates records for no reason. +//! An engine whose WAL records are deltas or appends against current state +//! needs one. The sparse-vector engine carries one as well: its checkpoint +//! names the records it holds, and deciding every record by that stamp keeps +//! replay in the order the live core applied them. //! -//! Six checkpointed engines deliberately have no field here: +//! Checkpointed engines with no field here: //! -//! * Sparse vector — `SparseVectorPut` is an upsert keyed by `doc_id` and -//! `SparseVectorDelete` is a no-op against an absent document, so a record -//! re-applied over the restored index reproduces it. //! * The sync idempotency gate — `SyncSeqAdvance` advances both its maps by //! max-wins, so re-folding a record already contained in the restored state //! cannot change it. What that restore needs instead is for replay to MERGE @@ -62,10 +60,12 @@ //! precisely because ids are not stable across restarts, and replay uses the //! same `add_node_label` / `remove_node_label` entry points as the live //! handler. -//! * The array engine — it carries its own per-array floor rather than one -//! here: each array's manifest records the `durable_lsn` its flushed segments -//! reach, and `replay_array_wal` gates on that. A shared engine-wide field -//! would be wrong for it, since arrays flush independently of one another. +//! * The array and timeseries engines. Each carries its own stamp per +//! artifact rather than one here: an array manifest carries the stamp of +//! the flush that last published it, and a timeseries partition carries the +//! stamp of the flush that wrote it. Arrays and timeseries collections flush +//! independently of one another, so an engine-wide field is wrong for both. +//! Both still decide a record through `ReplayStamp::skips`. //! * Full-text search — `FtsIndex` rewrites the surrogate's posting, length and //! stats entries wholesale, deriving the corpus-counter deltas from the prior //! doc-length row read in the same write transaction, so a re-applied record @@ -143,6 +143,24 @@ pub(in crate::data::executor) struct ReplayFloors { /// and a record class that tolerates gating does not need an exemption /// from it. pub(in crate::data::executor) columnar: ReplayFloor, + + /// Vector engine floor (HNSW, multi-vector and direct-row indexes), + /// populated by `CoreLoop::load_vector_checkpoints`. + /// + /// An HNSW insert appends a node and never dedups, so a record the + /// restored generation holds must not replay. The generation is published + /// whole under one manifest, so one engine-wide stamp describes every + /// index in it, including an index emptied or dropped since. + pub(in crate::data::executor) vector: ReplayFloor, + + /// Sparse-vector engine floor, populated by + /// `CoreLoop::load_sparse_vector_checkpoints`. + /// + /// A sparse put upserts by `doc_id`, so re-applying a record the restored + /// generation holds lands on the same postings. The stamp still decides + /// every record, so replay applies exactly the records the live core + /// applied after the generation was written, in LSN order. + pub(in crate::data::executor) sparse_vector: ReplayFloor, } /// What an engine's restored checkpoint holds. @@ -167,7 +185,8 @@ impl ReplayFloor { /// Whether a record at `record_lsn` is already folded into the restored /// checkpoint, or has a final outcome that is not an apply, and must - /// therefore NOT be replayed. Every KV and columnar skip site asks here. + /// therefore NOT be replayed. Every KV, columnar, vector and sparse-vector + /// skip site asks here. pub(in crate::data::executor) fn covers(&self, record_lsn: u64) -> bool { self.stamp .as_ref() diff --git a/nodedb/src/data/executor/replay_policy.rs b/nodedb/src/data/executor/replay_policy.rs index 5522cd0d0..92151a4f4 100644 --- a/nodedb/src/data/executor/replay_policy.rs +++ b/nodedb/src/data/executor/replay_policy.rs @@ -62,6 +62,15 @@ impl CoreLoop { self.redo_apply.scope.is_some() } + /// Whether the validate pass of a committed-redo apply is driving the + /// replay arms. That pass writes nothing. + pub(in crate::data::executor) fn validating_committed_redo(&self) -> bool { + self.redo_apply + .scope + .as_ref() + .is_some_and(|scope| scope.pass == RedoApplyPass::Validate) + } + /// Whether the arm must skip the write it reached because the apply is /// validating. Claims the sub-record when it is. Call once per /// sub-record, after its decode, routing and checks. @@ -128,6 +137,19 @@ impl CoreLoop { covered && !self.applying_committed_redo() } + /// Whether restart replay skips a vector-index record (HNSW, multi-vector, + /// direct row): the restored vector checkpoint holds it. Every vector + /// skip site asks here, and the answer is `ReplayStamp::skips`. + pub(in crate::data::executor) fn vector_replay_skips(&self, record_lsn: u64) -> bool { + self.replay_watermark_skips(self.floors.replay_floors.vector.covers(record_lsn)) + } + + /// Whether restart replay skips a sparse-vector record: the restored + /// sparse-vector checkpoint holds it. + pub(in crate::data::executor) fn sparse_vector_replay_skips(&self, record_lsn: u64) -> bool { + self.replay_watermark_skips(self.floors.replay_floors.sparse_vector.covers(record_lsn)) + } + /// A committed record cannot be applied. Restart replay stops recovery /// and never returns. A committed-redo apply records the error for its /// response and returns, and the caller skips the record. diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/format.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/format.rs index 80cd818b4..360e339ce 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/format.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/format.rs @@ -11,12 +11,14 @@ use serde::{Deserialize, Serialize}; +use crate::types::replay_stamp::ReplayStamp; + /// On-disk format version for the manifest and the generation it names. /// /// A manifest stamped with any other version is refused rather than misparsed. /// Refusing costs a WAL replay; misparsing would install indexes built from /// bytes this build cannot read. -pub(crate) const SPARSE_VECTOR_CKPT_FORMAT_VERSION: u16 = 1; +pub(crate) const SPARSE_VECTOR_CKPT_FORMAT_VERSION: u16 = 2; /// Names the live generation. Writing this file is what publishes a checkpoint. #[derive( @@ -43,6 +45,9 @@ pub(crate) struct SparseVectorCheckpointManifest { /// restart would have no last-known-durable point to clamp to and would /// pin truncation at zero. pub durable_through_lsn: u64, + /// The records every index in the generation holds. Restart replay skips + /// a sparse-vector record exactly when this stamp names it. + pub replay: ReplayStamp, } /// Encode a manifest publishing `generation`, for tests that need a live @@ -57,6 +62,7 @@ pub(crate) fn test_manifest_bytes(generation: u64) -> Vec { format_version: SPARSE_VECTOR_CKPT_FORMAT_VERSION, generation, durable_through_lsn: 0, + replay: ReplayStamp::default(), }) .expect("manifest encode is infallible for this fixed struct") } diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/load.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/load.rs index 1e80221c4..6f529ca67 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/load.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/load.rs @@ -31,9 +31,9 @@ impl CoreLoop { /// The restored generation's LSN becomes `sparse_vector_durable_lsn`, so a /// flush that fails before the first successful checkpoint of this process /// clamps to what the PREVIOUS process actually made durable instead of - /// pinning WAL truncation at zero. It installs no replay floor: every - /// sparse-vector WAL record is idempotent, so replay above and below the - /// stamp both reproduce the same indexes (see this module's `mod.rs`). + /// pinning WAL truncation at zero. Its replay stamp becomes the + /// sparse-vector replay floor: restart replay skips exactly the records + /// the restored indexes hold. /// /// # Fail-stop on corruption /// @@ -72,6 +72,10 @@ impl CoreLoop { // clamps truncation to, so claiming it over a half-restored generation // would authorise deleting the records that would have completed it. self.floors.sparse_vector_durable_lsn = Lsn::new(manifest.durable_through_lsn); + let replay_prefix = manifest.replay.prefix; + let applied_ranges = manifest.replay.applied_above.len(); + self.floors.sparse_vector_published_lsn = Lsn::new(replay_prefix); + self.floors.replay_floors.sparse_vector.set(manifest.replay); info!( core = self.core_id, @@ -79,6 +83,8 @@ impl CoreLoop { indexes, docs, durable_through_lsn = manifest.durable_through_lsn, + replay_prefix, + applied_ranges, "sparse vector checkpoint restored" ); Ok(()) @@ -208,6 +214,7 @@ mod tests { format_version: SPARSE_VECTOR_CKPT_FORMAT_VERSION, generation: 4, durable_through_lsn: 8_128, + replay: crate::types::replay_stamp::ReplayStamp::default(), }; let tmp = tempfile::tempdir().expect("tempdir"); let bytes = zerompk::to_msgpack_vec(&written).expect("encode"); @@ -240,6 +247,7 @@ mod tests { format_version: SPARSE_VECTOR_CKPT_FORMAT_VERSION + 1, generation: 1, durable_through_lsn: 5, + replay: crate::types::replay_stamp::ReplayStamp::default(), }; let tmp = tempfile::tempdir().expect("tempdir"); let bytes = zerompk::to_msgpack_vec(&written).expect("encode"); diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/manifest.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/manifest.rs index a4252e71a..c525f17a4 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/manifest.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/manifest.rs @@ -50,6 +50,13 @@ pub(crate) fn read_sparse_vector_manifest_at( expected: SPARSE_VECTOR_CKPT_FORMAT_VERSION, }); } + manifest + .replay + .validate() + .map_err(|source| CheckpointDecodeError::InvalidReplayStamp { + path: path.clone(), + source, + })?; Ok(Some(manifest)) } diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/mod.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/mod.rs index 145df3642..0fd2b6101 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/mod.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/mod.rs @@ -40,17 +40,13 @@ //! expressible, so a torn or abandoned write is inert garbage rather than a //! generation whose LSN overstates what is on disk. //! -//! ## Why no replay floor +//! ## Replay floor //! -//! Unlike KV — whose `kv_incr` / `kv_cas` records are deltas that double-count -//! if replayed over a checkpoint that already folded them in — every -//! sparse-vector record is idempotent: `SparseVectorPut` is an upsert keyed by -//! `doc_id` and `SparseVectorDelete` is a no-op against an absent document (see -//! `wal_replay_vector_extended.rs`, whose replay arms document exactly this and -//! take no watermark gate). Replaying records at or below the restored -//! generation's LSN therefore reproduces the same state rather than corrupting -//! it, so this engine needs no entry in `ReplayFloors`; restoring before replay -//! is the whole requirement. +//! The manifest carries the core's replay stamp, and a restart installs it as +//! the sparse-vector replay floor. A record the stamp names is skipped. Every +//! other record replays in LSN order on top of the restored generation, +//! including a lower-LSN record still in flight when the generation was +//! written, so the replayed indexes equal the live ones. mod format; mod load; diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs index 638014203..cc00db43d 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs @@ -13,6 +13,7 @@ use super::paths::{ }; use crate::data::executor::core_loop::CoreLoop; use crate::types::Lsn; +use crate::types::replay_stamp::ReplayStamp; impl CoreLoop { /// Flush every sparse-vector index on this core to disk and return the LSN @@ -23,15 +24,17 @@ impl CoreLoop { /// reported checkpoint LSN to the last LSN this engine was known durable /// through, so a failed flush costs WAL growth instead of data. /// - /// Stamping the generation with the core watermark rests on this: it runs - /// on the core's own thread between - /// tasks, so every sparse-vector write the core has admitted is already - /// folded into the in-memory indexes exported here. Where a sparse-vector - /// write did not itself raise the watermark, the stamp merely UNDERSTATES - /// this engine's durability, and understating is the safe direction — the - /// record replays idempotently on top of the restored index. - pub(in crate::data::executor) fn checkpoint_sparse_vector_indexes(&self) -> crate::Result { + /// The manifest carries the core's replay stamp: it runs on the core's own + /// thread between tasks, so every record the stamp names is already folded + /// into the in-memory indexes exported here. Restart replay skips exactly + /// those records. + pub(in crate::data::executor) fn checkpoint_sparse_vector_indexes( + &mut self, + ) -> crate::Result { let durable_through = self.watermark; + let replay = self.floors.applied_prefix.stamp()?; + let prefix = replay.prefix; + let applied_ranges = replay.applied_above.len(); let ckpt_dir = sparse_vector_ckpt_dir(&self.data_dir, self.core_id); std::fs::create_dir_all(&ckpt_dir).map_err(|e| storage_err(&ckpt_dir, "create dir", &e))?; @@ -53,7 +56,11 @@ impl CoreLoop { .map_err(|e| storage_err(&gen_dir, "create generation dir", &e))?; let written = self.write_sparse_vector_generation(&gen_dir)?; - self.publish_sparse_vector_generation(&ckpt_dir, generation, durable_through)?; + self.publish_sparse_vector_generation(&ckpt_dir, generation, durable_through, replay)?; + self.floors.sparse_vector_published_lsn = self + .floors + .sparse_vector_published_lsn + .max(Lsn::new(prefix)); // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what @@ -79,6 +86,8 @@ impl CoreLoop { generation, indexes = written, durable_through_lsn = durable_through.as_u64(), + replay_prefix = prefix, + applied_ranges, "sparse vector checkpoint published" ); Ok(durable_through) @@ -126,11 +135,13 @@ impl CoreLoop { ckpt_dir: &std::path::Path, generation: u64, durable_through: Lsn, + replay: ReplayStamp, ) -> crate::Result<()> { let manifest = SparseVectorCheckpointManifest { format_version: SPARSE_VECTOR_CKPT_FORMAT_VERSION, generation, durable_through_lsn: durable_through.as_u64(), + replay, }; let bytes = zerompk::to_msgpack_vec(&manifest).map_err(|e| crate::Error::Serialization { diff --git a/nodedb/src/data/executor/timeseries_checkpoint/flush.rs b/nodedb/src/data/executor/timeseries_checkpoint/flush.rs index d510d6187..afc4ee750 100644 --- a/nodedb/src/data/executor/timeseries_checkpoint/flush.rs +++ b/nodedb/src/data/executor/timeseries_checkpoint/flush.rs @@ -172,8 +172,8 @@ mod tests { /// Ingest one ILP line through the real dispatch path, exactly as the /// Control Plane does. `wal_lsn` is threaded BOTH on the op (the - /// partition's flush stamp, which replay's dedup gate reads) and on the - /// request (what `note_collection_write_lsn` raises the watermark from). + /// record the partition's replay stamp names) and on the request (what + /// `note_collection_write_lsn` raises the watermark from). fn ingest(&mut self, host: &str, value: f64, ts_ms: i64, wal_lsn: u64) { let line = format!("{COLL},host={host} value={value} {}\n", ts_ms * 1_000_000); let r = self.send( @@ -287,8 +287,7 @@ mod tests { assert_eq!( after.scan_hosts(), vec!["a".to_string(), "b".to_string()], - "every checkpointed row must come back from its on-disk partition — \ - pre-fix the checkpoint flushed nothing and both rows were gone" + "every checkpointed row must come back from its on-disk partition" ); } @@ -385,4 +384,213 @@ mod tests { draining before the write would have discarded it outright" ); } + + /// The ILP line `ingest` writes for one row. + fn ilp_line(host: &str, value: f64, ts_ms: i64) -> String { + format!("{COLL},host={host} value={value} {}\n", ts_ms * 1_000_000) + } + + /// The `TimeseriesBatch` WAL record a live ingest of one row appends. + fn ingest_record(host: &str, value: f64, ts_ms: i64, lsn: u64) -> nodedb_wal::WalRecord { + let payload = zerompk::to_msgpack_vec(&( + "timeseries".to_string(), + COLL.to_string(), + ilp_line(host, value, ts_ms).into_bytes(), + Option::::None, + "ilp".to_string(), + )) + .expect("encode timeseries tuple"); + wal_record( + nodedb_wal::record::RecordType::TimeseriesBatch, + lsn, + payload, + ) + } + + /// The `TimeseriesTruncate` WAL record a live truncate appends. + fn truncate_record(lsn: u64) -> nodedb_wal::WalRecord { + let payload = zerompk::to_msgpack_vec(&nodedb_types::columnar::ColumnarTruncateWalRecord { + collection: COLL.to_string(), + }) + .expect("encode truncate"); + wal_record( + nodedb_wal::record::RecordType::TimeseriesTruncate, + lsn, + payload, + ) + } + + fn wal_record( + record_type: nodedb_wal::record::RecordType, + lsn: u64, + payload: Vec, + ) -> nodedb_wal::WalRecord { + nodedb_wal::WalRecord::new(nodedb_wal::record::WalRecordArgs { + record_type: record_type as u32, + lsn, + tenant_id: TID, + vshard_id: 0, + database_id: DatabaseId::DEFAULT.as_u64(), + payload, + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record") + } + + /// Restart the way boot does: load the registries and stamps, raise the + /// floor to the WAL end, then replay `records`. + fn restart_and_replay(dir: &std::path::Path, records: &[nodedb_wal::WalRecord]) -> Core { + let mut core = Core::open_at(dir); + core.core.load_ts_registries().expect("load"); + let wal_end = records.iter().map(|r| r.header.lsn).max().unwrap_or(0); + core.core + .floors + .applied_prefix + .seed_replayed_through(Lsn::new(wal_end)); + core.core + .replay_timeseries_wal(records, 1, &nodedb_wal::TombstoneSet::new()); + core + } + + fn observe_floor(core: &mut Core, lsn: u64) { + core.core + .floors + .applied_prefix + .observe_outcome_floor(Lsn::new(lsn)); + } + + fn truncate(core: &mut Core, lsn: u64) { + let r = core.send( + PhysicalPlan::Timeseries(TimeseriesOp::Truncate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, COLL), + restart_identity: false, + }), + Some(lsn), + ); + assert_eq!(r.status, Status::Ok, "ts truncate: {r:?}"); + } + + /// Floor 10. B at lsn 30 applies, a checkpoint flushes it, then A at lsn + /// 20 applies. A restart replays A once and never re-applies B, and the + /// collection scans as it did live — across a second restart too. + #[test] + fn a_timeseries_write_in_flight_at_a_checkpoint_replays_once() { + let dir = tempfile::tempdir().expect("tempdir"); + let records = [ + ingest_record("a", 1.0, 1_000, 20), + ingest_record("b", 2.0, 2_000, 30), + ]; + + let mut before = Core::open_at(dir.path()); + observe_floor(&mut before, 10); + before.ingest("b", 2.0, 2_000, 30); + before + .core + .checkpoint_timeseries_memtables() + .expect("flush B"); + before.ingest("a", 1.0, 1_000, 20); + let live = before.scan_hosts(); + assert_eq!(live, vec!["a".to_string(), "b".to_string()]); + drop(before); + + let mut after = restart_and_replay(dir.path(), &records); + assert_eq!( + after.scan_hosts(), + live, + "replay equals live: A once, B once" + ); + after + .core + .checkpoint_timeseries_memtables() + .expect("flush A"); + drop(after); + + let mut again = restart_and_replay(dir.path(), &records); + assert_eq!( + again.scan_hosts(), + live, + "every record is named now: a second restart applies nothing" + ); + } + + /// A live ingest below the highest flushed LSN is a record that was still + /// on its way when the flush ran: it applies. A redelivery of a flushed + /// record is named by the stamp and writes nothing. + #[test] + fn a_live_ingest_below_the_highest_flushed_lsn_applies() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut core = Core::open_at(dir.path()); + observe_floor(&mut core, 10); + core.ingest("b", 2.0, 2_000, 30); + core.core + .checkpoint_timeseries_memtables() + .expect("flush B"); + + core.ingest("a", 1.0, 1_000, 20); + assert_eq!(core.scan_hosts(), vec!["a".to_string(), "b".to_string()]); + core.ingest("b", 2.0, 2_000, 30); + assert_eq!( + core.scan_hosts(), + vec!["a".to_string(), "b".to_string()], + "the redelivered flushed record writes nothing" + ); + } + + /// Restart replay asks the collection stamp, record by record. + #[test] + fn the_collection_stamp_gates_replay_record_by_record() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut core = Core::open_at(dir.path()); + let key = (DatabaseId::DEFAULT, TenantId::new(TID), COLL.to_string()); + core.core.ts_replay_stamps.insert( + key, + super::super::stamp::TsReplayStamp { + rows: crate::types::replay_stamp::ReplayStamp::through(5) + .union(&crate::types::replay_stamp::ReplayStamp::naming(10)), + truncates: crate::types::replay_stamp::ReplayStamp::default(), + }, + ); + let records = [ + ingest_record("h05", 1.0, 1_000, 5), + ingest_record("h07", 1.0, 2_000, 7), + ingest_record("h10", 1.0, 3_000, 10), + ingest_record("h11", 1.0, 4_000, 11), + ]; + core.core + .replay_timeseries_wal(&records, 1, &nodedb_wal::TombstoneSet::new()); + assert_eq!( + core.scan_hosts(), + vec!["h07".to_string(), "h11".to_string()] + ); + } + + /// X at lsn 10 applies, a truncate at lsn 20 removes it, then A at lsn 15 + /// applies. A restart must not re-apply the truncate over A, and must not + /// bring X back. + #[test] + fn a_write_in_flight_at_a_truncate_survives_replay() { + let dir = tempfile::tempdir().expect("tempdir"); + let records = [ + ingest_record("x", 1.0, 1_000, 10), + ingest_record("a", 2.0, 2_000, 15), + truncate_record(20), + ]; + + let mut before = Core::open_at(dir.path()); + before.ingest("x", 1.0, 1_000, 10); + truncate(&mut before, 20); + before.ingest("a", 2.0, 2_000, 15); + let live = before.scan_hosts(); + assert_eq!(live, vec!["a".to_string()]); + drop(before); + + let mut after = restart_and_replay(dir.path(), &records); + assert_eq!( + after.scan_hosts(), + live, + "the truncate took effect before the crash: replay skips it and \ + everything it removed, and applies the write that followed it" + ); + } } diff --git a/nodedb/src/data/executor/timeseries_checkpoint/load.rs b/nodedb/src/data/executor/timeseries_checkpoint/load.rs index 8a51dfe3c..3411cdc74 100644 --- a/nodedb/src/data/executor/timeseries_checkpoint/load.rs +++ b/nodedb/src/data/executor/timeseries_checkpoint/load.rs @@ -3,9 +3,9 @@ //! Rebuilding `ts_registries` from the partition directories on disk. //! //! The partitions ARE the timeseries checkpoint, so this is its loader: it makes -//! the flushed rows reachable again, and it installs the per-collection -//! `last_flushed_wal_lsn` gate that stops `replay_timeseries_wal` re-appending -//! records those partitions already contain. +//! the flushed rows reachable again, and it installs the per-collection replay +//! stamp (`stamp.rs`) that stops `replay_timeseries_wal` re-appending records +//! those partitions already contain. //! //! A partition directory is committed by its `partition.meta`, which //! `ColumnarSegmentWriter::write_partition` writes last and atomically. A @@ -14,7 +14,8 @@ //! `PartitionRegistry::cleanup_orphans` would remove it. A meta that is present //! but cannot be read or decoded is NOT ignored: that is corruption of a //! committed partition, and silently dropping it would under-restore the -//! collection while the LSN gate it carries says its records need no replay. +//! collection while its stamp says its records need no replay. The same holds +//! for a stamp file that is present but does not decode. use tracing::info; @@ -25,21 +26,16 @@ use crate::types::{DatabaseId, TenantId}; impl CoreLoop { /// Rebuild every timeseries collection's partition registry from disk. /// - /// Called once at boot, BEFORE `replay_all_wal`. Until it existed - /// `ts_registries` was populated only lazily — by the first SCAN of a - /// collection, via [`Self::ensure_ts_registry`] — which happens long after - /// replay. Replay's dedup gate reads those registries, so at replay time it - /// found no partitions, gated nothing, and re-appended every retained - /// `TimeseriesBatch` on top of the partition that already held it. A - /// timeseries ingest is an append and the scan reads partitions and memtable - /// together, so nothing masked the duplicate rows. + /// Called once at boot, BEFORE `replay_all_wal`. Replay's skip gate reads + /// the collection stamps this loads. Loaded lazily instead, at the first + /// scan, they would be absent at replay: every retained `TimeseriesBatch` + /// would append again on top of the partition that holds it, and the scan + /// reads partitions and memtable together, so nothing masks the duplicate. /// /// A collection whose registry cannot be rebuilt is fail-stop: a - /// `partition.meta` that exists but will not decode is corruption of a - /// committed partition this core is about to claim is durable, and the WAL - /// below its `last_flushed_wal_lsn` gate may already be gone. Skipping it - /// quietly would under-restore the collection while replay still trusts - /// the gate it never installed. + /// `partition.meta` or stamp that exists but will not decode is corruption + /// of committed state this core is about to claim is durable, and the WAL + /// behind it may already be gone. pub fn load_ts_registries(&mut self) -> crate::Result<()> { let ts_root = self.data_dir.join("ts"); if !ts_root.exists() { @@ -72,7 +68,8 @@ impl CoreLoop { Ok(()) } - /// Ensure the partition registry is loaded for one timeseries collection. + /// Ensure the partition registry and the replay stamp are loaded for one + /// timeseries collection. /// /// A no-op once the collection's registry is present, so the boot load above /// and the lazy first-scan path can both call it. @@ -97,6 +94,12 @@ impl CoreLoop { } let registry = read_registry(&ts_dir, self.segment_keks.ts_segment_kek.as_ref())?; + let stamp = super::stamp::read_collection_ts_stamp(&ts_dir, ®istry)?; + let merged = match self.ts_replay_stamps.get(&key) { + Some(held) => held.union(&stamp), + None => stamp, + }; + self.ts_replay_stamps.insert(key.clone(), merged); if registry.partition_count() > 0 { info!( collection, diff --git a/nodedb/src/data/executor/timeseries_checkpoint/mod.rs b/nodedb/src/data/executor/timeseries_checkpoint/mod.rs index 409ede566..da418cfb3 100644 --- a/nodedb/src/data/executor/timeseries_checkpoint/mod.rs +++ b/nodedb/src/data/executor/timeseries_checkpoint/mod.rs @@ -7,14 +7,12 @@ //! `columnar_memtables` holds one `ColumnarMemtable` per timeseries collection, //! and every ILP / JSON / msgpack ingest lands there. Those rows advance the //! core watermark (`execute_timeseries_ingest` calls `note_collection_write_lsn` -//! with the Control Plane's `wal_lsn`), so the periodic checkpoint reported them -//! as durable and the manager truncated the `TimeseriesBatch` records that were -//! their only copy. The memtable reached a segment only when the ingest path -//! crossed its 64 MiB threshold or when the idle timer in -//! `handlers/compact/maintenance.rs` happened to fire — neither of which is -//! ordered against the truncation the checkpoint authorises. A restart on a -//! collection ingesting below the threshold returned it missing every row since -//! the last idle flush. +//! with the Control Plane's `wal_lsn`). The periodic checkpoint reports them as +//! durable, and the manager then removes the `TimeseriesBatch` records below +//! that LSN. The checkpoint must therefore flush every memtable first. The +//! ingest path's 64 MiB threshold and the idle timer in +//! `handlers/compact/maintenance.rs` are not ordered against that removal, so +//! they cannot stand in for it. //! //! ## Why this reuses `flush_ts_collection` rather than writing a checkpoint blob //! @@ -33,8 +31,7 @@ //! A checkpoint blob would persist a second, redundant copy of state the engine //! already knows how to write and read back — and a worse one: it would bypass //! the column codecs, the sparse index, and the merge path that later compacts -//! those partitions. The bug was never a missing format; it was that the only -//! flush was a threshold and a timer. +//! those partitions. //! //! ## What LSN is durable after a flush //! @@ -44,40 +41,24 @@ //! every row with `lsn <= watermark` is in a memtable. Flushing every non-empty //! memtable therefore puts all of them in a partition. //! -//! The `last_flushed_wal_lsn` each partition records is a different and narrower -//! number: the highest `TimeseriesBatch` LSN folded into it -//! (`ts_max_ingested_lsn`), which is what replay's dedup gate compares against. -//! It is not the checkpoint's answer and must not be — the watermark is core-wide -//! across every engine, while that stamp is per-collection and counts only -//! timeseries records. -//! -//! A collection whose memtable is empty flushes nothing and leaves its last -//! partition's stamp where it stands. That is not a gap: an empty memtable means -//! every row ever ingested into it is already in a partition, so the watermark is -//! durable for that collection either way. -//! -//! ## Why no `ReplayFloors` field +//! ## What restart replay skips //! //! Timeseries replay is NOT idempotent — an ingest is an APPEND, and //! `raw_scan` reads partitions and memtable together, so re-applying a record -//! already folded into a partition shows every one of its rows twice. It needs a -//! floor, and it already HAS one, older and finer than `ReplayFloors`: -//! `replay_timeseries_wal` skips any record at or below the highest -//! `last_flushed_wal_lsn` across the collection's registered partitions. +//! already folded into a partition shows every one of its rows twice. Each +//! collection therefore carries a replay stamp (`stamp`): the records whose +//! rows a partition holds or a truncate removed. Replay skips exactly those. +//! A record in flight when a flush ran is not named, even when a higher LSN +//! is, so it replays once. //! -//! That gate is per-collection, which is what timeseries needs and what a -//! `ReplayFloors` field could not give it — those are engine-wide, because a KV -//! or columnar record can span two collections. A `TimeseriesBatch` names -//! exactly one. Adding a second, coarser floor beside the existing one would -//! gate records the partitions do not contain. +//! The stamp is per collection, which is what timeseries needs and what a +//! `ReplayFloors` field could not give it — those are engine-wide, because a +//! KV or columnar record can span two collections. A `TimeseriesBatch` names +//! exactly one. //! -//! What that gate DID lack is a boot-time load: `ts_registries` was populated -//! only lazily, by the first scan of a collection (`ensure_ts_registry`), which -//! runs long after `replay_all_wal`. So at replay the registry was empty, the -//! gate found no partitions, and every retained record replayed on top of the -//! partitions that already held it. [`CoreLoop::load_ts_registries`] closes that -//! by populating the registries before replay — which is exactly the role -//! `load_kv_checkpoints` plays for `ReplayFloors::kv`. +//! [`CoreLoop::load_ts_registries`] loads the stamps before replay, which is +//! the role `load_kv_checkpoints` plays for `ReplayFloors::kv`. mod flush; mod load; +pub mod stamp; diff --git a/nodedb/src/data/executor/timeseries_checkpoint/stamp.rs b/nodedb/src/data/executor/timeseries_checkpoint/stamp.rs new file mode 100644 index 000000000..950d0a99c --- /dev/null +++ b/nodedb/src/data/executor/timeseries_checkpoint/stamp.rs @@ -0,0 +1,324 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The replay stamp of a timeseries collection. +//! +//! ## What it names +//! +//! A timeseries ingest is an append, so restart replay must apply a record +//! exactly once. [`TsReplayStamp::rows`] names the records whose rows a +//! partition holds or a truncate removed. Replay skips exactly those. A record +//! still on its way to the core when a flush ran is not named, even when a +//! higher LSN is, so it replays once. +//! +//! [`TsReplayStamp::truncates`] names the truncates that took effect. Replay +//! skips a named truncate: re-applying it would remove rows written after it. +//! +//! ## Where it lives +//! +//! - Every partition directory carries the collection stamp as of its flush, +//! in `replay.stamp`. It is written before `partition.meta`, the partition's +//! commit point, so a committed partition always carries the stamp naming +//! its rows. +//! - The collection directory carries `replay.stamp` too. A truncate writes it +//! into the fresh directory it leaves behind. Retention writes it before it +//! removes a partition, so the names of the removed partition survive it. +//! +//! The collection stamp is the union of every copy. Stamps only grow, so the +//! union never names a record no copy names. +//! +//! ## Where a flush's stamp comes from +//! +//! A live flush takes the core's applied-prefix stamp, after the ingest that +//! filled the memtable noted its record. Restart replay applies the WAL in +//! LSN order, and the core's floor is already raised to the WAL end, so the +//! core stamp would name records replay has not reached. A flush during +//! restart replay therefore names the records through the replay cursor +//! instead: every record the timeseries pass has passed. + +use std::path::Path; + +use crate::data::executor::core_loop::CoreLoop; +use crate::engine::timeseries::partition_registry::PartitionRegistry; +use crate::types::replay_stamp::ReplayStamp; +use crate::types::{DatabaseId, Lsn, TenantId}; + +/// File name of a stamp, in a partition directory and in a collection +/// directory alike. Neither starts with `ts-`, so the partition scan and the +/// orphan sweep pass over it. +pub(in crate::data::executor) const TS_STAMP_FILE: &str = "replay.stamp"; + +/// Leading byte of an encoded stamp file. +const TS_STAMP_FORMAT_VERSION: u8 = 1; + +/// The key of one timeseries collection on a core. +pub(in crate::data::executor) type TsCollectionKey = (DatabaseId, TenantId, String); + +/// What a timeseries collection's durable state holds. +#[derive( + Debug, Clone, Default, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack, +)] +pub(in crate::data::executor) struct TsReplayStamp { + /// Records whose rows a partition holds or a truncate removed. + pub rows: ReplayStamp, + /// Truncates of the collection that took effect. + pub truncates: ReplayStamp, +} + +impl TsReplayStamp { + /// A stamp naming what `self` or `other` names. + pub(in crate::data::executor) fn union(&self, other: &TsReplayStamp) -> TsReplayStamp { + TsReplayStamp { + rows: self.rows.union(&other.rows), + truncates: self.truncates.union(&other.truncates), + } + } +} + +fn stamp_error(path: &Path, action: &str, detail: &dyn std::fmt::Display) -> crate::Error { + crate::Error::Storage { + engine: "timeseries".to_string(), + detail: format!( + "replay stamp: failed to {action} {}: {detail}", + path.display() + ), + } +} + +/// Write `stamp` into `dir` durably: data fsynced before the rename, `dir` +/// fsynced after it. +pub(in crate::data::executor) fn write_ts_stamp( + dir: &Path, + stamp: &TsReplayStamp, +) -> crate::Result<()> { + let mut bytes = vec![TS_STAMP_FORMAT_VERSION]; + let body = zerompk::to_msgpack_vec(stamp) + .map_err(|e| stamp_error(&dir.join(TS_STAMP_FILE), "encode", &e))?; + bytes.extend_from_slice(&body); + nodedb_wal::segment::atomic_write_fsync(dir, TS_STAMP_FILE, &bytes) + .map_err(|e| stamp_error(&dir.join(TS_STAMP_FILE), "write", &e)) +} + +/// Decode a stamp file's bytes. `path` names the file in errors. +pub(in crate::data::executor) fn decode_ts_stamp( + path: &Path, + bytes: &[u8], +) -> crate::Result { + let Some((&version, body)) = bytes.split_first() else { + return Err(stamp_error(path, "decode", &"the file is empty")); + }; + if version != TS_STAMP_FORMAT_VERSION { + return Err(stamp_error( + path, + "decode", + &format!("format version {version}, expected {TS_STAMP_FORMAT_VERSION}"), + )); + } + let stamp: TsReplayStamp = + zerompk::from_msgpack(body).map_err(|e| stamp_error(path, "decode", &e))?; + stamp + .rows + .validate() + .map_err(|e| stamp_error(path, "validate the row stamp of", &e))?; + stamp + .truncates + .validate() + .map_err(|e| stamp_error(path, "validate the truncate stamp of", &e))?; + Ok(stamp) +} + +/// Read the stamp in `dir`. A missing file names nothing. +pub(in crate::data::executor) fn read_ts_stamp(dir: &Path) -> crate::Result> { + let path = dir.join(TS_STAMP_FILE); + let bytes = match std::fs::read(&path) { + Ok(bytes) => bytes, + Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(None), + Err(e) => return Err(stamp_error(&path, "read", &e)), + }; + decode_ts_stamp(&path, &bytes).map(Some) +} + +/// The collection stamp of `ts_dir`: the union of its own file and the file +/// of every committed partition `registry` names. +pub(in crate::data::executor) fn read_collection_ts_stamp( + ts_dir: &Path, + registry: &PartitionRegistry, +) -> crate::Result { + let mut stamp = read_ts_stamp(ts_dir)?.unwrap_or_default(); + for (_, entry) in registry.iter() { + if let Some(partition) = read_ts_stamp(&ts_dir.join(&entry.dir_name))? { + stamp = stamp.union(&partition); + } + } + Ok(stamp) +} + +impl CoreLoop { + /// Whether restart replay skips the timeseries record at `lsn` for `key`: + /// the collection stamp names it. A committed-redo apply never skips + /// (`replay_watermark_skips`). + pub(in crate::data::executor) fn ts_replay_skips( + &self, + key: &TsCollectionKey, + lsn: u64, + ) -> bool { + self.replay_watermark_skips(self.ts_rows_named(key, lsn)) + } + + /// Whether `key`'s collection stamp names the record at `lsn` as held by + /// a partition or removed by a truncate. + pub(in crate::data::executor) fn ts_rows_named(&self, key: &TsCollectionKey, lsn: u64) -> bool { + self.ts_replay_stamps + .get(key) + .is_some_and(|stamp| stamp.rows.skips(lsn)) + } + + /// Whether `key`'s collection stamp names the truncate at `lsn` as taken + /// effect. + pub(in crate::data::executor) fn ts_truncate_named( + &self, + key: &TsCollectionKey, + lsn: u64, + ) -> bool { + self.ts_replay_stamps + .get(key) + .is_some_and(|stamp| stamp.truncates.skips(lsn)) + } + + /// The records this core applied, as a flush or truncate writing now + /// states them. See the module docs for the restart-replay cursor. + pub(in crate::data::executor) fn ts_applied_stamp(&self) -> crate::Result { + match self.ts_replay_cursor { + Some(through) => Ok(ReplayStamp::through(through)), + None => Ok(self.floors.applied_prefix.stamp()?), + } + } + + /// The stamp a flush of `key` writing now carries: the collection stamp, + /// plus every record this core applied. + pub(in crate::data::executor) fn ts_flush_stamp( + &self, + key: &TsCollectionKey, + ) -> crate::Result { + let applied = TsReplayStamp { + rows: self.ts_applied_stamp()?, + truncates: ReplayStamp::default(), + }; + Ok(self + .ts_replay_stamps + .get(key) + .map_or_else(|| applied.clone(), |stamp| stamp.union(&applied))) + } + + /// Record that a timeseries record at `lsn` applied. A flush after this + /// names the record. Called once the record's rows landed and before any + /// flush that can write them. + pub(in crate::data::executor) fn note_ts_record_applied(&mut self, lsn: u64) { + self.floors.applied_prefix.note_applied(Lsn::new(lsn)); + if let Some(cursor) = self.ts_replay_cursor.as_mut() { + *cursor = (*cursor).max(lsn); + } + } + + /// Write `key`'s collection stamp into its collection directory. Called + /// before a partition directory is removed, so the stamp names that + /// partition's records after it is gone. + pub(in crate::data::executor) fn persist_ts_collection_stamp( + &self, + key: &TsCollectionKey, + ) -> crate::Result<()> { + let Some(stamp) = self.ts_replay_stamps.get(key) else { + return Ok(()); + }; + let (database_id, tenant_id, collection) = key; + let dir = crate::data::executor::handlers::timeseries::paths::ts_collection_dir( + &self.data_dir, + database_id.as_u64(), + tenant_id.as_u64(), + collection, + ); + write_ts_stamp(&dir, stamp) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn named(prefix: u64, lsns: &[u64]) -> ReplayStamp { + let mut stamp = ReplayStamp::through(prefix); + for &lsn in lsns { + stamp = stamp.union(&ReplayStamp::naming(lsn)); + } + stamp + } + + #[test] + fn a_stamp_file_round_trips() { + let dir = tempfile::tempdir().expect("tempdir"); + let stamp = TsReplayStamp { + rows: named(5, &[9, 12]), + truncates: named(0, &[7]), + }; + write_ts_stamp(dir.path(), &stamp).expect("write"); + assert_eq!(read_ts_stamp(dir.path()).expect("read"), Some(stamp)); + let empty = tempfile::tempdir().expect("tempdir"); + assert_eq!(read_ts_stamp(empty.path()).expect("read"), None); + } + + #[test] + fn a_corrupt_stamp_file_is_an_error() { + let dir = tempfile::tempdir().expect("tempdir"); + std::fs::write(dir.path().join(TS_STAMP_FILE), [9u8, 1, 2]).expect("write"); + assert!(read_ts_stamp(dir.path()).is_err()); + std::fs::write(dir.path().join(TS_STAMP_FILE), []).expect("write"); + assert!(read_ts_stamp(dir.path()).is_err()); + } + + #[test] + fn the_collection_stamp_is_the_union_of_every_copy() { + use crate::engine::timeseries::partition_registry::PartitionEntry; + + let dir = tempfile::tempdir().expect("tempdir"); + write_ts_stamp( + dir.path(), + &TsReplayStamp { + rows: named(3, &[]), + truncates: named(0, &[2]), + }, + ) + .expect("collection stamp"); + let mut registry = PartitionRegistry::new( + nodedb_types::timeseries::TieredPartitionConfig::origin_defaults(), + ); + for (name, stamp) in [("ts-a", named(3, &[8])), ("ts-b", named(3, &[6]))] { + let partition = dir.path().join(name); + std::fs::create_dir_all(&partition).expect("partition dir"); + write_ts_stamp( + &partition, + &TsReplayStamp { + rows: stamp, + truncates: ReplayStamp::default(), + }, + ) + .expect("partition stamp"); + registry.insert_partition(PartitionEntry { + meta: nodedb_types::timeseries::PartitionMeta { + min_ts: 0, + max_ts: 0, + row_count: 1, + size_bytes: 0, + schema_version: 1, + state: nodedb_types::timeseries::PartitionState::Sealed, + interval_ms: 0, + last_flushed_wal_lsn: 0, + column_stats: std::collections::HashMap::new(), + max_system_ts: 0, + }, + dir_name: name.to_string(), + }); + } + let stamp = read_collection_ts_stamp(dir.path(), ®istry).expect("union"); + assert_eq!(stamp.rows, named(3, &[6, 8])); + assert_eq!(stamp.truncates, named(0, &[2])); + } +} diff --git a/nodedb/src/data/executor/vector_checkpoint/format.rs b/nodedb/src/data/executor/vector_checkpoint/format.rs index 6220c99d5..bfd3649f3 100644 --- a/nodedb/src/data/executor/vector_checkpoint/format.rs +++ b/nodedb/src/data/executor/vector_checkpoint/format.rs @@ -12,12 +12,14 @@ use serde::{Deserialize, Serialize}; +use crate::types::replay_stamp::ReplayStamp; + /// On-disk format version for the manifest and the generation it names. /// /// A manifest stamped with any other version is refused rather than misparsed. /// Refusing costs a WAL replay; misparsing would install indexes built from /// bytes this build cannot read. -pub(crate) const VECTOR_CKPT_FORMAT_VERSION: u16 = 1; +pub(crate) const VECTOR_CKPT_FORMAT_VERSION: u16 = 2; /// Names the live generation. Writing this file is what publishes a checkpoint. #[derive( @@ -42,6 +44,9 @@ pub(crate) struct VectorCheckpointManifest { /// flush clamps WAL truncation to. Without it the first failed flush after /// a restart would pin truncation at zero. pub durable_through_lsn: u64, + /// The records every index in the generation holds. Restart replay skips + /// a vector record exactly when this stamp names it. + pub replay: ReplayStamp, } /// Encode a manifest publishing `generation`, for tests that need a live @@ -56,6 +61,7 @@ pub(crate) fn test_manifest_bytes(generation: u64) -> Vec { format_version: VECTOR_CKPT_FORMAT_VERSION, generation, durable_through_lsn: 0, + replay: ReplayStamp::default(), }) .expect("manifest encode is infallible for this fixed struct") } diff --git a/nodedb/src/data/executor/vector_checkpoint/load.rs b/nodedb/src/data/executor/vector_checkpoint/load.rs index d32cfdf9a..07f2f84ef 100644 --- a/nodedb/src/data/executor/vector_checkpoint/load.rs +++ b/nodedb/src/data/executor/vector_checkpoint/load.rs @@ -68,7 +68,10 @@ impl CoreLoop { // clamps truncation to, so claiming it over a half-restored generation // would authorise deleting the records that would have completed it. self.floors.vector_durable_lsn = Lsn::new(manifest.durable_through_lsn); - self.floors.vector_published_lsn = Lsn::new(manifest.durable_through_lsn); + self.floors.vector_published_lsn = Lsn::new(manifest.replay.prefix); + let replay_prefix = manifest.replay.prefix; + let applied_ranges = manifest.replay.applied_above.len(); + self.floors.replay_floors.vector.set(manifest.replay); info!( core = self.core_id, @@ -76,6 +79,8 @@ impl CoreLoop { loaded, vectors, durable_through_lsn = manifest.durable_through_lsn, + replay_prefix, + applied_ranges, "vector checkpoint restored" ); Ok(()) @@ -180,6 +185,7 @@ mod tests { format_version: VECTOR_CKPT_FORMAT_VERSION + 1, generation: 0, durable_through_lsn: 5, + replay: crate::types::replay_stamp::ReplayStamp::default(), }) .expect("encode"); nodedb_wal::segment::write_checkpoint_framed(&ckpt_dir, VECTOR_CKPT_MANIFEST, &bytes) @@ -214,4 +220,70 @@ mod tests { .load_vector_checkpoints() .expect_err("a corrupt index file must fail the load, not skip it"); } + + /// A bare `VectorPut` record (collection, vector, dim) at `lsn`. + fn vector_put(lsn: u64, vector: Vec) -> nodedb_wal::WalRecord { + let dim = vector.len(); + nodedb_wal::WalRecord::new(nodedb_wal::record::WalRecordArgs { + record_type: nodedb_wal::record::RecordType::VectorPut as u32, + lsn, + tenant_id: 1, + vshard_id: 0, + database_id: 0, + payload: zerompk::to_msgpack_vec(&("emb", vector, dim)).expect("encode put"), + encryption_key: None, + preamble_bytes: None, + }) + .expect("build record") + } + + /// Record B at LSN 30 applies and the checkpoint is written while record A + /// at LSN 20 is still on its way. A applies after the checkpoint, and the + /// core stops before another one. Replay applies A once and skips B, so + /// the replayed index equals the live one. + #[test] + fn a_vector_write_in_flight_at_a_checkpoint_replays_once() { + use crate::engine::vector::hnsw::HnswParams; + + let dir = tempfile::tempdir().expect("tempdir"); + let key = CoreLoop::vector_index_key(0, 1, "emb", ""); + let live = { + let mut core = open_core_at(dir.path()); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(10)); + let mut collection = VectorCollection::new(3, HnswParams::default()); + collection.insert(vec![0.0, 0.0, 1.0]); + core.vector_collections.insert(key.clone(), collection); + core.floors.applied_prefix.note_applied(Lsn::new(30)); + core.checkpoint_vector_indexes().expect("checkpoint"); + if let Some(collection) = core.vector_collections.get_mut(&key) { + collection.insert(vec![1.0, 0.0, 0.0]); + } + core.floors.applied_prefix.note_applied(Lsn::new(20)); + core.vector_collections.get(&key).map(|c| c.len()) + }; + assert_eq!(live, Some(2)); + + let mut restored = open_core_at(dir.path()); + restored.load_vector_checkpoints().expect("load"); + assert_eq!( + restored.vector_collections.get(&key).map(|c| c.len()), + Some(1), + "the checkpoint holds B and not A" + ); + restored.replay_vector_wal( + &[ + vector_put(20, vec![1.0, 0.0, 0.0]), + vector_put(30, vec![0.0, 0.0, 1.0]), + ], + 1, + &nodedb_wal::TombstoneSet::new(), + ); + assert_eq!( + restored.vector_collections.get(&key).map(|c| c.len()), + live, + "replay applies A once and never applies B again" + ); + } } diff --git a/nodedb/src/data/executor/vector_checkpoint/manifest.rs b/nodedb/src/data/executor/vector_checkpoint/manifest.rs index bc9d84041..074aa8f07 100644 --- a/nodedb/src/data/executor/vector_checkpoint/manifest.rs +++ b/nodedb/src/data/executor/vector_checkpoint/manifest.rs @@ -46,6 +46,13 @@ pub(crate) fn read_vector_manifest_at( expected: VECTOR_CKPT_FORMAT_VERSION, }); } + manifest + .replay + .validate() + .map_err(|source| CheckpointDecodeError::InvalidReplayStamp { + path: path.clone(), + source, + })?; Ok(Some(manifest)) } diff --git a/nodedb/src/data/executor/vector_checkpoint/mod.rs b/nodedb/src/data/executor/vector_checkpoint/mod.rs index c520b82fa..2e58d0955 100644 --- a/nodedb/src/data/executor/vector_checkpoint/mod.rs +++ b/nodedb/src/data/executor/vector_checkpoint/mod.rs @@ -29,12 +29,14 @@ //! restores as "no vectors" instead of as last cycle's contents, and a torn or //! abandoned write is inert garbage rather than a half-published state. //! -//! ## Why no replay floor -//! -//! Vector WAL records are idempotent against a restored index — `VectorOp::Insert` -//! upserts by surrogate and a delete of an absent vector is a no-op — so replay -//! above and below the generation's stamp both reproduce the same index. The -//! stamp is carried anyway, because it is what a failed flush clamps WAL +//! ## Replay floor +//! +//! The manifest carries the core's replay stamp: every record the generation +//! holds. A restart installs it as the vector replay floor. An HNSW insert +//! appends a node and never dedups, so a record the stamp names must not +//! replay. Every other record replays in LSN order, including a lower-LSN +//! record still in flight when the generation was written. The manifest also +//! carries `durable_through_lsn`, the point a failed flush clamps WAL //! truncation to after a restart. mod build_completions; diff --git a/nodedb/src/data/executor/vector_checkpoint/publish.rs b/nodedb/src/data/executor/vector_checkpoint/publish.rs index 6eaa559d6..ed460b9fe 100644 --- a/nodedb/src/data/executor/vector_checkpoint/publish.rs +++ b/nodedb/src/data/executor/vector_checkpoint/publish.rs @@ -11,6 +11,7 @@ use super::format::{VECTOR_CKPT_FORMAT_VERSION, VectorCheckpointManifest}; use super::manifest::{read_vector_manifest_at, storage_err}; use super::paths::VECTOR_CKPT_MANIFEST; use crate::types::Lsn; +use crate::types::replay_stamp::ReplayStamp; /// The generation number the next publish under `ckpt_dir` must use. /// @@ -24,18 +25,21 @@ pub(crate) fn next_generation(ckpt_dir: &std::path::Path) -> crate::Result /// Publish a written generation by atomically replacing the manifest. /// /// This single write is the commit point of the whole checkpoint: before it -/// nothing changed; after it the entire generation is live at one LSN. It also +/// nothing changed; after it the entire generation is live, holding the +/// records `replay` names. It also /// fsyncs `ckpt_dir`, the same directory holding the `gen-{n}/` entry, so that /// entry cannot still be pending when the manifest naming it becomes visible. pub(crate) fn publish_vector_generation( ckpt_dir: &std::path::Path, generation: u64, durable_through: Lsn, + replay: ReplayStamp, ) -> crate::Result<()> { let manifest = VectorCheckpointManifest { format_version: VECTOR_CKPT_FORMAT_VERSION, generation, durable_through_lsn: durable_through.as_u64(), + replay, }; let bytes = zerompk::to_msgpack_vec(&manifest).map_err(|e| crate::Error::Serialization { format: "msgpack".to_string(), diff --git a/nodedb/src/data/executor/vector_checkpoint/write.rs b/nodedb/src/data/executor/vector_checkpoint/write.rs index 8eaaac560..65c6220e9 100644 --- a/nodedb/src/data/executor/vector_checkpoint/write.rs +++ b/nodedb/src/data/executor/vector_checkpoint/write.rs @@ -10,6 +10,7 @@ use super::paths::{vector_ckpt_dir, vector_ckpt_gen_dir}; use super::publish::publish_vector_generation; use crate::data::executor::checkpoint_outcome::CheckpointOutcome; use crate::data::executor::core_loop::CoreLoop; +use crate::types::Lsn; impl CoreLoop { /// Flush every vector index to disk and report the LSN they are now durable @@ -41,13 +42,17 @@ impl CoreLoop { /// the watermark only after the collection has already been mutated, so /// every write with `lsn <= watermark` is in the bytes written below. /// - /// Every published generation raises `vector_published_lsn`, whichever - /// caller published it: restart restores from the newest generation, so - /// a committed record applied at or below it must reach a newer one - /// (`redo_apply::cover`). `vector_durable_lsn` stays the caller's to - /// raise. + /// The manifest carries the core's replay stamp: the records every index + /// in the generation holds. Restart replay skips exactly those. + /// + /// Every published generation raises `vector_published_lsn` to the + /// stamp's prefix, whichever caller published it: restart restores from + /// the newest generation, so a committed record applied at or below it + /// must reach a newer one (`redo_apply::cover`). `vector_durable_lsn` + /// stays the caller's to raise. pub(crate) fn checkpoint_vector_indexes(&mut self) -> crate::Result { let durable_lsn = self.watermark; + let replay = self.floors.applied_prefix.stamp()?; let ckpt_dir = vector_ckpt_dir(&self.data_dir, self.core_id); std::fs::create_dir_all(&ckpt_dir).map_err(|e| storage_err(&ckpt_dir, "create dir", &e))?; @@ -69,8 +74,10 @@ impl CoreLoop { .map_err(|e| storage_err(&gen_dir, "create generation dir", &e))?; let files_written = self.write_vector_generation(&gen_dir)?; - publish_vector_generation(&ckpt_dir, generation, durable_lsn)?; - self.floors.vector_published_lsn = self.floors.vector_published_lsn.max(durable_lsn); + let prefix = Lsn::new(replay.prefix); + let applied_ranges = replay.applied_above.len(); + publish_vector_generation(&ckpt_dir, generation, durable_lsn, replay)?; + self.floors.vector_published_lsn = self.floors.vector_published_lsn.max(prefix); // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what @@ -97,6 +104,8 @@ impl CoreLoop { files_written, total = self.vector_collections.len(), durable_through_lsn = durable_lsn.as_u64(), + replay_prefix = prefix.as_u64(), + applied_ranges, "vector checkpoint published" ); Ok(CheckpointOutcome { diff --git a/nodedb/src/data/executor/wal_replay/array.rs b/nodedb/src/data/executor/wal_replay/array.rs index 9e6c22472..1c63e111f 100644 --- a/nodedb/src/data/executor/wal_replay/array.rs +++ b/nodedb/src/data/executor/wal_replay/array.rs @@ -2,18 +2,21 @@ //! Array engine WAL replay: rebuilds tile state after crash. //! -//! ## The durable watermark +//! ## The replay stamp //! -//! Each array's manifest carries a `durable_lsn` — the highest LSN whose cells -//! are already inside a flushed, on-disk segment (`Manifest::add_segment` -//! raises it). Replay skips every record at or below it, which is what makes -//! replaying the same retained tail twice a no-op and, more importantly, what -//! stops a cell version that a bitemporal audit purge physically removed from a -//! segment being re-materialised out of a still-retained `ArrayPut`. Without -//! the gate the purge is silently undone on the next boot. +//! Each array's manifest carries the `ReplayStamp` of its newest flush: the +//! records whose cells are already inside a flushed, on-disk segment. Replay +//! skips exactly the records that stamp names (`array_replay_skips`). A +//! record in flight when the flush ran is not named, even when a higher LSN +//! is, so it replays once. //! -//! Records ABOVE the watermark are the tail the segments have not absorbed and -//! must be re-applied into the memtable. +//! Re-applying a named record would write its tile version into a second +//! segment on the next flush. After a bitemporal audit purge it would also +//! re-materialise the cell version the purge removed from a segment. +//! +//! Replay never flushes. Its records land in the memtable, and the first +//! live threshold flush or checkpoint flush after replay writes them with a +//! stamp that names them. use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::transaction::undo::UndoEntry; @@ -92,17 +95,37 @@ impl CoreLoop { self.record_redo_capture(captured) } - /// The LSN this array's flushed segments are already durable through. - /// - /// `0` when the store is not open, which gates nothing — the safe - /// direction, matching every other engine's unset replay floor. - pub(in crate::data::executor) fn array_durable_lsn( + /// Whether restart replay skips the record at `record_lsn` for + /// `array_id`: the stamp in the array's manifest names it. A + /// committed-redo apply never skips (`replay_watermark_skips`). + pub(in crate::data::executor) fn array_replay_skips( + &self, + array_id: &nodedb_array::types::ArrayId, + record_lsn: u64, + ) -> bool { + self.replay_watermark_skips(self.array_stamp_names(array_id, record_lsn)) + } + + /// Whether the stamp in `array_id`'s manifest names the record at + /// `record_lsn`. An array whose store is not open names nothing, which + /// replays the record: the safe direction. + pub(in crate::data::executor) fn array_stamp_names( &self, array_id: &nodedb_array::types::ArrayId, - ) -> u64 { + record_lsn: u64, + ) -> bool { + self.array_engine + .store(array_id) + .is_ok_and(|store| store.manifest().replay.skips(record_lsn)) + } + + /// Whether `array_id`'s manifest stamp names an LSN above `record_lsn`. + /// A replayed record it holds was in flight when the flush that wrote the + /// stamp ran. + fn array_stamp_passes(&self, array_id: &nodedb_array::types::ArrayId, record_lsn: u64) -> bool { self.array_engine .store(array_id) - .map_or(0, |store| store.manifest().durable_lsn) + .is_ok_and(|store| store.manifest().replay.highest() > record_lsn) } pub fn replay_array_wal( @@ -117,6 +140,8 @@ impl CoreLoop { let mut puts = 0usize; let mut deletes = 0usize; let mut skipped = 0usize; + // Records replayed below the highest LSN their array's stamp names. + let mut in_flight = 0usize; for record in records { let logical_type = record.logical_record_type(); @@ -189,17 +214,14 @@ impl CoreLoop { continue; } } - if self - .replay_watermark_skips(record_lsn <= self.array_durable_lsn(&payload.array_id)) - { + if self.array_replay_skips(&payload.array_id, record_lsn) { skipped += 1; continue; } if self.claim_for_validation() { continue; } - let installing = self.recording_redo_undo(); - if installing + if self.recording_redo_undo() && !self.record_array_tiles_undo( &payload.array_id, payload @@ -214,18 +236,10 @@ impl CoreLoop { } let cell_count = payload.cells.len(); let prov = payload.provenance.clone(); - // An install flushes once the whole record landed, so a - // rollback finds its cells in the memtable. - let stamped = if installing { - self.array_engine.put_cells_unflushed( - &payload.array_id, - payload.cells, - record_lsn, - ) - } else { + let passed = self.array_stamp_passes(&payload.array_id, record_lsn); + let stamped = self.array_engine - .put_cells(&payload.array_id, payload.cells, record_lsn) - }; + .put_cells(&payload.array_id, payload.cells, record_lsn); if let Err(e) = stamped { self.replay_record_unapplied( "array", @@ -237,6 +251,7 @@ impl CoreLoop { continue; } puts += cell_count; + in_flight += usize::from(passed); self.note_redo_array_written(&payload.array_id); // Rebuild the per-core HWM frontier from the WAL record's // provenance. No fence check here — replay records are already @@ -295,16 +310,14 @@ impl CoreLoop { continue; } } - if self.replay_watermark_skips(record_lsn <= self.array_durable_lsn(&payload.array_id)) - { + if self.array_replay_skips(&payload.array_id, record_lsn) { skipped += 1; continue; } if self.claim_for_validation() { continue; } - let installing = self.recording_redo_undo(); - if installing + if self.recording_redo_undo() && !self.record_array_tiles_undo( &payload.array_id, payload @@ -319,16 +332,10 @@ impl CoreLoop { } let cell_count = payload.cells.len(); let prov = payload.provenance.clone(); - let stamped = if installing { - self.array_engine.delete_cells_unflushed( - &payload.array_id, - payload.cells, - record_lsn, - ) - } else { + let passed = self.array_stamp_passes(&payload.array_id, record_lsn); + let stamped = self.array_engine - .delete_cells(&payload.array_id, payload.cells, record_lsn) - }; + .delete_cells(&payload.array_id, payload.cells, record_lsn); if let Err(e) = stamped { self.replay_record_unapplied( "array", @@ -340,6 +347,7 @@ impl CoreLoop { continue; } deletes += cell_count; + in_flight += usize::from(passed); self.note_redo_array_written(&payload.array_id); if let Some(p) = &prov { self.sync_commit(p); @@ -352,6 +360,7 @@ impl CoreLoop { puts, deletes, skipped, + in_flight, "WAL array replay complete" ); } diff --git a/nodedb/src/data/executor/wal_replay_columnar_dml.rs b/nodedb/src/data/executor/wal_replay_columnar_dml.rs index 548584098..25b0e0632 100644 --- a/nodedb/src/data/executor/wal_replay_columnar_dml.rs +++ b/nodedb/src/data/executor/wal_replay_columnar_dml.rs @@ -38,10 +38,10 @@ //! state and must still replay. Without the gate the checkpoint would duplicate //! rows; without the replay above it the checkpoint would lose them. //! -//! Note the gate is NOT the `last_flushed_wal_lsn` watermark of the timeseries -//! profile: that field exists only on `nodedb_types::timeseries::PartitionMeta`, -//! used by the separate `ts_registries` / bucketed-partition machinery, which -//! this op pair never targets. +//! Note the gate is NOT the timeseries collection stamp +//! (`timeseries_checkpoint::stamp`): that stamp covers the separate +//! `ts_registries` / bucketed-partition machinery, which this op pair never +//! targets. use super::core_loop::CoreLoop; use crate::bridge::envelope::{PhysicalPlan, Status}; diff --git a/nodedb/src/data/executor/wal_replay_columnar_truncate.rs b/nodedb/src/data/executor/wal_replay_columnar_truncate.rs index edaab36bc..8ec0a0fe7 100644 --- a/nodedb/src/data/executor/wal_replay_columnar_truncate.rs +++ b/nodedb/src/data/executor/wal_replay_columnar_truncate.rs @@ -7,9 +7,15 @@ //! //! Both records route through the live handlers, so replay leaves the //! engines exactly as the live truncate did. A `TimeseriesBatch` record -//! at or below a collection's truncate LSN describes a row the truncate -//! removed; `replay_timeseries_wal` skips it through [`TruncateFloors`] -//! rather than ingesting it only to wipe it again at the truncate's LSN. +//! below a collection's truncate LSN describes a row the truncate removed; +//! `replay_timeseries_wal` skips it through [`TruncateFloors`] rather than +//! ingesting it only to wipe it again at the truncate's LSN. +//! +//! A timeseries truncate that took effect before the crash is named by its +//! collection's replay stamp. Replay never re-applies it: it would remove +//! rows written after it. It imposes no floor either. The stamp names every +//! record it removed, and a record it does not name applied after the +//! truncate and survives it. use std::collections::HashMap; @@ -18,33 +24,55 @@ use nodedb_wal::WalRecord; use nodedb_wal::record::RecordType; use super::core_loop::CoreLoop; +use super::timeseries_checkpoint::stamp::TsReplayStamp; use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use nodedb_physical::physical_plan::{ColumnarOp, TimeseriesOp}; -/// Highest truncate LSN per collection on one core, read off the WAL before -/// the engine pass replays it. +/// The key of one columnar-family collection on a core. +type CollectionKey = (DatabaseId, TenantId, String); + +/// The truncates a replay pass must honour, read off the WAL before the +/// engine pass replays it. #[derive(Debug, Default)] pub(in crate::data::executor) struct TruncateFloors { - floors: HashMap<(DatabaseId, TenantId, String), u64>, + /// Highest LSN of a truncate that has not taken effect, per collection. + /// Its replay removes every earlier record of the collection. + floors: HashMap, + /// Timeseries truncates the collection stamp names and the pass has not + /// reached yet. A sub-record at such an LSN precedes the truncate in its + /// redo group, so the truncate removed its rows. + named_ahead: HashMap>, +} + +/// The collection a truncate record names, or `None` for any other record. +fn truncate_key(record: &WalRecord) -> Option { + let is_truncate = matches!( + RecordType::from_raw(record.logical_record_type()), + Some(RecordType::ColumnarTruncate) | Some(RecordType::TimeseriesTruncate) + ); + if !is_truncate { + return None; + } + let payload = zerompk::from_msgpack::(&record.payload).ok()?; + Some(( + DatabaseId::new(record.header.database_id), + TenantId::new(record.header.tenant_id), + payload.collection, + )) } impl TruncateFloors { - /// Collect the truncate records that route to `core_id`. + /// Collect the truncate records that route to `core_id`. `stamps` holds + /// the timeseries collection stamps restored from disk. pub(in crate::data::executor) fn collect( records: &[WalRecord], num_cores: usize, core_id: usize, + stamps: &HashMap, ) -> Self { - let mut floors = HashMap::new(); + let mut this = Self::default(); for record in records { - let is_truncate = matches!( - RecordType::from_raw(record.logical_record_type()), - Some(RecordType::ColumnarTruncate) | Some(RecordType::TimeseriesTruncate) - ); - if !is_truncate { - continue; - } let target_core = if num_cores > 0 { record.header.vshard_id as usize % num_cores } else { @@ -53,35 +81,50 @@ impl TruncateFloors { if target_core != core_id { continue; } - let Ok(payload) = zerompk::from_msgpack::(&record.payload) - else { + let Some(key) = truncate_key(record) else { continue; }; - let key = ( - DatabaseId::new(record.header.database_id), - TenantId::new(record.header.tenant_id), - payload.collection, - ); - let floor = floors.entry(key).or_insert(0u64); - *floor = (*floor).max(record.header.lsn); + let lsn = record.header.lsn; + let took_effect = stamps + .get(&key) + .is_some_and(|stamp| stamp.truncates.skips(lsn)); + if took_effect { + this.named_ahead.entry(key).or_default().push(lsn); + } else { + let floor = this.floors.entry(key).or_insert(0u64); + *floor = (*floor).max(lsn); + } + } + this + } + + /// The pass reached the truncate `record`: sub-records after it in its + /// redo group apply. + pub(in crate::data::executor) fn pass(&mut self, record: &WalRecord) { + let Some(key) = truncate_key(record) else { + return; + }; + if let Some(ahead) = self.named_ahead.get_mut(&key) { + ahead.retain(|lsn| *lsn != record.header.lsn); } - Self { floors } } - /// Whether a record at `lsn` for `key` precedes a truncate of that - /// collection and is therefore already removed. + /// Whether a record at `lsn` for `key` is removed by a truncate of that + /// collection. /// - /// Strictly below: a record at the truncate's own LSN is a sibling - /// sub-record of the same transaction redo group. Replay applies a group's - /// sub-records in the order the transaction wrote them, so a row staged - /// before the truncate is removed by the truncate's own replay, and a row - /// staged after it must apply. - pub(in crate::data::executor) fn covers( - &self, - key: &(DatabaseId, TenantId, String), - lsn: u64, - ) -> bool { + /// Strictly below a truncate that has not taken effect: a record at the + /// truncate's own LSN is a sibling sub-record of the same transaction + /// redo group. Replay applies a group's sub-records in the order the + /// transaction wrote them, so a row staged before the truncate is removed + /// by the truncate's own replay, and a row staged after it must apply. A + /// truncate that took effect is not replayed, so a sibling ahead of it is + /// skipped here instead. + pub(in crate::data::executor) fn covers(&self, key: &CollectionKey, lsn: u64) -> bool { self.floors.get(key).is_some_and(|floor| lsn < *floor) + || self + .named_ahead + .get(key) + .is_some_and(|ahead| ahead.contains(&lsn)) } /// Same as [`Self::covers`], keyed by the collection's parts. The key @@ -94,7 +137,7 @@ impl TruncateFloors { collection: &str, lsn: u64, ) -> bool { - if self.floors.is_empty() { + if self.floors.is_empty() && self.named_ahead.is_empty() { return false; } self.covers(&(database_id, tenant_id, collection.to_string()), lsn) @@ -206,9 +249,8 @@ impl CoreLoop { } /// Replay one `TimeseriesTruncate` record through the live handler. - /// Gated by the partitions on disk: a partition flushed at or above this - /// LSN was written after the truncate, so the truncate already happened - /// and re-applying it would remove rows written after it. + /// Gated by the collection's replay stamp: a truncate it names already + /// took effect, and re-applying it would remove rows written after it. pub(in crate::data::executor) fn replay_timeseries_truncate( &mut self, payload: &[u8], @@ -239,12 +281,7 @@ impl CoreLoop { } let tid = TenantId::new(tenant_id); let key = (database_id, tid, record.collection.clone()); - let flushed_after_truncate = self.ts_registries.get(&key).is_some_and(|registry| { - registry - .iter() - .any(|(_, e)| e.meta.last_flushed_wal_lsn >= record_lsn) - }); - if self.replay_watermark_skips(flushed_after_truncate) { + if self.replay_watermark_skips(self.ts_truncate_named(&key, record_lsn)) { return false; } if self.claim_for_validation() { @@ -314,7 +351,7 @@ mod tests { #[test] fn a_row_at_the_truncates_own_lsn_is_not_covered() { let records = vec![truncate_record(RecordType::ColumnarTruncate, 9, 0, "c")]; - let floors = TruncateFloors::collect(&records, 1, 0); + let floors = TruncateFloors::collect(&records, 1, 0, &HashMap::new()); let c = (DatabaseId::DEFAULT, TenantId::new(1), "c".to_string()); assert!(floors.covers(&c, 8)); assert!( @@ -332,7 +369,7 @@ mod tests { // Routes to another core. truncate_record(RecordType::TimeseriesTruncate, 50, 1, "ts"), ]; - let floors = TruncateFloors::collect(&records, 2, 0); + let floors = TruncateFloors::collect(&records, 2, 0, &HashMap::new()); let c = (DatabaseId::DEFAULT, TenantId::new(1), "c".to_string()); let ts = (DatabaseId::DEFAULT, TenantId::new(1), "ts".to_string()); assert!(floors.covers(&c, 8)); @@ -345,4 +382,43 @@ mod tests { assert!(floors.covers_collection(DatabaseId::DEFAULT, TenantId::new(1), "c", 1)); assert!(!floors.covers_collection(DatabaseId::DEFAULT, TenantId::new(1), "other", 1)); } + + /// A truncate the collection stamp names took effect before the crash. + /// It sets no floor, so a record below it that applied after it + /// replays. A sibling sub-record at its own LSN is skipped until the pass + /// reaches the truncate, and applies after it. + #[test] + fn a_named_truncate_sets_no_floor_and_skips_only_the_siblings_ahead_of_it() { + let ts = (DatabaseId::DEFAULT, TenantId::new(1), "ts".to_string()); + let stamps = HashMap::from([( + ts.clone(), + TsReplayStamp { + rows: crate::types::replay_stamp::ReplayStamp::default(), + truncates: crate::types::replay_stamp::ReplayStamp::naming(20), + }, + )]); + let truncate = truncate_record(RecordType::TimeseriesTruncate, 20, 0, "ts"); + let mut floors = TruncateFloors::collect(std::slice::from_ref(&truncate), 1, 0, &stamps); + + assert!( + !floors.covers(&ts, 15), + "a record in flight at the truncate replays" + ); + assert!( + floors.covers(&ts, 20), + "a sibling ahead of the truncate is skipped" + ); + floors.pass(&truncate); + assert!( + !floors.covers(&ts, 20), + "a sibling after the truncate applies" + ); + + let unnamed = + TruncateFloors::collect(std::slice::from_ref(&truncate), 1, 0, &HashMap::new()); + assert!( + unnamed.covers(&ts, 15), + "a truncate that did not take effect floors" + ); + } } diff --git a/nodedb/src/data/executor/wal_replay_redo_document.rs b/nodedb/src/data/executor/wal_replay_redo_document.rs index 0a0c1ac80..7a6da42e8 100644 --- a/nodedb/src/data/executor/wal_replay_redo_document.rs +++ b/nodedb/src/data/executor/wal_replay_redo_document.rs @@ -747,17 +747,13 @@ mod tests { ); } - /// The raw engine op `VectorCollection::insert` is still append-only (it - /// never dedups), but cross-boot (checkpoint) idempotency holds: the replay - /// gate (`checkpoint_wal_lsn`) is frozen during a replay pass — - /// `note_checkpoint_lsn` only advances the running `applied_wal_lsn` max, - /// so sibling sub-records sharing one `TransactionRedo` LSN all apply - /// instead of the first one gating the rest. The gate only moves at a - /// checkpoint save (folding `applied_wal_lsn` in) and load (exposing it), - /// which is what a real reboot does. This test reproduces that: replay - /// once, round-trip the collection through a checkpoint to install the - /// persisted watermark as the gate, then replay again and assert the - /// record is skipped, leaving ONE copy, not two. + /// The raw engine op `VectorCollection::insert` is append-only (it never + /// dedups), so cross-boot idempotency rests on the restored checkpoint's + /// stamp: a record it names is skipped. The stamp is set only at load, so + /// sibling sub-records sharing one `TransactionRedo` LSN all apply within + /// one replay pass. This test replays once, installs a stamp naming the + /// record as a restored checkpoint does, then replays again and asserts + /// the record is skipped, leaving one copy, not two. #[test] fn redo_vector_insert_idempotent_on_double_replay() { let mut h = make_core(); @@ -768,28 +764,11 @@ mod tests { .replay_transaction_redo_wal(std::slice::from_ref(&record), 1, &tomb) .expect("redo replay must succeed"); - // Simulate a checkpoint capture + reboot: saving folds the applied - // watermark into the persisted gate, and restoring exposes it — so the - // straddling record is now gated on the second replay. + // A checkpoint written after that replay names the record. let key = CoreLoop::vector_index_key(0, 7, "emb", ""); - let bytes = h - .core - .vector_collections - .get(&key) - .expect("collection present after first replay") - .checkpoint_to_bytes(None) - .unwrap(); - let memory = nodedb_mem::ScopedMemory::new( - h.core.governor.clone(), - nodedb_types::DatabaseId::new(0), - crate::types::TenantId::new(7), - nodedb_mem::EngineId::Vector, + h.core.floors.replay_floors.vector.set( + crate::data::executor::applied_prefix::ReplayStamp::through(record.header.lsn), ); - let restored = crate::engine::vector::collection::VectorCollection::from_checkpoint( - &bytes, None, memory, - ) - .expect("decode checkpoint"); - h.core.vector_collections.insert(key.clone(), restored); h.core .replay_transaction_redo_wal(std::slice::from_ref(&record), 1, &tomb) @@ -799,8 +778,7 @@ mod tests { assert_eq!( len, Some(1), - "checkpoint-LSN gate makes replay idempotent: a record at or below the \ - collection's recorded watermark is skipped on re-replay" + "a record the restored checkpoint's stamp names is skipped on re-replay" ); } diff --git a/nodedb/src/data/executor/wal_replay_vector.rs b/nodedb/src/data/executor/wal_replay_vector.rs index e02eba487..bcd168f68 100644 --- a/nodedb/src/data/executor/wal_replay_vector.rs +++ b/nodedb/src/data/executor/wal_replay_vector.rs @@ -65,6 +65,18 @@ impl CoreLoop { continue; } + // The restored vector checkpoint holds this put, delete or index + // drop. Its stamp names exactly the records the live core applied + // before the checkpoint, so every other record replays, in LSN + // order, on top of it: a lower-LSN record still in flight at the + // checkpoint applies here as it applied live, after the higher + // ones. `VectorParams` is configuration the catalog seeds at boot, + // not index content the checkpoint holds, so it always replays. + if !is_vector_params && self.vector_replay_skips(record_lsn) { + skipped += 1; + continue; + } + if is_index_drop { // Applied in LSN order, so it wipes the params / puts that // preceded it and leaves a later re-CREATE to rebuild. @@ -135,26 +147,12 @@ impl CoreLoop { skipped += 1; continue; } - // Checkpoint watermark gate: a restored checkpoint already - // contains every write at or below its `checkpoint_wal_lsn`. - // Re-applying a straddling-segment record would append a - // duplicate HNSW node (`insert_with_surrogate` never dedups), - // so skip it. Records above the watermark are the WAL tail - // the checkpoint has not yet absorbed and must replay. let insert_index_key = CoreLoop::vector_index_key( database_id, tenant_id, &collection, &field_name, ); - if self.replay_watermark_skips( - self.vector_collections - .get(&insert_index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { - skipped += 1; - continue; - } let surrogate = nodedb_types::Surrogate::new(surrogate_u32); if !self.redo_vector_prelude( RedoVectorWrite { @@ -222,12 +220,6 @@ impl CoreLoop { skipped += 1; continue; } - // Advance the (possibly freshly created) collection's - // watermark so the next checkpoint records this replayed - // write and a subsequent restart does not re-apply it. - if let Some(coll) = self.vector_collections.get_mut(&insert_index_key) { - coll.note_checkpoint_lsn(record_lsn); - } inserted += 1; } else if let Ok((collection, vector, dim, field_name, doc_id)) = zerompk::from_msgpack::<(String, Vec, usize, String, Option)>( @@ -267,15 +259,6 @@ impl CoreLoop { &collection, &field_name, ); - // Checkpoint watermark gate (see the surrogate arm above). - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { - skipped += 1; - continue; - } let params = self .vector_params .get(&index_key) @@ -318,7 +301,6 @@ impl CoreLoop { // local-id-only and bind to `Surrogate::ZERO`. let _ = doc_id; index.insert_with_surrogate(vector, nodedb_types::Surrogate::ZERO); - index.note_checkpoint_lsn(record_lsn); inserted += 1; } else if let Ok((collection, vector, dim)) = zerompk::from_msgpack::<(String, Vec, usize)>(&record.payload) @@ -352,15 +334,6 @@ impl CoreLoop { } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, ""); - // Checkpoint watermark gate (see the surrogate arm above). - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { - skipped += 1; - continue; - } let params = self .vector_params .get(&index_key) @@ -398,7 +371,6 @@ impl CoreLoop { continue; } index.insert(vector); - index.note_checkpoint_lsn(record_lsn); inserted += 1; } else if let Ok((collection, vectors, dim)) = zerompk::from_msgpack::<(String, Vec>, usize)>(&record.payload) @@ -409,15 +381,6 @@ impl CoreLoop { } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, ""); - // Checkpoint watermark gate (see the surrogate arm above). - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { - skipped += 1; - continue; - } if !self.redo_vector_prelude( RedoVectorWrite { index_key: &index_key, @@ -453,7 +416,6 @@ impl CoreLoop { for vector in vectors { index.insert(vector); } - index.note_checkpoint_lsn(record_lsn); inserted += 1; } } else if is_vector_delete { @@ -487,6 +449,8 @@ impl CoreLoop { #[cfg(test)] mod tests { use super::*; + use crate::data::executor::applied_prefix::ReplayStamp; + use crate::data::executor::applied_prefix::stamp::LsnRange; use crate::engine::vector::collection::VectorCollection; use crate::engine::vector::hnsw::HnswParams; use std::sync::Arc; @@ -549,35 +513,21 @@ mod tests { .expect("wal record") } - /// Simulate `load_vector_checkpoints` restoring a checkpoint that already - /// contains one vector, stamped with watermark `lsn`. + /// Simulate `load_vector_checkpoints` restoring a checkpoint that holds + /// one vector of `collection` and names the records in `stamp`. fn restore_checkpoint( core: &mut CoreLoop, tenant_id: u64, collection: &str, vector: Vec, - lsn: u64, + stamp: ReplayStamp, ) { let dim = vector.len(); let mut coll = VectorCollection::new(dim, HnswParams::default()); coll.insert(vector); - coll.note_checkpoint_lsn(lsn); - // Round-trip through a checkpoint so the persisted watermark becomes the - // replay gate (`checkpoint_wal_lsn`): save folds the applied watermark - // into it, and load exposes it — faithfully simulating a restored - // checkpoint (the gate is set only by load/save, never by a live - // `note_checkpoint_lsn`, which feeds the separate applied watermark). - let bytes = coll.checkpoint_to_bytes(None).unwrap(); - let memory = nodedb_mem::ScopedMemory::new( - core.governor.clone(), - nodedb_types::DatabaseId::new(0), - crate::types::TenantId::new(tenant_id), - nodedb_mem::EngineId::Vector, - ); - let coll = - VectorCollection::from_checkpoint(&bytes, None, memory).expect("decode checkpoint"); let key = CoreLoop::vector_index_key(0, tenant_id, collection, ""); core.vector_collections.insert(key, coll); + core.floors.replay_floors.vector.set(stamp); } fn coll_len(core: &CoreLoop, tenant_id: u64, collection: &str) -> Option { @@ -585,69 +535,70 @@ mod tests { core.vector_collections.get(&key).map(|c| c.len()) } - /// The regression: a WAL record at LSN N whose write the restored checkpoint - /// (watermark N) already absorbed must NOT be replayed — otherwise the - /// straddling segment's record appends a duplicate HNSW node. Before the - /// checkpoint-LSN gate this left TWO copies. + fn replay(core: &mut CoreLoop, records: &[nodedb_wal::WalRecord]) { + core.replay_vector_wal(records, 1, &nodedb_wal::TombstoneSet::new()); + } + + /// A record the restored checkpoint's stamp names is not replayed: an HNSW + /// insert never dedups, so replaying it appends a second node. #[test] - fn straddling_record_not_reapplied_over_checkpoint() { + fn a_record_the_stamp_names_is_not_reapplied() { let mut h = make_core(); - restore_checkpoint(&mut h.core, 7, "emb", vec![1.0, 2.0, 3.0], 10); - let rec = vector_put_record(10, 7, "emb", vec![1.0, 2.0, 3.0]); - h.core.replay_vector_wal( - std::slice::from_ref(&rec), - 1, - &nodedb_wal::TombstoneSet::new(), + restore_checkpoint( + &mut h.core, + 7, + "emb", + vec![1.0, 2.0, 3.0], + ReplayStamp::through(10), ); - assert_eq!( - coll_len(&h.core, 7, "emb"), - Some(1), - "a record at/below the restored checkpoint watermark must be skipped exactly once" + replay( + &mut h.core, + &[vector_put_record(10, 7, "emb", vec![1.0, 2.0, 3.0])], ); + assert_eq!(coll_len(&h.core, 7, "emb"), Some(1)); } - /// A record above the restored watermark is the genuine WAL tail the - /// checkpoint has not absorbed and MUST replay. + /// A record above the stamp's prefix and outside its applied ranges is the + /// WAL tail the checkpoint does not hold, and it replays. #[test] - fn record_above_watermark_still_replays() { + fn a_record_the_stamp_does_not_name_replays() { let mut h = make_core(); - restore_checkpoint(&mut h.core, 7, "emb", vec![1.0, 2.0, 3.0], 10); - let rec = vector_put_record(11, 7, "emb", vec![4.0, 5.0, 6.0]); - h.core.replay_vector_wal( - std::slice::from_ref(&rec), - 1, - &nodedb_wal::TombstoneSet::new(), + restore_checkpoint( + &mut h.core, + 7, + "emb", + vec![1.0, 2.0, 3.0], + ReplayStamp::through(10), ); - assert_eq!( - coll_len(&h.core, 7, "emb"), - Some(2), - "a record above the watermark is the WAL tail and must replay" + replay( + &mut h.core, + &[vector_put_record(11, 7, "emb", vec![4.0, 5.0, 6.0])], ); + assert_eq!(coll_len(&h.core, 7, "emb"), Some(2)); } - /// A checkpoint restored for collection A must not suppress replay of a - /// record for collection B, even when B's record LSN is below A's watermark. + /// Record 7 was still on its way when record 10 applied and the checkpoint + /// was written. The checkpoint holds 10 and not 7. A stamp at the highest + /// applied LSN would skip 7 and lose its vector. #[test] - fn checkpoint_watermark_is_per_collection() { + fn a_record_in_flight_below_an_applied_one_replays_once() { let mut h = make_core(); - restore_checkpoint(&mut h.core, 7, "col_a", vec![1.0, 2.0, 3.0], 10); - // Collection B has no checkpoint; its record at LSN 5 (below A's - // watermark of 10) must still replay. - let rec = vector_put_record(5, 7, "col_b", vec![7.0, 8.0, 9.0]); - h.core.replay_vector_wal( - std::slice::from_ref(&rec), - 1, - &nodedb_wal::TombstoneSet::new(), + let stamp = ReplayStamp { + prefix: 5, + applied_above: vec![LsnRange { start: 10, end: 10 }], + }; + restore_checkpoint(&mut h.core, 7, "emb", vec![1.0, 2.0, 3.0], stamp); + replay( + &mut h.core, + &[ + vector_put_record(7, 7, "emb", vec![4.0, 5.0, 6.0]), + vector_put_record(10, 7, "emb", vec![1.0, 2.0, 3.0]), + ], ); assert_eq!( - coll_len(&h.core, 7, "col_b"), - Some(1), - "collection A's watermark must not gate collection B's records" - ); - assert_eq!( - coll_len(&h.core, 7, "col_a"), - Some(1), - "collection A must be untouched by B's replay" + coll_len(&h.core, 7, "emb"), + Some(2), + "record 7 applies once and record 10 does not apply again" ); } } diff --git a/nodedb/src/data/executor/wal_replay_vector_direct.rs b/nodedb/src/data/executor/wal_replay_vector_direct.rs index 48499d36a..571348ee7 100644 --- a/nodedb/src/data/executor/wal_replay_vector_direct.rs +++ b/nodedb/src/data/executor/wal_replay_vector_direct.rs @@ -46,11 +46,7 @@ impl CoreLoop { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { + if self.vector_replay_skips(record_lsn) { return false; } if !self.redo_vector_targets_prelude( @@ -103,9 +99,6 @@ impl CoreLoop { ); return false; } - if let Some(coll) = self.vector_collections.get_mut(&index_key) { - coll.note_checkpoint_lsn(record_lsn); - } true } @@ -135,11 +128,7 @@ impl CoreLoop { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { + if self.vector_replay_skips(record_lsn) { return false; } if self.claim_for_validation() { @@ -176,9 +165,6 @@ impl CoreLoop { ); return false; } - if let Some(coll) = self.vector_collections.get_mut(&index_key) { - coll.note_checkpoint_lsn(record_lsn); - } true } @@ -216,11 +202,7 @@ impl CoreLoop { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { + if self.vector_replay_skips(record_lsn) { return false; } if !self.redo_vector_targets_prelude( @@ -283,9 +265,6 @@ impl CoreLoop { ); return false; } - if let Some(coll) = self.vector_collections.get_mut(&index_key) { - coll.note_checkpoint_lsn(record_lsn); - } true } } diff --git a/nodedb/src/data/executor/wal_replay_vector_extended.rs b/nodedb/src/data/executor/wal_replay_vector_extended.rs index 78dccdd7c..0c9a96e7e 100644 --- a/nodedb/src/data/executor/wal_replay_vector_extended.rs +++ b/nodedb/src/data/executor/wal_replay_vector_extended.rs @@ -5,23 +5,20 @@ //! insert/delete, and multi-vector (ColBERT-style) insert/delete. //! //! Runs during startup after `replay_vector_wal` (so `VectorParams` are -//! already registered) and after the vector / sparse checkpoints are loaded -//! (so the per-collection checkpoint watermark gates re-application). +//! already registered) and after the vector and sparse checkpoints are loaded +//! (so their replay stamps gate re-application). //! -//! ## Watermark discipline +//! ## Stamp discipline //! -//! Direct-upsert and multi-vector writes append HNSW nodes, which are **not** +//! Direct-upsert and multi-vector writes append HNSW nodes, which are not //! idempotent under double replay (`insert_with_surrogate` / -//! `insert_multi_vector` never dedup). They are therefore gated by the -//! per-collection `checkpoint_wal_lsn`: a restored checkpoint already contains -//! every write at or below its watermark, so records at/below it are skipped; -//! records above it are the WAL tail the checkpoint has not absorbed and are -//! applied, after which the watermark is advanced. +//! `insert_multi_vector` never dedup). A record the restored vector +//! checkpoint's stamp names is skipped (`vector_replay_skips`). Every other +//! record replays, including a lower-LSN record that was still in flight when +//! the checkpoint was written. //! -//! Sparse insert/delete need no watermark: the sparse index upserts by -//! `doc_id` (a re-inserted document replaces its own postings) and delete of -//! an absent document is a no-op, so full re-application over a restored -//! checkpoint reproduces the exact same state. +//! Sparse insert and delete are gated the same way by the sparse-vector +//! checkpoint's stamp (`sparse_vector_replay_skips`). use nodedb_physical::physical_plan::VectorOp; use nodedb_wal::record::RecordType; @@ -202,13 +199,9 @@ impl CoreLoop { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - // Watermark gate: a restored checkpoint already holds every write at or - // below its watermark; re-applying would append a duplicate HNSW node. - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { + // The restored checkpoint holds this write; re-applying it appends a + // duplicate HNSW node. + if self.vector_replay_skips(record_lsn) { return false; } let surrogate = nodedb_types::Surrogate::new(surrogate_u32); @@ -282,12 +275,6 @@ impl CoreLoop { ); return false; } - // Advance the (possibly freshly created) collection's watermark so the - // next checkpoint records this replayed write and a later restart does - // not re-apply it. - if let Some(coll) = self.vector_collections.get_mut(&index_key) { - coll.note_checkpoint_lsn(record_lsn); - } true } @@ -317,11 +304,7 @@ impl CoreLoop { } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field_name); - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { + if self.vector_replay_skips(record_lsn) { return false; } let document_surrogate = nodedb_types::Surrogate::new(doc_surrogate_u32); @@ -378,9 +361,6 @@ impl CoreLoop { ); return false; } - if let Some(coll) = self.vector_collections.get_mut(&index_key) { - coll.note_checkpoint_lsn(record_lsn); - } true } @@ -412,11 +392,7 @@ impl CoreLoop { } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field_name); - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { + if self.vector_replay_skips(record_lsn) { return false; } let document_surrogate = nodedb_types::Surrogate::new(doc_surrogate_u32); @@ -459,9 +435,6 @@ impl CoreLoop { &field_name, document_surrogate, ); - if let Some(coll) = self.vector_collections.get_mut(&index_key) { - coll.note_checkpoint_lsn(record_lsn); - } true } } @@ -582,11 +555,10 @@ mod tests { ); } - /// A `DirectUpsert` record at or below the collection's restored checkpoint - /// watermark must NOT be re-applied (no duplicate HNSW node); one above it - /// must replay. + /// A `DirectUpsert` record the restored checkpoint's stamp names is not + /// re-applied (no duplicate HNSW node); one it does not name replays. #[test] - fn direct_upsert_watermark_gates_replay() { + fn direct_upsert_stamp_gates_replay() { // Record LSNs are 1 (below/at) and 2 (above) after two appends. let records = append_via_autocommit(&[ direct_upsert_plan(1, vec![1.0, 2.0, 3.0]), @@ -595,25 +567,14 @@ mod tests { assert_eq!(records.len(), 2); let mut h = make_core(); - // Simulate a restored checkpoint holding the first write, watermarked at - // its LSN (1). The second write (LSN 2) is the un-absorbed WAL tail. + // A restored checkpoint holding the first write, whose stamp names its + // LSN. The second write is the WAL tail the checkpoint does not hold. let mut coll = VectorCollection::new(3, HnswParams::default()); coll.insert_with_surrogate(vec![1.0, 2.0, 3.0], Surrogate::new(1)); - coll.note_checkpoint_lsn(records[0].header.lsn); - // Round-trip through a checkpoint so the persisted watermark becomes the - // replay gate (`checkpoint_wal_lsn`): a live `note_checkpoint_lsn` only - // feeds the applied watermark, which save folds into the gate and load - // restores — the faithful shape of a restored checkpoint. - let bytes = coll.checkpoint_to_bytes(None).unwrap(); - let memory = nodedb_mem::ScopedMemory::new( - h.core.governor.clone(), - DatabaseId::DEFAULT, - TenantId::new(TID), - nodedb_mem::EngineId::Vector, - ); - let coll = - VectorCollection::from_checkpoint(&bytes, None, memory).expect("decode checkpoint"); h.core.vector_collections.insert(du_index_key(), coll); + h.core.floors.replay_floors.vector.set( + crate::data::executor::applied_prefix::ReplayStamp::through(records[0].header.lsn), + ); h.core .replay_vector_extended_wal(&records, 1, &nodedb_wal::TombstoneSet::new()); @@ -626,7 +587,7 @@ mod tests { assert_eq!( coll.len(), 2, - "record at/below watermark skipped, record above replayed (no duplicate)" + "the record the stamp names is skipped and the other replays (no duplicate)" ); } @@ -660,6 +621,29 @@ mod tests { assert_eq!(idx.doc_count(), 1, "the sparse document must be recovered"); } + /// A sparse insert the restored sparse-vector checkpoint's stamp names is + /// not replayed. + #[test] + fn a_sparse_insert_the_stamp_names_is_skipped() { + let plan = PhysicalPlan::Vector(VectorOp::SparseInsert { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "sc"), + field_name: "sv".into(), + doc_id: "d1".into(), + entries: vec![(10, 0.5)], + }); + let records = append_via_autocommit(&[plan]); + let mut h = make_core(); + h.core.floors.replay_floors.sparse_vector.set( + crate::data::executor::applied_prefix::ReplayStamp::through(records[0].header.lsn), + ); + h.core + .replay_vector_extended_wal(&records, 1, &nodedb_wal::TombstoneSet::new()); + assert!( + !h.core.sparse_vector_indexes.contains_key(&sparse_key("sv")), + "the checkpoint holds the insert, so replay does not apply it again" + ); + } + #[test] fn sparse_delete_survives_wal_replay() { let insert = PhysicalPlan::Vector(VectorOp::SparseInsert { diff --git a/nodedb/src/data/executor/wal_replay_vector_resolved.rs b/nodedb/src/data/executor/wal_replay_vector_resolved.rs index ad124d0cc..d67b30766 100644 --- a/nodedb/src/data/executor/wal_replay_vector_resolved.rs +++ b/nodedb/src/data/executor/wal_replay_vector_resolved.rs @@ -45,11 +45,7 @@ impl CoreLoop { return false; } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field); - if self.replay_watermark_skips( - self.vector_collections - .get(&index_key) - .is_some_and(|existing| record_lsn <= existing.checkpoint_wal_lsn()), - ) { + if self.vector_replay_skips(record_lsn) { return false; } let surrogates: Vec = mutations @@ -123,9 +119,6 @@ impl CoreLoop { if !touched.is_empty() { self.finish_vector_direct_write(&task, &index_key, tenant_id, &collection, &touched); } - if let Some(coll) = self.vector_collections.get_mut(&index_key) { - coll.note_checkpoint_lsn(record_lsn); - } true } } diff --git a/nodedb/src/data/executor/wal_replay_vector_sparse.rs b/nodedb/src/data/executor/wal_replay_vector_sparse.rs index 435f73a56..9700d8b43 100644 --- a/nodedb/src/data/executor/wal_replay_vector_sparse.rs +++ b/nodedb/src/data/executor/wal_replay_vector_sparse.rs @@ -1,8 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! WAL replay for sparse-vector inserts and deletes. Both upsert or remove by -//! `doc_id`, so full re-application over a restored checkpoint reproduces the -//! same state and no watermark gates them. +//! WAL replay for sparse-vector inserts and deletes. +//! +//! Both upsert or remove by `doc_id`. A record the restored sparse-vector +//! checkpoint's stamp names is skipped, and every other record replays in LSN +//! order on top of the checkpoint, the order the live core applied it in. use nodedb_physical::physical_plan::VectorOp; @@ -37,8 +39,8 @@ impl CoreLoop { }]); } - /// Replay one `SparseVectorPut` record. Idempotent upsert-by-`doc_id`, so - /// no watermark gate is required. + /// Replay one `SparseVectorPut` record, unless the restored checkpoint + /// holds it. pub(in crate::data::executor) fn replay_sparse_put( &mut self, payload: &[u8], @@ -62,6 +64,9 @@ impl CoreLoop { if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { return false; } + if self.sparse_vector_replay_skips(record_lsn) { + return false; + } if self.applying_committed_redo() && let Err(e) = nodedb_types::SparseVector::from_entries(entries.clone()) { @@ -112,8 +117,8 @@ impl CoreLoop { true } - /// Replay one `SparseVectorDelete` record. Idempotent (an absent document - /// is a no-op), so no watermark gate is required. + /// Replay one `SparseVectorDelete` record, unless the restored checkpoint + /// holds it. pub(in crate::data::executor) fn replay_sparse_delete( &mut self, payload: &[u8], @@ -137,6 +142,9 @@ impl CoreLoop { if tombstones.is_tombstoned(tenant_id, &collection, record_lsn) { return false; } + if self.sparse_vector_replay_skips(record_lsn) { + return false; + } if self.claim_for_validation() { return false; } diff --git a/nodedb/src/data/runtime/boot_restore.rs b/nodedb/src/data/runtime/boot_restore.rs index c8fe07af0..8412cb5a2 100644 --- a/nodedb/src/data/runtime/boot_restore.rs +++ b/nodedb/src/data/runtime/boot_restore.rs @@ -87,20 +87,15 @@ pub(super) fn load_boot_checkpoints(core: &mut CoreLoop) -> crate::Result<()> { core.load_graph_label_checkpoint()?; // Timeseries needs no checkpoint loader either — its checkpoint IS // the on-disk L1 partitions its flush writes. What it does need is - // its partition REGISTRIES rebuilt from them, because that is where - // replay's per-collection dedup gate lives: a `TimeseriesBatch` at - // or below a partition's `last_flushed_wal_lsn` is already in that - // partition and must not replay. The registries were previously - // built lazily by the first scan of a collection, which happens long - // after `replay_all_wal` — so replay saw no partitions, gated - // nothing, and re-appended every retained record on top of the - // partition that already held it. A timeseries ingest is an append, - // so nothing masked the duplicate rows. A committed `partition.meta` - // that will not decode is fail-stop: it is corruption of state this - // core is about to claim is durable, and skipping it quietly would - // under-restore the collection while leaving its records un-gated. An - // UNCOMMITTED partition directory (no `partition.meta` at all — the - // remains of an interrupted flush) is still a legitimate, silent skip. + // its partition registries and replay stamps rebuilt from them, because + // replay's per-collection skip gate reads the stamps: a + // `TimeseriesBatch` the stamp names is already in a partition, or a + // truncate removed it, and must not replay. A timeseries ingest is an + // append, so nothing masks a duplicate. A committed `partition.meta` or + // stamp that will not decode is fail-stop: it is corruption of state + // this core is about to claim is durable. An UNCOMMITTED partition + // directory (no `partition.meta` at all — the remains of an interrupted + // flush) is a legitimate, silent skip. core.load_ts_registries()?; Ok(()) } diff --git a/nodedb/src/engine/array/compact.rs b/nodedb/src/engine/array/compact.rs index 594e88d8b..ca333bb71 100644 --- a/nodedb/src/engine/array/compact.rs +++ b/nodedb/src/engine/array/compact.rs @@ -66,7 +66,13 @@ mod tests { let mut e = ArrayEngine::new(cfg).unwrap(); e.open_array(aid(), schema(), 0x1).unwrap(); for i in 0..4 { - put_one(&mut e, i, 0, i, (i as u64) + 1); + let lsn = (i as u64) + 1; + put_one(&mut e, i, 0, i, lsn); + e.flush( + &aid(), + crate::types::replay_stamp::ReplayStamp::through(lsn), + ) + .unwrap(); } assert_eq!(e.store(&aid()).unwrap().manifest().segments.len(), 4); let merged = e.maybe_compact(&aid(), None, 0).unwrap(); diff --git a/nodedb/src/engine/array/compaction/merger.rs b/nodedb/src/engine/array/compaction/merger.rs index e0de44c7c..e78d2a44c 100644 --- a/nodedb/src/engine/array/compaction/merger.rs +++ b/nodedb/src/engine/array/compaction/merger.rs @@ -399,7 +399,7 @@ impl MergedTile { #[cfg(test)] mod tests { use crate::engine::array::engine::{ArrayEngine, ArrayEngineConfig}; - use crate::engine::array::test_support::{aid, put_one, schema}; + use crate::engine::array::test_support::{aid, flush_if_full, put_one, schema}; use crate::engine::array::wal::ArrayPutCell; use nodedb_array::types::cell_value::value::CellValue; use nodedb_array::types::coord::value::CoordValue; @@ -419,6 +419,7 @@ mod tests { lsn, ) .unwrap(); + flush_if_full(e, &aid(), lsn); } #[test] @@ -456,9 +457,11 @@ mod tests { cfg.flush_cell_threshold = 1; let mut e = ArrayEngine::new(cfg).unwrap(); e.open_array(aid(), schema(), 0x1).unwrap(); - // Four auto-flushes → four L0 segments → the picker fires. + // Four threshold flushes → four L0 segments → the picker fires. for i in 0..4 { - put_one(&mut e, i, 0, i, (i as u64) + 1); + let lsn = (i as u64) + 1; + put_one(&mut e, i, 0, i, lsn); + flush_if_full(&mut e, &aid(), lsn); } assert!(e.maybe_compact(&aid(), None, 0).unwrap()); @@ -472,6 +475,7 @@ mod tests { // The next flush must not be able to claim that name. put_one(&mut e, 5, 0, 5, 5); + flush_if_full(&mut e, &aid(), 5); let m = e.store(&aid()).unwrap().manifest(); let ids: Vec<&str> = m.segments.iter().map(|s| s.id.as_str()).collect(); @@ -505,7 +509,8 @@ mod tests { // Segment 1: live cell at (1,0) put_one(&mut e, 1, 0, 10, 1); - e.flush(&aid(), 2).unwrap(); + e.flush(&aid(), crate::types::replay_stamp::ReplayStamp::through(2)) + .unwrap(); // Segment 2: tombstone at (2,0) system=200 e.delete_cells( @@ -518,7 +523,8 @@ mod tests { 3, ) .unwrap(); - e.flush(&aid(), 4).unwrap(); + e.flush(&aid(), crate::types::replay_stamp::ReplayStamp::through(4)) + .unwrap(); // Segment 3: GDPR erasure at (3,0) system=300 e.gdpr_erase_cell( @@ -528,11 +534,13 @@ mod tests { 5, ) .unwrap(); - e.flush(&aid(), 6).unwrap(); + e.flush(&aid(), crate::types::replay_stamp::ReplayStamp::through(6)) + .unwrap(); // Segment 4: another live cell to reach the L0_TRIGGER threshold. put_one(&mut e, 4, 0, 40, 7); - e.flush(&aid(), 8).unwrap(); + e.flush(&aid(), crate::types::replay_stamp::ReplayStamp::through(8)) + .unwrap(); loop { if !e.maybe_compact(&aid(), None, 0).unwrap() { diff --git a/nodedb/src/engine/array/engine.rs b/nodedb/src/engine/array/engine.rs index da695ad7d..047a820f0 100644 --- a/nodedb/src/engine/array/engine.rs +++ b/nodedb/src/engine/array/engine.rs @@ -85,6 +85,13 @@ impl ArrayEngine { &self.cfg } + /// Set the memtable cell count at which [`Self::needs_flush`] reports an + /// array full. + #[cfg(test)] + pub(crate) fn set_flush_cell_threshold(&mut self, cells: usize) { + self.cfg.flush_cell_threshold = cells; + } + /// Open or attach to an array. /// /// Idempotent: if the array is already open with the same @@ -364,12 +371,16 @@ mod tests { let mut e = ArrayEngine::new(ArrayEngineConfig::new(dir.path().to_path_buf())).unwrap(); e.open_array(aid.clone(), schema(), 0xBEEF).unwrap(); put_one(&mut e, 1, 1, 7, 1); - e.flush(&aid, 2).unwrap(); + e.flush(&aid, crate::types::replay_stamp::ReplayStamp::through(2)) + .unwrap(); } let mut e = ArrayEngine::new(ArrayEngineConfig::new(dir.path().to_path_buf())).unwrap(); e.open_array(aid.clone(), schema(), 0xBEEF).unwrap(); let m = e.store(&aid).unwrap().manifest(); assert_eq!(m.segments.len(), 1); - assert!(m.durable_lsn > 0); + assert_eq!( + m.replay, + crate::types::replay_stamp::ReplayStamp::through(2) + ); } } diff --git a/nodedb/src/engine/array/flush.rs b/nodedb/src/engine/array/flush.rs index 5500d9bac..88b4ff9ea 100644 --- a/nodedb/src/engine/array/flush.rs +++ b/nodedb/src/engine/array/flush.rs @@ -11,25 +11,31 @@ use nodedb_array::types::{ArrayId, TileId}; use super::engine::{ArrayEngine, ArrayEngineError, ArrayEngineResult}; use super::memtable::{Memtable, TileBuffer}; use super::store::SegmentRef; +use crate::types::replay_stamp::ReplayStamp; impl ArrayEngine { - /// Flush the array's memtable to a new on-disk segment using a - /// caller-supplied LSN as the segment's flush watermark. A no-op if - /// the memtable is empty. + /// Flush the array's memtable to a new on-disk segment and publish the + /// manifest naming it with `stamp`. A no-op if the memtable is empty. + /// + /// `stamp` must name every record whose cells the memtable holds, and no + /// record it does not: restart replay skips exactly the records it names. + /// A core passes its applied-prefix stamp, taken after it noted every + /// record it applied to this array. /// /// The memtable is cleared LAST, once the segment and the manifest naming - /// it are both on disk. Draining it up front — as this did while the only - /// caller was an explicit `NDARRAY_FLUSH` — means an encode or write failure - /// takes the cells out of memory without putting them anywhere: reads stop - /// returning them for the rest of the process's life, and only a restart's - /// WAL replay brings them back. Now every failure path leaves the memtable - /// exactly as it was, so a failed flush costs nothing but the retry, and the - /// caller's clamped checkpoint LSN keeps the WAL records that back it. - pub fn flush(&mut self, id: &ArrayId, wal_lsn: u64) -> ArrayEngineResult> { + /// it are both on disk. Every failure path leaves the memtable exactly as + /// it was: reads keep returning the cells, and a failed flush costs only + /// the retry. The caller's clamped checkpoint LSN keeps the WAL records + /// that back the cells. + pub fn flush( + &mut self, + id: &ArrayId, + stamp: ReplayStamp, + ) -> ArrayEngineResult> { let Some(prepared) = self.prepare_flush(id)? else { return Ok(None); }; - let seg_ref = self.install_flushed_segment(id, prepared, wal_lsn)?; + let seg_ref = self.install_flushed_segment(id, prepared, stamp)?; self.store_mut(id)?.memtable = Memtable::new(); Ok(Some(seg_ref)) } @@ -61,7 +67,7 @@ impl ArrayEngine { &mut self, id: &ArrayId, prepared: PreparedFlush, - flush_lsn: u64, + stamp: ReplayStamp, ) -> ArrayEngineResult { let store = self.store_mut(id)?; let root = store.root().to_path_buf(); @@ -86,10 +92,21 @@ impl ArrayEngine { min_tile: prepared.min_tile.unwrap_or_else(|| TileId::snapshot(0)), max_tile: prepared.max_tile.unwrap_or_else(|| TileId::snapshot(0)), tile_count: prepared.tile_count, - flush_lsn, + flush_lsn: stamp.highest(), }; + // The segment and the stamp naming its records publish together: the + // manifest write below is the commit point for both. store.install_segment(seg_ref.clone())?; - store.persist_manifest()?; + let previous = std::mem::replace(&mut store.manifest_mut().replay, stamp); + if let Err(e) = store.persist_manifest() { + // The manifest on disk names neither the segment nor the stamp, + // and the memtable still holds the cells. Withdraw both from + // memory too, so reads do not see the cells twice and a later + // publish does not claim records this one never made durable. + store.manifest_mut().replay = previous; + store.replace_segments(std::slice::from_ref(&seg_ref.id), Vec::new())?; + return Err(e.into()); + } Ok(seg_ref) } } @@ -141,6 +158,7 @@ fn build_segment_from_memtable<'a>( mod tests { use crate::engine::array::engine::{ArrayEngine, ArrayEngineConfig}; use crate::engine::array::test_support::{aid, put_one, schema}; + use crate::types::replay_stamp::ReplayStamp; use tempfile::TempDir; #[test] @@ -149,7 +167,10 @@ mod tests { let mut e = ArrayEngine::new(ArrayEngineConfig::new(dir.path().to_path_buf())).unwrap(); e.open_array(aid(), schema(), 0xCAFE).unwrap(); put_one(&mut e, 1, 2, 10, 1); - let seg = e.flush(&aid(), 7).unwrap().expect("non-empty flush"); + let seg = e + .flush(&aid(), ReplayStamp::through(7)) + .unwrap() + .expect("non-empty flush"); assert_eq!(seg.level, 0); assert_eq!(seg.tile_count, 1); assert_eq!(seg.flush_lsn, 7); @@ -161,7 +182,7 @@ mod tests { let dir = TempDir::new().unwrap(); let mut e = ArrayEngine::new(ArrayEngineConfig::new(dir.path().to_path_buf())).unwrap(); e.open_array(aid(), schema(), 0x1).unwrap(); - assert!(e.flush(&aid(), 1).unwrap().is_none()); + assert!(e.flush(&aid(), ReplayStamp::through(1)).unwrap().is_none()); } /// A flush that cannot write its segment must leave the memtable untouched. @@ -181,7 +202,7 @@ mod tests { std::fs::remove_dir_all(&root).unwrap(); assert!( - e.flush(&aid(), 7).is_err(), + e.flush(&aid(), ReplayStamp::through(7)).is_err(), "a flush that cannot write its segment must report the failure, not \ swallow it — the caller clamps its checkpoint LSN on this Err" ); @@ -193,11 +214,36 @@ mod tests { // And the retry, once the directory is back, still publishes it. std::fs::create_dir_all(&root).unwrap(); - let seg = e.flush(&aid(), 7).unwrap().expect("retry must flush"); + let seg = e + .flush(&aid(), ReplayStamp::through(7)) + .unwrap() + .expect("retry must flush"); assert_eq!(seg.tile_count, 1); assert!( e.store(&aid()).unwrap().memtable.is_empty(), "a SUCCESSFUL flush must clear the memtable — the cells are durable now" ); } + + /// A flush publishes the caller's stamp in the manifest, and a reopened + /// store reads it back: it is what restart replay decides records by. + #[test] + fn a_flush_publishes_its_stamp_in_the_manifest() { + let dir = TempDir::new().unwrap(); + let stamp = ReplayStamp { + prefix: 5, + applied_above: vec![crate::types::replay_stamp::LsnRange { start: 9, end: 9 }], + }; + { + let mut e = ArrayEngine::new(ArrayEngineConfig::new(dir.path().to_path_buf())).unwrap(); + e.open_array(aid(), schema(), 0xCAFE).unwrap(); + put_one(&mut e, 1, 2, 10, 9); + let seg = e.flush(&aid(), stamp.clone()).unwrap().expect("flush"); + assert_eq!(seg.flush_lsn, 9); + assert_eq!(e.store(&aid()).unwrap().manifest().replay, stamp); + } + let mut e = ArrayEngine::new(ArrayEngineConfig::new(dir.path().to_path_buf())).unwrap(); + e.open_array(aid(), schema(), 0xCAFE).unwrap(); + assert_eq!(e.store(&aid()).unwrap().manifest().replay, stamp); + } } diff --git a/nodedb/src/engine/array/mod.rs b/nodedb/src/engine/array/mod.rs index 64ab7157d..967c59a0d 100644 --- a/nodedb/src/engine/array/mod.rs +++ b/nodedb/src/engine/array/mod.rs @@ -4,9 +4,9 @@ //! //! Lives in the Data Plane: `!Send`, no tokio. Persistence routes through //! the [`wal::ArrayWalAppender`] trait, which Origin wires to the real -//! group-committed WAL writer. Recovery replays WAL records past the -//! last `ArrayFlush` watermark; flushed segments are durable on disk and -//! mmap'd by the segment store on open. +//! group-committed WAL writer. Recovery replays every WAL record the +//! manifest's replay stamp does not name; flushed segments are durable on +//! disk and mmap'd by the segment store on open. pub mod compact; pub mod compaction; diff --git a/nodedb/src/engine/array/purge/execute.rs b/nodedb/src/engine/array/purge/execute.rs index ab632c825..05ea30f39 100644 --- a/nodedb/src/engine/array/purge/execute.rs +++ b/nodedb/src/engine/array/purge/execute.rs @@ -344,7 +344,11 @@ mod tests { e.open_array(test_aid(), schema.clone(), 0x1).unwrap(); put_cell(&mut e, 1, 10, 100, 1); put_cell(&mut e, 1, 60, 600, 2); - e.flush(&test_aid(), 3).unwrap(); + e.flush( + &test_aid(), + crate::types::replay_stamp::ReplayStamp::through(3), + ) + .unwrap(); let horizon = 500; let dropped = run_purge(&mut e, horizon); @@ -367,7 +371,11 @@ mod tests { put_cell(&mut e, 2, 20, 100, 1); e.gdpr_erase_cell(&test_aid(), vec![CoordValue::Int64(2)], 200, 2) .unwrap(); - e.flush(&test_aid(), 3).unwrap(); + e.flush( + &test_aid(), + crate::types::replay_stamp::ReplayStamp::through(3), + ) + .unwrap(); let dropped = run_purge(&mut e, 500); @@ -391,7 +399,11 @@ mod tests { e.open_array(test_aid(), schema.clone(), 0x1).unwrap(); put_cell(&mut e, 3, 30, 100, 1); put_cell(&mut e, 3, 35, 700, 2); - e.flush(&test_aid(), 3).unwrap(); + e.flush( + &test_aid(), + crate::types::replay_stamp::ReplayStamp::through(3), + ) + .unwrap(); let horizon = 500; let d1 = run_purge(&mut e, horizon); diff --git a/nodedb/src/engine/array/purge/plan.rs b/nodedb/src/engine/array/purge/plan.rs index 3aca0edd8..31e4607b3 100644 --- a/nodedb/src/engine/array/purge/plan.rs +++ b/nodedb/src/engine/array/purge/plan.rs @@ -342,6 +342,15 @@ mod tests { lsn, ) .unwrap(); + // The threshold flush the executor runs after every write. + if engine.needs_flush(&test_aid()).unwrap() { + engine + .flush( + &test_aid(), + crate::types::replay_stamp::ReplayStamp::through(lsn), + ) + .unwrap(); + } } #[test] diff --git a/nodedb/src/engine/array/recovery.rs b/nodedb/src/engine/array/recovery.rs index fdce74e11..0f31ff185 100644 --- a/nodedb/src/engine/array/recovery.rs +++ b/nodedb/src/engine/array/recovery.rs @@ -4,10 +4,9 @@ //! //! Recovery is driven by the engine on `open`: the caller streams //! decoded WAL records (filtered to array record types) into -//! [`Recovery::apply_record`]. Records with LSN <= the manifest's -//! `durable_lsn` are skipped — the segment they belong to is already on -//! disk. Records with LSN greater than the durable watermark are -//! re-applied to the live memtable. +//! [`Recovery::apply_record`]. A record the manifest's replay stamp names is +//! skipped — a segment already holds it. Every other record is re-applied +//! to the live memtable. //! //! The recovery layer is intentionally pure with respect to WAL I/O: it //! takes already-decoded payloads. The engine open path reads the @@ -48,9 +47,9 @@ pub enum RecoveryRecord { lsn: u64, payload: ArrayDeletePayload, }, - /// Flush watermarks update the durable_lsn on the matching store. - /// The segment itself is already mmap'd at startup time; this - /// record simply tells us "WAL records up to this LSN are durable". + /// A flush record. The segment it wrote is already mmap'd at startup, + /// and the manifest's stamp already names the records that segment + /// holds, so the record changes nothing here. Flush { lsn: u64, array: nodedb_array::types::ArrayId, @@ -71,10 +70,9 @@ impl<'a> Recovery<'a> { } pub fn apply_record(&mut self, rec: RecoveryRecord) -> Result<(), RecoveryError> { - let durable = self.store.manifest().durable_lsn; match rec { RecoveryRecord::Put { lsn, payload } => { - if lsn <= durable { + if self.store.manifest().replay.skips(lsn) { self.stats.puts_skipped += 1; return Ok(()); } @@ -96,7 +94,7 @@ impl<'a> Recovery<'a> { self.stats.puts_applied += 1; } RecoveryRecord::Delete { lsn, payload } => { - if lsn <= durable { + if self.store.manifest().replay.skips(lsn) { self.stats.deletes_skipped += 1; return Ok(()); } @@ -111,10 +109,7 @@ impl<'a> Recovery<'a> { stamp_delete_cells(self.store, payload.cells, lsn)?; self.stats.deletes_applied += 1; } - RecoveryRecord::Flush { lsn, .. } => { - let m = self.store.manifest_mut(); - m.durable_lsn = m.durable_lsn.max(lsn); - } + RecoveryRecord::Flush { .. } => {} } Ok(()) } diff --git a/nodedb/src/engine/array/rollback.rs b/nodedb/src/engine/array/rollback.rs index 0995ca173..fe3a8fbd1 100644 --- a/nodedb/src/engine/array/rollback.rs +++ b/nodedb/src/engine/array/rollback.rs @@ -5,9 +5,8 @@ //! Every cell write lands in the memtable tile its coordinate and system //! time map to. [`ArrayEngine::snapshot_tiles`] copies those tiles before the //! write, and [`ArrayEngine::restore_tiles`] puts them back. The write must -//! not flush in between: [`ArrayEngine::put_cells_unflushed`] and -//! [`ArrayEngine::delete_cells_unflushed`] stamp the memtable without the -//! threshold flush, and [`ArrayEngine::flush_if_full`] runs it afterwards. +//! not flush in between. `put_cells` and `delete_cells` never flush, so the +//! caller holds the flush until the whole record landed. use nodedb_array::tile::tile_id_for_cell; use nodedb_array::types::coord::value::CoordValue; @@ -15,8 +14,6 @@ use nodedb_array::types::{ArrayId, TileId}; use super::engine::{ArrayEngine, ArrayEngineResult}; use super::memtable::TileBuffer; -use super::wal::{ArrayDeleteCell, ArrayPutCell}; -use super::write::{stamp_delete_cells, stamp_put_cells}; /// The memtable tiles a write touches, as they were before it. #[derive(Debug)] @@ -57,36 +54,4 @@ impl ArrayEngine { } Ok(()) } - - /// Stamp `cells` into the memtable without the threshold flush. - pub fn put_cells_unflushed( - &mut self, - id: &ArrayId, - cells: Vec, - wal_lsn: u64, - ) -> ArrayEngineResult<()> { - if cells.is_empty() { - return Ok(()); - } - stamp_put_cells(self.store_mut(id)?, cells, wal_lsn) - } - - /// Stamp tombstones for `cells` into the memtable without the threshold - /// flush. - pub fn delete_cells_unflushed( - &mut self, - id: &ArrayId, - cells: Vec, - wal_lsn: u64, - ) -> ArrayEngineResult<()> { - if cells.is_empty() { - return Ok(()); - } - stamp_delete_cells(self.store_mut(id)?, cells, wal_lsn) - } - - /// Run the threshold flush the unflushed writes skipped. - pub fn flush_if_full(&mut self, id: &ArrayId) -> ArrayEngineResult<()> { - self.maybe_flush(id) - } } diff --git a/nodedb/src/engine/array/store/catalog/scan.rs b/nodedb/src/engine/array/store/catalog/scan.rs index 13e179fa6..cac15a69b 100644 --- a/nodedb/src/engine/array/store/catalog/scan.rs +++ b/nodedb/src/engine/array/store/catalog/scan.rs @@ -298,6 +298,8 @@ mod tests { e.set_kek(kek.clone()); e.open_array(aid(), engine_schema(), 0xBEEF).unwrap(); put_one(&mut e, 1, 1, 7, 1); + e.flush(&aid(), crate::types::replay_stamp::ReplayStamp::through(1)) + .unwrap(); assert_eq!(e.store(&aid()).unwrap().manifest().segments.len(), 1); } @@ -362,7 +364,8 @@ mod tests { ); // Flushed to a segment: still found. - e.flush(&aid(), 2).unwrap(); + e.flush(&aid(), crate::types::replay_stamp::ReplayStamp::through(2)) + .unwrap(); assert!( e.store(&aid()) .unwrap() diff --git a/nodedb/src/engine/array/store/manifest.rs b/nodedb/src/engine/array/store/manifest.rs index 52ae20f44..b88407066 100644 --- a/nodedb/src/engine/array/store/manifest.rs +++ b/nodedb/src/engine/array/store/manifest.rs @@ -13,6 +13,8 @@ use serde::{Deserialize, Serialize}; use nodedb_array::types::TileId; +use crate::types::replay_stamp::{InvalidReplayStamp, ReplayStamp}; + const MANIFEST_FILENAME: &str = "manifest.ndam"; #[derive( @@ -32,9 +34,8 @@ pub struct SegmentRef { pub min_tile: TileId, pub max_tile: TileId, pub tile_count: u32, - /// LSN watermark recorded by the `ArrayFlush` record that produced - /// this segment. Recovery uses it to skip already-durable WAL - /// records. + /// The highest LSN the stamp of the flush that wrote this segment named. + /// Informational: replay decides a record by the manifest's stamp. pub flush_lsn: u64, } @@ -44,10 +45,16 @@ pub struct SegmentRef { pub struct Manifest { pub schema_hash: u64, pub segments: Vec, - /// Highest WAL LSN reflected in any segment in this manifest. - /// Recovery replays WAL records strictly greater than this LSN - /// into the live memtable. - pub durable_lsn: u64, + /// The records the segments this manifest names hold: the replay stamp + /// of the flush that last published it. Restart replay skips an array + /// record exactly when this stamp names it. + /// + /// A newer flush's stamp names every record an older one named, because + /// the core's applied set only grows and its outcome floor only rises. So + /// the newest stamp replaces the older one rather than merging with it. + /// Compaction and purge rewrite segments without changing which records + /// they hold, so they leave the stamp alone. + pub replay: ReplayStamp, } #[derive(Debug, thiserror::Error)] @@ -58,6 +65,11 @@ pub enum ManifestError { Decode { detail: String }, #[error("manifest encode failed: {detail}")] Encode { detail: String }, + #[error("manifest carries an invalid replay stamp: {source}")] + InvalidReplayStamp { + #[source] + source: InvalidReplayStamp, + }, } impl Manifest { @@ -65,7 +77,7 @@ impl Manifest { Self { schema_hash, segments: Vec::new(), - durable_lsn: 0, + replay: ReplayStamp::default(), } } @@ -80,6 +92,11 @@ impl Manifest { zerompk::from_msgpack(&bytes).map_err(|e| ManifestError::Decode { detail: format!("{path:?}: {e}"), })?; + // `skips` relies on the stamp's shape; a stamp that breaks it + // could skip a record no segment holds. + m.replay + .validate() + .map_err(|source| ManifestError::InvalidReplayStamp { source })?; Ok(m) } Err(nodedb_wal::WalError::Io(e)) if e.kind() == std::io::ErrorKind::NotFound => { @@ -114,7 +131,6 @@ impl Manifest { } pub fn append(&mut self, seg: SegmentRef) { - self.durable_lsn = self.durable_lsn.max(seg.flush_lsn); self.segments.push(seg); } @@ -122,10 +138,7 @@ impl Manifest { /// compaction merger after it has produced a replacement segment. pub fn replace(&mut self, removed: &[String], added: Vec) { self.segments.retain(|s| !removed.contains(&s.id)); - for seg in added { - self.durable_lsn = self.durable_lsn.max(seg.flush_lsn); - self.segments.push(seg); - } + self.segments.extend(added); } pub fn segments_at_level(&self, level: u8) -> impl Iterator { @@ -163,7 +176,7 @@ mod tests { let loaded = Manifest::load_or_new(dir.path(), 0xCAFE).unwrap(); assert_eq!(loaded.schema_hash, 0xCAFE); assert_eq!(loaded.segments.len(), 1); - assert_eq!(loaded.durable_lsn, 5); + assert_eq!(loaded.replay, ReplayStamp::default()); } #[test] @@ -171,17 +184,31 @@ mod tests { let dir = TempDir::new().unwrap(); let m = Manifest::load_or_new(dir.path(), 0x1).unwrap(); assert!(m.segments.is_empty()); - assert_eq!(m.durable_lsn, 0); + assert_eq!(m.replay, ReplayStamp::default()); } #[test] - fn replace_swaps_segments_and_keeps_max_lsn() { + fn replace_swaps_segments_and_keeps_the_stamp() { let mut m = Manifest::new(0x1); + m.replay = ReplayStamp::through(2); m.append(seg("a", 0, 1)); m.append(seg("b", 0, 2)); m.replace(&["a".into(), "b".into()], vec![seg("c", 1, 2)]); assert_eq!(m.segments.len(), 1); assert_eq!(m.segments[0].id, "c"); - assert_eq!(m.durable_lsn, 2); + assert_eq!(m.replay, ReplayStamp::through(2)); + } + + #[test] + fn the_stamp_round_trips_through_persist() { + let dir = TempDir::new().unwrap(); + let mut m = Manifest::new(0xCAFE); + m.replay = ReplayStamp { + prefix: 5, + applied_above: vec![crate::types::replay_stamp::LsnRange { start: 9, end: 9 }], + }; + m.persist(dir.path()).unwrap(); + let loaded = Manifest::load_or_new(dir.path(), 0xCAFE).unwrap(); + assert_eq!(loaded.replay, m.replay); } } diff --git a/nodedb/src/engine/array/test_support.rs b/nodedb/src/engine/array/test_support.rs index 423874022..09696c733 100644 --- a/nodedb/src/engine/array/test_support.rs +++ b/nodedb/src/engine/array/test_support.rs @@ -19,6 +19,7 @@ use nodedb_types::TenantId; use super::engine::ArrayEngine; use super::wal::ArrayPutCell; +use crate::types::replay_stamp::ReplayStamp; pub(super) fn schema() -> Arc { Arc::new( @@ -59,3 +60,11 @@ pub(super) fn put_one(e: &mut ArrayEngine, x: i64, y: i64, v: i64, lsn: u64) { ) .unwrap(); } + +/// Flush `id` once its memtable reached the threshold, stamped through +/// `lsn`: what the executor does after every write it applies. +pub(super) fn flush_if_full(e: &mut ArrayEngine, id: &ArrayId, lsn: u64) { + if e.needs_flush(id).unwrap() { + e.flush(id, ReplayStamp::through(lsn)).unwrap(); + } +} diff --git a/nodedb/src/engine/array/wal.rs b/nodedb/src/engine/array/wal.rs index d8a2fcdf2..85e8706b2 100644 --- a/nodedb/src/engine/array/wal.rs +++ b/nodedb/src/engine/array/wal.rs @@ -8,10 +8,9 @@ //! single array. Batched so the recovery path can rebuild a memtable //! without one syscall per cell. //! * [`ArrayDeletePayload`] — a batch of point deletes for a single array. -//! * [`ArrayFlushPayload`] — emitted *after* the engine has fsync'd a new -//! segment file. Replay treats it as a watermark: any earlier -//! `ArrayPut`/`ArrayDelete` whose LSN <= this record's LSN is already -//! captured in the segment and must not be reapplied. +//! * [`ArrayFlushPayload`] — an explicit flush request. Replay applies +//! nothing for it: the manifest's replay stamp, not this record, decides +//! which `ArrayPut`/`ArrayDelete` records a segment already holds. //! //! All three are zerompk-encoded — JSON is reserved for the API //! boundary, never used between planes. LSNs are allocated by the diff --git a/nodedb/src/engine/array/write.rs b/nodedb/src/engine/array/write.rs index 9d6943d23..0e279f1e7 100644 --- a/nodedb/src/engine/array/write.rs +++ b/nodedb/src/engine/array/write.rs @@ -14,6 +14,10 @@ impl ArrayEngine { /// Stamps the memtable with a caller-supplied LSN. The Control Plane /// allocates the LSN at WAL-append time and the Data Plane just /// stamps; this is the only write path for the engine. + /// + /// Never flushes. A flush must stamp the records its segment holds, and + /// only the caller knows them: it checks [`Self::needs_flush`] and + /// flushes with its core's replay stamp. pub fn put_cells( &mut self, id: &ArrayId, @@ -23,9 +27,7 @@ impl ArrayEngine { if cells.is_empty() { return Ok(()); } - stamp_put_cells(self.store_mut(id)?, cells, wal_lsn)?; - self.maybe_flush(id)?; - Ok(()) + stamp_put_cells(self.store_mut(id)?, cells, wal_lsn) } /// Stamps memtable tombstones (or GDPR erasure markers) with a @@ -40,9 +42,7 @@ impl ArrayEngine { if cells.is_empty() { return Ok(()); } - stamp_delete_cells(self.store_mut(id)?, cells, wal_lsn)?; - self.maybe_flush(id)?; - Ok(()) + stamp_delete_cells(self.store_mut(id)?, cells, wal_lsn) } /// GDPR-erase a single cell at `(array, coord, system_from_ms)`. @@ -74,16 +74,10 @@ impl ArrayEngine { ) } - pub(super) fn maybe_flush(&mut self, id: &ArrayId) -> ArrayEngineResult<()> { - let stats = self.store(id)?.memtable.stats(); - if stats.cell_count >= self.cfg.flush_cell_threshold { - // Auto-flush in the threshold path uses an LSN of 0 — the - // caller's WAL-side flush record (if any) carries the real - // watermark; auto-flushes are always followed by another - // explicit Flush from the Control Plane in production. - self.flush(id, 0)?; - } - Ok(()) + /// Whether the array's memtable holds at least `flush_cell_threshold` + /// live cells. + pub fn needs_flush(&self, id: &ArrayId) -> ArrayEngineResult { + Ok(self.store(id)?.memtable.stats().cell_count >= self.cfg.flush_cell_threshold) } } @@ -142,17 +136,20 @@ mod tests { use nodedb_array::types::coord::value::CoordValue; use tempfile::TempDir; + /// A put reports the threshold and leaves the flush to its caller, which + /// alone knows the stamp the segment must carry. #[test] - fn auto_flush_triggers_at_threshold() { + fn a_put_reports_the_threshold_and_never_flushes() { let dir = TempDir::new().unwrap(); let mut cfg = ArrayEngineConfig::new(dir.path().to_path_buf()); cfg.flush_cell_threshold = 2; let mut e = ArrayEngine::new(cfg).unwrap(); e.open_array(aid(), schema(), 0x1).unwrap(); - for i in 0..2 { - put_one(&mut e, i, i, i, (i as u64) + 1); - } - assert!(!e.store(&aid()).unwrap().manifest().segments.is_empty()); + put_one(&mut e, 0, 0, 0, 1); + assert!(!e.needs_flush(&aid()).unwrap()); + put_one(&mut e, 1, 1, 1, 2); + assert!(e.needs_flush(&aid()).unwrap()); + assert!(e.store(&aid()).unwrap().manifest().segments.is_empty()); } #[test] @@ -170,7 +167,10 @@ mod tests { valid_until_ms: i64::MAX, }]; e.put_cells(&aid(), cells, 42).unwrap(); - let seg = e.flush(&aid(), 43).unwrap().unwrap(); + let seg = e + .flush(&aid(), crate::types::replay_stamp::ReplayStamp::through(43)) + .unwrap() + .unwrap(); assert_eq!(seg.flush_lsn, 43); let pred = MbrQueryPredicate::default(); diff --git a/nodedb/src/engine/timeseries/columnar_memtable/memtable/ingest.rs b/nodedb/src/engine/timeseries/columnar_memtable/memtable/ingest.rs index ab45c7e2d..886182e30 100644 --- a/nodedb/src/engine/timeseries/columnar_memtable/memtable/ingest.rs +++ b/nodedb/src/engine/timeseries/columnar_memtable/memtable/ingest.rs @@ -21,10 +21,10 @@ impl ColumnarMemtable { /// the ceiling lives at the record boundary in the live ingest handler's /// admission gate, never here. NOTE: if a structured-`TimeseriesWalBatch` /// ingest producer is ever added, its replay must gain that same - /// record-boundary flush, or a mid-record replay flush would stamp the - /// partition with an LSN covering rows it does not hold. - /// `replay_timeseries_wal` advances `ts_max_ingested_lsn` only AFTER a - /// record is applied, which is what keeps that stamp honest. + /// record-boundary flush, or a mid-record replay flush would write a + /// partition holding part of a record. `replay_timeseries_wal` moves its + /// replay cursor past a record only AFTER the record is applied, which + /// keeps the partition stamp honest. pub fn ingest_metric(&mut self, series_id: SeriesId, sample: MetricSample) -> IngestResult { // Push to timestamp column. if let ColumnData::Timestamp(ref mut v) = self.columns[self.schema.timestamp_idx] { diff --git a/nodedb/src/storage/snapshot_executor.rs b/nodedb/src/storage/snapshot_executor.rs index 34b5c54ec..cee92648f 100644 --- a/nodedb/src/storage/snapshot_executor.rs +++ b/nodedb/src/storage/snapshot_executor.rs @@ -246,10 +246,12 @@ fn restore_vector_checkpoints( .map_err(crate::Error::Wal)?; total_vectors += 1; } + // The snapshot holds every record through its LSN. crate::data::executor::vector_checkpoint::publish_vector_generation( &ckpt_dir, generation, snapshot_lsn, + crate::types::replay_stamp::ReplayStamp::through(snapshot_lsn.as_u64()), )?; info!( diff --git a/nodedb/src/types/mod.rs b/nodedb/src/types/mod.rs index 223c48a03..f5f8b38f9 100644 --- a/nodedb/src/types/mod.rs +++ b/nodedb/src/types/mod.rs @@ -3,6 +3,7 @@ pub mod consistency; pub mod id; pub mod lsn; +pub mod replay_stamp; pub mod snapshot; pub use consistency::ReadConsistency; diff --git a/nodedb/src/data/executor/applied_prefix/stamp.rs b/nodedb/src/types/replay_stamp.rs similarity index 63% rename from nodedb/src/data/executor/applied_prefix/stamp.rs rename to nodedb/src/types/replay_stamp.rs index 53daaf026..f226eecc1 100644 --- a/nodedb/src/data/executor/applied_prefix/stamp.rs +++ b/nodedb/src/types/replay_stamp.rs @@ -36,7 +36,7 @@ use serde::{Deserialize, Serialize}; zerompk::ToMessagePack, zerompk::FromMessagePack, )] -pub(crate) struct LsnRange { +pub struct LsnRange { pub start: u64, pub end: u64, } @@ -53,7 +53,7 @@ pub(crate) struct LsnRange { zerompk::ToMessagePack, zerompk::FromMessagePack, )] -pub(crate) struct ReplayStamp { +pub struct ReplayStamp { /// The outcome floor when the artifact was written. pub prefix: u64, /// The LSNs above `prefix` applied before the artifact was written, as @@ -63,7 +63,7 @@ pub(crate) struct ReplayStamp { /// Why a decoded [`ReplayStamp`] cannot describe an artifact. #[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] -pub(crate) enum InvalidReplayStamp { +pub enum InvalidReplayStamp { /// A range ends before it starts. #[error("applied range [{start}, {end}] ends before it starts")] Inverted { start: u64, end: u64 }, @@ -77,18 +77,32 @@ pub(crate) enum InvalidReplayStamp { impl ReplayStamp { /// A stamp holding every record at or below `prefix` and nothing above. - pub(crate) fn through(prefix: u64) -> Self { + pub fn through(prefix: u64) -> Self { Self { prefix, applied_above: Vec::new(), } } + /// A stamp naming the record at `lsn` and nothing else above LSN 0. + pub fn naming(lsn: u64) -> Self { + if lsn == 0 { + return Self::through(0); + } + Self { + prefix: 0, + applied_above: vec![LsnRange { + start: lsn, + end: lsn, + }], + } + } + /// Whether restart replay skips the record at `record_lsn`: the artifact /// already holds it, or it has a final outcome that is not an apply. /// /// Every other record replays. - pub(crate) fn skips(&self, record_lsn: u64) -> bool { + pub fn skips(&self, record_lsn: u64) -> bool { if record_lsn <= self.prefix { return true; } @@ -102,8 +116,56 @@ impl ReplayStamp { .is_some_and(|range| record_lsn <= range.end) } + /// The highest LSN the stamp names: the end of its last applied range, or + /// its prefix when it names nothing above the prefix. + pub fn highest(&self) -> u64 { + self.applied_above + .last() + .map_or(self.prefix, |range| range.end.max(self.prefix)) + } + + /// A stamp naming every record `self` or `other` names. + /// + /// Two artifacts of one collection written at different times each name + /// the records they hold. The union is what the collection holds across + /// both. + pub fn union(&self, other: &ReplayStamp) -> ReplayStamp { + let mut prefix = self.prefix.max(other.prefix); + let mut ranges: Vec = self + .applied_above + .iter() + .chain(&other.applied_above) + .filter(|range| range.end > prefix) + .map(|range| LsnRange { + start: range.start.max(prefix.saturating_add(1)), + end: range.end, + }) + .collect(); + ranges.sort_unstable_by_key(|range| range.start); + let mut merged: Vec = Vec::with_capacity(ranges.len()); + for range in ranges { + match merged.last_mut() { + Some(last) if range.start <= last.end.saturating_add(1) => { + last.end = last.end.max(range.end); + } + _ => merged.push(range), + } + } + // A range that starts right above the prefix extends it. + if merged + .first() + .is_some_and(|first| first.start == prefix.saturating_add(1)) + { + prefix = merged.remove(0).end; + } + ReplayStamp { + prefix, + applied_above: merged, + } + } + /// Check the shape [`Self::skips`] relies on. - pub(crate) fn validate(&self) -> Result<(), InvalidReplayStamp> { + pub fn validate(&self) -> Result<(), InvalidReplayStamp> { let mut previous_end: Option = None; for range in &self.applied_above { if range.end < range.start { @@ -175,6 +237,42 @@ mod tests { assert!(!ReplayStamp::default().skips(1)); } + #[test] + fn the_highest_named_lsn_is_the_last_range_end_or_the_prefix() { + assert_eq!(stamp(10, &[(12, 13), (20, 21)]).highest(), 21); + assert_eq!(ReplayStamp::through(7).highest(), 7); + } + + #[test] + fn a_union_names_what_either_stamp_names_and_nothing_else() { + let a = stamp(10, &[(14, 15), (30, 30)]); + let b = stamp(12, &[(16, 18), (25, 26), (40, 41)]); + let u = a.union(&b); + assert_eq!(u, stamp(12, &[(14, 18), (25, 26), (30, 30), (40, 41)])); + assert!(u.validate().is_ok()); + for lsn in 0..=45 { + assert_eq!(u.skips(lsn), a.skips(lsn) || b.skips(lsn), "lsn {lsn}"); + } + } + + #[test] + fn a_naming_stamp_names_its_one_record() { + let stamp = ReplayStamp::naming(9); + assert!(stamp.validate().is_ok()); + assert!(stamp.skips(9)); + assert!(!stamp.skips(8) && !stamp.skips(10)); + } + + #[test] + fn a_union_folds_ranges_that_reach_the_prefix_into_it() { + let u = stamp(10, &[(11, 12), (20, 20)]).union(&stamp(5, &[(8, 14)])); + assert_eq!(u, stamp(14, &[(20, 20)])); + assert_eq!( + ReplayStamp::through(9).union(&ReplayStamp::default()), + ReplayStamp::through(9) + ); + } + #[test] fn a_malformed_stamp_is_refused() { assert!(stamp(10, &[(12, 13), (20, 20)]).validate().is_ok()); diff --git a/nodedb/tests/crash_replay_stamp.rs b/nodedb/tests/crash_replay_stamp.rs index e18f0d8fb..837bd0eea 100644 --- a/nodedb/tests/crash_replay_stamp.rs +++ b/nodedb/tests/crash_replay_stamp.rs @@ -18,6 +18,11 @@ //! 4. After restart, replay must apply A: the stamp does not name it. A stamp //! that holds only the highest applied LSN skips A, and A's row is lost. //! +//! An array or timeseries checkpoint writes only a collection that holds +//! unwritten state. Those cases seed the held collection in boot 1 and start +//! boot 2's checkpoints after A is minted, so the checkpoint that names B also +//! writes the held collection. +//! //! Requires `--features failpoints`. #![cfg(feature = "failpoints")] @@ -28,8 +33,9 @@ use std::time::{Duration, Instant}; use crash_harness::{CrashHarness, diagnostics}; -/// One checkpoint per second, so several are written while A is parked. -const CHECKPOINT_INTERVAL_SECS: &str = "1"; +/// Boot 1 writes no checkpoint, so a seeded write stays in memory until the +/// kill. +const QUIET_CHECKPOINT_INTERVAL_SECS: &str = "3600"; /// How long the test waits for a checkpoint whose stamp names a B: thirty /// checkpoint cycles at one per second. A is parked until the test releases @@ -47,22 +53,32 @@ struct Case { applied: &'static str, create_held: &'static str, create_applied: &'static str, + /// A write to the held collection in boot 1, still in memory at the kill. + /// Boot 2 replays it, so the held collection has state for boot 2's + /// checkpoint to write while A is parked. + seed_held: Option<&'static str>, /// Write A. insert_held: &'static str, /// Reads A's row back as one column. read_held: &'static str, - /// The value `read_held` returns once A applied. + /// The value `read_held` returns once A applied. Two numbers compare by + /// value, so `8` equals `8.0`. held_value: &'static str, /// Write B number `n`. insert_applied: fn(usize) -> String, /// Reads every B row as one column. read_applied: &'static str, - /// The engine's checkpoint module, as a `RUST_LOG` target. - log_target: &'static str, - /// The log message of a published checkpoint. + /// Boot 2's checkpoint interval. It must pass after A is minted when the + /// engine writes a collection only while it holds unwritten state. + checkpoint_interval_secs: &'static str, + /// `RUST_LOG` directives that enable the `published` and `restored` lines. + log_directives: &'static str, + /// The log message of a published checkpoint. Its `applied_ranges` field + /// counts the ranges its stamp names above the prefix. published: &'static str, - /// The log message of a checkpoint restored at boot. - restored: &'static str, + /// The boot-3 log message that proves this run reproduced the in-flight + /// write, and the numeric field that must be above zero on it. + restored: (&'static str, &'static str), } #[tokio::test(flavor = "multi_thread")] @@ -74,14 +90,16 @@ async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { WITH (engine='kv')", create_applied: "CREATE COLLECTION stamp_kv_hi (k STRING PRIMARY KEY, v STRING) \ WITH (engine='kv')", + seed_held: None, insert_held: "INSERT INTO stamp_kv_lo (k, v) VALUES ('held', 'a')", read_held: "SELECT v FROM stamp_kv_lo WHERE k = 'held'", held_value: "a", insert_applied: |n| format!("INSERT INTO stamp_kv_hi (k, v) VALUES ('k{n:03}', 'v{n}')"), read_applied: "SELECT v FROM stamp_kv_hi", - log_target: "nodedb::data::executor::kv_checkpoint", + checkpoint_interval_secs: "1", + log_directives: "nodedb::data::executor::kv_checkpoint=info", published: "KV checkpoint published", - restored: "KV checkpoint restored", + restored: ("KV checkpoint restored", "applied_ranges"), }) .await; } @@ -95,14 +113,82 @@ async fn a_columnar_write_in_flight_at_a_checkpoint_survives_kill_9() { WITH (engine='columnar')", create_applied: "CREATE COLLECTION stamp_col_hi COLUMNS (id TEXT, v TEXT) \ WITH (engine='columnar')", + seed_held: None, insert_held: "INSERT INTO stamp_col_lo (id, v) VALUES ('held', 'a')", read_held: "SELECT v FROM stamp_col_lo WHERE id = 'held'", held_value: "a", insert_applied: |n| format!("INSERT INTO stamp_col_hi (id, v) VALUES ('r{n:03}', 'v{n}')"), read_applied: "SELECT v FROM stamp_col_hi", - log_target: "nodedb::data::executor::columnar_checkpoint", + checkpoint_interval_secs: "1", + log_directives: "nodedb::data::executor::columnar_checkpoint=info", published: "columnar checkpoint published", - restored: "columnar checkpoint restored", + restored: ("columnar checkpoint restored", "applied_ranges"), + }) + .await; +} + +/// An array flush writes only an array whose memtable holds cells, and it +/// stamps that array's manifest. So the held array carries a boot-1 cell into +/// boot 2's memtable, and boot 2's first checkpoint runs after A is minted. +/// That checkpoint flushes the held array with a stamp naming B writes above +/// A. Boot 3 replays A below that stamp's highest LSN (`in_flight`). +#[tokio::test(flavor = "multi_thread")] +async fn an_array_write_in_flight_at_a_checkpoint_survives_kill_9() { + run(Case { + held: "stamp_arr_lo", + applied: "stamp_arr_hi", + create_held: "CREATE ARRAY stamp_arr_lo DIMS (k INT64 [0..15]) ATTRS (v FLOAT64) \ + TILE_EXTENTS (16) CELL_ORDER ROW_MAJOR", + create_applied: "CREATE ARRAY stamp_arr_hi DIMS (k INT64 [0..1023]) ATTRS (v FLOAT64) \ + TILE_EXTENTS (64) CELL_ORDER ROW_MAJOR", + seed_held: Some("INSERT INTO ARRAY stamp_arr_lo COORDS (0) VALUES (1.0)"), + insert_held: "INSERT INTO ARRAY stamp_arr_lo COORDS (1) VALUES (7.0)", + read_held: "SELECT * FROM ARRAY_AGG('stamp_arr_lo', 'v', 'sum')", + held_value: "8", + insert_applied: |n| format!("INSERT INTO ARRAY stamp_arr_hi COORDS ({n}) VALUES ({n}.0)"), + read_applied: "SELECT * FROM ARRAY_AGG('stamp_arr_hi', 'v', 'sum')", + checkpoint_interval_secs: "10", + log_directives: "nodedb::data::executor::array_checkpoint=info,\ + nodedb::data::executor::wal_replay::array=info", + published: "array checkpoint flushed", + restored: ("WAL array replay complete", "in_flight"), + }) + .await; +} + +/// A timeseries checkpoint flushes only a collection whose memtable holds +/// rows, and stamps that collection's partition. So the held collection +/// carries a boot-1 row into boot 2's memtable, and boot 2's first checkpoint +/// runs after A is minted. That checkpoint flushes the held collection with a +/// stamp naming B writes above A. Boot 3 replays A below that stamp's highest +/// LSN (`in_flight`). +#[tokio::test(flavor = "multi_thread")] +async fn a_timeseries_write_in_flight_at_a_checkpoint_survives_kill_9() { + run(Case { + held: "stamp_ts_lo", + applied: "stamp_ts_hi", + create_held: "CREATE COLLECTION stamp_ts_lo \ + COLUMNS (id TEXT, ts BIGINT TIME_KEY, value FLOAT) \ + WITH (engine='timeseries')", + create_applied: "CREATE COLLECTION stamp_ts_hi \ + COLUMNS (id TEXT, ts BIGINT TIME_KEY, value FLOAT) \ + WITH (engine='timeseries')", + seed_held: Some("INSERT INTO stamp_ts_lo (id, ts, value) VALUES ('seed', 1000, 1.0)"), + insert_held: "INSERT INTO stamp_ts_lo (id, ts, value) VALUES ('held', 2000, 7.0)", + read_held: "SELECT value FROM stamp_ts_lo WHERE id = 'held'", + held_value: "7", + insert_applied: |n| { + format!( + "INSERT INTO stamp_ts_hi (id, ts, value) VALUES ('r{n:03}', {}, {n}.0)", + 1_000 + n + ) + }, + read_applied: "SELECT id FROM stamp_ts_hi", + checkpoint_interval_secs: "10", + log_directives: "nodedb::data::executor::handlers::timeseries::flush=info,\ + nodedb::data::executor::handlers::timeseries_wal=info", + published: "timeseries columnar flush complete", + restored: ("WAL timeseries replay complete", "in_flight"), }) .await; } @@ -110,17 +196,27 @@ async fn a_columnar_write_in_flight_at_a_checkpoint_survives_kill_9() { async fn run(case: Case) { let mut h = CrashHarness::new() .standalone() - .with_env("NODEDB_CHECKPOINT_INTERVAL_SECS", CHECKPOINT_INTERVAL_SECS) - .with_env("RUST_LOG", &format!("warn,{}=info", case.log_target)); + .with_env( + "NODEDB_CHECKPOINT_INTERVAL_SECS", + QUIET_CHECKPOINT_INTERVAL_SECS, + ) + .with_env("RUST_LOG", &format!("warn,{}", case.log_directives)); h.spawn(); h.wait_ready(); h.exec(case.create_held).await; h.exec(case.create_applied).await; + if let Some(seed) = case.seed_held { + h.exec(seed).await; + } // Boot 2 arms the gate and the abort, keyed to the held collection. Both // match only a request carrying a WAL LSN, so boot itself passes them. h.kill_9(); let release = h.data_dir().join("release-held-write"); + h.set_env( + "NODEDB_CHECKPOINT_INTERVAL_SECS", + case.checkpoint_interval_secs, + ); h.set_env( "NODEDB_FAILPOINTS", &format!( @@ -153,7 +249,10 @@ async fn run(case: Case) { applied += 1; tokio::time::sleep(Duration::from_millis(200)).await; let log = boot_section(&h.server_log(), 2); - if applied_ranges(&log, case.published).iter().any(|n| *n > 0) { + if log_field(&log, case.published, "applied_ranges") + .iter() + .any(|n| *n > 0) + { break; } assert!( @@ -192,21 +291,23 @@ async fn run(case: Case) { h.clear_env("NODEDB_FAILPOINTS"); h.reopen(); - let ranges = applied_ranges(&boot_section(&h.server_log(), 3), case.restored); + let (restored, field) = case.restored; + let proof = log_field(&boot_section(&h.server_log(), 3), restored, field); assert!( - ranges.iter().any(|n| *n > 0), - "the restored generation must name an applied LSN above its prefix, or this run did \ - not reproduce the in-flight write (restored stamps: {ranges:?}).{}\n{}", + proof.iter().any(|n| *n > 0), + "no {restored} line has {field} above zero, so this run did not reproduce the \ + in-flight write (values: {proof:?}).{}\n{}", h.keep_data_dir_note(), diagnostics::log_tail_section(&h.server_log()) ); let held = h.query_col_idx(case.read_held, 0).await; - assert_eq!( - held, - vec![case.held_value.to_string()], - "write A to {} applied after the checkpoint and before the crash; replay must \ - apply it, never skip it as covered by a higher applied LSN", + assert!( + held.len() == 1 && same_value(&held[0], case.held_value), + "read {held:?}, expected [{}]: write A to {} applied after the checkpoint and \ + before the crash; replay must apply it, never skip it as covered by a higher \ + applied LSN", + case.held_value, case.held ); let mut replayed = h.query_col_idx(case.read_applied, 0).await; @@ -232,13 +333,23 @@ fn boot_section(log: &str, n: u32) -> String { } } -/// The `applied_ranges` field of every log line carrying `message`. -fn applied_ranges(log: &str, message: &str) -> Vec { +/// Whether two read values are equal: by value when both are numbers, by +/// text otherwise. +fn same_value(read: &str, expected: &str) -> bool { + match (read.parse::(), expected.parse::()) { + (Ok(a), Ok(b)) => a == b, + _ => read == expected, + } +} + +/// The numeric `field` of every log line carrying `message`. +fn log_field(log: &str, message: &str, field: &str) -> Vec { + let key = format!("{field}="); strip_ansi(log) .lines() .filter(|line| line.contains(message)) .filter_map(|line| { - let rest = line.split_once("applied_ranges=")?.1; + let rest = line.split_once(key.as_str())?.1; let digits: String = rest.chars().take_while(char::is_ascii_digit).collect(); digits.parse().ok() }) @@ -267,7 +378,13 @@ fn strip_ansi(text: &str) -> String { fn log_fields_are_read_through_colour_codes() { let log = "INFO KV checkpoint published \u{1b}[3mapplied_ranges\u{1b}[0m\u{1b}[2m=\u{1b}[0m2\n\ INFO KV checkpoint published applied_ranges=0\n"; - assert_eq!(applied_ranges(log, "KV checkpoint published"), vec![2, 0]); + assert_eq!( + log_field(log, "KV checkpoint published", "applied_ranges"), + vec![2, 0] + ); + assert!(same_value("8.0", "8")); + assert!(!same_value("8.5", "8")); + assert!(same_value("a", "a")); let booted = "=== crash harness boot 1 (pid 1) ===\nfirst-line\n\ === crash harness boot 2 (pid 2) ===\nsecond-line\n"; assert!(boot_section(booted, 2).contains("second-line")); diff --git a/nodedb/tests/inproc/cases/bitemporal_array_basic.rs b/nodedb/tests/inproc/cases/bitemporal_array_basic.rs index 135d910f1..066b6dbd3 100644 --- a/nodedb/tests/inproc/cases/bitemporal_array_basic.rs +++ b/nodedb/tests/inproc/cases/bitemporal_array_basic.rs @@ -233,7 +233,8 @@ fn tombstone_appended_not_in_place_visible_after_flush() { } // Force flush so both tiles land in a segment file. - e.flush(&aid(), 3).unwrap(); + e.flush(&aid(), nodedb::types::replay_stamp::ReplayStamp::through(3)) + .unwrap(); // After flush, confirm the manifest contains a segment spanning sys=200. let store = e.store(&aid()).unwrap(); diff --git a/nodedb/tests/inproc/cases/bitemporal_array_compaction.rs b/nodedb/tests/inproc/cases/bitemporal_array_compaction.rs index 0ac4f0735..254c7d2d2 100644 --- a/nodedb/tests/inproc/cases/bitemporal_array_compaction.rs +++ b/nodedb/tests/inproc/cases/bitemporal_array_compaction.rs @@ -51,6 +51,19 @@ fn put_cell(e: &mut ArrayEngine, x: i64, v: i64, sys_ms: i64, lsn: u64) { lsn, ) .unwrap(); + flush_if_full(e, lsn); +} + +/// The threshold flush the executor runs after every write it applies, +/// stamped through the write's LSN. +fn flush_if_full(e: &mut ArrayEngine, lsn: u64) { + if e.needs_flush(&aid()).unwrap() { + e.flush( + &aid(), + nodedb::types::replay_stamp::ReplayStamp::through(lsn), + ) + .unwrap(); + } } #[test] @@ -131,7 +144,8 @@ fn truncated_before_horizon_flag_set_when_cutoff_predates_all_data() { let mut e = ArrayEngine::new(cfg).unwrap(); e.open_array(aid(), schema(), 0xCAFE).unwrap(); put_cell(&mut e, 0, 99, 500, 1); - e.flush(&aid(), 2).unwrap(); + e.flush(&aid(), nodedb::types::replay_stamp::ReplayStamp::through(2)) + .unwrap(); let store = e.store(&aid()).unwrap(); let (rows, truncated) = store.scan_tiles_at(50, None).unwrap(); @@ -333,6 +347,7 @@ fn valid_time_filter_works_after_flush() { 1, ) .unwrap(); + flush_if_full(&mut e, 1); e.put_cells( &aid(), @@ -347,6 +362,7 @@ fn valid_time_filter_works_after_flush() { 2, ) .unwrap(); + flush_if_full(&mut e, 2); assert!( e.store(&aid()).unwrap().manifest().segments.len() >= 2, diff --git a/nodedb/tests/inproc/cases/bitemporal_array_gdpr.rs b/nodedb/tests/inproc/cases/bitemporal_array_gdpr.rs index f20538d01..5d7641b76 100644 --- a/nodedb/tests/inproc/cases/bitemporal_array_gdpr.rs +++ b/nodedb/tests/inproc/cases/bitemporal_array_gdpr.rs @@ -81,6 +81,19 @@ fn put_cell(e: &mut ArrayEngine, x: i64, v: i64, sys: i64, lsn: u64) { lsn, ) .unwrap(); + flush_if_full(e, lsn); +} + +/// The threshold flush the executor runs after every write it applies, +/// stamped through the write's LSN. +fn flush_if_full(e: &mut ArrayEngine, lsn: u64) { + if e.needs_flush(&aid()).unwrap() { + e.flush( + &aid(), + nodedb::types::replay_stamp::ReplayStamp::through(lsn), + ) + .unwrap(); + } } fn tombstone_cell(e: &mut ArrayEngine, x: i64, sys: i64, lsn: u64) { @@ -94,11 +107,13 @@ fn tombstone_cell(e: &mut ArrayEngine, x: i64, sys: i64, lsn: u64) { lsn, ) .unwrap(); + flush_if_full(e, lsn); } fn erase_cell(e: &mut ArrayEngine, x: i64, sys: i64, lsn: u64) { e.gdpr_erase_cell(&aid(), vec![CoordValue::Int64(x)], sys, lsn) .unwrap(); + flush_if_full(e, lsn); } fn coord_x(x: i64) -> Vec { @@ -182,17 +197,17 @@ fn gdpr_erasure_persists_through_flush_and_blocks_reads() { use nodedb_array::tile::sparse_tile::RowKind; let dir = TempDir::new().unwrap(); - // flush_cell_threshold=1 forces an auto-flush after each write. + // flush_cell_threshold=1 makes `flush_if_full` flush after each write. let mut e = open_engine_with_threshold(&dir, 1); // Live cell at x=0, sys=100. put_cell(&mut e, 0, 111, 100, 1); - // Auto-flush creates segment 1. + // The threshold flush creates segment 1. // Live cell at x=5, sys=200 — creates segment 2. put_cell(&mut e, 5, 222, 200, 2); - // GDPR erase x=0 at sys=300 — auto-flush creates segment 3. + // GDPR erase x=0 at sys=300 — the threshold flush creates segment 3. erase_cell(&mut e, 0, 300, 3); // Write a 4th entry to push us over the L0 compaction trigger. @@ -273,13 +288,14 @@ fn gdpr_erasure_physically_dropped_outside_retention_horizon() { let mut e = ArrayEngine::new(cfg).unwrap(); e.open_array(aid(), schema(), SCHEMA_HASH).unwrap(); - // Write two live cells; auto-flush at threshold=1 creates one segment each. + // Write two live cells; the threshold flush at 1 creates one segment each. put_cell(&mut e, 0, 111, 100, 1); put_cell(&mut e, 5, 222, 200, 2); // GDPR-erase x=0 at sys=300, then manually flush the erasure tile. erase_cell(&mut e, 0, 300, 3); - e.flush(&aid(), 4).unwrap(); + e.flush(&aid(), nodedb::types::replay_stamp::ReplayStamp::through(4)) + .unwrap(); let seg_count_before = e.store(&aid()).unwrap().manifest().segments.len(); assert!( diff --git a/nodedb/tests/inproc/cases/bitemporal_array_recovery.rs b/nodedb/tests/inproc/cases/bitemporal_array_recovery.rs index decba8a8b..43c2de7e5 100644 --- a/nodedb/tests/inproc/cases/bitemporal_array_recovery.rs +++ b/nodedb/tests/inproc/cases/bitemporal_array_recovery.rs @@ -222,8 +222,16 @@ fn follower_replay_yields_identical_state() { } // Flush both so the segment-scan path is also exercised. - e1.flush(&aid(), 10).unwrap(); - e2.flush(&aid(), 10).unwrap(); + e1.flush( + &aid(), + nodedb::types::replay_stamp::ReplayStamp::through(10), + ) + .unwrap(); + e2.flush( + &aid(), + nodedb::types::replay_stamp::ReplayStamp::through(10), + ) + .unwrap(); let s1 = e1.store(&aid()).unwrap(); let s2 = e2.store(&aid()).unwrap(); diff --git a/nodedb/tests/inproc/cases/bitemporal_array_retention.rs b/nodedb/tests/inproc/cases/bitemporal_array_retention.rs index acae1375c..fb9ea39b1 100644 --- a/nodedb/tests/inproc/cases/bitemporal_array_retention.rs +++ b/nodedb/tests/inproc/cases/bitemporal_array_retention.rs @@ -55,6 +55,19 @@ fn put_cell(e: &mut ArrayEngine, x: i64, v: i64, sys_ms: i64, lsn: u64) { lsn, ) .unwrap(); + flush_if_full(e, lsn); +} + +/// The threshold flush the executor runs after every write it applies, +/// stamped through the write's LSN. +fn flush_if_full(e: &mut ArrayEngine, lsn: u64) { + if e.needs_flush(&aid()).unwrap() { + e.flush( + &aid(), + nodedb::types::replay_stamp::ReplayStamp::through(lsn), + ) + .unwrap(); + } } fn delete_cell(e: &mut ArrayEngine, x: i64, sys_ms: i64, lsn: u64) { @@ -68,11 +81,13 @@ fn delete_cell(e: &mut ArrayEngine, x: i64, sys_ms: i64, lsn: u64) { lsn, ) .unwrap(); + flush_if_full(e, lsn); } fn gdpr_erase(e: &mut ArrayEngine, x: i64, sys_ms: i64, lsn: u64) { e.gdpr_erase_cell(&aid(), vec![CoordValue::Int64(x)], sys_ms, lsn) .unwrap(); + flush_if_full(e, lsn); } fn compact_all(e: &mut ArrayEngine, audit_retain_ms: Option, now_ms: i64) { diff --git a/nodedb/tests/inproc/cases/bitemporal_array_temporal_purge.rs b/nodedb/tests/inproc/cases/bitemporal_array_temporal_purge.rs index 27a2b6604..43a888dcb 100644 --- a/nodedb/tests/inproc/cases/bitemporal_array_temporal_purge.rs +++ b/nodedb/tests/inproc/cases/bitemporal_array_temporal_purge.rs @@ -51,8 +51,8 @@ fn aid() -> ArrayId { ArrayId::new(TENANT, ARRAY_NAME) } -/// Open an engine with `flush_cell_threshold = 1` so every `put_cells` -/// call lands in a fresh segment. This makes it easy to produce multiple +/// Open an engine with `flush_cell_threshold = 1` so every `put` lands in +/// a fresh segment through `flush_if_full`. This makes it easy to produce multiple /// system-time versions of the same coordinate in separate segments, /// which exercises the cross-segment retention logic in `plan.rs`. fn open_engine(dir: &TempDir) -> ArrayEngine { @@ -77,11 +77,25 @@ fn put(e: &mut ArrayEngine, x: i64, v: i64, sys_ms: i64, lsn: u64) { lsn, ) .unwrap(); + flush_if_full(e, lsn); +} + +/// The threshold flush the executor runs after every write it applies, +/// stamped through the write's LSN. +fn flush_if_full(e: &mut ArrayEngine, lsn: u64) { + if e.needs_flush(&aid()).unwrap() { + e.flush( + &aid(), + nodedb::types::replay_stamp::ReplayStamp::through(lsn), + ) + .unwrap(); + } } fn erase(e: &mut ArrayEngine, x: i64, sys_ms: i64, lsn: u64) { e.gdpr_erase_cell(&aid(), vec![CoordValue::Int64(x)], sys_ms, lsn) .unwrap(); + flush_if_full(e, lsn); } fn ceiling(e: &ArrayEngine, x: i64, sys: i64) -> CeilingResult { @@ -241,7 +255,8 @@ fn temporal_purge_drops_gdpr_erased_cells_outright() { put(&mut e, 0, 42, 100, 1); // Live(x=0) at T=100 erase(&mut e, 0, 200, 2); // GdprErased(x=0) at T=200 - e.flush(&aid(), 3).unwrap(); // ensure erasure tile-version lands in a segment + e.flush(&aid(), nodedb::types::replay_stamp::ReplayStamp::through(3)) + .unwrap(); // ensure erasure tile-version lands in a segment let dropped = e.temporal_purge(TENANT, DATABASE, ARRAY_NAME, 250).unwrap(); // Both tile-versions are outside horizon=250 and are candidates for diff --git a/nodedb/tests/inproc/cases/bitemporal_array_variant_cube.rs b/nodedb/tests/inproc/cases/bitemporal_array_variant_cube.rs index 668e79b96..f25ddf983 100644 --- a/nodedb/tests/inproc/cases/bitemporal_array_variant_cube.rs +++ b/nodedb/tests/inproc/cases/bitemporal_array_variant_cube.rs @@ -149,8 +149,7 @@ fn genomics_variant_reclassification_bitemporal() { let cells: Vec = (0_i64..100) .map(|s| variant_cell(s, 1 /* pathogenic */, day_1_ms, day_1_ms)) .collect(); - // Write in batches of 20 to avoid any single call with 100 cells - // (stays under any auto-flush threshold for the test engine config). + // Write in batches of 20, one WAL LSN per batch. for (i, chunk) in cells.chunks(20).enumerate() { let lsn = (i as u64) + 1; e.put_cells(&aid(), chunk.to_vec(), lsn).unwrap(); From 2ccb891caa79859c0e3fe761f978138b7ec3b52d Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 15:45:37 +0800 Subject: [PATCH 31/64] feat(executor): report the outcome floor, not the watermark, for durable LSN MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The watermark passes a record still on its way to a core, since records apply out of Raft-commit order rather than LSN order. Every checkpoint and snapshot used to clamp against the watermark, so a truncation run or a checkpoint published while such a record was in flight could authorize deleting the only copy of it. Every engine's durable-LSN contributor, the whole-checkpoint ceiling, and the core snapshot now report `checkpoint_floor()` — the core's outcome floor — instead. Remove the `redo_apply::cover` republish mechanism entirely: it used to re-publish a checkpoint artifact after a record applied below its stamp. That race is now closed at the source instead of patched after the fact — the write funnel and the Calvin scheduler park a record on a per-collection fail-gate (`fail_gate::before_dispatch`, `fail_gate::holds_flush`) so the outcome floor cannot pass a record before it reaches a core, using `PhysicalPlan::named_collections` to name every collection a multi-collection redo or flush touches. Timeseries settle drops its separate below-stamp flush check for the same reason. Have the checkpoint task wait for the gateway's startup phase before its first cycle, since boot recovery reads the WAL on disk until then and a cycle that runs earlier can delete segments a replaying reader still needs. Extend the crash-replay-stamp test with a WAL-truncation case and a Calvin variant, and add crash-harness helpers for reading WAL segment names and boot-log fields back out of the server log. --- .config/nextest.toml | 2 +- .../src/physical_plan/collection.rs | 63 +++ nodedb/src/control/checkpoint_task.rs | 21 + .../calvin/scheduler/driver/core/deferred.rs | 17 + nodedb/src/control/fail_gate.rs | 28 +- nodedb/src/data/executor/array_checkpoint.rs | 110 +++-- .../executor/columnar_checkpoint/format.rs | 8 +- .../data/executor/columnar_checkpoint/load.rs | 20 +- .../executor/columnar_checkpoint/write.rs | 10 +- .../src/data/executor/core_loop/accessors.rs | 12 + .../core_loop/checkpoint_floors/init.rs | 5 - .../core_loop/checkpoint_floors/state.rs | 23 - .../src/data/executor/core_loop/fail_stop.rs | 8 +- nodedb/src/data/executor/core_loop/tick.rs | 18 +- .../executor/graph_label_checkpoint/load.rs | 4 + .../executor/graph_label_checkpoint/write.rs | 13 +- .../handlers/control/checkpoint_crdt.rs | 18 +- .../control/checkpoint_durable_lsn.rs | 105 +++- .../executor/handlers/control/snapshot.rs | 21 +- .../handlers/timeseries_wal_payload.rs | 4 +- .../handlers/transaction/redo_apply/cover.rs | 464 ------------------ .../handlers/transaction/redo_apply/entry.rs | 224 ++++++--- .../handlers/transaction/redo_apply/mod.rs | 2 - .../handlers/transaction/redo_apply/settle.rs | 45 +- .../handlers/transaction/redo_apply/state.rs | 6 +- .../src/data/executor/kv_checkpoint/format.rs | 8 +- .../src/data/executor/kv_checkpoint/load.rs | 12 +- .../src/data/executor/kv_checkpoint/write.rs | 10 +- nodedb/src/data/executor/snapshot.rs | 2 +- .../sparse_vector_checkpoint/format.rs | 15 +- .../executor/sparse_vector_checkpoint/load.rs | 19 +- .../sparse_vector_checkpoint/write.rs | 12 +- .../data/executor/spatial_checkpoint/write.rs | 11 +- .../data/executor/sync_hwm_checkpoint/load.rs | 4 + .../executor/sync_hwm_checkpoint/write.rs | 12 +- .../executor/timeseries_checkpoint/flush.rs | 19 +- .../executor/timeseries_checkpoint/mod.rs | 11 +- .../data/executor/vector_checkpoint/format.rs | 14 +- .../data/executor/vector_checkpoint/load.rs | 5 +- .../data/executor/vector_checkpoint/mod.rs | 5 +- .../executor/vector_checkpoint/publish.rs | 3 - .../data/executor/vector_checkpoint/write.rs | 33 +- nodedb/src/data/snapshot.rs | 4 +- nodedb/src/storage/snapshot_executor.rs | 22 +- .../tests/crash_checkpoint_truncate_window.rs | 23 +- nodedb/tests/crash_harness/log_fields.rs | 95 ++++ nodedb/tests/crash_harness/mod.rs | 13 + nodedb/tests/crash_harness/pgwire.rs | 6 +- nodedb/tests/crash_harness/wal_truncation.rs | 53 ++ nodedb/tests/crash_replay_stamp.rs | 227 ++++++--- nodedb/tests/crash_replay_stamp_calvin.rs | 215 ++++++++ nodedb/tests/crash_wal_truncation.rs | 20 +- 52 files changed, 1188 insertions(+), 936 deletions(-) delete mode 100644 nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs create mode 100644 nodedb/tests/crash_harness/log_fields.rs create mode 100644 nodedb/tests/crash_harness/wal_truncation.rs create mode 100644 nodedb/tests/crash_replay_stamp_calvin.rs diff --git a/.config/nextest.toml b/.config/nextest.toml index cc1394558..591f66946 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -147,7 +147,7 @@ slow-timeout = { period = "30s", terminate-after = 8 } # hide the cause rather than fix it. The tail this costs is bounded — roughly a # dozen crash/shutdown tests at about ten seconds each. [[profile.default.overrides]] -filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | binary(crash_replay_stamp) | binary(crash_kv_atomic_autocommit) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' +filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | binary(crash_replay_stamp) | binary(crash_replay_stamp_calvin) | binary(crash_kv_atomic_autocommit) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' test-group = 'server-process' threads-required = 'num-test-threads' diff --git a/nodedb-physical/src/physical_plan/collection.rs b/nodedb-physical/src/physical_plan/collection.rs index 5ab0855c4..a5891fbe8 100644 --- a/nodedb-physical/src/physical_plan/collection.rs +++ b/nodedb-physical/src/physical_plan/collection.rs @@ -154,4 +154,67 @@ impl PhysicalPlan { PhysicalPlan::ClusterArray(op) => Some(op.array_id().name.as_str()), } } + + /// Every user collection this plan names: each collection a committed + /// redo install writes, or else the one [`Self::collection`] reports. + /// + /// A committed-redo apply and a Calvin flush install one record that can + /// write several collections, so [`Self::collection`] reports none for + /// them. A caller that keys on a collection name uses this instead. + pub fn named_collections(&self) -> Vec<&str> { + if let PhysicalPlan::Meta( + MetaOp::ApplyTransactionRedo { collections, .. } + | MetaOp::CalvinFlush { collections, .. }, + ) = self + { + collections.iter().map(String::as_str).collect() + } else { + self.collection().into_iter().collect() + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{DatabaseId, QualifiedCollection}; + + use crate::physical_plan::KvOp; + + #[test] + fn a_redo_install_names_every_collection_it_writes() { + let collections = vec!["a".to_string(), "b".to_string()]; + let redo = PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { + redo: Vec::new(), + collections: collections.clone(), + sum_targets: Vec::new(), + }); + let flush = PhysicalPlan::Meta(MetaOp::CalvinFlush { + epoch: 1, + position: 0, + redo: Vec::new(), + collections, + sum_targets: Vec::new(), + }); + for plan in [redo, flush] { + assert_eq!(plan.collection(), None); + assert_eq!(plan.named_collections(), vec!["a", "b"]); + } + } + + #[test] + fn a_single_collection_plan_names_its_collection() { + let get = PhysicalPlan::Kv(KvOp::Get { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + key: Vec::new(), + rls_filters: Vec::new(), + surrogate_ceiling: None, + }); + assert_eq!(get.named_collections(), vec!["users"]); + assert!( + PhysicalPlan::Meta(MetaOp::Checkpoint) + .named_collections() + .is_empty() + ); + } } diff --git a/nodedb/src/control/checkpoint_task.rs b/nodedb/src/control/checkpoint_task.rs index 0aabfebe1..694ada1ed 100644 --- a/nodedb/src/control/checkpoint_task.rs +++ b/nodedb/src/control/checkpoint_task.rs @@ -10,6 +10,7 @@ use tracing::{info, warn}; use super::checkpoint_manager::{ CheckpointCycleInputs, CheckpointManagerConfig, run_checkpoint_cycle, }; +use super::startup::StartupPhase; /// Spawn the checkpoint manager as a background Tokio task. /// @@ -33,6 +34,26 @@ pub fn spawn_checkpoint_task( "checkpoint_manager::final", ); tokio::spawn(async move { + // Boot recovery reads the WAL on disk until the gateway opens: data + // groups replay their Raft logs against it, and each Calvin scheduler + // scans it for applied markers. A replayed core reports its floor at + // once, so a cycle in that window can delete segments those readers + // still need. The first cycle waits for the gateway. + let started = tokio::select! { + ready = shared.startup.await_phase(StartupPhase::GatewayEnable) => ready.map_err(|error| { + warn!(%error, "checkpoint manager not started: startup did not complete"); + }), + // The WAL on disk is intact, so restart replay covers what a final + // cycle would have made redundant. + _ = guard.await_signal() => { + info!("shutdown before startup completed: no final checkpoint"); + Err(()) + } + }; + if started.is_err() { + guard.report_drained(); + return; + } info!( interval_secs = config.interval.as_secs(), "checkpoint manager started" diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs index 5d69915c5..1863be958 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs @@ -202,6 +202,23 @@ impl Scheduler { /// One send attempt: register, dispatch, and cancel on refusal. fn send_once(&mut self, txn_id: TxnId, step: DispatchStep, request: Request) -> Attempt { + // A crash test holds one collection's flush here: the redo record is + // appended and no core holds the flush. The flush waits in the + // re-send queue, as at capacity, so the scheduler keeps running. + #[cfg(feature = "failpoints")] + if step == DispatchStep::Flush && crate::control::fail_gate::holds_flush(&request.plan) { + tracing::info!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + "calvin: flush held at a fail point" + ); + return Attempt::Capacity(Box::new(DeferredDispatch { + txn_id, + step, + request, + })); + } let request_id = request.request_id; let resp_rx = self.shared.tracker.register(request_id); let result = match self.shared.dispatcher.lock() { diff --git a/nodedb/src/control/fail_gate.rs b/nodedb/src/control/fail_gate.rs index 838babaaa..1322a9d28 100644 --- a/nodedb/src/control/fail_gate.rs +++ b/nodedb/src/control/fail_gate.rs @@ -34,14 +34,30 @@ pub(crate) async fn wait(name: &str) { /// The write funnel's gate: after a logged write's WAL record is appended and /// before any core holds the request. Named per collection, /// `funnel::before_dispatch::`, and only a write carrying a WAL -/// LSN reaches it. +/// LSN reaches it. A committed redo parks on the gate of each collection it +/// writes. pub(crate) async fn before_dispatch(plan: &PhysicalPlan, wal_lsn: Option) { if wal_lsn.is_none() { return; } - let name = format!( - "funnel::before_dispatch::{}", - plan.collection().unwrap_or_default() - ); - wait(&name).await; + for collection in plan.named_collections() { + wait(&format!("funnel::before_dispatch::{collection}")).await; + } +} + +/// The Calvin scheduler's flush gate: after a committed transaction's redo +/// record is appended and before its flush reaches a core. Named per +/// collection, `calvin::before_flush::`. +/// +/// The scheduler cannot park on a timer, so the gate answers without waiting: +/// `true` while the file armed for one of the flush's collections is absent. +/// The scheduler then keeps the flush in its re-send queue and asks again on +/// its next pass. +pub(crate) fn holds_flush(plan: &PhysicalPlan) -> bool { + plan.named_collections().iter().any(|collection| { + matches!( + lookup(&format!("calvin::before_flush::{collection}")), + Some(FailAction::WaitForFile(path)) if !path.exists() + ) + }) } diff --git a/nodedb/src/data/executor/array_checkpoint.rs b/nodedb/src/data/executor/array_checkpoint.rs index 4937841a9..958d7b12e 100644 --- a/nodedb/src/data/executor/array_checkpoint.rs +++ b/nodedb/src/data/executor/array_checkpoint.rs @@ -44,8 +44,10 @@ //! task runs, so every record whose cells the memtable holds is named. A //! record the stamp does not name was not applied yet, and replay applies it. //! -//! The flush also reports the core watermark as this engine's durable point, -//! the LSN the checkpoint manager may truncate below. +//! The flush also reports the checkpoint floor (`checkpoint_floor`) as this +//! engine's durable point, the LSN the checkpoint manager may truncate below. +//! It is the stamp's prefix: a record above it can be absent from the +//! applied ranges, so the WAL keeps it. //! //! An array whose memtable is empty flushes nothing and leaves its manifest's //! stamp where it stands. That is not a gap: an empty memtable means every @@ -75,7 +77,7 @@ impl CoreLoop { /// Flush every array open on this core to disk and return the LSN the array /// engine is now durable through. /// - /// Returns `Ok(watermark)` only once every array's segment AND manifest have + /// Returns `Ok(floor)` only once every array's segment AND manifest have /// landed. Any failure returns `Err` — the caller must then clamp the /// reported checkpoint LSN to the last LSN the arrays were known durable /// through, so a failed flush costs WAL growth instead of the cells it could @@ -92,7 +94,7 @@ impl CoreLoop { /// here to lose. Its segments are on disk and its store is opened lazily by /// the first read (`ensure_array_open`) or by replay. pub(in crate::data::executor) fn checkpoint_array_engines(&mut self) -> crate::Result { - let durable_through = self.watermark; + let durable_through = self.checkpoint_floor(); let stamp = self.floors.applied_prefix.stamp()?; // Collected first: `flush` takes `&mut self.array_engine`, so the id @@ -106,7 +108,7 @@ impl CoreLoop { let mut first_error: Option = None; for id in &ids { // `Ok(None)` = empty memtable, nothing to write; see the module docs - // for why the watermark is still durable for that array. + // for why the floor is still durable for that array. match self.array_engine.flush(id, stamp.clone()) { Ok(Some(_)) => flushed += 1, Ok(None) => {} @@ -298,6 +300,17 @@ mod tests { self.put_at(id, x, y, v, 1, wal_lsn); } + /// Every record through `lsn` has a final outcome: the watermark and + /// the outcome floor both reach it, as a live core's do once those + /// writes answered. + fn settle_through(&mut self, lsn: u64) { + self.core.advance_watermark(Lsn::new(lsn)); + self.core + .floors + .applied_prefix + .observe_outcome_floor(Lsn::new(lsn)); + } + /// `put` with an explicit system time, for bitemporal versions. fn put_at(&mut self, id: &ArrayId, x: i64, y: i64, v: i64, sys_ms: i64, wal_lsn: u64) { let cells = vec![cell(x, y, v, sys_ms)]; @@ -402,7 +415,7 @@ mod tests { vec![(1, 2, 30), (9, 9, 40)], "both cells must be live in the memtable before any flush" ); - before.core.advance_watermark(Lsn::new(20)); + before.settle_through(20); let reported = before .core @@ -429,37 +442,37 @@ mod tests { ); } - /// A flush with an empty memtable must still report the watermark: every + /// A flush with an empty memtable must still report the floor: every /// cell it holds is already in a segment, so clamping there would pin WAL /// truncation for no reason. #[test] - fn empty_memtable_reports_the_watermark() { + fn empty_memtable_reports_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let id = aid(); let mut core = Core::open_at(dir.path()); core.open_array(&id); core.put(&id, 1, 1, 7, 5); - core.core.advance_watermark(Lsn::new(5)); + core.settle_through(5); core.core.checkpoint_array_engines().expect("first flush"); - core.core.advance_watermark(Lsn::new(900)); + core.settle_through(900); assert_eq!( core.core.checkpoint_array_engines().expect("second flush"), Lsn::new(900), "nothing was written since the last flush, so the array engine is \ - durable through the current watermark" + durable through the current floor" ); } - /// A core with no arrays open reports the watermark rather than clamping — + /// A core with no arrays open reports the floor rather than clamping — /// it holds no array state at all, so it can never be the reason the WAL /// must be kept. #[test] - fn no_arrays_reports_the_watermark() { + fn no_arrays_reports_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = Core::open_at(dir.path()); - core.core.advance_watermark(Lsn::new(42)); + core.settle_through(42); assert_eq!( core.core.checkpoint_array_engines().expect("flush"), Lsn::new(42) @@ -477,7 +490,7 @@ mod tests { let mut core = Core::open_at(dir.path()); core.open_array(&id); core.put(&id, 1, 2, 30, 10); - core.core.advance_watermark(Lsn::new(10)); + core.settle_through(10); core.core.checkpoint_array_engines().expect("flush"); core.put(&id, 3, 3, 50, 11); @@ -488,15 +501,14 @@ mod tests { ); } - /// A committed record applied at an LSN the array's manifest stamp - /// already names is flushed into a segment: restart replay skips it, so - /// the segment is its only copy. + /// A committed record applied after a flush that named a higher LSN is + /// not named by that flush's manifest stamp. The flush needs no republish: + /// restart replay applies the record, and skips the flushed one. #[test] - fn a_committed_record_the_manifest_stamp_names_is_flushed() { + fn a_committed_record_applied_after_a_flush_replays_after_a_restart() { use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; use crate::engine::array::wal::{ArrayPutPayload, encode_put_with_version}; use crate::wal::{RedoRecord, RedoSubRecord}; - use nodedb_wal::record::RecordType; let dir = tempfile::tempdir().expect("tempdir"); let id = aid(); @@ -505,17 +517,10 @@ mod tests { before.open_array(&id); before.put(&id, 1, 2, 30, 100); before.core.advance_watermark(Lsn::new(100)); - // The outcome floor passed lsn 50 while that record was still on its - // way to this core, so the flush's stamp prefix names it. - before - .core - .floors - .applied_prefix - .observe_outcome_floor(Lsn::new(100)); before .core .checkpoint_array_engines() - .expect("flush stamped through lsn 100"); + .expect("flush naming lsn 100"); let payload = encode_put_with_version(&ArrayPutPayload { array_id: id.clone(), @@ -552,15 +557,56 @@ mod tests { }, ); assert_eq!(response.status, Status::Ok, "apply: {response:?}"); + assert!( + !before + .core + .array_engine + .store(&id) + .expect("open") + .manifest() + .replay + .skips(50), + "the flush written before the record applied does not name it" + ); + let live = before.slice_all(&id); drop(before); + let redo_record = WalRecord::new(WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn: 50, + tenant_id: TID, + vshard_id: 0, + database_id: DatabaseId::DEFAULT.as_u64(), + payload: redo, + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record"); let mut after = Core::open_at(dir.path()); after.open_array(&id); + after + .core + .floors + .applied_prefix + .seed_replayed_through(Lsn::new(100)); + after + .core + .replay_transaction_redo_wal( + &[redo_record, put_record(&id, 1, 2, 30, 1, 100)], + 1, + &TombstoneSet::new(), + ) + .expect("replay"); + assert_eq!(after.slice_all(&id), live); + assert_eq!(live, vec![(1, 2, 30), (9, 9, 40)]); + after + .core + .checkpoint_array_engines() + .expect("flush the replayed cell"); assert_eq!( - after.slice_all(&id), - vec![(1, 2, 30), (9, 9, 40)], - "the cell committed at lsn 50 survives a restart that skips every \ - record the stamp names" + segment_tiles(&after, &id), + 2, + "the replayed segment holds the lsn-50 cell alone; lsn 100 is not repeated" ); } diff --git a/nodedb/src/data/executor/columnar_checkpoint/format.rs b/nodedb/src/data/executor/columnar_checkpoint/format.rs index bb7216a29..3a5f303a3 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/format.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/format.rs @@ -12,7 +12,7 @@ use crate::data::executor::applied_prefix::ReplayStamp; /// A file stamped with any other version is refused rather than misparsed. /// Refusing costs a WAL replay; misparsing would install wrong rows AND a floor /// that suppresses the records which would have corrected them. -pub(crate) const COLUMNAR_CKPT_FORMAT_VERSION: u16 = 2; +pub(crate) const COLUMNAR_CKPT_FORMAT_VERSION: u16 = 3; /// Names the live generation. Writing this file is what publishes a checkpoint. #[derive( @@ -30,10 +30,8 @@ pub(crate) struct ColumnarCheckpointManifest { pub format_version: u16, /// Which `gen-{n}/` directory holds the live collection files. pub generation: u64, - /// The LSN this core reports as the columnar engine's truncation floor when - /// the generation is restored. It gates no replay: [`Self::replay`] does. - pub durable_through_lsn: u64, - /// The records the generation holds. WAL replay skips exactly the columnar + /// The records the generation holds. Its prefix is also the LSN this core + /// reports as the columnar engine's truncation floor once it is restored. WAL replay skips exactly the columnar /// records [`ReplayStamp::skips`] names and replays every other one. /// /// A single highest-applied LSN cannot state this. LSNs are node-global and diff --git a/nodedb/src/data/executor/columnar_checkpoint/load.rs b/nodedb/src/data/executor/columnar_checkpoint/load.rs index 25620d7c5..d1e4be334 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/load.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/load.rs @@ -107,8 +107,7 @@ impl CoreLoop { // Claimed only once every engine is in: the floor suppresses WAL // records, so claiming it over a half-restored generation would turn a // recoverable read failure into permanent data loss. - self.floors.columnar_durable_lsn = Lsn::new(manifest.durable_through_lsn); - self.floors.columnar_published_lsn = Lsn::new(manifest.replay.prefix); + self.floors.columnar_durable_lsn = Lsn::new(manifest.replay.prefix); let replay_prefix = manifest.replay.prefix; let applied_ranges = manifest.replay.applied_above.len(); self.floors.replay_floors.columnar.set(manifest.replay); @@ -119,7 +118,6 @@ impl CoreLoop { collections, segments, geometry_rows, - durable_through_lsn = manifest.durable_through_lsn, replay_prefix, applied_ranges, "columnar checkpoint restored" @@ -683,7 +681,7 @@ mod tests { /// the highest applied LSN, since a record below that can still be on its /// way. Too wide a gate drops a write; too narrow re-applies a folded one. #[test] - fn the_stamp_becomes_the_restored_floor_and_the_watermark_the_durable_lsn() { + fn the_stamp_becomes_the_restored_floor_and_its_prefix_the_durable_lsn() { let dir = tempfile::tempdir().expect("tempdir"); let coll = "ck_lsn"; @@ -700,8 +698,9 @@ mod tests { .expect("checkpoint must publish"); assert_eq!( reported, - Lsn::new(900), - "a successful flush reports the watermark" + Lsn::new(890), + "a successful flush reports the outcome floor, never the watermark: the \ + record at 891 can still be on its way" ); drop(core); @@ -714,8 +713,7 @@ mod tests { .load_columnar_checkpoints() .expect("checkpoint load must succeed"); - assert_eq!(restored.floors.columnar_durable_lsn, Lsn::new(900)); - assert_eq!(restored.floors.columnar_published_lsn, Lsn::new(890)); + assert_eq!(restored.floors.columnar_durable_lsn, Lsn::new(890)); let floor = &restored.floors.replay_floors.columnar; assert!(floor.covers(890), "the prefix is folded in"); assert!( @@ -816,6 +814,9 @@ mod tests { &[(1, "a", Surrogate(901)), (2, "b", Surrogate(902))], ); core.watermark = Lsn::new(100); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(100)); core.checkpoint_columnar_engines() .expect("first checkpoint must publish"); @@ -826,6 +827,9 @@ mod tests { .delete(&Value::Integer(1)) .expect("delete"); core.watermark = Lsn::new(200); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(200)); core.checkpoint_columnar_engines() .expect("second checkpoint must publish"); drop(core); diff --git a/nodedb/src/data/executor/columnar_checkpoint/write.rs b/nodedb/src/data/executor/columnar_checkpoint/write.rs index 4a3d518c7..bdf89a45a 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/write.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/write.rs @@ -47,11 +47,8 @@ impl CoreLoop { /// replays. That is safe and stays safe: it matched nothing against the /// state that the export captured, so re-executing the same predicate /// against that same restored state matches nothing again. - /// - /// Every published generation raises `columnar_published_lsn` to the - /// stamp's prefix (see `redo_apply::cover`). pub(in crate::data::executor) fn checkpoint_columnar_engines(&mut self) -> crate::Result { - let durable_through = self.watermark; + let durable_through = self.checkpoint_floor(); let replay = self.floors.applied_prefix.stamp()?; let ckpt_dir = columnar_ckpt_dir(&self.data_dir, self.core_id); @@ -76,8 +73,7 @@ impl CoreLoop { let written = self.write_columnar_generation(&gen_dir)?; let prefix = Lsn::new(replay.prefix); let applied_ranges = replay.applied_above.len(); - self.publish_columnar_generation(&ckpt_dir, generation, durable_through, replay)?; - self.floors.columnar_published_lsn = self.floors.columnar_published_lsn.max(prefix); + self.publish_columnar_generation(&ckpt_dir, generation, replay)?; // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what @@ -194,13 +190,11 @@ impl CoreLoop { &self, ckpt_dir: &std::path::Path, generation: u64, - durable_through: Lsn, replay: ReplayStamp, ) -> crate::Result<()> { let manifest = ColumnarCheckpointManifest { format_version: COLUMNAR_CKPT_FORMAT_VERSION, generation, - durable_through_lsn: durable_through.as_u64(), replay, }; let bytes = diff --git a/nodedb/src/data/executor/core_loop/accessors.rs b/nodedb/src/data/executor/core_loop/accessors.rs index 0eb82331b..604f4c425 100644 --- a/nodedb/src/data/executor/core_loop/accessors.rs +++ b/nodedb/src/data/executor/core_loop/accessors.rs @@ -24,6 +24,18 @@ impl CoreLoop { self.watermark = lsn; } + /// The LSN a checkpoint written now makes durable, and the lowest LSN + /// the WAL must keep above: the outcome floor this core read. Every + /// record at or below it has a final outcome, so a checkpoint written + /// between tasks holds each one this core applied, and a refused one has + /// a durable abort marker. A record above it can still be on its way to + /// this core, and restart replay must reach it. The core watermark is no + /// such bound: records apply out of LSN order, so it can pass a record + /// that has not applied yet. + pub(in crate::data::executor) fn checkpoint_floor(&self) -> Lsn { + self.floors.applied_prefix.outcome_floor() + } + /// Merge reconstructed sync HWM state from WAL replay into this core's gate. /// /// Called by `replay_all_wal` with the maps built by diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs index ce7457459..1fcbb5b18 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/init.rs @@ -42,11 +42,6 @@ impl CheckpointFloors { // holds none — so both stay at zero until this process's own flush // succeeds. Vector restores its LSN from its manifest at load. vector_durable_lsn: Lsn::ZERO, - // No generation is published until one is loaded or written. - vector_published_lsn: Lsn::ZERO, - sparse_vector_published_lsn: Lsn::ZERO, - kv_published_lsn: Lsn::ZERO, - columnar_published_lsn: Lsn::ZERO, crdt_durable_lsn: Lsn::ZERO, spatial_durable_lsn: Lsn::ZERO, replay_floors: ReplayFloors::default(), diff --git a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs index b8a244e51..446fe59da 100644 --- a/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs +++ b/nodedb/src/data/executor/core_loop/checkpoint_floors/state.rs @@ -123,29 +123,6 @@ pub(in crate::data::executor) struct CheckpointFloors { /// the watermark — is what the core may report. pub(in crate::data::executor) vector_durable_lsn: Lsn, - /// Replay-stamp prefix of the newest KV checkpoint generation on disk, - /// restored at boot from the manifest. Restart replay skips every KV - /// record at or below it, so a committed record applied at or below it - /// must be published again (`redo_apply::cover`). Never a truncation - /// floor. - pub(in crate::data::executor) kv_published_lsn: Lsn, - - /// Replay-stamp prefix of the newest columnar checkpoint generation on - /// disk, whichever flush published it, restored at boot from the - /// manifest. Same rule as `kv_published_lsn`. - pub(in crate::data::executor) columnar_published_lsn: Lsn, - - /// Replay-stamp prefix of the newest vector checkpoint generation on - /// disk, whichever flush published it, restored at boot from the - /// manifest. Same rule as `kv_published_lsn`. Never a truncation floor: - /// that is `vector_durable_lsn`. - pub(in crate::data::executor) vector_published_lsn: Lsn, - - /// Replay-stamp prefix of the newest sparse-vector checkpoint generation - /// on disk, restored at boot from the manifest. Same rule as - /// `kv_published_lsn`. - pub(in crate::data::executor) sparse_vector_published_lsn: Lsn, - /// Highest LSN the CRDT engines are known to be durable through OUTSIDE the /// WAL (i.e. in `{data_dir}/crdt-ckpt/`), advanced only by a fully /// successful `checkpoint_crdt_engines`. diff --git a/nodedb/src/data/executor/core_loop/fail_stop.rs b/nodedb/src/data/executor/core_loop/fail_stop.rs index c097782ab..f5e75ce79 100644 --- a/nodedb/src/data/executor/core_loop/fail_stop.rs +++ b/nodedb/src/data/executor/core_loop/fail_stop.rs @@ -4,8 +4,8 @@ //! //! A core's state is unknown when a rollback fails part way, or when a //! committed record installed and the work after its install failed: a -//! memtable flush that drained rows, or the republish of an artifact that -//! must hold the record. Restart replay rebuilds the state from the WAL. +//! memtable flush that drained rows, a vector seal, or a truncate's removal. +//! Restart replay rebuilds the state from the WAL. //! Until then the core must not serve the state it holds. //! //! The first cause wins. It logs one ERROR, files one recorder report, @@ -29,8 +29,8 @@ use super::CoreLoop; pub(in crate::data::executor) enum FailStopCause { /// A rollback of an undo log failed part way. RollbackFailed, - /// A committed record installed, and a flush or artifact republish owed - /// after its install failed. Neither can be rolled back. + /// A committed record installed, and the settle owed after its install + /// failed. It cannot be rolled back. PostInstallFailed, } diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index 70c64323e..aea4f4a43 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -113,15 +113,15 @@ impl CoreLoop { self.floors.applied_prefix.note_applied(lsn); } // A crash test kills the process here: one collection's logged - // write applied, and its response never leaves the core. A task - // with no WAL record never matches. - crate::fail_point!(&match task.wal_lsn() { - Some(_) => format!( - "core::after_apply::{}", - task.plan().collection().unwrap_or_default() - ), - None => String::new(), - }); + // write applied, and its response never leaves the core. A + // committed redo matches each collection it writes. A task with + // no WAL record never matches. + #[cfg(feature = "failpoints")] + if task.wal_lsn().is_some() { + for collection in task.plan().named_collections() { + crate::fail_point!(&format!("core::after_apply::{collection}")); + } + } task.state = TaskState::Completed; // A failed rollback leaves this core's state unknown. self.fail_stop_on_rollback_failure(&resp); diff --git a/nodedb/src/data/executor/graph_label_checkpoint/load.rs b/nodedb/src/data/executor/graph_label_checkpoint/load.rs index 52eaa0927..8b5198d7c 100644 --- a/nodedb/src/data/executor/graph_label_checkpoint/load.rs +++ b/nodedb/src/data/executor/graph_label_checkpoint/load.rs @@ -252,6 +252,10 @@ mod tests { csr.add_node_label("carol", "Bot").expect("label carol"); } before.advance_watermark(Lsn::new(900)); + before + .floors + .applied_prefix + .observe_outcome_floor(Lsn::new(900)); let reported = before .checkpoint_graph_labels() diff --git a/nodedb/src/data/executor/graph_label_checkpoint/write.rs b/nodedb/src/data/executor/graph_label_checkpoint/write.rs index c777f2fc0..5a4613a52 100644 --- a/nodedb/src/data/executor/graph_label_checkpoint/write.rs +++ b/nodedb/src/data/executor/graph_label_checkpoint/write.rs @@ -16,7 +16,7 @@ impl CoreLoop { /// Flush this core's CSR node labels to disk and return the LSN they are now /// durable through. /// - /// Returns `Ok(watermark)` only once the state file has landed and been + /// Returns `Ok(floor)` only once the state file has landed and been /// fsynced. Any failure returns `Err` — the caller must then clamp the /// reported checkpoint LSN to the last LSN the labels were known durable /// through, so a failed flush costs WAL growth instead of the @@ -26,11 +26,10 @@ impl CoreLoop { /// is intact and live, after it the new one is. There is no window in which /// half a core's partitions are published. /// - /// Stamping with the core watermark rests on this: the checkpoint runs - /// on the core's own thread between tasks, and a label write reaches - /// `note_write_lsn` (which raises the watermark) only after - /// `add_node_label` / `remove_node_label` has already mutated the bitset. So - /// every label change with `lsn <= watermark` is in the export below. + /// The reported LSN is the checkpoint floor (`checkpoint_floor`): the + /// checkpoint runs on the core's own thread between tasks, so every record + /// at or below that outcome floor that this core applied is in the export + /// below. /// /// Edges are deliberately absent from the export. They are committed to the /// redb `EdgeStore` at apply time and the whole CSR is rebuilt from it in @@ -38,7 +37,7 @@ impl CoreLoop { /// second copy of state that cannot be lost — and a stale one, since the /// rebuild would overwrite it on the next boot regardless. pub(in crate::data::executor) fn checkpoint_graph_labels(&self) -> crate::Result { - let durable_through = self.watermark; + let durable_through = self.checkpoint_floor(); // Sorted at both levels so identical label state always encodes to // identical bytes. diff --git a/nodedb/src/data/executor/handlers/control/checkpoint_crdt.rs b/nodedb/src/data/executor/handlers/control/checkpoint_crdt.rs index 1c149223c..1ba23aa7a 100644 --- a/nodedb/src/data/executor/handlers/control/checkpoint_crdt.rs +++ b/nodedb/src/data/executor/handlers/control/checkpoint_crdt.rs @@ -35,7 +35,7 @@ impl CoreLoop { /// A `TenantCrdtEngine` is a set of in-memory `LoroDoc`s with no store /// behind them. `load_crdt_checkpoints` reads these files back at boot and /// WAL replay re-imports the deltas above them; there is no third source. - /// So a flush that failed while the core still reported its watermark would + /// So a flush that failed while the core still reported its floor would /// authorise deleting the delta records that are the only remaining copy of /// the state this flush did not write — the documents come back at whatever /// version the last SUCCESSFUL checkpoint captured, with every edit since @@ -45,13 +45,14 @@ impl CoreLoop { /// caller clamps the reported checkpoint LSN to the last LSN the CRDT /// engines were known durable through. /// - /// Stamping with the core watermark rests on this: the checkpoint runs - /// on the core's own thread between tasks, and a delta apply raises the - /// watermark only after the `LoroDoc` has already imported it. + /// The reported LSN is the checkpoint floor (`checkpoint_floor`): the + /// checkpoint runs on the core's own thread between tasks, so every record + /// at or below that outcome floor that this core applied is in the export + /// below. pub(in crate::data::executor) fn checkpoint_crdt_engines( &self, ) -> crate::Result { - let durable_lsn = self.watermark; + let durable_lsn = self.checkpoint_floor(); let ckpt_dir = crdt_ckpt_dir(&self.data_dir, self.core_id); std::fs::create_dir_all(&ckpt_dir).map_err(|e| storage_err(&ckpt_dir, "create dir", &e))?; @@ -149,10 +150,9 @@ mod tests { use crate::types::TenantId; /// State that has left the core between two cycles must not reload at boot. - /// The flush reports the core watermark either way, so the WAL records that - /// removed it are already deletable — under the previous flat layout the - /// file stayed reachable and the collection came back at every boot, - /// forever. + /// The flush reports the core's floor either way, so the WAL records that + /// removed it can be deleted. The previous generation's file must not stay + /// reachable, or the collection comes back at every boot. #[test] fn state_dropped_between_cycles_does_not_survive_the_next_one() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs b/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs index e0d81e860..02deb33f0 100644 --- a/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs +++ b/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs @@ -13,7 +13,7 @@ //! violating. On success the engine's `*_durable_lsn` advances to the flushed //! point and that point is returned. On FAILURE the error is surfaced — logged //! at `warn` naming the clamp it caused, never swallowed — and the LAST-KNOWN -//! durable LSN is returned instead of the watermark. The reported LSN authorises +//! durable LSN is returned instead of the checkpoint floor. The reported LSN authorises //! `WalManager::truncate_before` to unlink segments below it, so a flush that //! failed must never widen that authority over the very state it failed to //! write. Clamping costs WAL growth until the next cycle succeeds; not clamping @@ -26,7 +26,7 @@ //! POSTINGS / DOC_LENGTHS / DOC_TERMS / STATS straight into the same redb `Database` the //! `sparse` engine commits to — bypassing the LSM memtable precisely so the //! index is atomic with the document write. redb commits durably, so an FTS -//! write at or below the watermark is already on stable storage in a store that +//! write at or below the floor is already on stable storage in a store that //! is not the WAL. There is nothing to flush and therefore nothing to clamp. use tracing::warn; @@ -397,9 +397,61 @@ mod tests { .expect("CoreLoop::open") } + /// Records apply out of LSN order. The record at 20 is still on its way + /// while 30 applied, so the watermark is 30 and the outcome floor 10. + /// Every engine, and the whole checkpoint, reports the floor: truncating + /// below 30 would delete the record at 20, which no checkpoint holds. + /// Once the floor passes 30, the reported LSN follows it. + #[test] + fn a_checkpoint_never_reports_past_a_record_in_flight() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut core = open_core(dir.path()); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(10)); + core.floors.applied_prefix.note_applied(Lsn::new(30)); + core.watermark = Lsn::new(30); + + let reported = [ + core.checkpoint_kv_durable_lsn(), + core.checkpoint_sparse_vector_durable_lsn(), + core.checkpoint_sync_hwm_durable_lsn(), + core.checkpoint_columnar_durable_lsn(), + core.checkpoint_graph_label_durable_lsn(), + core.checkpoint_array_durable_lsn(), + core.checkpoint_ts_durable_lsn(), + core.checkpoint_vector_durable_lsn(), + core.checkpoint_crdt_durable_lsn(), + core.checkpoint_spatial_durable_lsn(), + ]; + assert!( + reported.iter().all(|lsn| *lsn == Lsn::new(10)), + "every engine reports the outcome floor, not the watermark: {reported:?}" + ); + let checkpoint_lsn = |core: &mut CoreLoop| { + let response = core + .execute_checkpoint(&crate::data::executor::core_loop::tests::make_default_task()); + let bytes: [u8; 8] = response.payload.as_bytes()[..8] + .try_into() + .expect("an 8-byte LSN"); + Lsn::new(u64::from_le_bytes(bytes)) + }; + assert_eq!(checkpoint_lsn(&mut core), Lsn::new(10)); + + // The record at 20 applied and answered: the floor passes 30. + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(30)); + assert_eq!( + checkpoint_lsn(&mut core), + Lsn::new(30), + "truncation advances once the in-flight record settles" + ); + } + /// The clamp is the whole point of this module: on flush failure the - /// contributor must return the LAST-KNOWN durable LSN, never the watermark. - /// Returning the watermark would widen `WalManager::truncate_before` over + /// contributor must return the LAST-KNOWN durable LSN, never the floor. + /// Returning the floor would widen `WalManager::truncate_before` over /// exactly the state the flush just failed to write. #[test] fn columnar_flush_failure_clamps_to_the_last_known_durable_lsn() { @@ -416,7 +468,7 @@ mod tests { core.checkpoint_columnar_durable_lsn(), Lsn::ZERO, "a fresh core has flushed nothing, so a failed flush must clamp to \ - zero rather than authorise truncating up to the watermark" + zero rather than authorise truncating up to the floor" ); assert_eq!(core.floors.columnar_durable_lsn, Lsn::ZERO); } @@ -425,10 +477,13 @@ mod tests { /// LSN — the two must not drift, since the field is what a later failure /// clamps back to. #[test] - fn columnar_flush_success_advances_and_returns_the_watermark() { + fn columnar_flush_success_advances_and_returns_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = open_core(dir.path()); core.watermark = Lsn::new(750); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(750)); assert_eq!(core.checkpoint_columnar_durable_lsn(), Lsn::new(750)); assert_eq!(core.floors.columnar_durable_lsn, Lsn::new(750)); @@ -455,16 +510,19 @@ mod tests { core.checkpoint_graph_label_durable_lsn(), Lsn::ZERO, "a fresh core has flushed nothing, so a failed flush must clamp to \ - zero rather than authorise truncating up to the watermark" + zero rather than authorise truncating up to the floor" ); assert_eq!(core.floors.graph_label_durable_lsn, Lsn::ZERO); } #[test] - fn graph_label_flush_success_advances_and_returns_the_watermark() { + fn graph_label_flush_success_advances_and_returns_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = open_core(dir.path()); core.watermark = Lsn::new(750); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(750)); assert_eq!(core.checkpoint_graph_label_durable_lsn(), Lsn::new(750)); assert_eq!(core.floors.graph_label_durable_lsn, Lsn::new(750)); @@ -536,10 +594,13 @@ mod tests { } #[test] - fn array_flush_success_advances_and_returns_the_watermark() { + fn array_flush_success_advances_and_returns_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = open_core(dir.path()); core.watermark = Lsn::new(750); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(750)); open_array_with_a_pending_cell(&mut core); assert_eq!(core.checkpoint_array_durable_lsn(), Lsn::new(750)); @@ -607,10 +668,13 @@ mod tests { } #[test] - fn timeseries_flush_success_advances_and_returns_the_watermark() { + fn timeseries_flush_success_advances_and_returns_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = open_core(dir.path()); core.watermark = Lsn::new(750); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(750)); ts_memtable_with_a_pending_row(&mut core); assert_eq!(core.checkpoint_ts_durable_lsn(), Lsn::new(750)); @@ -660,16 +724,19 @@ mod tests { core.checkpoint_vector_durable_lsn(), Lsn::ZERO, "a fresh core has flushed nothing, so a failed flush must clamp to \ - zero rather than authorise truncating up to the watermark" + zero rather than authorise truncating up to the floor" ); assert_eq!(core.floors.vector_durable_lsn, Lsn::ZERO); } #[test] - fn vector_flush_success_advances_and_returns_the_watermark() { + fn vector_flush_success_advances_and_returns_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = open_core(dir.path()); core.watermark = Lsn::new(750); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(750)); vector_collection_with_a_pending_vector(&mut core); assert_eq!(core.checkpoint_vector_durable_lsn(), Lsn::new(750)); @@ -699,16 +766,19 @@ mod tests { core.checkpoint_crdt_durable_lsn(), Lsn::ZERO, "a fresh core has flushed nothing, so a failed flush must clamp to \ - zero rather than authorise truncating up to the watermark" + zero rather than authorise truncating up to the floor" ); assert_eq!(core.floors.crdt_durable_lsn, Lsn::ZERO); } #[test] - fn crdt_flush_success_advances_and_returns_the_watermark() { + fn crdt_flush_success_advances_and_returns_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = open_core(dir.path()); core.watermark = Lsn::new(750); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(750)); core.get_crdt_engine( nodedb_types::DatabaseId::DEFAULT, nodedb_types::TenantId::new(1), @@ -763,16 +833,19 @@ mod tests { core.checkpoint_spatial_durable_lsn(), Lsn::ZERO, "a fresh core has flushed nothing, so a failed flush must clamp to \ - zero rather than authorise truncating up to the watermark" + zero rather than authorise truncating up to the floor" ); assert_eq!(core.floors.spatial_durable_lsn, Lsn::ZERO); } #[test] - fn spatial_flush_success_advances_and_returns_the_watermark() { + fn spatial_flush_success_advances_and_returns_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = open_core(dir.path()); core.watermark = Lsn::new(750); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(750)); spatial_index_with_a_pending_entry(&mut core); assert_eq!(core.checkpoint_spatial_durable_lsn(), Lsn::new(750)); diff --git a/nodedb/src/data/executor/handlers/control/snapshot.rs b/nodedb/src/data/executor/handlers/control/snapshot.rs index 9445b9595..f31eb166d 100644 --- a/nodedb/src/data/executor/handlers/control/snapshot.rs +++ b/nodedb/src/data/executor/handlers/control/snapshot.rs @@ -334,19 +334,20 @@ impl CoreLoop { /// > recoverable WITHOUT the WAL — either flushed by this checkpoint, or /// > held in a durable store that is not the WAL. /// - /// Reporting the watermark unconditionally violated that rule for any engine - /// whose flush was partial or absent, and deleted the only copy of its - /// state. The LSN is therefore computed as `min(watermark, every engine's - /// reported durable LSN)`: a flush that fails clamps the LSN instead of - /// silently widening the deletion. + /// The ceiling is the checkpoint floor (`checkpoint_floor`), never the core + /// watermark. Records apply out of LSN order, so the watermark can pass a + /// record still on its way to this core; truncating below it would delete + /// that record's only copy. The LSN is `min(floor, every engine's reported + /// durable LSN)`: a flush that fails clamps the LSN instead of widening the + /// deletion. pub(in crate::data::executor) fn execute_checkpoint( &mut self, task: &ExecutionTask, ) -> Response { // Every engine on this core whose flush can fail contributes the LSN it - // is durable through. The watermark is the ceiling — no engine can be - // durable past writes this core has not seen — and every contribution - // can only pull it DOWN. + // is durable through. The checkpoint floor is the ceiling — a record + // above it can still be on its way to this core — and every + // contribution can only pull it DOWN. // // No engine is left contributing nothing on the grounds that its flush // reports no LSN; that gap is closed. The only state on this core with @@ -456,13 +457,13 @@ impl CoreLoop { tracing::warn!(error = %e, "CSR compaction rejected by memory governor during snapshot; skipping"); } - // 8. Clamp the watermark down to every engine's durable LSN. `min` and + // 8. Clamp the floor down to every engine's durable LSN. `min` and // not `max`: the reported LSN authorises deletion, so where the // engines disagree the WAL must keep whatever the least-durable one // still needs. let checkpoint_lsn = durable_lsns .iter() - .fold(self.watermark, |acc, lsn| acc.min(*lsn)) + .fold(self.checkpoint_floor(), |acc, lsn| acc.min(*lsn)) .as_u64(); // The checkpoint coordinator is deliberately NOT told about this LSN. diff --git a/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs b/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs index ac551f807..645e9f0d9 100644 --- a/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs +++ b/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs @@ -73,7 +73,7 @@ impl CoreLoop { let sample_count = batch.samples.len(); if self.recording_redo_undo() { // The install settles the budget once the whole record landed. - self.note_redo_timeseries_written(key, record_lsn); + self.note_redo_timeseries_written(key); } else { // Re-charge the engine memory budget to the memtable's // resident footprint after replaying these samples. The @@ -153,7 +153,7 @@ impl CoreLoop { return 0; } if installing { - self.note_redo_timeseries_written((db_id, tid, collection.to_string()), record_lsn); + self.note_redo_timeseries_written((db_id, tid, collection.to_string())); } if format == "ilp-msgpack" { return zerompk::from_msgpack::>(payload).map_or(0, |rows| rows.len()); diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs deleted file mode 100644 index e58b3d183..000000000 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/cover.rs +++ /dev/null @@ -1,464 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Keep every published replay stamp true after a committed record applies -//! below its prefix. -//! -//! Restart replay skips a record at or below the prefix of an engine's -//! published replay stamp: the vector, KV and columnar checkpoint manifests, -//! an array manifest. The prefix claims that the published artifact holds -//! every record at or below it. -//! Records apply in the order Raft -//! commits them, not in LSN order, so a record can apply after an artifact -//! was published at a higher LSN. That record lives only in memory, and the -//! claim is false for it. Publishing the artifact again, now holding the -//! record, makes the claim true. The install settles a timeseries partition -//! stamp the same way (`settle`). - -use nodedb_types::columnar::{COLUMNAR_IMAGE_KIND, ColumnarImageWalRecord}; -use nodedb_wal::record::RecordType; - -use crate::bridge::envelope::ErrorCode; -use crate::data::executor::core_loop::CoreLoop; -use crate::types::Lsn; -use crate::wal::{RedoRecord, RedoSubRecord}; - -use super::sub_ops::kv_ops; - -/// The engine-wide checkpoints a redo record writes into. -pub(super) struct WrittenEngines { - /// A vector index the vector checkpoint publishes. - pub vectors: bool, - /// A sparse-vector index the sparse-vector checkpoint publishes. - pub sparse_vectors: bool, - /// A KV collection the KV checkpoint publishes. - pub kv: bool, - /// A columnar collection the columnar checkpoint publishes. - pub columnar: bool, -} - -impl WrittenEngines { - pub(super) fn of(redo: &RedoRecord) -> Self { - Self { - vectors: redo.ops.iter().any(writes_vector_index), - sparse_vectors: redo.ops.iter().any(writes_sparse_vector_index), - kv: !kv_ops(&redo.ops).is_empty() - || redo - .ops - .iter() - .any(|op| is_kv_truncate(op) || is_kv_ttl(op)), - columnar: redo.ops.iter().any(writes_columnar), - } - } -} - -fn writes_vector_index(op: &RedoSubRecord) -> bool { - matches!( - RecordType::from_raw(op.record_type), - Some( - RecordType::VectorPut - | RecordType::VectorDelete - | RecordType::VectorDirectUpsert - | RecordType::VectorDirectUpdate - | RecordType::VectorDirectDelete - | RecordType::VectorDirectTruncate - | RecordType::VectorResolvedDirectWrite - | RecordType::MultiVectorPut - | RecordType::MultiVectorDelete - ) - ) -} - -fn writes_sparse_vector_index(op: &RedoSubRecord) -> bool { - matches!( - RecordType::from_raw(op.record_type), - Some(RecordType::SparseVectorPut | RecordType::SparseVectorDelete) - ) -} - -fn is_kv_truncate(op: &RedoSubRecord) -> bool { - RecordType::from_raw(op.record_type) == Some(RecordType::Delete) - && zerompk::from_msgpack::<(String, String)>(&op.payload) - .is_ok_and(|(disc, _)| disc == "kv_truncate") -} - -/// A `kv_expire` or `kv_persist` sub-record: both lead with their -/// discriminator, and the collection follows it. -fn is_kv_ttl(op: &RedoSubRecord) -> bool { - RecordType::from_raw(op.record_type) == Some(RecordType::Put) - && (zerompk::from_msgpack::<(String, String, Vec)>(&op.payload) - .is_ok_and(|(disc, ..)| disc == "kv_persist") - || zerompk::from_msgpack::<(String, String, Vec, u64, u64)>(&op.payload) - .is_ok_and(|(disc, ..)| disc == "kv_expire")) -} - -fn writes_columnar(op: &RedoSubRecord) -> bool { - match RecordType::from_raw(op.record_type) { - Some(RecordType::ColumnarTruncate) => true, - Some(RecordType::TimeseriesBatch) => { - zerompk::from_msgpack::(&op.payload) - .is_ok_and(|record| record.kind == COLUMNAR_IMAGE_KIND) - } - _ => false, - } -} - -impl CoreLoop { - /// Record that the open redo-apply scope wrote cells to `array_id`. - pub(in crate::data::executor) fn note_redo_array_written( - &mut self, - array_id: &nodedb_array::types::ArrayId, - ) { - if let Some(scope) = self.redo_apply.scope.as_mut() - && !scope.arrays_written.contains(array_id) - { - scope.arrays_written.push(array_id.clone()); - } - } - - /// Publish again every artifact whose stamp names `lsn` but does not hold - /// the record just applied at it. - pub(super) fn cover_applied_record( - &mut self, - lsn: Lsn, - engines: &WrittenEngines, - arrays: &[nodedb_array::types::ArrayId], - ) -> Result<(), ErrorCode> { - if engines.vectors && lsn <= self.floors.vector_published_lsn { - let published = self.floors.vector_published_lsn; - self.checkpoint_vector_indexes() - .map_err(|error| republish_error("vector checkpoint", lsn, published, &error))?; - } - if engines.sparse_vectors && lsn <= self.floors.sparse_vector_published_lsn { - let published = self.floors.sparse_vector_published_lsn; - self.checkpoint_sparse_vector_indexes().map_err(|error| { - republish_error("sparse-vector checkpoint", lsn, published, &error) - })?; - } - if engines.kv && lsn <= self.floors.kv_published_lsn { - let published = self.floors.kv_published_lsn; - self.checkpoint_kv_engines() - .map_err(|error| republish_error("KV checkpoint", lsn, published, &error))?; - } - if engines.columnar && lsn <= self.floors.columnar_published_lsn { - let published = self.floors.columnar_published_lsn; - self.checkpoint_columnar_engines() - .map_err(|error| republish_error("columnar checkpoint", lsn, published, &error))?; - } - for array_id in arrays { - if !self.array_stamp_names(array_id, lsn.as_u64()) { - continue; - } - self.flush_array(array_id) - .map_err(|error| ErrorCode::Internal { - detail: format!( - "a record applied at lsn {} that array '{}' manifest stamp names \ - could not be flushed: {error}", - lsn.as_u64(), - array_id.name - ), - })?; - } - Ok(()) - } -} - -fn republish_error(artifact: &str, lsn: Lsn, published: Lsn, error: &crate::Error) -> ErrorCode { - ErrorCode::Internal { - detail: format!( - "a record applied at lsn {} below the {artifact} at lsn {} could not be \ - published again: {error}", - lsn.as_u64(), - published.as_u64() - ), - } -} - -#[cfg(test)] -mod tests { - use nodedb_types::sync::wire::SyncProvenance; - use nodedb_types::{DatabaseId, Surrogate}; - - use super::*; - use crate::bridge::envelope::Status; - use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; - use crate::data::executor::handlers::transaction::redo_apply::CommittedRedo; - use crate::engine::vector::collection::VectorCollection; - use crate::engine::vector::hnsw::HnswParams; - use crate::wal::RedoSubRecord; - - const TID: u64 = 1; - - fn vector_put(collection: &str, surrogate: u32) -> RedoSubRecord { - RedoSubRecord { - record_type: RecordType::VectorPut as u32, - payload: zerompk::to_msgpack_vec(&( - collection, - vec![0.5f32, 0.5], - 2usize, - "", - None::, - surrogate, - None::, - )) - .expect("encode vector put"), - } - } - - #[test] - fn a_vector_record_applied_below_a_published_checkpoint_is_published_again() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let key = CoreLoop::vector_index_key(DatabaseId::DEFAULT.as_u64(), TID, "docs", ""); - let mut collection = VectorCollection::new(2, HnswParams::default()); - collection.insert_with_surrogate(vec![0.1, 0.9], Surrogate::new(1)); - core.vector_collections.insert(key.clone(), collection); - core.advance_watermark(Lsn::new(100)); - // The published stamp's prefix is what restart replay skips through. - core.floors - .applied_prefix - .observe_outcome_floor(Lsn::new(100)); - core.checkpoint_vector_indexes() - .expect("publish at lsn 100"); - - // Committed at lsn 50, applied after the generation at lsn 100. - let mut task = make_default_task(); - task.wal_lsn = Some(Lsn::new(50)); - let redo = RedoRecord { - version: 1, - ops: vec![vector_put("docs", 2)], - calvin_stamp: None, - } - .to_bytes() - .expect("encode redo"); - let response = core.execute_apply_transaction_redo( - &task, - TID, - CommittedRedo { - redo: &redo, - collections: &["docs".to_string()], - sum_targets: &[], - }, - ); - assert_eq!(response.status, Status::Ok, "apply: {response:?}"); - drop(core); - - // Restart replay skips lsn 50, which the restored stamp's prefix names, - // so the published generation is the only copy of the second vector. - let dir_path = dir.path().to_path_buf(); - let (mut restored, _tx2, _rx2) = make_core_with_dir(&dir_path); - restored.load_vector_checkpoints().expect("load"); - assert_eq!( - restored.vector_collections.get(&key).map(|c| c.len()), - Some(2), - "the vector applied below the checkpoint is in the newest generation" - ); - } - - fn kv_put(collection: &str, key: &[u8], value: &[u8], surrogate: u32) -> RedoSubRecord { - RedoSubRecord { - record_type: RecordType::Put as u32, - payload: zerompk::to_msgpack_vec(&( - "kv_put", - collection, - key.to_vec(), - value.to_vec(), - 0u64, - None::, - surrogate, - )) - .expect("encode kv put"), - } - } - - #[test] - fn a_kv_record_applied_below_a_published_checkpoint_is_published_again() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let _prior = core.kv_engine.put(crate::engine::kv::KvPutParams { - database_id: DatabaseId::DEFAULT.as_u64(), - tenant_id: TID, - collection: "cache", - key: b"a", - value: b"1", - ttl_ms: 0, - now_ms: 0, - surrogate: Surrogate::new(1), - }); - core.advance_watermark(Lsn::new(100)); - // The published stamp's prefix is what restart replay skips through. - core.floors - .applied_prefix - .observe_outcome_floor(Lsn::new(100)); - core.checkpoint_kv_engines().expect("publish at lsn 100"); - - let mut task = make_default_task(); - task.wal_lsn = Some(Lsn::new(50)); - let redo = RedoRecord { - version: 1, - ops: vec![kv_put("cache", b"b", b"2", 2)], - calvin_stamp: None, - } - .to_bytes() - .expect("encode redo"); - let response = core.execute_apply_transaction_redo( - &task, - TID, - CommittedRedo { - redo: &redo, - collections: &["cache".to_string()], - sum_targets: &[], - }, - ); - assert_eq!(response.status, Status::Ok, "apply: {response:?}"); - drop(core); - - let dir_path = dir.path().to_path_buf(); - let (mut restored, _tx2, _rx2) = make_core_with_dir(&dir_path); - restored.load_kv_checkpoints().expect("load"); - let now = crate::engine::kv::current_ms(); - assert_eq!( - restored - .kv_engine - .get(DatabaseId::DEFAULT.as_u64(), TID, "cache", b"b", now), - Some(b"2".to_vec()), - "the KV row applied below the checkpoint is in the newest generation" - ); - } - - /// A record applied below a published checkpoint that cannot be published - /// again leaves the checkpoint's claim false for it. The work cannot be - /// rolled back, so the core fail-stops and the record's events stay unsent. - #[test] - fn a_failed_republish_fail_stops_the_core() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let (mut producers, mut consumers) = - crate::event::bus::create_event_bus_with_capacity(1, 64); - core.set_event_producer(producers.pop().expect("producer")); - core.floors.kv_published_lsn = Lsn::new(100); - // A file where the checkpoint directory belongs makes the publish fail. - let ckpt_dir = core - .data_dir - .join("kv-ckpt") - .join(format!("core-{}", core.core_id)); - std::fs::create_dir_all(ckpt_dir.parent().expect("parent")).expect("kv-ckpt dir"); - std::fs::write(&ckpt_dir, b"not a directory").expect("block the checkpoint dir"); - - let mut task = make_default_task(); - task.wal_lsn = Some(Lsn::new(50)); - let redo = RedoRecord { - version: 1, - ops: vec![kv_put("cache", b"b", b"2", 2)], - calvin_stamp: None, - } - .to_bytes() - .expect("encode redo"); - let response = core.execute_apply_transaction_redo( - &task, - TID, - CommittedRedo { - redo: &redo, - collections: &["cache".to_string()], - sum_targets: &[], - }, - ); - - assert_eq!(response.status, Status::Error); - assert!(core.fail_stop.is_stopped(), "the core fail-stops"); - assert!( - consumers[0].try_recv().is_none(), - "no event leaves for a record whose post-install work failed" - ); - } - - #[test] - fn a_columnar_record_applied_below_a_published_checkpoint_is_published_again() { - use nodedb_types::Value; - use nodedb_types::columnar::{ColumnDef, ColumnType, ColumnarImageWalRow, ColumnarSchema}; - - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let key = ( - DatabaseId::DEFAULT, - crate::types::TenantId::new(TID), - "m".to_string(), - ); - let schema = ColumnarSchema { - columns: vec![ - ColumnDef::required("id", ColumnType::Int64).with_primary_key(), - ColumnDef::required("v", ColumnType::Int64), - ], - version: 1, - }; - let mut engine = nodedb_columnar::MutationEngine::new("m".to_string(), schema); - engine - .insert_with_surrogate(&[Value::Integer(1), Value::Integer(10)], Surrogate::new(1)) - .expect("seed row"); - core.columnar_engines.insert(key.clone(), engine); - core.advance_watermark(Lsn::new(100)); - // The published stamp's prefix is what restart replay skips through. - core.floors - .applied_prefix - .observe_outcome_floor(Lsn::new(100)); - core.checkpoint_columnar_engines() - .expect("publish at lsn 100"); - - let mut image = std::collections::HashMap::new(); - image.insert("id".to_string(), Value::Integer(2)); - image.insert("v".to_string(), Value::Integer(20)); - let record = ColumnarImageWalRecord { - kind: COLUMNAR_IMAGE_KIND.to_string(), - collection: "m".to_string(), - schema_bytes: Vec::new(), - rows: vec![ColumnarImageWalRow { - surrogate: 2, - prior_pk_msgpack: Vec::new(), - image_msgpack: nodedb_types::value_to_msgpack(&Value::Object(image)) - .expect("encode image"), - }], - }; - let mut task = make_default_task(); - task.wal_lsn = Some(Lsn::new(50)); - let redo = RedoRecord { - version: 1, - ops: vec![RedoSubRecord { - record_type: RecordType::TimeseriesBatch as u32, - payload: zerompk::to_msgpack_vec(&record).expect("encode image record"), - }], - calvin_stamp: None, - } - .to_bytes() - .expect("encode redo"); - let response = core.execute_apply_transaction_redo( - &task, - TID, - CommittedRedo { - redo: &redo, - collections: &["m".to_string()], - sum_targets: &[], - }, - ); - assert_eq!(response.status, Status::Ok, "apply: {response:?}"); - drop(core); - - let dir_path = dir.path().to_path_buf(); - let (mut restored, _tx2, _rx2) = make_core_with_dir(&dir_path); - restored.load_columnar_checkpoints().expect("load"); - let mut ids: Vec = restored - .columnar_engines - .get(&key) - .expect("engine restored") - .scan_memtable_rows() - .map(|row| row[0].clone()) - .collect(); - ids.sort_by_key(|value| match value { - Value::Integer(i) => *i, - _ => i64::MAX, - }); - assert_eq!( - ids, - vec![Value::Integer(1), Value::Integer(2)], - "the columnar row applied below the checkpoint is in the newest generation" - ); - } -} diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs index b75fe007a..e1e4d718c 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs @@ -43,7 +43,6 @@ use crate::data::executor::task::ExecutionTask; use crate::types::TenantId; use crate::wal::RedoRecord; -use super::cover::WrittenEngines; use super::passes::{RedoTarget, final_refusal}; use super::state::{RedoApplyPass, RedoApplyScope}; use super::sub_ops::{document_ops, kv_ops, label_ops}; @@ -68,7 +67,7 @@ impl CoreLoop { } /// Install one committed redo record at the LSN the request carries: - /// validate every sub-record, install with undo, then settle and cover. + /// validate every sub-record, install with undo, then settle. /// Every committed transaction installs here, whichever path committed it, /// and restart replay drives the same arms over the same record. pub(in crate::data::executor) fn install_committed_redo( @@ -134,17 +133,18 @@ impl CoreLoop { Ok(scope) => scope, Err(refusal) => return self.response_error(task, refusal.into_code()), }; - // The install applied the record: a flush the settle runs, and a - // checkpoint the cover writes, name it. + // The install applied the record: every flush the settle runs, and + // every later checkpoint, names it. No artifact written before this + // point names it: its stamp prefix is an outcome floor the record was + // above, since the record's dispatcher window held the floor below it + // until this apply answers, and its applied ranges name only records + // applied before. So restart replay applies it, and nothing written + // earlier needs publishing again. self.floors.applied_prefix.note_applied(lsn); - // Settle, then publish again every artifact whose stamp prefix covers - // the record, so restart replay does not skip it. Neither step can be - // rolled back once it started, so a failure leaves live state restart - // replay does not rebuild: the core fail-stops. The funnel keeps the - // record for restart replay. - let settled = self.settle_redo_install(task, &mut scope).and_then(|()| { - self.cover_applied_record(lsn, &WrittenEngines::of(&redo), &scope.arrays_written) - }); + // The settle cannot be rolled back once it started, so a failure + // leaves live state restart replay does not rebuild: the core + // fail-stops. The funnel keeps the record for restart replay. + let settled = self.settle_redo_install(task, &mut scope); if let Err(error) = settled { self.fail_stop_core( FailStopCause::PostInstallFailed, @@ -156,7 +156,7 @@ impl CoreLoop { return self.response_error(task, error); } // Write versions and the watermark move only once the record is - // settled and covered. + // settled. for version in std::mem::take(&mut scope.write_versions) { self.publish_write_version( version.db, @@ -166,7 +166,7 @@ impl CoreLoop { version.lsn, ); } - // Events leave only once the record is settled and covered. Every + // Events leave only once the record is settled. Every // one names the record's LSN: the install held the watermark back. for mut event in std::mem::take(&mut scope.pending_events) { event.lsn = lsn; @@ -379,82 +379,158 @@ mod tests { } } - /// Restart replay skips a timeseries record the collection stamp names. - /// A committed redo applied online after a flush whose stamp prefix - /// passed its LSN must still install, and must flush so the stamp's - /// claim holds for it on the next restart. + /// A committed KV record applied after a checkpoint that named a higher + /// LSN is not named by that checkpoint: nothing publishes it again, and a + /// restart replays it over the restored generation. #[test] - fn an_online_redo_apply_below_a_flushed_partition_stamp_still_installs() { + fn a_kv_redo_applied_after_a_checkpoint_replays_after_a_restart() { let dir = tempfile::tempdir().expect("tempdir"); + let redo_record = |lsn: u64, redo: &[u8]| { + WalRecord::new(WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn, + tenant_id: TID, + vshard_id: 0, + database_id: 0, + payload: redo.to_vec(), + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record") + }; + let later = redo_bytes(vec![kv_put("cache", b"k2", b"v2", 22)]); + let earlier = redo_bytes(vec![kv_put("cache", b"k1", b"v1", 21)]); + let collections = vec!["cache".to_string()]; + let install = |core: &mut CoreLoop, lsn: u64, redo: &[u8]| { + let response = core.execute_apply_transaction_redo( + &task_at(Some(lsn)), + TID, + CommittedRedo { + redo, + collections: &collections, + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + }; + + { + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(10)); + install(&mut core, 100, &later); + core.checkpoint_kv_engines().expect("checkpoint"); + install(&mut core, 50, &earlier); + } + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + core.load_kv_checkpoints().expect("load"); + assert!( + !core.floors.replay_floors.kv.covers(50), + "the checkpoint written before lsn 50 applied does not name it" + ); + core.floors + .applied_prefix + .seed_replayed_through(Lsn::new(100)); + core.replay_transaction_redo_wal( + &[redo_record(50, &earlier), redo_record(100, &later)], + 1, + &nodedb_wal::TombstoneSet::new(), + ) + .expect("replay"); + let now_ms = crate::engine::kv::current_ms(); + for (key, value) in [(&b"k1"[..], &b"v1"[..]), (&b"k2"[..], &b"v2"[..])] { + assert_eq!( + core.kv_engine.get(0, TID, "cache", key, now_ms).as_deref(), + Some(value), + "{key:?} is present after the restart" + ); + } + } + + /// A committed redo record applied after a flush that named a higher LSN + /// is not named by that flush's stamp: its dispatcher window held the + /// outcome floor below it. The flush needs no republish, and a restart + /// replays the record once, beside the flushed one it does not repeat. + #[test] + fn a_redo_applied_below_a_flushed_record_replays_once_after_a_restart() { + let dir = tempfile::tempdir().expect("tempdir"); let tenant = TenantId::new(TID); let key = ( crate::types::DatabaseId::DEFAULT, tenant, "metrics".to_string(), ); - - // A write at LSN 100 lands, the outcome floor passes it, and a flush - // stamps through LSN 100. - let earlier = RedoRecord { - version: 1, - ops: vec![metric_samples_sub("metrics", 1_700_000_000_000, 1.0)], - calvin_stamp: None, + let redo_record = |lsn: u64, redo: &[u8]| { + WalRecord::new(WalRecordArgs { + record_type: RecordType::TransactionRedo as u32, + lsn, + tenant_id: TID, + vshard_id: 0, + database_id: 0, + payload: redo.to_vec(), + encryption_key: None, + preamble_bytes: None, + }) + .expect("wal record") }; - let record = WalRecord::new(WalRecordArgs { - record_type: RecordType::TransactionRedo as u32, - lsn: 100, - tenant_id: TID, - vshard_id: 0, - database_id: 0, - payload: earlier.to_bytes().expect("encode redo"), - encryption_key: None, - preamble_bytes: None, - }) - .expect("wal record"); + let rows = |core: &CoreLoop| { + core.columnar_memtables + .get(&key) + .map_or(0, |m| m.row_count()) + + core + .ts_registries + .get(&key) + .map_or(0, |registry| registry.total_row_count()) + }; + let later = redo_bytes(vec![metric_samples_sub("metrics", 1_700_000_000_000, 1.0)]); + let earlier = redo_bytes(vec![metric_samples_sub("metrics", 1_700_000_000_001, 2.0)]); + let collections = vec!["metrics".to_string()]; + let install = |core: &mut CoreLoop, lsn: u64, redo: &[u8]| { + let response = core.execute_apply_transaction_redo( + &task_at(Some(lsn)), + TID, + CommittedRedo { + redo, + collections: &collections, + sum_targets: &[], + }, + ); + assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); + }; + + { + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + core.floors + .applied_prefix + .observe_outcome_floor(crate::types::Lsn::new(10)); + // The record at 100 applies first and flushes. + install(&mut core, 100, &later); + core.flush_ts_collection(tenant, crate::types::DatabaseId::DEFAULT, "metrics", 0) + .expect("flush"); + let stamp = &core.ts_replay_stamps.get(&key).expect("stamp").rows; + assert!(stamp.skips(100) && !stamp.skips(50), "{stamp:?}"); + // The record at 50 applies afterwards and stays in the memtable. + install(&mut core, 50, &earlier); + assert_eq!(rows(&core), 2); + } + + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + core.load_ts_registries().expect("load"); + core.floors + .applied_prefix + .seed_replayed_through(crate::types::Lsn::new(100)); core.replay_transaction_redo_wal( - std::slice::from_ref(&record), + &[redo_record(50, &earlier), redo_record(100, &later)], 1, &nodedb_wal::TombstoneSet::new(), ) - .expect("seed replay"); - core.floors - .applied_prefix - .observe_outcome_floor(crate::types::Lsn::new(100)); - core.flush_ts_collection(tenant, crate::types::DatabaseId::DEFAULT, "metrics", 0) - .expect("flush the seeded partition"); - - // A committed redo minted at LSN 50 applies afterwards. - let redo = redo_bytes(vec![metric_samples_sub("metrics", 1_700_000_000_001, 2.0)]); - let collections = vec!["metrics".to_string()]; - let response = core.execute_apply_transaction_redo( - &task_at(Some(50)), - TID, - CommittedRedo { - redo: &redo, - collections: &collections, - sum_targets: &[], - }, - ); - assert_eq!(response.status, Status::Ok, "{:?}", response.error_code); - - let memtable_rows = core - .columnar_memtables - .get(&key) - .map_or(0, |m| m.row_count()); - let flushed_rows = core - .ts_registries - .get(&key) - .map_or(0, |registry| registry.total_row_count()); + .expect("replay"); assert_eq!( - memtable_rows + flushed_rows, + rows(&core), 2, - "the online apply installs its sample despite the higher partition stamp" - ); - assert_eq!( - memtable_rows, 0, - "the sample applied below the stamp is flushed, so a restart that skips its \ - record still finds it on disk" + "the record at 50 replays once; the flushed record at 100 does not repeat" ); } diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs index 59c80d1a1..77877a411 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/mod.rs @@ -4,7 +4,6 @@ //! the core that owns its vShard, through the WAL replay arms. //! //! - [`entry`]: the `MetaOp::ApplyTransactionRedo` handler. -//! - [`cover`]: republishing an engine artifact a record applied below. //! - [`validate`]: commit-boundary checks that run before any write. //! - [`passes`]: the validate and install passes over the replay arms. //! - [`settle`]: the work an install defers until every sub-record landed. @@ -20,7 +19,6 @@ #[cfg(test)] mod calvin_fold_tests; -mod cover; mod document; mod entry; mod events; diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs index 6f304c403..c22708f4e 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs @@ -6,7 +6,7 @@ //! the undo would restore in memory, a vector seal moves inserted nodes out of //! the growing segment, and a truncate that removes files cannot be renamed //! back. Once the install succeeded, this runs them in one step. The record's -//! events wait until this step and the cover step both succeeded. +//! events wait until this step succeeded. //! //! A failure here comes after every sub-record landed. The undo cannot //! reverse a partial flush, and a flush failure is an I/O failure no validate @@ -30,21 +30,27 @@ impl CoreLoop { } } - /// Record that the install ingested rows into the timeseries collection - /// `key` at `lsn`. - pub(in crate::data::executor) fn note_redo_timeseries_written( + /// Record that the open redo-apply scope wrote cells to `array_id`, so + /// the settle runs its threshold flush. + pub(in crate::data::executor) fn note_redo_array_written( &mut self, - key: CollectionKey, - lsn: u64, + array_id: &nodedb_array::types::ArrayId, ) { + if let Some(scope) = self.redo_apply.scope.as_mut() + && !scope.arrays_written.contains(array_id) + { + scope.arrays_written.push(array_id.clone()); + } + } + + /// Record that the install ingested rows into the timeseries collection + /// `key`. + pub(in crate::data::executor) fn note_redo_timeseries_written(&mut self, key: CollectionKey) { if let Some(scope) = self.redo_apply.scope.as_mut() && scope.pass == RedoApplyPass::Install - && !scope - .timeseries_written - .iter() - .any(|(seen, _)| *seen == key) + && !scope.timeseries_written.contains(&key) { - scope.timeseries_written.push((key, lsn)); + scope.timeseries_written.push(key); } } @@ -69,8 +75,8 @@ impl CoreLoop { ) })?; } - for (key, lsn) in std::mem::take(&mut scope.timeseries_written) { - self.settle_redo_timeseries(key, lsn)?; + for key in std::mem::take(&mut scope.timeseries_written) { + self.settle_redo_timeseries(key)?; } for array_id in &scope.arrays_written { self.flush_array_if_full(array_id) @@ -108,13 +114,9 @@ impl CoreLoop { /// Settle one timeseries collection the install ingested into: charge /// the memory budget for its memtable, and flush it when it is over its - /// soft limit or when the collection stamp already names the record. - /// - /// Restart replay skips every record the collection's replay stamp - /// names. A record applied after a stamp's prefix passed its LSN sits - /// only in the memtable, so the claim is false for it until the memtable - /// flushes too, with a stamp that names it. - fn settle_redo_timeseries(&mut self, key: CollectionKey, lsn: u64) -> Result<(), ErrorCode> { + /// soft limit. The install noted the record before this runs, so the + /// flush's stamp names the whole record (`group_flush`). + fn settle_redo_timeseries(&mut self, key: CollectionKey) -> Result<(), ErrorCode> { let (database_id, tid, collection) = key; self.recharge_ts_memtable_budget(tid, database_id, &collection); self.checkpoint_coordinator.mark_dirty("timeseries", 1); @@ -125,8 +127,7 @@ impl CoreLoop { .columnar_memtables .get(&memtable_key) .is_some_and(|mt| mt.memory_bytes() >= self.ts_tuning.memtable_budget_bytes); - let below_stamp = self.ts_rows_named(&memtable_key, lsn); - if !over_budget && !below_stamp { + if !over_budget { return Ok(()); } let now_ms = self.ingest_now_ms(); diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs index 10c6c19c8..88868c8fc 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs @@ -129,9 +129,9 @@ pub(in crate::data::executor) struct RedoApplyScope { pub(in crate::data::executor) arrays_written: Vec, /// Columnar collections the install wrote, flushed once it succeeded. pub(in crate::data::executor) columnar_written: Vec, - /// Timeseries collections the install ingested into, with the record's - /// LSN, settled once it succeeded. - pub(in crate::data::executor) timeseries_written: Vec<(CollectionKey, u64)>, + /// Timeseries collections the install ingested into, settled once it + /// succeeded. + pub(in crate::data::executor) timeseries_written: Vec, /// Events the install's writes raised, sent once it succeeded. pub(in crate::data::executor) pending_events: Vec, /// Write versions the install's writes produced, published once the diff --git a/nodedb/src/data/executor/kv_checkpoint/format.rs b/nodedb/src/data/executor/kv_checkpoint/format.rs index 8e4c4894f..38ec0307f 100644 --- a/nodedb/src/data/executor/kv_checkpoint/format.rs +++ b/nodedb/src/data/executor/kv_checkpoint/format.rs @@ -14,7 +14,7 @@ use super::index_format::KvCheckpointIndexes; /// A file stamped with any other version is refused rather than misparsed. /// Refusing costs a WAL replay; misparsing would install wrong rows AND a floor /// that suppresses the records which would have corrected them. -pub(crate) const KV_CKPT_FORMAT_VERSION: u16 = 3; +pub(crate) const KV_CKPT_FORMAT_VERSION: u16 = 4; /// Names the live generation. Writing this file is what publishes a checkpoint. #[derive( @@ -32,10 +32,8 @@ pub(crate) struct KvCheckpointManifest { pub format_version: u16, /// Which `gen-{n}/` directory holds the live collection files. pub generation: u64, - /// The LSN this core reports as the KV engine's truncation floor when - /// the generation is restored. It gates no replay: [`Self::replay`] does. - pub durable_through_lsn: u64, - /// The records the generation holds. WAL replay skips exactly the KV + /// The records the generation holds. Its prefix is also the LSN this core + /// reports as the KV engine's truncation floor once it is restored. WAL replay skips exactly the KV /// records [`ReplayStamp::skips`] names and replays every other one. /// /// A single highest-applied LSN cannot state this. LSNs are node-global and diff --git a/nodedb/src/data/executor/kv_checkpoint/load.rs b/nodedb/src/data/executor/kv_checkpoint/load.rs index d235224bc..242c75731 100644 --- a/nodedb/src/data/executor/kv_checkpoint/load.rs +++ b/nodedb/src/data/executor/kv_checkpoint/load.rs @@ -13,7 +13,6 @@ use super::paths::{kv_ckpt_dir, kv_ckpt_gen_dir, parse_kv_ckpt_stem}; use crate::data::executor::checkpoint_decode_error::CheckpointDecodeError; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::snapshot::restore::database_id_from_qualified; -use crate::types::Lsn; impl CoreLoop { /// Load the KV checkpoint from disk on startup, BEFORE WAL replay. @@ -69,8 +68,8 @@ impl CoreLoop { // Claimed only once every row AND every registration is in: the floor // suppresses WAL records, so claiming it over a half-restored generation // would turn a recoverable read failure into permanent data loss. - self.floors.kv_published_lsn = Lsn::new(manifest.replay.prefix); let replay_prefix = manifest.replay.prefix; + self.floors.kv_durable_lsn = crate::types::Lsn::new(replay_prefix); let applied_ranges = manifest.replay.applied_above.len(); self.floors.replay_floors.kv.set(manifest.replay); @@ -80,7 +79,6 @@ impl CoreLoop { collections, rows, indexes, - durable_through_lsn = manifest.durable_through_lsn, replay_prefix, applied_ranges, "KV checkpoint restored" @@ -384,7 +382,7 @@ mod tests { /// The manifest is the only record of what a generation holds, and the /// entire replay floor rests on it: it must survive the round-trip exactly. #[test] - fn manifest_roundtrips_generation_lsn_and_stamp() { + fn manifest_roundtrips_generation_and_stamp() { let replay = crate::data::executor::applied_prefix::ReplayStamp { prefix: 4_200, applied_above: vec![crate::data::executor::applied_prefix::stamp::LsnRange { @@ -395,7 +393,6 @@ mod tests { let written = KvCheckpointManifest { format_version: KV_CKPT_FORMAT_VERSION, generation: 9, - durable_through_lsn: 4_242, replay: replay.clone(), }; let tmp = tempfile::tempdir().expect("tempdir"); @@ -406,10 +403,6 @@ mod tests { let read_back = nodedb_wal::segment::read_checkpoint_framed(&path).expect("read"); let decoded: KvCheckpointManifest = zerompk::from_msgpack(&read_back).expect("decode"); - assert_eq!( - decoded.durable_through_lsn, 4_242, - "the manifest must report exactly the LSN it was written with" - ); assert_eq!(decoded.generation, 9); assert_eq!(decoded.format_version, KV_CKPT_FORMAT_VERSION); assert_eq!(decoded.replay, replay, "the stamp must survive exactly"); @@ -510,7 +503,6 @@ mod tests { let manifest = KvCheckpointManifest { format_version: KV_CKPT_FORMAT_VERSION, generation: 0, - durable_through_lsn: 10, replay: crate::data::executor::applied_prefix::ReplayStamp { prefix: 10, applied_above: vec![crate::data::executor::applied_prefix::stamp::LsnRange { diff --git a/nodedb/src/data/executor/kv_checkpoint/write.rs b/nodedb/src/data/executor/kv_checkpoint/write.rs index 2280134bb..3f97fc551 100644 --- a/nodedb/src/data/executor/kv_checkpoint/write.rs +++ b/nodedb/src/data/executor/kv_checkpoint/write.rs @@ -43,11 +43,8 @@ impl CoreLoop { /// already restored is a no-op (`add_index` reports the field as already /// indexed and skips the backfill), and replaying a drop whose registration /// the export therefore never saw is a no-op too. - /// - /// Every published generation raises `kv_published_lsn` to the stamp's - /// prefix (see `redo_apply::cover`). pub(in crate::data::executor) fn checkpoint_kv_engines(&mut self) -> crate::Result { - let durable_through = self.watermark; + let durable_through = self.checkpoint_floor(); let replay = self.floors.applied_prefix.stamp()?; let ckpt_dir = kv_ckpt_dir(&self.data_dir, self.core_id); @@ -72,8 +69,7 @@ impl CoreLoop { let written = self.write_kv_generation(&gen_dir)?; let prefix = Lsn::new(replay.prefix); let applied_ranges = replay.applied_above.len(); - self.publish_kv_generation(&ckpt_dir, generation, durable_through, replay)?; - self.floors.kv_published_lsn = self.floors.kv_published_lsn.max(prefix); + self.publish_kv_generation(&ckpt_dir, generation, replay)?; // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what @@ -174,13 +170,11 @@ impl CoreLoop { &self, ckpt_dir: &std::path::Path, generation: u64, - durable_through: Lsn, replay: ReplayStamp, ) -> crate::Result<()> { let manifest = KvCheckpointManifest { format_version: KV_CKPT_FORMAT_VERSION, generation, - durable_through_lsn: durable_through.as_u64(), replay, }; let bytes = diff --git a/nodedb/src/data/executor/snapshot.rs b/nodedb/src/data/executor/snapshot.rs index 2f98bb1fe..dce50850b 100644 --- a/nodedb/src/data/executor/snapshot.rs +++ b/nodedb/src/data/executor/snapshot.rs @@ -71,7 +71,7 @@ impl CoreLoop { } Ok(CoreSnapshot { - watermark: self.watermark.as_u64(), + watermark: self.checkpoint_floor().as_u64(), sparse_documents, sparse_indexes, edges, diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/format.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/format.rs index 360e339ce..abb3c840f 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/format.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/format.rs @@ -18,7 +18,7 @@ use crate::types::replay_stamp::ReplayStamp; /// A manifest stamped with any other version is refused rather than misparsed. /// Refusing costs a WAL replay; misparsing would install indexes built from /// bytes this build cannot read. -pub(crate) const SPARSE_VECTOR_CKPT_FORMAT_VERSION: u16 = 2; +pub(crate) const SPARSE_VECTOR_CKPT_FORMAT_VERSION: u16 = 3; /// Names the live generation. Writing this file is what publishes a checkpoint. #[derive( @@ -37,16 +37,10 @@ pub(crate) struct SparseVectorCheckpointManifest { pub format_version: u16, /// Which `gen-{n}/` directory holds the live per-index files. pub generation: u64, - /// The LSN every index in that generation is durable THROUGH (inclusive). - /// - /// This is the value `execute_checkpoint` folds into the minimum it reports - /// to the checkpoint manager, and the value a restart restores - /// `sparse_vector_durable_lsn` from — without it, the first flush after a - /// restart would have no last-known-durable point to clamp to and would - /// pin truncation at zero. - pub durable_through_lsn: u64, /// The records every index in the generation holds. Restart replay skips - /// a sparse-vector record exactly when this stamp names it. + /// a sparse-vector record exactly when this stamp names it. Its prefix + /// restores `sparse_vector_durable_lsn`, the point a failed flush after a + /// restart clamps WAL truncation to. pub replay: ReplayStamp, } @@ -61,7 +55,6 @@ pub(crate) fn test_manifest_bytes(generation: u64) -> Vec { zerompk::to_msgpack_vec(&SparseVectorCheckpointManifest { format_version: SPARSE_VECTOR_CKPT_FORMAT_VERSION, generation, - durable_through_lsn: 0, replay: ReplayStamp::default(), }) .expect("manifest encode is infallible for this fixed struct") diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/load.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/load.rs index 6f529ca67..c8b65c2fa 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/load.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/load.rs @@ -71,10 +71,9 @@ impl CoreLoop { // Claimed only once every index is in: this LSN is what a failed flush // clamps truncation to, so claiming it over a half-restored generation // would authorise deleting the records that would have completed it. - self.floors.sparse_vector_durable_lsn = Lsn::new(manifest.durable_through_lsn); + self.floors.sparse_vector_durable_lsn = Lsn::new(manifest.replay.prefix); let replay_prefix = manifest.replay.prefix; let applied_ranges = manifest.replay.applied_above.len(); - self.floors.sparse_vector_published_lsn = Lsn::new(replay_prefix); self.floors.replay_floors.sparse_vector.set(manifest.replay); info!( @@ -82,7 +81,6 @@ impl CoreLoop { generation = manifest.generation, indexes, docs, - durable_through_lsn = manifest.durable_through_lsn, replay_prefix, applied_ranges, "sparse vector checkpoint restored" @@ -209,12 +207,11 @@ mod tests { /// through, and both the reported checkpoint LSN and the post-restart clamp /// rest on it: it must survive the round-trip exactly. #[test] - fn manifest_roundtrips_generation_and_lsn() { + fn manifest_roundtrips_generation_and_stamp() { let written = SparseVectorCheckpointManifest { format_version: SPARSE_VECTOR_CKPT_FORMAT_VERSION, generation: 4, - durable_through_lsn: 8_128, - replay: crate::types::replay_stamp::ReplayStamp::default(), + replay: crate::types::replay_stamp::ReplayStamp::through(8_128), }; let tmp = tempfile::tempdir().expect("tempdir"); let bytes = zerompk::to_msgpack_vec(&written).expect("encode"); @@ -229,8 +226,9 @@ mod tests { .expect("manifest must read") .expect("manifest file exists, so this must be Some"); assert_eq!( - decoded.durable_through_lsn, 8_128, - "the manifest must report exactly the LSN it was written with" + decoded.replay, + crate::types::replay_stamp::ReplayStamp::through(8_128), + "the manifest must report exactly the stamp it was written with" ); assert_eq!(decoded.generation, 4); } @@ -246,7 +244,6 @@ mod tests { let written = SparseVectorCheckpointManifest { format_version: SPARSE_VECTOR_CKPT_FORMAT_VERSION + 1, generation: 1, - durable_through_lsn: 5, replay: crate::types::replay_stamp::ReplayStamp::default(), }; let tmp = tempfile::tempdir().expect("tempdir"); @@ -351,6 +348,10 @@ mod tests { index_with(&[("doc-c", &[(3, 1.0)])]), ); before.advance_watermark(Lsn::new(1_234)); + before + .floors + .applied_prefix + .observe_outcome_floor(Lsn::new(1_234)); let reported = before .checkpoint_sparse_vector_indexes() diff --git a/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs b/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs index cc00db43d..d2b61e59a 100644 --- a/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs +++ b/nodedb/src/data/executor/sparse_vector_checkpoint/write.rs @@ -19,7 +19,7 @@ impl CoreLoop { /// Flush every sparse-vector index on this core to disk and return the LSN /// the sparse-vector engine is now durable through. /// - /// Returns `Ok(watermark)` only once a manifest naming a COMPLETE generation + /// Returns `Ok(floor)` only once a manifest naming a COMPLETE generation /// has landed. Any failure returns `Err` — the caller must then clamp the /// reported checkpoint LSN to the last LSN this engine was known durable /// through, so a failed flush costs WAL growth instead of data. @@ -31,7 +31,7 @@ impl CoreLoop { pub(in crate::data::executor) fn checkpoint_sparse_vector_indexes( &mut self, ) -> crate::Result { - let durable_through = self.watermark; + let durable_through = self.checkpoint_floor(); let replay = self.floors.applied_prefix.stamp()?; let prefix = replay.prefix; let applied_ranges = replay.applied_above.len(); @@ -56,11 +56,7 @@ impl CoreLoop { .map_err(|e| storage_err(&gen_dir, "create generation dir", &e))?; let written = self.write_sparse_vector_generation(&gen_dir)?; - self.publish_sparse_vector_generation(&ckpt_dir, generation, durable_through, replay)?; - self.floors.sparse_vector_published_lsn = self - .floors - .sparse_vector_published_lsn - .max(Lsn::new(prefix)); + self.publish_sparse_vector_generation(&ckpt_dir, generation, replay)?; // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what @@ -134,13 +130,11 @@ impl CoreLoop { &self, ckpt_dir: &std::path::Path, generation: u64, - durable_through: Lsn, replay: ReplayStamp, ) -> crate::Result<()> { let manifest = SparseVectorCheckpointManifest { format_version: SPARSE_VECTOR_CKPT_FORMAT_VERSION, generation, - durable_through_lsn: durable_through.as_u64(), replay, }; let bytes = diff --git a/nodedb/src/data/executor/spatial_checkpoint/write.rs b/nodedb/src/data/executor/spatial_checkpoint/write.rs index b861068d8..f68a6cc4a 100644 --- a/nodedb/src/data/executor/spatial_checkpoint/write.rs +++ b/nodedb/src/data/executor/spatial_checkpoint/write.rs @@ -46,11 +46,12 @@ impl CoreLoop { /// drop geometry entries while the rows they point at survive — a spatial /// predicate silently stops matching rows a full scan still returns. /// - /// Stamping with the core watermark rests on this: the checkpoint runs - /// on the core's own thread between tasks, and a geometry write raises - /// the watermark only after the R-tree has already been mutated. + /// The reported LSN is the checkpoint floor (`checkpoint_floor`): the + /// checkpoint runs on the core's own thread between tasks, so every record + /// at or below that outcome floor that this core applied is in the export + /// below. pub(crate) fn checkpoint_spatial_indexes(&self) -> crate::Result { - let durable_lsn = self.watermark; + let durable_lsn = self.checkpoint_floor(); let ckpt_dir = spatial_ckpt_dir(&self.data_dir, self.core_id); std::fs::create_dir_all(&ckpt_dir).map_err(|e| storage_err(&ckpt_dir, "create dir", &e))?; @@ -274,7 +275,7 @@ mod tests { /// An index that has disappeared from the map — dropped, or evicted with /// its collection — must not restore the previous generation's geometry. - /// The flush still reports the watermark, so the WAL records that built + /// The flush still reports its floor, so the WAL records that built /// those entries are already deletable; leaving the old file reachable /// would resurrect geometry for rows that no longer exist. #[test] diff --git a/nodedb/src/data/executor/sync_hwm_checkpoint/load.rs b/nodedb/src/data/executor/sync_hwm_checkpoint/load.rs index f5849c2eb..3e97e5925 100644 --- a/nodedb/src/data/executor/sync_hwm_checkpoint/load.rs +++ b/nodedb/src/data/executor/sync_hwm_checkpoint/load.rs @@ -136,6 +136,10 @@ mod tests { assert_eq!(before.sync_admit(&other_stream), SyncAdmit::Apply); before.sync_commit(&other_stream); before.advance_watermark(Lsn::new(900)); + before + .floors + .applied_prefix + .observe_outcome_floor(Lsn::new(900)); let reported = before .checkpoint_sync_hwm() diff --git a/nodedb/src/data/executor/sync_hwm_checkpoint/write.rs b/nodedb/src/data/executor/sync_hwm_checkpoint/write.rs index aabe52f2e..5c47cdbcf 100644 --- a/nodedb/src/data/executor/sync_hwm_checkpoint/write.rs +++ b/nodedb/src/data/executor/sync_hwm_checkpoint/write.rs @@ -14,7 +14,7 @@ impl CoreLoop { /// Flush this core's sync gate to disk and return the LSN it is now durable /// through. /// - /// Returns `Ok(watermark)` only once the state file has landed and been + /// Returns `Ok(floor)` only once the state file has landed and been /// fsynced. Any failure returns `Err` — the caller must then clamp the /// reported checkpoint LSN to the last LSN the gate was known durable /// through, so a failed flush costs WAL growth instead of a gate that comes @@ -24,12 +24,12 @@ impl CoreLoop { /// is intact and live, after it the new one is. There is no window in which /// half a gate is published. /// - /// Stamping with the core watermark rests on this: the checkpoint runs - /// on the core's own thread between tasks, and `sync_commit` advances - /// the HWM only after the frame's `SyncSeqAdvance` record is durable, so - /// every advance the core has admitted is already in the maps exported here. + /// The reported LSN is the checkpoint floor (`checkpoint_floor`): the + /// checkpoint runs on the core's own thread between tasks, so every record + /// at or below that outcome floor that this core applied is in the export + /// below. pub(in crate::data::executor) fn checkpoint_sync_hwm(&self) -> crate::Result { - let durable_through = self.watermark; + let durable_through = self.checkpoint_floor(); // Sorted so identical gate state always encodes to identical bytes. let mut hwm: Vec<(u64, u64, u64)> = self diff --git a/nodedb/src/data/executor/timeseries_checkpoint/flush.rs b/nodedb/src/data/executor/timeseries_checkpoint/flush.rs index afc4ee750..5c6ae485f 100644 --- a/nodedb/src/data/executor/timeseries_checkpoint/flush.rs +++ b/nodedb/src/data/executor/timeseries_checkpoint/flush.rs @@ -11,7 +11,7 @@ impl CoreLoop { /// Flush every timeseries memtable on this core to an L1 partition and /// return the LSN the timeseries engine is now durable through. /// - /// Returns `Ok(watermark)` only once every collection's partition — columns, + /// Returns `Ok(floor)` only once every collection's partition — columns, /// symbol dictionaries, sparse index, and the `partition.meta` that commits /// them — has landed. Any failure returns `Err`; the caller must then clamp /// the reported checkpoint LSN to the last LSN timeseries was known durable @@ -31,7 +31,7 @@ impl CoreLoop { pub(in crate::data::executor) fn checkpoint_timeseries_memtables( &mut self, ) -> crate::Result { - let durable_through = self.watermark; + let durable_through = self.checkpoint_floor(); // Collected first: `flush_ts_collection` takes `&mut self`, so the // memtable-map iterator cannot stay borrowed across the loop. Empty @@ -265,6 +265,7 @@ mod tests { vec!["a".to_string(), "b".to_string()], "both rows must be live in the memtable before any flush" ); + observe_floor(&mut before, 20); let reported = before .core @@ -291,14 +292,14 @@ mod tests { ); } - /// A core with no timeseries memtables reports the watermark rather than + /// A core with no timeseries memtables reports the floor rather than /// clamping — it holds no timeseries state at all, so it can never be the /// reason the WAL must be kept. #[test] - fn no_memtables_reports_the_watermark() { + fn no_memtables_reports_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = Core::open_at(dir.path()); - core.core.advance_watermark(Lsn::new(42)); + observe_floor(&mut core, 42); assert_eq!( core.core .checkpoint_timeseries_memtables() @@ -307,11 +308,11 @@ mod tests { ); } - /// A flush with an empty memtable must still report the watermark: every row + /// A flush with an empty memtable must still report the floor: every row /// it held is already in a partition, so clamping there would pin WAL /// truncation for no reason. #[test] - fn empty_memtable_reports_the_watermark() { + fn empty_memtable_reports_the_floor() { let dir = tempfile::tempdir().expect("tempdir"); let mut core = Core::open_at(dir.path()); core.ingest("a", 1.0, 1_000, 5); @@ -319,14 +320,14 @@ mod tests { .checkpoint_timeseries_memtables() .expect("first flush"); - core.core.advance_watermark(Lsn::new(900)); + observe_floor(&mut core, 900); assert_eq!( core.core .checkpoint_timeseries_memtables() .expect("second flush"), Lsn::new(900), "nothing was ingested since the last flush, so the timeseries engine \ - is durable through the current watermark" + is durable through the current floor" ); } diff --git a/nodedb/src/data/executor/timeseries_checkpoint/mod.rs b/nodedb/src/data/executor/timeseries_checkpoint/mod.rs index da418cfb3..bf886ff4b 100644 --- a/nodedb/src/data/executor/timeseries_checkpoint/mod.rs +++ b/nodedb/src/data/executor/timeseries_checkpoint/mod.rs @@ -35,11 +35,12 @@ //! //! ## What LSN is durable after a flush //! -//! The core watermark. A timeseries row reaches the memtable before its ingest -//! returns, and only then does `note_collection_write_lsn` raise the watermark, -//! so on this core's own thread — where the checkpoint runs, between tasks — -//! every row with `lsn <= watermark` is in a memtable. Flushing every non-empty -//! memtable therefore puts all of them in a partition. +//! The checkpoint floor (`checkpoint_floor`): the outcome floor this core +//! read. Every record at or below it has a final outcome, so on this core's +//! own thread — where the checkpoint runs, between tasks — each one this core +//! applied is in a memtable or a partition. Flushing every non-empty memtable +//! therefore puts all of them in a partition. A record above the floor can +//! still be on its way, so the WAL keeps it. //! //! ## What restart replay skips //! diff --git a/nodedb/src/data/executor/vector_checkpoint/format.rs b/nodedb/src/data/executor/vector_checkpoint/format.rs index bfd3649f3..59284e9b6 100644 --- a/nodedb/src/data/executor/vector_checkpoint/format.rs +++ b/nodedb/src/data/executor/vector_checkpoint/format.rs @@ -19,7 +19,7 @@ use crate::types::replay_stamp::ReplayStamp; /// A manifest stamped with any other version is refused rather than misparsed. /// Refusing costs a WAL replay; misparsing would install indexes built from /// bytes this build cannot read. -pub(crate) const VECTOR_CKPT_FORMAT_VERSION: u16 = 2; +pub(crate) const VECTOR_CKPT_FORMAT_VERSION: u16 = 3; /// Names the live generation. Writing this file is what publishes a checkpoint. #[derive( @@ -37,15 +37,10 @@ pub(crate) struct VectorCheckpointManifest { pub format_version: u16, /// Which `gen-{n}/` directory holds the live per-index files. pub generation: u64, - /// The LSN every index in that generation is durable THROUGH (inclusive). - /// - /// Recorded so a restart knows what the PREVIOUS process actually made - /// durable: it restores `vector_durable_lsn`, which is the point a failed - /// flush clamps WAL truncation to. Without it the first failed flush after - /// a restart would pin truncation at zero. - pub durable_through_lsn: u64, /// The records every index in the generation holds. Restart replay skips - /// a vector record exactly when this stamp names it. + /// a vector record exactly when this stamp names it. Its prefix restores + /// `vector_durable_lsn`, the point a failed flush after a restart clamps + /// WAL truncation to. pub replay: ReplayStamp, } @@ -60,7 +55,6 @@ pub(crate) fn test_manifest_bytes(generation: u64) -> Vec { zerompk::to_msgpack_vec(&VectorCheckpointManifest { format_version: VECTOR_CKPT_FORMAT_VERSION, generation, - durable_through_lsn: 0, replay: ReplayStamp::default(), }) .expect("manifest encode is infallible for this fixed struct") diff --git a/nodedb/src/data/executor/vector_checkpoint/load.rs b/nodedb/src/data/executor/vector_checkpoint/load.rs index 07f2f84ef..58cc4c97f 100644 --- a/nodedb/src/data/executor/vector_checkpoint/load.rs +++ b/nodedb/src/data/executor/vector_checkpoint/load.rs @@ -67,8 +67,7 @@ impl CoreLoop { // Claimed only once every index is in: this LSN is what a failed flush // clamps truncation to, so claiming it over a half-restored generation // would authorise deleting the records that would have completed it. - self.floors.vector_durable_lsn = Lsn::new(manifest.durable_through_lsn); - self.floors.vector_published_lsn = Lsn::new(manifest.replay.prefix); + self.floors.vector_durable_lsn = Lsn::new(manifest.replay.prefix); let replay_prefix = manifest.replay.prefix; let applied_ranges = manifest.replay.applied_above.len(); self.floors.replay_floors.vector.set(manifest.replay); @@ -78,7 +77,6 @@ impl CoreLoop { generation = manifest.generation, loaded, vectors, - durable_through_lsn = manifest.durable_through_lsn, replay_prefix, applied_ranges, "vector checkpoint restored" @@ -184,7 +182,6 @@ mod tests { let bytes = zerompk::to_msgpack_vec(&VectorCheckpointManifest { format_version: VECTOR_CKPT_FORMAT_VERSION + 1, generation: 0, - durable_through_lsn: 5, replay: crate::types::replay_stamp::ReplayStamp::default(), }) .expect("encode"); diff --git a/nodedb/src/data/executor/vector_checkpoint/mod.rs b/nodedb/src/data/executor/vector_checkpoint/mod.rs index 2e58d0955..77eb27d7a 100644 --- a/nodedb/src/data/executor/vector_checkpoint/mod.rs +++ b/nodedb/src/data/executor/vector_checkpoint/mod.rs @@ -35,9 +35,8 @@ //! holds. A restart installs it as the vector replay floor. An HNSW insert //! appends a node and never dedups, so a record the stamp names must not //! replay. Every other record replays in LSN order, including a lower-LSN -//! record still in flight when the generation was written. The manifest also -//! carries `durable_through_lsn`, the point a failed flush clamps WAL -//! truncation to after a restart. +//! record still in flight when the generation was written. The stamp's prefix +//! is also the point a failed flush clamps WAL truncation to after a restart. mod build_completions; mod format; diff --git a/nodedb/src/data/executor/vector_checkpoint/publish.rs b/nodedb/src/data/executor/vector_checkpoint/publish.rs index ed460b9fe..caf615fba 100644 --- a/nodedb/src/data/executor/vector_checkpoint/publish.rs +++ b/nodedb/src/data/executor/vector_checkpoint/publish.rs @@ -10,7 +10,6 @@ use super::format::{VECTOR_CKPT_FORMAT_VERSION, VectorCheckpointManifest}; use super::manifest::{read_vector_manifest_at, storage_err}; use super::paths::VECTOR_CKPT_MANIFEST; -use crate::types::Lsn; use crate::types::replay_stamp::ReplayStamp; /// The generation number the next publish under `ckpt_dir` must use. @@ -32,13 +31,11 @@ pub(crate) fn next_generation(ckpt_dir: &std::path::Path) -> crate::Result pub(crate) fn publish_vector_generation( ckpt_dir: &std::path::Path, generation: u64, - durable_through: Lsn, replay: ReplayStamp, ) -> crate::Result<()> { let manifest = VectorCheckpointManifest { format_version: VECTOR_CKPT_FORMAT_VERSION, generation, - durable_through_lsn: durable_through.as_u64(), replay, }; let bytes = zerompk::to_msgpack_vec(&manifest).map_err(|e| crate::Error::Serialization { diff --git a/nodedb/src/data/executor/vector_checkpoint/write.rs b/nodedb/src/data/executor/vector_checkpoint/write.rs index 65c6220e9..8f1233571 100644 --- a/nodedb/src/data/executor/vector_checkpoint/write.rs +++ b/nodedb/src/data/executor/vector_checkpoint/write.rs @@ -29,7 +29,7 @@ impl CoreLoop { /// and nothing on that path puts a row in `sparse` for the rebuild to find. /// Those vectors exist in exactly two places — this checkpoint and the /// `VectorOp::Insert` WAL records — so a flush that fails while still - /// letting the core report its watermark deletes the only surviving copy. + /// letting the core report its floor deletes the only surviving copy. /// /// The failure is therefore all-or-nothing by construction: any index that /// cannot be published returns `Err`, and the caller clamps the reported @@ -37,21 +37,17 @@ impl CoreLoop { /// partial success cannot be expressed, because the LSN it would justify /// does not exist. /// - /// Stamping with the core watermark rests on this: the checkpoint runs - /// on the core's own thread between tasks, and a vector write raises - /// the watermark only after the collection has already been mutated, so - /// every write with `lsn <= watermark` is in the bytes written below. + /// The reported LSN is the checkpoint floor (`checkpoint_floor`): the + /// checkpoint runs on the core's own thread between tasks, so every record + /// at or below that outcome floor that this core applied is in the export + /// below. /// /// The manifest carries the core's replay stamp: the records every index /// in the generation holds. Restart replay skips exactly those. /// - /// Every published generation raises `vector_published_lsn` to the - /// stamp's prefix, whichever caller published it: restart restores from - /// the newest generation, so a committed record applied at or below it - /// must reach a newer one (`redo_apply::cover`). `vector_durable_lsn` - /// stays the caller's to raise. + /// `vector_durable_lsn` stays the caller's to raise. pub(crate) fn checkpoint_vector_indexes(&mut self) -> crate::Result { - let durable_lsn = self.watermark; + let durable_lsn = self.checkpoint_floor(); let replay = self.floors.applied_prefix.stamp()?; let ckpt_dir = vector_ckpt_dir(&self.data_dir, self.core_id); @@ -76,8 +72,7 @@ impl CoreLoop { let files_written = self.write_vector_generation(&gen_dir)?; let prefix = Lsn::new(replay.prefix); let applied_ranges = replay.applied_above.len(); - publish_vector_generation(&ckpt_dir, generation, durable_lsn, replay)?; - self.floors.vector_published_lsn = self.floors.vector_published_lsn.max(prefix); + publish_vector_generation(&ckpt_dir, generation, replay)?; // The previous generation is now unreachable. Removing it reclaims disk // but is NOT required for correctness — the manifest alone decides what @@ -177,12 +172,9 @@ mod tests { /// The resurrection guard. A collection checkpointed at generation N and /// then EMPTIED by deletes must not come back at boot: the deletes are - /// acknowledged and the checkpoint still reports the watermark, so the WAL - /// records that would have re-deleted the vectors are already gone. - /// - /// Before generations this failed — cycle N+1 skipped the empty collection, - /// the flat directory kept cycle N's populated file, and every deleted - /// vector reappeared on restart. + /// acknowledged and the checkpoint still reports its floor, so the WAL + /// records that would have re-deleted the vectors can be gone. Each cycle + /// therefore writes a new generation that omits the empty collection. #[test] fn emptied_collection_does_not_resurrect_the_previous_generation() { let dir = tempfile::tempdir().expect("tempdir"); @@ -258,6 +250,9 @@ mod tests { core.vector_collections .insert(collection_key(), collection_with_one_vector()); core.advance_watermark(Lsn::new(4_242)); + core.floors + .applied_prefix + .observe_outcome_floor(Lsn::new(4_242)); let outcome = core .checkpoint_vector_indexes() .expect("flush must publish"); diff --git a/nodedb/src/data/snapshot.rs b/nodedb/src/data/snapshot.rs index 3b2f35888..749bb861d 100644 --- a/nodedb/src/data/snapshot.rs +++ b/nodedb/src/data/snapshot.rs @@ -101,7 +101,9 @@ pub struct TenantKvPair { zerompk::FromMessagePack, )] pub struct CoreSnapshot { - /// Core/vShard watermark LSN. + /// The core's checkpoint floor when the snapshot was taken: every record + /// at or below it that the core applied is in the snapshot. A record + /// above it can be missing, so restore replays the WAL above it. pub watermark: u64, /// All documents from SparseEngine. diff --git a/nodedb/src/storage/snapshot_executor.rs b/nodedb/src/storage/snapshot_executor.rs index cee92648f..7062fb4c0 100644 --- a/nodedb/src/storage/snapshot_executor.rs +++ b/nodedb/src/storage/snapshot_executor.rs @@ -7,7 +7,7 @@ //! 1. Validate the restore plan via `dry_run_restore()`. //! 2. For each core, load its `CoreSnapshot` from the object store. //! 3. Apply the snapshot to the core's engines (redb, HNSW, CRDT). -//! 4. Replay WAL records from `snapshot_lsn` to `target_lsn`. +//! 4. Replay WAL records above the lowest core floor to `target_lsn`. //! //! ## Offline restore //! @@ -83,8 +83,7 @@ pub async fn execute_restore( for core_id in 0..manifest.num_cores { let core_snap = load_core_snapshot(snapshot_store, prefix, core_id, Some(encryption_key)).await?; - let (docs, vectors) = - restore_core_state(data_dir, core_id, &core_snap, manifest.meta.end_lsn)?; + let (docs, vectors) = restore_core_state(data_dir, core_id, &core_snap)?; total_docs += docs; total_vectors += vectors; @@ -97,19 +96,21 @@ pub async fn execute_restore( ); } + // The lowest core floor: every core needs the WAL above its own floor. + let replay_from = manifest.meta.begin_lsn.as_u64(); let wal_to_replay: Vec<_> = wal_records .iter() - .filter(|r| r.header.lsn > snapshot_lsn) + .filter(|r| r.header.lsn > replay_from) .collect(); let wal_count = wal_to_replay.len() as u64; if wal_count > 0 { info!( records = wal_count, - from_lsn = snapshot_lsn + 1, + from_lsn = replay_from + 1, "replaying WAL records after snapshot" ); - write_restore_marker(restore_store, snapshot_lsn).await?; + write_restore_marker(restore_store, replay_from).await?; } let result = RestoreResult { @@ -141,7 +142,6 @@ fn restore_core_state( data_dir: &Path, core_id: usize, snap: &CoreSnapshot, - snapshot_lsn: Lsn, ) -> crate::Result<(u64, u64)> { let sparse_path = data_dir.join(format!("sparse/core-{core_id}.redb")); if let Some(parent) = sparse_path.parent() { @@ -159,8 +159,11 @@ fn restore_core_state( let edge_store = crate::engine::graph::edge_store::store::EdgeStore::open(&graph_path)?; restore_edge_data(&edge_store, &snap.edges)?; - let vectors = restore_vector_checkpoints(data_dir, core_id, &snap.hnsw_indexes, snapshot_lsn)?; - restore_crdt_checkpoints(data_dir, core_id, &snap.crdt_snapshots, snapshot_lsn)?; + // The core's own floor, not the snapshot's: a record above it can be + // missing from this core's state, and replay must reach it. + let core_floor = Lsn::new(snap.watermark); + let vectors = restore_vector_checkpoints(data_dir, core_id, &snap.hnsw_indexes, core_floor)?; + restore_crdt_checkpoints(data_dir, core_id, &snap.crdt_snapshots, core_floor)?; Ok((snap.sparse_documents.len() as u64, vectors)) } @@ -250,7 +253,6 @@ fn restore_vector_checkpoints( crate::data::executor::vector_checkpoint::publish_vector_generation( &ckpt_dir, generation, - snapshot_lsn, crate::types::replay_stamp::ReplayStamp::through(snapshot_lsn.as_u64()), )?; diff --git a/nodedb/tests/crash_checkpoint_truncate_window.rs b/nodedb/tests/crash_checkpoint_truncate_window.rs index 9fd20dbd0..56dfcb7ac 100644 --- a/nodedb/tests/crash_checkpoint_truncate_window.rs +++ b/nodedb/tests/crash_checkpoint_truncate_window.rs @@ -65,6 +65,9 @@ async fn acknowledged_rows_survive_a_crash_between_checkpoint_marker_and_truncat // The next checkpoint cycle writes its marker and dies on the spot. h.await_self_crash(CRASH_TIMEOUT); + // Boot 2 runs disarmed, so its checkpoint cycles run to completion and the + // read below cannot race an abort. + h.clear_env("NODEDB_FAILPOINTS"); h.reopen(); let recovered = h @@ -80,17 +83,19 @@ async fn acknowledged_rows_survive_a_crash_between_checkpoint_marker_and_truncat — recovery treated the marker as proof of a truncation that never ran (got {recovered:?})" ); - // A freshly-replayed core reports checkpoint LSN 0 until its own new - // write, or every checkpoint cycle logs "skipping" and the window under - // test never reopens. Key is outside `row%` so it's not a canary. - h.exec("INSERT INTO ckpt_window (k, v) VALUES ('trigger', 'post-restart')") - .await; - - // The fail point is still armed, so it aborts again — proving the second - // cycle actually reached the same window. - h.await_self_crash(CRASH_TIMEOUT); + // Boot 3 is armed again. A replayed core reports its floor at once, so + // its first cycle reaches the same window with no new write. The cycle + // waits for the gateway, so the boot reports ready first. + h.kill_9(); + h.set_env( + "NODEDB_FAILPOINTS", + "checkpoint::after_marker_before_truncate=abort", + ); h.reopen(); + h.await_self_crash(CRASH_TIMEOUT); + h.clear_env("NODEDB_FAILPOINTS"); + h.reopen(); let after = h .query_col( "SELECT v FROM ckpt_window WHERE k LIKE 'row%' ORDER BY k", diff --git a/nodedb/tests/crash_harness/log_fields.rs b/nodedb/tests/crash_harness/log_fields.rs new file mode 100644 index 000000000..3a7d6e0f1 --- /dev/null +++ b/nodedb/tests/crash_harness/log_fields.rs @@ -0,0 +1,95 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Read facts back from the server log a crash test captured. +//! +//! The harness appends every boot's output to one file and marks the start of +//! each boot. A crash test proves what a boot did from the lines it logged. + +/// The server output of boot `n`, from its harness marker to the next one. +pub fn boot_section(log: &str, n: u32) -> String { + let marker = format!("=== crash harness boot {n} (pid"); + let Some(start) = log.find(&marker) else { + return String::new(); + }; + let rest = &log[start..]; + let next = format!("=== crash harness boot {} (pid", n + 1); + match rest.find(&next) { + Some(end) => rest[..end].to_string(), + None => rest.to_string(), + } +} + +/// Whether two read values are equal: by value when both are numbers, by +/// text otherwise. +pub fn same_value(read: &str, expected: &str) -> bool { + match (read.parse::(), expected.parse::()) { + (Ok(a), Ok(b)) => a == b, + _ => read == expected, + } +} + +/// The numeric `field` of every log line carrying `message`. +pub fn log_field(log: &str, message: &str, field: &str) -> Vec { + let key = format!("{field}="); + strip_ansi(log) + .lines() + .filter(|line| line.contains(message)) + .filter_map(|line| { + let rest = line.split_once(key.as_str())?.1; + let digits: String = rest.chars().take_while(char::is_ascii_digit).collect(); + digits.parse().ok() + }) + .collect() +} + +/// The number of log lines carrying any of `messages`. +pub fn count_lines(log: &str, messages: &[&str]) -> usize { + strip_ansi(log) + .lines() + .filter(|line| messages.iter().any(|message| line.contains(message))) + .count() +} + +/// `text` without terminal colour escape sequences. +pub fn strip_ansi(text: &str) -> String { + let mut out = String::with_capacity(text.len()); + let mut chars = text.chars(); + while let Some(c) = chars.next() { + if c == '\u{1b}' { + for next in chars.by_ref() { + if next == 'm' { + break; + } + } + } else { + out.push(c); + } + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn log_fields_are_read_through_colour_codes() { + let log = "INFO KV checkpoint published \u{1b}[3mapplied_ranges\u{1b}[0m\u{1b}[2m=\u{1b}[0m2\n\ + INFO KV checkpoint published applied_ranges=0\n"; + assert_eq!( + log_field(log, "KV checkpoint published", "applied_ranges"), + vec![2, 0] + ); + assert_eq!(count_lines(log, &["published", "absent"]), 2); + assert_eq!(strip_ansi("\u{1b}[3ma\u{1b}[0m"), "a"); + assert!(same_value("8.0", "8")); + assert!(!same_value("8.5", "8")); + assert!(same_value("a", "a")); + let booted = "=== crash harness boot 1 (pid 1) ===\nfirst-line\n\ + === crash harness boot 2 (pid 2) ===\nsecond-line\n"; + assert!(boot_section(booted, 2).contains("second-line")); + assert!(!boot_section(booted, 2).contains("first-line")); + assert!(boot_section(booted, 1).contains("first-line")); + assert!(!boot_section(booted, 1).contains("second-line")); + } +} diff --git a/nodedb/tests/crash_harness/mod.rs b/nodedb/tests/crash_harness/mod.rs index e0344b820..7b67b8b36 100644 --- a/nodedb/tests/crash_harness/mod.rs +++ b/nodedb/tests/crash_harness/mod.rs @@ -19,6 +19,10 @@ use std::time::{Duration, Instant}; pub mod diagnostics; // `wait_ready` and its bind-collision respawn. mod boot; +// Boot sections and numeric fields read back from the server log. +pub mod log_fields; +// WAL segment names and checkpoint truncation lines. +pub mod wal_truncation; // The ILP client helper lives in `nodedb-test-support` and is imported // directly by tests that need it, not re-exported here. mod pgwire; @@ -262,6 +266,15 @@ impl CrashHarness { names } + /// The WAL segment the server appends to now. Segment names carry their + /// zero-padded first LSN, so the last name is the active segment. + pub fn active_wal_segment(&self) -> String { + self.wal_segments() + .last() + .cloned() + .unwrap_or_else(|| panic!("no WAL segment on disk after an acknowledged write")) + } + /// Path the server's stdout/stderr is appended to across every spawn. pub fn server_log_path(&self) -> std::path::PathBuf { self.data_dir_path.join("server.log") diff --git a/nodedb/tests/crash_harness/pgwire.rs b/nodedb/tests/crash_harness/pgwire.rs index 8f9f64f8a..d7bc08621 100644 --- a/nodedb/tests/crash_harness/pgwire.rs +++ b/nodedb/tests/crash_harness/pgwire.rs @@ -169,10 +169,12 @@ impl CrashHarness { /// peeking at server-internal state. The probe write lands in /// `__crash_harness_calvin_probe`, a throwaway collection scoped to /// this call and never referenced by any caller's own assertions, so it - /// cannot perturb a row count a test checks elsewhere. + /// cannot perturb a row count a test checks elsewhere. The probe is + /// idempotent, so a test can call it again after a restart on the same + /// data directory. pub async fn wait_for_calvin_ready(&self, timeout: Duration) { self.exec( - "CREATE COLLECTION __crash_harness_calvin_probe \ + "CREATE COLLECTION IF NOT EXISTS __crash_harness_calvin_probe \ COLUMNS (id TEXT, ts BIGINT TIME_KEY, v FLOAT) \ WITH (engine='timeseries')", ) diff --git a/nodedb/tests/crash_harness/wal_truncation.rs b/nodedb/tests/crash_harness/wal_truncation.rs new file mode 100644 index 000000000..46dac26cc --- /dev/null +++ b/nodedb/tests/crash_harness/wal_truncation.rs @@ -0,0 +1,53 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Observe WAL truncation from the segment names on disk and the checkpoint +//! manager's log lines. + +use super::log_fields::{log_field, strip_ansi}; + +/// The checkpoint manager's line for a cycle that wrote its marker. A debug +/// line. +pub const MARKER_WRITTEN: &str = "checkpoint WAL marker written"; +/// The line for a cycle whose truncation step removed segments. +pub const WAL_TRUNCATED: &str = "WAL truncated after checkpoint"; +/// The line for a cycle whose truncation step removed nothing. A debug line. +pub const NOTHING_TRUNCATED: &str = "checkpoint complete (no segments to truncate)"; + +/// The first LSN of a WAL segment, from its `wal-.seg` name. +pub fn segment_first_lsn(segment: &str) -> u64 { + segment + .trim_start_matches("wal-") + .trim_end_matches(".seg") + .parse() + .unwrap_or_else(|e| panic!("WAL segment name {segment} carries no first LSN: {e}")) +} + +/// Whether a checkpoint whose marker is at or above `lsn` finished its +/// truncation step in `log`. Cycles run one at a time, so the first +/// truncation line after that marker belongs to the same cycle. +pub fn truncation_finished_from(log: &str, lsn: u64) -> bool { + let log = strip_ansi(log); + let mut lines = log.lines(); + let marked = lines.by_ref().any(|line| { + log_field(line, MARKER_WRITTEN, "marker_lsn") + .first() + .is_some_and(|marker| *marker >= lsn) + }); + marked && lines.any(|line| line.contains(WAL_TRUNCATED) || line.contains(NOTHING_TRUNCATED)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_cycle_counts_only_once_its_truncation_step_finished() { + let marker = "DEBUG checkpoint WAL marker written marker_lsn=40 checkpoint_lsn=9"; + let done = "DEBUG checkpoint complete (no segments to truncate)"; + assert!(!truncation_finished_from(marker, 40)); + assert!(truncation_finished_from(&format!("{marker}\n{done}"), 40)); + assert!(!truncation_finished_from(&format!("{marker}\n{done}"), 41)); + assert!(!truncation_finished_from(&format!("{done}\n{marker}"), 40)); + assert_eq!(segment_first_lsn("wal-00000000000000000019.seg"), 19); + } +} diff --git a/nodedb/tests/crash_replay_stamp.rs b/nodedb/tests/crash_replay_stamp.rs index 837bd0eea..7d7fbe442 100644 --- a/nodedb/tests/crash_replay_stamp.rs +++ b/nodedb/tests/crash_replay_stamp.rs @@ -23,6 +23,11 @@ //! boot 2's checkpoints after A is minted, so the checkpoint that names B also //! writes the held collection. //! +//! The WAL-truncation case seals segments below A in boot 1, and the segment +//! that holds A's record while A is parked. Truncation must remove segments +//! below A and keep A's segment, and must remove it once A settled after the +//! restart. +//! //! Requires `--features failpoints`. #![cfg(feature = "failpoints")] @@ -31,20 +36,34 @@ mod crash_harness; use std::time::{Duration, Instant}; +use crash_harness::log_fields::{boot_section, log_field, same_value}; +use crash_harness::wal_truncation::{WAL_TRUNCATED, segment_first_lsn, truncation_finished_from}; use crash_harness::{CrashHarness, diagnostics}; /// Boot 1 writes no checkpoint, so a seeded write stays in memory until the /// kill. const QUIET_CHECKPOINT_INTERVAL_SECS: &str = "3600"; -/// How long the test waits for a checkpoint whose stamp names a B: thirty -/// checkpoint cycles at one per second. A is parked until the test releases -/// it, so this bounds only the wait for the checkpoint manager. -const STAMP_DEADLINE: Duration = Duration::from_secs(30); +/// How long the test waits for a checkpoint whose stamp names a B, for +/// truncation runs, or for a segment to go: thirty checkpoint cycles at one +/// per second. A is parked until the test releases it, so this bounds only the +/// wait for the checkpoint manager. +const CHECKPOINT_DEADLINE: Duration = Duration::from_secs(30); /// How long the process may take to abort once A is released. const CRASH_TIMEOUT: Duration = Duration::from_secs(60); +/// The smallest WAL segment target the config accepts, in whole MiB. +const WAL_SEGMENT_TARGET_MB: &str = "1"; + +/// One filler value. Five of them hold 2.5 MiB, which seals the segment that +/// holds A's record under a 1 MiB target. +const FILLER_VALUE_BYTES: usize = 512 * 1024; +const FILLER_ROWS: usize = 5; + +/// The collection the filler goes to. The test never reads it back. +const FILLER: &str = "stamp_trunc_fill"; + /// One engine's run of the in-flight sequence. struct Case { /// The collection write A goes to. @@ -79,6 +98,8 @@ struct Case { /// The boot-3 log message that proves this run reproduced the in-flight /// write, and the numeric field that must be above zero on it. restored: (&'static str, &'static str), + /// Seal A's segment and wait for truncation runs while A is parked. + wal_truncation: bool, } #[tokio::test(flavor = "multi_thread")] @@ -100,6 +121,7 @@ async fn a_kv_write_in_flight_at_a_checkpoint_survives_kill_9() { log_directives: "nodedb::data::executor::kv_checkpoint=info", published: "KV checkpoint published", restored: ("KV checkpoint restored", "applied_ranges"), + wal_truncation: false, }) .await; } @@ -123,6 +145,7 @@ async fn a_columnar_write_in_flight_at_a_checkpoint_survives_kill_9() { log_directives: "nodedb::data::executor::columnar_checkpoint=info", published: "columnar checkpoint published", restored: ("columnar checkpoint restored", "applied_ranges"), + wal_truncation: false, }) .await; } @@ -152,6 +175,7 @@ async fn an_array_write_in_flight_at_a_checkpoint_survives_kill_9() { nodedb::data::executor::wal_replay::array=info", published: "array checkpoint flushed", restored: ("WAL array replay complete", "in_flight"), + wal_truncation: false, }) .await; } @@ -189,6 +213,36 @@ async fn a_timeseries_write_in_flight_at_a_checkpoint_survives_kill_9() { nodedb::data::executor::handlers::timeseries_wal=info", published: "timeseries columnar flush complete", restored: ("WAL timeseries replay complete", "in_flight"), + wal_truncation: false, + }) + .await; +} + +/// A checkpoint that runs while A is parked reports a floor below A. So a +/// truncation run keeps the segment that holds A's record, even after filler +/// writes above A seal it. A floor taken from the highest applied LSN lets the +/// run remove that segment, and A's row is lost with it. +#[tokio::test(flavor = "multi_thread")] +async fn a_kv_write_in_flight_at_a_wal_truncation_survives_kill_9() { + run(Case { + held: "stamp_trunc_lo", + applied: "stamp_trunc_hi", + create_held: "CREATE COLLECTION stamp_trunc_lo (k STRING PRIMARY KEY, v STRING) \ + WITH (engine='kv')", + create_applied: "CREATE COLLECTION stamp_trunc_hi (k STRING PRIMARY KEY, v STRING) \ + WITH (engine='kv')", + seed_held: None, + insert_held: "INSERT INTO stamp_trunc_lo (k, v) VALUES ('held', 'a')", + read_held: "SELECT v FROM stamp_trunc_lo WHERE k = 'held'", + held_value: "a", + insert_applied: |n| format!("INSERT INTO stamp_trunc_hi (k, v) VALUES ('k{n:03}', 'v{n}')"), + read_applied: "SELECT v FROM stamp_trunc_hi", + checkpoint_interval_secs: "1", + log_directives: "nodedb::data::executor::kv_checkpoint=info,\ + nodedb::control::checkpoint_manager=debug", + published: "KV checkpoint published", + restored: ("KV checkpoint restored", "applied_ranges"), + wal_truncation: true, }) .await; } @@ -201,10 +255,21 @@ async fn run(case: Case) { QUIET_CHECKPOINT_INTERVAL_SECS, ) .with_env("RUST_LOG", &format!("warn,{}", case.log_directives)); + if case.wal_truncation { + h.set_env("NODEDB_WAL_SEGMENT_TARGET_MB", WAL_SEGMENT_TARGET_MB); + } h.spawn(); h.wait_ready(); h.exec(case.create_held).await; h.exec(case.create_applied).await; + if case.wal_truncation { + h.exec(&format!( + "CREATE COLLECTION {FILLER} (k STRING PRIMARY KEY, v STRING) WITH (engine='kv')" + )) + .await; + // Sealed segments below A, so the floor A holds lets truncation run. + write_filler(&h, "below").await; + } if let Some(seed) = case.seed_held { h.exec(seed).await; } @@ -242,7 +307,7 @@ async fn run(case: Case) { }); // Apply B writes until a checkpoint names one of them above its prefix. - let deadline = Instant::now() + STAMP_DEADLINE; + let deadline = Instant::now() + CHECKPOINT_DEADLINE; let mut applied = 0usize; loop { h.exec(&(case.insert_applied)(applied)).await; @@ -257,13 +322,18 @@ async fn run(case: Case) { } assert!( Instant::now() < deadline, - "no {} named an applied LSN above its prefix within {STAMP_DEADLINE:?}: write A \ - never parked, or no checkpoint ran while it was.{}\n{}", + "no {} named an applied LSN above its prefix within {CHECKPOINT_DEADLINE:?}: \ + write A never parked, or no checkpoint ran while it was.{}\n{}", case.published, h.keep_data_dir_note(), diagnostics::log_tail_section(&h.server_log()) ); } + let held_segment = if case.wal_truncation { + Some(truncate_while_held(&h).await) + } else { + None + }; assert!( !held_task.is_finished(), "write A finished before its release: the gate never parked it" @@ -301,6 +371,18 @@ async fn run(case: Case) { diagnostics::log_tail_section(&h.server_log()) ); + assert_restored(&h, &case, &live).await; + + if let Some(segment) = held_segment { + truncation_advances_once_settled(&mut h, &segment).await; + h.kill_9(); + h.reopen(); + assert_restored(&h, &case, &live).await; + } +} + +/// A's row is back, and every B write is present once. +async fn assert_restored(h: &CrashHarness, case: &Case, live: &[String]) { let held = h.query_col_idx(case.read_held, 0).await; assert!( held.len() == 1 && same_value(&held[0], case.held_value), @@ -319,76 +401,79 @@ async fn run(case: Case) { ); } -/// The server output of boot `n`, from its harness marker to the next one. -fn boot_section(log: &str, n: u32) -> String { - let marker = format!("=== crash harness boot {n} (pid"); - let Some(start) = log.find(&marker) else { - return String::new(); - }; - let rest = &log[start..]; - let next = format!("=== crash harness boot {} (pid", n + 1); - match rest.find(&next) { - Some(end) => rest[..end].to_string(), - None => rest.to_string(), +/// Filler rows that seal the active WAL segment. +async fn write_filler(h: &CrashHarness, tag: &str) { + let filler = "x".repeat(FILLER_VALUE_BYTES); + for i in 0..FILLER_ROWS { + h.exec(&format!( + "INSERT INTO {FILLER} (k, v) VALUES ('{tag}{i}', '{filler}')" + )) + .await; } } -/// Whether two read values are equal: by value when both are numbers, by -/// text otherwise. -fn same_value(read: &str, expected: &str) -> bool { - match (read.parse::(), expected.parse::()) { - (Ok(a), Ok(b)) => a == b, - _ => read == expected, - } -} - -/// The numeric `field` of every log line carrying `message`. -fn log_field(log: &str, message: &str, field: &str) -> Vec { - let key = format!("{field}="); - strip_ansi(log) - .lines() - .filter(|line| line.contains(message)) - .filter_map(|line| { - let rest = line.split_once(key.as_str())?.1; - let digits: String = rest.chars().take_while(char::is_ascii_digit).collect(); - digits.parse().ok() - }) - .collect() -} +/// Seal the segment that holds A's record, then wait while A is parked for a +/// checkpoint that ran after the seal and a truncation that removed segments. +/// Returns the name of A's segment. +/// +/// A parked after its append and before any B write, and the B writes are too +/// small to fill a segment. So the active segment now holds A's record. +async fn truncate_while_held(h: &CrashHarness) -> String { + let held_segment = h.active_wal_segment(); + write_filler(h, "above").await; + let active = h.active_wal_segment(); + assert_ne!( + active, held_segment, + "the filler did not seal A's segment. Truncation never removes the active \ + segment, so this run proves nothing" + ); -/// `text` without terminal colour escape sequences. -fn strip_ansi(text: &str) -> String { - let mut out = String::with_capacity(text.len()); - let mut chars = text.chars(); - while let Some(c) = chars.next() { - if c == '\u{1b}' { - for next in chars.by_ref() { - if next == 'm' { - break; - } - } - } else { - out.push(c); + // Every record in the active segment is above A. A marker at or above its + // first LSN comes from a checkpoint that ran after the seal. + let sealed_at = segment_first_lsn(&active); + let deadline = Instant::now() + CHECKPOINT_DEADLINE; + loop { + let log = boot_section(&h.server_log(), 2); + let ran_after_seal = truncation_finished_from(&log, sealed_at); + let truncated = !log_field(&log, WAL_TRUNCATED, "segments_deleted").is_empty(); + if ran_after_seal && truncated { + break; } + assert!( + Instant::now() < deadline, + "within {CHECKPOINT_DEADLINE:?} no checkpoint finished truncation after the filler \ + sealed A's segment ({ran_after_seal}), or no truncation removed a segment below A \ + ({truncated}).{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + tokio::time::sleep(Duration::from_millis(250)).await; } - out + let segments = h.wal_segments(); + assert!( + segments.contains(&held_segment), + "a truncation removed {held_segment} while write A in it was in flight: \ + truncation must stay below the lowest checkpoint floor. Segments: {segments:?}" + ); + held_segment } -#[test] -fn log_fields_are_read_through_colour_codes() { - let log = "INFO KV checkpoint published \u{1b}[3mapplied_ranges\u{1b}[0m\u{1b}[2m=\u{1b}[0m2\n\ - INFO KV checkpoint published applied_ranges=0\n"; - assert_eq!( - log_field(log, "KV checkpoint published", "applied_ranges"), - vec![2, 0] - ); - assert!(same_value("8.0", "8")); - assert!(!same_value("8.5", "8")); - assert!(same_value("a", "a")); - let booted = "=== crash harness boot 1 (pid 1) ===\nfirst-line\n\ - === crash harness boot 2 (pid 2) ===\nsecond-line\n"; - assert!(boot_section(booted, 2).contains("second-line")); - assert!(!boot_section(booted, 2).contains("first-line")); - assert!(boot_section(booted, 1).contains("first-line")); - assert!(!boot_section(booted, 1).contains("second-line")); +/// After the restart A is applied, so truncation must remove A's segment. +/// A new write lets the Event Plane persist a watermark above it too. +async fn truncation_advances_once_settled(h: &mut CrashHarness, segment: &str) { + h.exec(&format!( + "INSERT INTO {FILLER} (k, v) VALUES ('settled', 's')" + )) + .await; + let deadline = Instant::now() + CHECKPOINT_DEADLINE; + while h.wal_segments().iter().any(|name| name == segment) { + assert!( + Instant::now() < deadline, + "truncation never removed {segment} after write A settled: the floor held \ + below a record that has its outcome.{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + tokio::time::sleep(Duration::from_millis(250)).await; + } } diff --git a/nodedb/tests/crash_replay_stamp_calvin.rs b/nodedb/tests/crash_replay_stamp_calvin.rs new file mode 100644 index 000000000..ad521502e --- /dev/null +++ b/nodedb/tests/crash_replay_stamp_calvin.rs @@ -0,0 +1,215 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A Calvin transaction's redo record still in flight when a checkpoint is +//! written survives a crash. +//! +//! The server runs the default single-node Calvin stack. A transaction that +//! writes two collections on different vShards commits through the Calvin +//! scheduler. Each vShard's scheduler appends the transaction's redo record, +//! then sends a flush that installs it on the core. The sequence: +//! +//! 1. Transaction A writes the held collection and a peer collection on +//! another vShard. The held vShard's scheduler appends A's redo record and +//! holds its flush at a fail point. No core installed the record. +//! 2. Writes B, autocommit `INSERT`s into a third collection, apply with +//! higher LSNs until a KV checkpoint's replay stamp names one of them above +//! its prefix. That proves the checkpoint was written while A's record was +//! appended and not applied. +//! 3. The test releases the flush. The core installs A's record, and the +//! process aborts before the flush response leaves. +//! 4. After restart, replay must apply A's record. The Calvin scheduler reads +//! the same record as the transaction's applied marker and never runs the +//! transaction again, so replay is the only way A's row returns. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +mod crash_harness; + +use std::time::{Duration, Instant}; + +use crash_harness::log_fields::{boot_section, count_lines, log_field}; +use crash_harness::{CrashHarness, diagnostics}; +use nodedb_types::id::{DatabaseId, VShardId}; + +/// How long the test waits for the held flush, and for a checkpoint whose +/// stamp names a B: thirty checkpoint cycles at one per second. +const CHECKPOINT_DEADLINE: Duration = Duration::from_secs(30); + +/// How long the process may take to abort once the flush is released. The +/// scheduler sends a held flush again on its next pass. +const CRASH_TIMEOUT: Duration = Duration::from_secs(60); + +/// How long the Calvin sequencer may take to elect its leader after a boot. +const CALVIN_READY_TIMEOUT: Duration = Duration::from_secs(30); + +/// The scheduler's log line for a flush held at the fail point. +const FLUSH_HELD: &str = "calvin: flush held at a fail point"; + +const LOG_DIRECTIVES: &str = "warn,nodedb::data::executor::kv_checkpoint=info,\ + nodedb::control::cluster::calvin::scheduler::driver::core=info"; + +/// Three KV collection names on three different vShards: held, peer and +/// applied. +fn collections_on_distinct_vshards() -> [String; 3] { + let mut taken: Vec = Vec::with_capacity(3); + ["calvin_held", "calvin_peer", "calvin_applied"].map(|prefix| { + let (name, vshard) = (0..512u32) + .map(|i| { + let name = format!("{prefix}_{i}"); + let vshard = + VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + (name, vshard) + }) + .find(|(_, vshard)| !taken.contains(vshard)) + .unwrap_or_else(|| panic!("no {prefix} name on a free vShard in 512 tries")); + taken.push(vshard); + name + }) +} + +#[tokio::test(flavor = "multi_thread")] +async fn a_calvin_redo_in_flight_at_a_checkpoint_survives_kill_9() { + let [held, peer, applied] = collections_on_distinct_vshards(); + let h = CrashHarness::new(); + let release = h.data_dir().join("release-held-flush"); + let mut h = h + .with_env("NODEDB_CHECKPOINT_INTERVAL_SECS", "1") + .with_env("RUST_LOG", LOG_DIRECTIVES) + .with_env( + "NODEDB_FAILPOINTS", + &format!( + "calvin::before_flush::{held}=wait_file({}),core::after_apply::{held}=abort", + release.display() + ), + ); + h.spawn(); + h.wait_ready(); + h.wait_for_calvin_ready(CALVIN_READY_TIMEOUT).await; + for name in [&held, &peer, &applied] { + h.exec(&format!( + "CREATE COLLECTION {name} (k STRING PRIMARY KEY, v STRING) WITH (engine='kv')" + )) + .await; + } + + // Transaction A. Its COMMIT waits for the held flush, so it runs on its + // own connection. + let conn_str = h.pgwire_conn_str(); + let statements = [ + "BEGIN".to_string(), + format!("INSERT INTO {held} (k, v) VALUES ('held', 'a')"), + format!("INSERT INTO {peer} (k, v) VALUES ('peer', 'p')"), + "COMMIT".to_string(), + ]; + let txn_task = tokio::spawn(async move { + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .map_err(|e| e.to_string())?; + tokio::spawn(connection); + for statement in &statements { + client + .simple_query(statement) + .await + .map_err(|e| format!("{statement}: {e}"))?; + } + Ok::<(), String>(()) + }); + + // A's redo record is appended before its flush is sent, so a held flush + // proves the record is in the WAL and not installed. + let deadline = Instant::now() + CHECKPOINT_DEADLINE; + while count_lines(&boot_section(&h.server_log(), 1), &[FLUSH_HELD]) == 0 { + assert!( + Instant::now() < deadline && !txn_task.is_finished(), + "the flush of transaction A was never held: A did not commit through the \ + Calvin scheduler.{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + + // Apply B writes until a checkpoint names one of them above its prefix. + let deadline = Instant::now() + CHECKPOINT_DEADLINE; + let mut written = 0usize; + loop { + h.exec(&format!( + "INSERT INTO {applied} (k, v) VALUES ('k{written:03}', 'v{written}')" + )) + .await; + written += 1; + tokio::time::sleep(Duration::from_millis(200)).await; + let log = boot_section(&h.server_log(), 1); + if log_field(&log, "KV checkpoint published", "applied_ranges") + .iter() + .any(|n| *n > 0) + { + break; + } + assert!( + Instant::now() < deadline, + "no KV checkpoint named an applied LSN above its prefix within \ + {CHECKPOINT_DEADLINE:?} while A's flush was held.{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + } + assert!( + !txn_task.is_finished(), + "transaction A finished before its flush was released" + ); + let read_applied = format!("SELECT v FROM {applied}"); + let mut live = h.query_col_idx(&read_applied, 0).await; + live.sort(); + + // Release the flush. The core installs A's record, and the process aborts + // before the flush response leaves. + std::fs::write(&release, b"release").expect("create the release file"); + h.await_self_crash(CRASH_TIMEOUT); + let marker = format!("fail_point aborting process: core::after_apply::{held}"); + assert!( + h.server_log().contains(&marker), + "the process exited, but not after A's flush installed its record.{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + // A's client lost its connection with the process; its result says nothing. + let _ = txn_task.await; + + h.clear_env("NODEDB_FAILPOINTS"); + h.reopen(); + h.wait_for_calvin_ready(CALVIN_READY_TIMEOUT).await; + + let proof = log_field( + &boot_section(&h.server_log(), 2), + "KV checkpoint restored", + "applied_ranges", + ); + assert!( + proof.iter().any(|n| *n > 0), + "no KV checkpoint restored with applied_ranges above zero, so this run did not \ + reproduce the in-flight record (values: {proof:?}).{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ); + + for (name, key, value) in [(&held, "held", "a"), (&peer, "peer", "p")] { + let read = h + .query_col_idx(&format!("SELECT v FROM {name} WHERE k = '{key}'"), 0) + .await; + assert_eq!( + read, + vec![value.to_string()], + "transaction A's row in {name} is missing: its redo record applied after the \ + checkpoint and before the crash, and replay must apply it" + ); + } + let mut replayed = h.query_col_idx(&read_applied, 0).await; + replayed.sort(); + assert_eq!( + replayed, live, + "the replayed state of {applied} must equal the live state: every B write once" + ); +} diff --git a/nodedb/tests/crash_wal_truncation.rs b/nodedb/tests/crash_wal_truncation.rs index 99d49fa91..84fe3ec93 100644 --- a/nodedb/tests/crash_wal_truncation.rs +++ b/nodedb/tests/crash_wal_truncation.rs @@ -44,18 +44,6 @@ fn filler_value() -> String { "x".repeat(FILLER_VALUE_BYTES) } -/// The segment the write that just returned landed in. Names are -/// zero-padded `wal-<20-digit first LSN>.seg`, so `last()` is the active -/// segment. Snapshotting before the filler write means the segment cannot -/// yet be truncated, so its later disappearance is unambiguous. -fn active_segment(h: &CrashHarness) -> String { - let segments = h.wal_segments(); - segments - .last() - .unwrap_or_else(|| panic!("no WAL segments on disk after an acknowledged write")) - .clone() -} - /// Block until `segment` has been unlinked, panicking if it never is. The /// only thing distinguishing "checkpoint restored the row" from "WAL record /// was still there all along". @@ -92,7 +80,7 @@ async fn kv_row_survives_wal_segment_truncation() { // The segment holding the canary's WAL record, captured while it is still // the active one and therefore provably not yet truncated. - let canary_segment = active_segment(&h); + let canary_segment = h.active_wal_segment(); // Live sanity BEFORE anything else: the row reads back now, so a failure // after the restart is attributable to recovery and not to test setup. @@ -115,7 +103,7 @@ async fn kv_row_survives_wal_segment_truncation() { .await; } assert_ne!( - active_segment(&h), + h.active_wal_segment(), canary_segment, "filler writes did not rotate the WAL — the canary's segment is still active and \ truncation would skip it, making this test vacuous" @@ -157,7 +145,7 @@ async fn columnar_row_survives_wal_segment_truncation() { h.exec("INSERT INTO trunc_columnar (id, region, payload) VALUES ('canary', 'us', 'small')") .await; - let canary_segment = active_segment(&h); + let canary_segment = h.active_wal_segment(); let live = h .query_col( @@ -179,7 +167,7 @@ async fn columnar_row_survives_wal_segment_truncation() { .await; } assert_ne!( - active_segment(&h), + h.active_wal_segment(), canary_segment, "filler writes did not rotate the WAL — the canary's segment is still active and \ truncation would skip it, making this test vacuous" From 9acdde93c14612a61770ac107b03cea96b0bd5d0 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 15:45:42 +0800 Subject: [PATCH 32/64] test(cluster): accept a rebalanced learner as a follower in a PK-read case The 4th node's group role can flip from learner to voting follower at any point after it joins, since the rebalancer promotes it on its own schedule. Assert on "replicates without leading" instead of pinning the role to learner, so a promotion that lands mid-test does not fail an otherwise-correct read path. --- .../common_suite/cases/calvin_multishard_pk_read.rs | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs index c49979c32..cfbe7fd9b 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_multishard_pk_read.rs @@ -116,13 +116,15 @@ async fn cross_node_pk_read_from_learner_node_after_calvin_commit() { .position(|n| n.node_id == learner_id) .expect("learner present in cluster"); - // The 4th node joins `col_a`'s group as a non-voting learner: it - // applies the replicated log but never coordinated this transaction, - // so its catalog holds no binding the coordinator minted. + // The 4th node joins `col_a`'s group after it was mounted: it applies + // the replicated log but never coordinated this transaction, so its + // catalog holds no binding the coordinator minted. It joins as a + // learner, and the rebalancer can promote it to a voter at any point, + // so either role holds the premise. It must not lead the group. let status = fx.cluster.nodes[reader].group_status_line(gid_a); assert!( - status.contains("role=Learner"), - "node {learner_id} must join {col_a}'s group {gid_a} as a learner: {status}" + status.contains("role=Learner") || status.contains("role=Follower"), + "node {learner_id} must replicate {col_a}'s group {gid_a} without leading it: {status}" ); // Coordinator is one of the original 3, chosen by the fixture default From 0a1331e2d70c7308a0bfba73f303dc7d37bd0109 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Fri, 25 Sep 2026 21:29:57 +0800 Subject: [PATCH 33/64] feat(control): fence reads and writes on stale authorization state Add an authorization lease and fence so a node never plans or serves a statement against roles, grants, policies, or a permission tree it knows are behind an already-acknowledged change. `auth_lease` tracks a Raft-backed lease with a renewal loop, Calvin-ack coverage, and leadership/holder bookkeeping; `auth_fence` composes it with a read-index gate and the permission-tree's synced state to answer whether this node's authorization view is current. A statement that cannot be planned yet fails with `AuthorizationStateBehind` and is retried by the client rather than run against stale policy. Extend `RaftReadGate`/`read_index` with a general-purpose read-index probe usable outside the auth path, and give `multi_raft` explicit `proposals` and `status` modules plus applied-ack tracking so a lease renewal (and the wider metadata path) can observe apply progress. Split `metadata_proposer.rs` into a directory (`catalog`, `ddl_prepare`, `handle`, `replicated_entries`, `timeouts`) and give the permission tree its own `reload`, `sources`, and `sync_state` modules alongside a heavier `cache`/`event_handler`. Add `catalog_entry` authorization hooks, a `ddl_authorization` session gate, and a `local_read` dispatch path that consult the fence before serving. Consolidate the Calvin-only fields scattered across `SharedState` (applied epoch, counters, apply results, lock managers, promotion channels, autocommit lock sequence) into a single `CalvinLocalState`, and split `SharedState`'s test constructors out of `init.rs` into `init_variants.rs` now that `authorization_fence` and the consolidated Calvin state are part of construction. Alongside this, give the Event Plane's non-idempotent sinks (audit, CDC, CRDT, key-space) durable per-sink ledgers (`sink_ledger`) keyed by a monotonic record number (`record_numbering`), tracked through a `progress` cursor, and split `consumer.rs` into a directory (`delivery`, `drain`, `fail_stop`, `handle`, `pipeline`, `recovery`, `replay`, `run`). `WriteEvent` carries an optional durable record so a consumer can resume a sink exactly where its ledger left off instead of replaying from the watermark. Cover the new paths with cluster tests for cross-node and lease- partitioned permission-tree reads, wire-protocol tests for the authorization lease, fence, and permission-tree restart/support paths, and update existing Calvin, event-plane, and admission-fence tests for the consolidated state and the new `WriteEvent` field. --- .../single_node_calvin_hot_key_reservation.rs | 4 +- .../cases/single_node_calvin_two_phase.rs | 5 +- .../single_node_calvin_write_versions.rs | 5 +- .../cases/sql_cluster_cross_node_dml.rs | 4 + .../misc_suite/cases/cluster_triggers.rs | 1 + .../permission_tree_cross_node.rs | 201 +++++ .../permission_tree_lease_partition.rs | 224 +++++ nodedb-cluster/src/calvin/applied_acks.rs | 87 ++ nodedb-cluster/src/calvin/completion.rs | 3 + nodedb-cluster/src/calvin/mod.rs | 2 + .../calvin/sequencer/state_machine/apply.rs | 9 +- nodedb-cluster/src/lib.rs | 19 +- nodedb-cluster/src/multi_raft/core.rs | 245 +---- nodedb-cluster/src/multi_raft/mod.rs | 11 +- nodedb-cluster/src/multi_raft/proposals.rs | 149 +++ nodedb-cluster/src/multi_raft/status.rs | 97 ++ nodedb-cluster/src/raft_loop/auth_lease.rs | 37 + .../src/raft_loop/auth_lease_hook.rs | 23 + nodedb-cluster/src/raft_loop/builder.rs | 12 + .../src/raft_loop/handle_rpc/dispatch.rs | 5 + nodedb-cluster/src/raft_loop/loop_core.rs | 6 + nodedb-cluster/src/raft_loop/mod.rs | 4 + nodedb-cluster/src/raft_loop/read_index.rs | 109 +++ .../src/raft_loop/tick/apply_committed.rs | 10 +- nodedb-cluster/src/rpc_codec/auth_lease.rs | 221 +++++ nodedb-cluster/src/rpc_codec/discriminants.rs | 18 + nodedb-cluster/src/rpc_codec/mod.rs | 7 + nodedb-cluster/src/rpc_codec/raft_rpc.rs | 30 +- nodedb-cluster/src/rpc_codec/read_index.rs | 130 +++ nodedb-cluster/src/transport/client/mod.rs | 1 + nodedb-cluster/src/transport/client/send.rs | 1 + nodedb-cluster/src/transport/client/sever.rs | 46 + .../src/transport/client/transport.rs | 3 + .../node/lifecycle/spawn_full.rs | 12 + .../src/pgwire_harness/restart.rs | 6 + nodedb/src/bootstrap/cluster_ready.rs | 30 + nodedb/src/bootstrap/mod.rs | 1 + nodedb/src/bootstrap/permission_tree_load.rs | 24 + .../control/catalog_entry/authorization.rs | 124 +++ nodedb/src/control/catalog_entry/mod.rs | 1 + .../catalog_entry/post_apply/collection.rs | 46 + .../control/catalog_entry/post_apply/sync.rs | 15 +- .../calvin/scheduler/applied_mirror.rs | 145 +++ .../driver/core/commit_resolve/apply_tail.rs | 3 +- .../driver/core/commit_resolve/verdict.rs | 6 +- .../driver/core/commit_resolve/vote.rs | 3 +- .../scheduler/driver/core/completion_route.rs | 3 +- .../calvin/scheduler/driver/core/deferred.rs | 3 +- .../calvin/scheduler/driver/core/process.rs | 8 +- .../calvin/scheduler/driver/core/scheduler.rs | 22 +- .../cluster/calvin/scheduler/driver/types.rs | 2 +- .../control/cluster/calvin/scheduler/mod.rs | 2 + nodedb/src/control/cluster/read_index.rs | 64 +- .../src/control/cluster/snapshot_applier.rs | 4 + .../control/cluster/start_raft/loop_build.rs | 29 + .../cluster/start_raft/observability.rs | 30 +- .../cluster/start_raft/proposer_wiring.rs | 95 +- .../src/control/cluster/start_raft_helpers.rs | 8 +- nodedb/src/control/event_trigger.rs | 1 + nodedb/src/control/gateway/sql_execute.rs | 9 +- nodedb/src/control/metadata_proposer.rs | 710 --------------- .../src/control/metadata_proposer/catalog.rs | 208 +++++ .../control/metadata_proposer/ddl_prepare.rs | 192 ++++ .../src/control/metadata_proposer/handle.rs | 87 ++ nodedb/src/control/metadata_proposer/mod.rs | 38 + .../metadata_proposer/replicated_entries.rs | 195 ++++ .../src/control/metadata_proposer/timeouts.rs | 21 + .../control/planner/calvin/dependent_recon.rs | 17 +- .../control/planner/calvin/submit/local.rs | 20 +- .../control/security/auth_fence/cluster.rs | 106 +++ nodedb/src/control/security/auth_fence/mod.rs | 11 + .../control/security/auth_fence/read_index.rs | 251 ++++++ .../src/control/security/auth_fence/state.rs | 154 ++++ .../control/security/auth_fence/tree_defs.rs | 163 ++++ .../src/control/security/auth_fence/view.rs | 85 ++ .../control/security/auth_lease/barrier.rs | 166 ++++ .../security/auth_lease/calvin_acks.rs | 116 +++ .../control/security/auth_lease/coverage.rs | 196 ++++ .../src/control/security/auth_lease/holder.rs | 92 ++ .../control/security/auth_lease/leadership.rs | 58 ++ nodedb/src/control/security/auth_lease/mod.rs | 21 + .../control/security/auth_lease/renew_loop.rs | 113 +++ .../control/security/auth_lease/service.rs | 275 ++++++ .../src/control/security/auth_lease/status.rs | 120 +++ .../src/control/security/auth_lease/table.rs | 274 ++++++ .../src/control/security/auth_lease/timing.rs | 97 ++ nodedb/src/control/security/mod.rs | 2 + .../control/security/permission_tree/cache.rs | 189 +++- .../security/permission_tree/event_handler.rs | 168 +++- .../control/security/permission_tree/mod.rs | 6 +- .../security/permission_tree/reload.rs | 143 +++ .../security/permission_tree/sources.rs | 175 ++++ .../security/permission_tree/sync_state.rs | 198 ++++ .../control/security/permission_tree/types.rs | 2 +- .../server/dispatch_utils/local_read.rs | 111 +++ .../src/control/server/dispatch_utils/mod.rs | 2 + .../submit_write/funnel/driver.rs | 23 +- .../src/control/server/http/routes/health.rs | 66 ++ .../src/control/server/http/routes/metrics.rs | 7 + .../server/native/dispatch/sql_admin.rs | 8 +- .../server/pgwire/handler/cursor_query.rs | 5 +- .../server/pgwire/handler/prepared/parser.rs | 7 +- .../server/pgwire/handler/routing/planning.rs | 13 +- .../server/pgwire/handler/session_explain.rs | 5 +- .../control/server/pgwire/types/error_map.rs | 4 + .../neutral/collection/dml/parse/dispatch.rs | 5 +- .../shared/ddl/neutral/permission_tree.rs | 35 + .../server/shared/ddl/neutral/planning.rs | 5 +- .../src/control/server/shared/ddl/result.rs | 9 + .../control/server/shared/plan_admission.rs | 3 +- .../shared/session/ddl_authorization.rs | 45 + .../server/shared/session/ddl_flush.rs | 14 +- .../control/server/shared/session/hot_key.rs | 1 + .../server/shared/session/lifecycle.rs | 9 +- .../src/control/server/shared/session/mod.rs | 1 + .../control/server/shared/session/read_set.rs | 1 + .../control/server/shared/session/state.rs | 2 +- .../server/shared/write_admission/gate.rs | 15 +- nodedb/src/control/state/calvin_apply.rs | 4 +- nodedb/src/control/state/calvin_local.rs | 96 ++ nodedb/src/control/state/fields.rs | 80 +- nodedb/src/control/state/init.rs | 173 +--- .../src/control/state/init_prod/auth_parts.rs | 135 +++ .../src/control/state/init_prod/bootstrap.rs | 17 +- nodedb/src/control/state/init_prod/mod.rs | 1 + nodedb/src/control/state/init_prod/open.rs | 126 +-- nodedb/src/control/state/init_variants.rs | 157 ++++ nodedb/src/control/state/mod.rs | 3 + nodedb/src/control/system_txn/scope.rs | 3 +- nodedb/src/control/trigger/batch/collector.rs | 2 +- .../src/data/executor/core_loop/deferred.rs | 3 + .../src/data/executor/core_loop/event_emit.rs | 4 + nodedb/src/error/types.rs | 6 + nodedb/src/error_classify.rs | 1 + nodedb/src/event/audit_dml/consumer.rs | 29 +- nodedb/src/event/bus.rs | 42 +- nodedb/src/event/cdc/buffer.rs | 33 + nodedb/src/event/cdc/router.rs | 48 +- nodedb/src/event/consumer.rs | 852 ------------------ nodedb/src/event/consumer/delivery.rs | 157 ++++ nodedb/src/event/consumer/drain.rs | 262 ++++++ nodedb/src/event/consumer/fail_stop.rs | 104 +++ nodedb/src/event/consumer/handle.rs | 84 ++ nodedb/src/event/consumer/mod.rs | 12 + nodedb/src/event/consumer/pipeline.rs | 182 ++++ nodedb/src/event/consumer/recovery.rs | 112 +++ nodedb/src/event/consumer/replay.rs | 35 + nodedb/src/event/consumer/run.rs | 840 +++++++++++++++++ nodedb/src/event/consumer_helpers.rs | 305 +------ nodedb/src/event/crdt_sync/packager.rs | 44 +- nodedb/src/event/mod.rs | 3 + nodedb/src/event/plane.rs | 24 +- nodedb/src/event/progress.rs | 51 ++ nodedb/src/event/record_numbering.rs | 114 +++ nodedb/src/event/sink_ledger/audit.rs | 120 +++ nodedb/src/event/sink_ledger/boot.rs | 31 + nodedb/src/event/sink_ledger/cdc.rs | 325 +++++++ nodedb/src/event/sink_ledger/crdt.rs | 144 +++ nodedb/src/event/sink_ledger/key.rs | 149 +++ nodedb/src/event/sink_ledger/ledgers.rs | 59 ++ nodedb/src/event/sink_ledger/mod.rs | 15 + nodedb/src/event/streaming_mv/applied.rs | 105 +++ nodedb/src/event/streaming_mv/mod.rs | 1 + nodedb/src/event/streaming_mv/persist.rs | 140 ++- nodedb/src/event/streaming_mv/registry.rs | 8 + nodedb/src/event/trigger/dispatcher/batch.rs | 2 +- nodedb/src/event/types.rs | 28 + nodedb/src/event/wal_replay.rs | 8 +- nodedb/src/event/wal_replay_parse.rs | 14 +- nodedb/src/event/watermark.rs | 6 + nodedb/tests/inproc/cases/bitemporal_cdc.rs | 1 + nodedb/tests/inproc/cases/cdc_arc_fanout.rs | 1 + nodedb/tests/inproc/cases/event_trigger.rs | 1 + .../inproc/cases/shutdown_event_plane.rs | 9 +- .../inproc/cases/write_admission_fence.rs | 8 +- .../wire/cases/healthz_authorization_lease.rs | 63 ++ nodedb/tests/wire/cases/mod.rs | 4 + .../tests/wire/cases/permission_tree_fence.rs | 99 ++ .../wire/cases/permission_tree_restart.rs | 67 ++ .../wire/cases/permission_tree_support.rs | 91 ++ ...ssion_plan_cache_permission_tree_revoke.rs | 92 +- 181 files changed, 10509 insertions(+), 2716 deletions(-) create mode 100644 nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs create mode 100644 nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs create mode 100644 nodedb-cluster/src/calvin/applied_acks.rs create mode 100644 nodedb-cluster/src/multi_raft/proposals.rs create mode 100644 nodedb-cluster/src/multi_raft/status.rs create mode 100644 nodedb-cluster/src/raft_loop/auth_lease.rs create mode 100644 nodedb-cluster/src/raft_loop/auth_lease_hook.rs create mode 100644 nodedb-cluster/src/raft_loop/read_index.rs create mode 100644 nodedb-cluster/src/rpc_codec/auth_lease.rs create mode 100644 nodedb-cluster/src/rpc_codec/read_index.rs create mode 100644 nodedb-cluster/src/transport/client/sever.rs create mode 100644 nodedb/src/bootstrap/permission_tree_load.rs create mode 100644 nodedb/src/control/catalog_entry/authorization.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs delete mode 100644 nodedb/src/control/metadata_proposer.rs create mode 100644 nodedb/src/control/metadata_proposer/catalog.rs create mode 100644 nodedb/src/control/metadata_proposer/ddl_prepare.rs create mode 100644 nodedb/src/control/metadata_proposer/handle.rs create mode 100644 nodedb/src/control/metadata_proposer/mod.rs create mode 100644 nodedb/src/control/metadata_proposer/replicated_entries.rs create mode 100644 nodedb/src/control/metadata_proposer/timeouts.rs create mode 100644 nodedb/src/control/security/auth_fence/cluster.rs create mode 100644 nodedb/src/control/security/auth_fence/mod.rs create mode 100644 nodedb/src/control/security/auth_fence/read_index.rs create mode 100644 nodedb/src/control/security/auth_fence/state.rs create mode 100644 nodedb/src/control/security/auth_fence/tree_defs.rs create mode 100644 nodedb/src/control/security/auth_fence/view.rs create mode 100644 nodedb/src/control/security/auth_lease/barrier.rs create mode 100644 nodedb/src/control/security/auth_lease/calvin_acks.rs create mode 100644 nodedb/src/control/security/auth_lease/coverage.rs create mode 100644 nodedb/src/control/security/auth_lease/holder.rs create mode 100644 nodedb/src/control/security/auth_lease/leadership.rs create mode 100644 nodedb/src/control/security/auth_lease/mod.rs create mode 100644 nodedb/src/control/security/auth_lease/renew_loop.rs create mode 100644 nodedb/src/control/security/auth_lease/service.rs create mode 100644 nodedb/src/control/security/auth_lease/status.rs create mode 100644 nodedb/src/control/security/auth_lease/table.rs create mode 100644 nodedb/src/control/security/auth_lease/timing.rs create mode 100644 nodedb/src/control/security/permission_tree/reload.rs create mode 100644 nodedb/src/control/security/permission_tree/sources.rs create mode 100644 nodedb/src/control/security/permission_tree/sync_state.rs create mode 100644 nodedb/src/control/server/dispatch_utils/local_read.rs create mode 100644 nodedb/src/control/server/shared/session/ddl_authorization.rs create mode 100644 nodedb/src/control/state/calvin_local.rs create mode 100644 nodedb/src/control/state/init_prod/auth_parts.rs create mode 100644 nodedb/src/control/state/init_variants.rs delete mode 100644 nodedb/src/event/consumer.rs create mode 100644 nodedb/src/event/consumer/delivery.rs create mode 100644 nodedb/src/event/consumer/drain.rs create mode 100644 nodedb/src/event/consumer/fail_stop.rs create mode 100644 nodedb/src/event/consumer/handle.rs create mode 100644 nodedb/src/event/consumer/mod.rs create mode 100644 nodedb/src/event/consumer/pipeline.rs create mode 100644 nodedb/src/event/consumer/recovery.rs create mode 100644 nodedb/src/event/consumer/replay.rs create mode 100644 nodedb/src/event/consumer/run.rs create mode 100644 nodedb/src/event/progress.rs create mode 100644 nodedb/src/event/record_numbering.rs create mode 100644 nodedb/src/event/sink_ledger/audit.rs create mode 100644 nodedb/src/event/sink_ledger/boot.rs create mode 100644 nodedb/src/event/sink_ledger/cdc.rs create mode 100644 nodedb/src/event/sink_ledger/crdt.rs create mode 100644 nodedb/src/event/sink_ledger/key.rs create mode 100644 nodedb/src/event/sink_ledger/ledgers.rs create mode 100644 nodedb/src/event/sink_ledger/mod.rs create mode 100644 nodedb/src/event/streaming_mv/applied.rs create mode 100644 nodedb/tests/wire/cases/healthz_authorization_lease.rs create mode 100644 nodedb/tests/wire/cases/permission_tree_fence.rs create mode 100644 nodedb/tests/wire/cases/permission_tree_restart.rs create mode 100644 nodedb/tests/wire/cases/permission_tree_support.rs diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs index ed3354781..38b7ba642 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs @@ -55,7 +55,8 @@ fn other_vshard_collection(exclude_vshard: u32) -> String { fn reservation_count(node: &TestClusterNode, vshard: u32, key: &LockKey) -> usize { let managers = node .shared - .calvin_lock_managers + .calvin + .lock_managers .lock() .unwrap_or_else(|p| p.into_inner()); let Some(lm) = managers.get(&vshard) else { @@ -112,6 +113,7 @@ async fn hot_key_read_reservation_installs_self_upgrades_and_releases() { { let mut table = node .shared + .calvin .hot_key_table .lock() .unwrap_or_else(|p| p.into_inner()); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_two_phase.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_two_phase.rs index 9bc0d3913..8c7653806 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_two_phase.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_two_phase.rs @@ -7,7 +7,7 @@ //! //! The staged buffer + verdict-driven flush/drop live on the `!Send` Data-Plane //! core, so this asserts the flush FIRED via the node-global -//! `calvin_counters.commits_flushed` counter (incremented once per staged apply the +//! `calvin.counters.commits_flushed` counter (incremented once per staged apply the //! per-vShard scheduler resolved to commit) plus the functional proof that the //! flushed write is visible. @@ -38,7 +38,8 @@ fn sequencer_leader(node: &TestClusterNode) -> u64 { /// flushing their commit-pending buffer to base. fn commits_flushed(node: &TestClusterNode) -> u64 { node.shared - .calvin_counters + .calvin + .counters .commits_flushed .load(Ordering::Relaxed) } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_write_versions.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_write_versions.rs index aeb3cbfc9..b4ee53f00 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_write_versions.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_write_versions.rs @@ -6,7 +6,7 @@ //! //! The version index itself lives on the `!Send` Data-Plane core and its //! readers are test-only, so this asserts the recording FIRED via the -//! node-global `calvin_counters.write_versions_recorded` counter, which the per-vShard +//! node-global `calvin.counters.write_versions_recorded` counter, which the per-vShard //! scheduler increments once per committed Calvin apply for which it dispatched //! a write-version record op (at the CalvinApplied WAL LSN). The counter is the //! standalone-observable proof that a cross-shard-committed write now advances @@ -40,7 +40,8 @@ fn sequencer_leader(node: &TestClusterNode) -> u64 { /// the per-core write-version index. fn write_versions_recorded(node: &TestClusterNode) -> u64 { node.shared - .calvin_counters + .calvin + .counters .write_versions_recorded .load(Ordering::Relaxed) } diff --git a/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs b/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs index c37a5423f..388047548 100644 --- a/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs +++ b/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs @@ -45,6 +45,10 @@ mod graph_traverse_reverse_cross_node; mod join_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/native_implicit_edge_delete_cross_node.rs"] mod native_implicit_edge_delete_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs"] +mod permission_tree_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs"] +mod permission_tree_lease_partition; #[path = "../../sql_cluster_cross_node_dml_tests/schema_objects.rs"] mod schema_objects; #[path = "../../sql_cluster_cross_node_dml_tests/select_remote_stream_cross_node.rs"] diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs index 799b46536..5ef85e352 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs @@ -183,6 +183,7 @@ fn event_source_preserved_through_write_event() { op: WriteOp::Insert, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("doc-1")), lsn: Lsn::new(100), + record: None, tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), source: EventSource::User, diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs new file mode 100644 index 000000000..e77268c93 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs @@ -0,0 +1,201 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A permission-tree revoke acknowledged on one node binds the next +//! statement on every other node. +//! +//! A grant is a row in the tree's permission table. Node B learns of it when +//! its own replica applies the write and its Event Plane updates its +//! permission cache. Node A acknowledges the write only after every node +//! holding an authorization lease reported that coverage, or its lease +//! expired. So B's first statement after the acknowledgement either sees the +//! change or is refused with a retryable error. It never plans against the +//! grant before the revoke. +//! +//! The test repeats grant and revoke several rounds and reads on every other +//! node right after each acknowledgement, with no wait in between. + +use std::time::{Duration, Instant}; + +use crate::common::cluster_harness::TestCluster; + +const PROBE_USER: &str = "ptx_probe"; +const SELECT_DOCS: &str = "SELECT id FROM ptx_docs ORDER BY id"; +const ROUNDS: usize = 5; + +/// SQLSTATE of a statement refused because this node's authorization state +/// is behind. The client retries it. +const AUTHORIZATION_BEHIND: &str = "55P03"; + +/// The outcome of one probe read. +#[derive(Debug, PartialEq)] +enum Read { + Rows(Vec), + Refused, +} + +async fn probe_read(client: &tokio_postgres::Client) -> Read { + match client.simple_query(SELECT_DOCS).await { + Ok(messages) => Read::Rows( + messages + .into_iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => { + Some(row.get(0).unwrap_or("").to_string()) + } + _ => None, + }) + .collect(), + ), + Err(error) => { + let code = error.code().map(|code| code.code().to_string()); + assert_eq!( + code.as_deref(), + Some(AUTHORIZATION_BEHIND), + "a probe read failed with an error other than a retryable refusal: {error:?}" + ); + Read::Refused + } + } +} + +/// Wait until `probe` reads exactly `expected`. A retryable refusal and a +/// replica still catching up are retried. A denial fails at once, in +/// [`probe_read`]. +async fn await_rows(probe: &tokio_postgres::Client, expected: &[&str], what: &str) { + let expected: Vec = expected.iter().map(|row| row.to_string()).collect(); + let deadline = Instant::now() + Duration::from_secs(10); + loop { + let read = probe_read(probe).await; + if read == Read::Rows(expected.clone()) { + return; + } + assert!( + Instant::now() < deadline, + "{what}: last read {read:?}, expected {expected:?}" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } +} + +async fn connect_probe( + pg_addr: std::net::SocketAddr, +) -> (tokio_postgres::Client, tokio::task::JoinHandle<()>) { + let conn_str = format!( + "host={} port={} user={PROBE_USER} dbname=default", + pg_addr.ip(), + pg_addr.port() + ); + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .unwrap_or_else(|e| panic!("connect as {PROBE_USER} to {pg_addr}: {e}")); + let handle = tokio::spawn(async move { + let _ = connection.await; + }); + (client, handle) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn revoke_on_one_node_binds_the_next_statement_on_every_other_node() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + for sql in [ + "CREATE COLLECTION ptx_docs (id TEXT PRIMARY KEY, title TEXT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION ptx_grants", + "CREATE ROLE ptx_role", + // `readwrite` is the built-in role that grants Read on every + // collection. An unknown name such as `read_write` is a custom role + // with no permissions, and every read would be denied. + "CREATE USER ptx_probe WITH PASSWORD 'ptx-probe-password' ROLE readwrite", + "GRANT ROLE ptx_role TO ptx_probe", + ] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + let writer = &cluster.nodes[0]; + for sql in [ + "INSERT INTO ptx_docs (id, title) VALUES ('d1', 'Doc One')", + "INSERT INTO ptx_docs (id, title) VALUES ('d2', 'Doc Two')", + ] { + writer + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + // The probe's base Read is in effect on every node, the writer included, + // before the tree narrows it. A denial here is a setup error, not a + // tree result. + for (index, node) in cluster.nodes.iter().enumerate() { + let (probe, handle) = connect_probe(node.pg_addr).await; + await_rows( + &probe, + &["d1", "d2"], + &format!("node {index}: base Read before the tree"), + ) + .await; + drop(probe); + handle.abort(); + } + + cluster + .exec_ddl_on_any_leader( + "ALTER COLLECTION ptx_docs SET PERMISSION_TREE = '{\ + \"resource_column\":\"id\",\ + \"graph_index\":\"ptx_docs_tree\",\ + \"permission_table\":\"ptx_grants\"\ + }'", + ) + .await + .expect("set permission tree"); + + let mut probes = Vec::new(); + for node in &cluster.nodes[1..] { + probes.push(connect_probe(node.pg_addr).await); + } + + // Reads that planned rather than being refused. A run of refusals only + // proves nothing was served stale, so the test also requires answers. + let mut answered = 0usize; + for round in 0..ROUNDS { + writer + .exec( + "INSERT INTO ptx_grants (resource_id, grantee, level, inherited) \ + VALUES ('d1', 'ptx_role', 'viewer', false)", + ) + .await + .unwrap_or_else(|e| panic!("round {round}: grant: {e}")); + for (index, (probe, _)) in probes.iter().enumerate() { + let read = probe_read(probe).await; + answered += usize::from(read != Read::Refused); + assert!( + read == Read::Rows(vec!["d1".to_string()]) || read == Read::Refused, + "round {round}: node {} planned against the state before the grant: {read:?}", + index + 1 + ); + } + + writer + .exec("DELETE FROM ptx_grants WHERE resource_id = 'd1' AND grantee = 'ptx_role'") + .await + .unwrap_or_else(|e| panic!("round {round}: revoke: {e}")); + for (index, (probe, _)) in probes.iter().enumerate() { + let read = probe_read(probe).await; + answered += usize::from(read != Read::Refused); + assert!( + read == Read::Rows(Vec::new()) || read == Read::Refused, + "round {round}: node {} served d1 after the revoke was acknowledged: {read:?}", + index + 1 + ); + } + } + + assert!(answered > 0, "every probe read was refused"); + + for (probe, handle) in probes { + drop(probe); + handle.abort(); + } + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs new file mode 100644 index 000000000..4136d21b0 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A partitioned node loses its authorization lease, and a revoke +//! acknowledged meanwhile never plans on it. +//! +//! Node B is severed from the other two nodes. It cannot renew its lease, so +//! the revoke written on another node is acknowledged once B's lease lapses. +//! From then on B refuses every permission-checked statement with a +//! retryable error. After the partition heals, B renews only once its own +//! state covers the revoke, so its first answer shows the revoke. + +use std::time::{Duration, Instant}; + +use crate::common::cluster_harness::TestCluster; + +const SELECT_DOCS: &str = "SELECT id FROM ptl_docs ORDER BY id"; +const AUTHORIZATION_BEHIND: &str = "55P03"; + +#[derive(Debug, PartialEq)] +enum Read { + Rows(Vec), + Refused, +} + +async fn probe_read(client: &tokio_postgres::Client) -> Read { + match client.simple_query(SELECT_DOCS).await { + Ok(messages) => Read::Rows( + messages + .into_iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => { + Some(row.get(0).unwrap_or("").to_string()) + } + _ => None, + }) + .collect(), + ), + Err(error) => { + assert_eq!( + error.code().map(|code| code.code().to_string()).as_deref(), + Some(AUTHORIZATION_BEHIND), + "a probe read failed with an error other than a retryable refusal: {error:?}" + ); + Read::Refused + } + } +} + +async fn connect_probe( + pg_addr: std::net::SocketAddr, +) -> (tokio_postgres::Client, tokio::task::JoinHandle<()>) { + let conn_str = format!( + "host={} port={} user=ptl_probe dbname=default", + pg_addr.ip(), + pg_addr.port() + ); + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .unwrap_or_else(|e| panic!("connect as ptl_probe to {pg_addr}: {e}")); + let handle = tokio::spawn(async move { + let _ = connection.await; + }); + (client, handle) +} + +/// Sever node `b` from every other node, both ways, or heal it. +fn partition(cluster: &TestCluster, b: usize, severed: bool) { + let b_id = cluster.nodes[b].node_id; + let b_transport = cluster.nodes[b] + .shared + .cluster_transport + .as_ref() + .expect("cluster transport"); + for (index, node) in cluster.nodes.iter().enumerate() { + if index == b { + continue; + } + let transport = node + .shared + .cluster_transport + .as_ref() + .expect("cluster transport"); + if severed { + transport.sever(b_id); + b_transport.sever(node.node_id); + } else { + transport.heal(b_id); + b_transport.heal(node.node_id); + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_partitioned_node_refuses_until_it_covers_the_revoke() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + for sql in [ + "CREATE COLLECTION ptl_docs (id TEXT PRIMARY KEY, title TEXT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION ptl_grants", + "CREATE ROLE ptl_role", + // `readwrite` is the built-in role that grants Read on every + // collection. An unknown name such as `read_write` is a custom role + // with no permissions, and every read would be denied. + "CREATE USER ptl_probe WITH PASSWORD 'ptl-probe-password' ROLE readwrite", + "GRANT ROLE ptl_role TO ptl_probe", + ] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + // B leads neither the metadata group nor the group homing the grants, + // so both keep a quorum while B is cut off. + let grants_group = cluster.nodes[0] + .group_id_for_collection("ptl_grants") + .expect("grants group"); + let metadata_leader = cluster.nodes[0].metadata_group_leader(); + let grants_leader = cluster.nodes[0] + .all_group_leaders() + .into_iter() + .find(|(group, _)| *group == grants_group) + .map(|(_, leader)| leader) + .unwrap_or(0); + let b = cluster + .nodes + .iter() + .position(|node| node.node_id != metadata_leader && node.node_id != grants_leader) + .expect("a node that leads neither group"); + let writer = &cluster.nodes[(b + 1) % cluster.nodes.len()]; + + writer + .exec("INSERT INTO ptl_docs (id, title) VALUES ('d1', 'Doc One')") + .await + .expect("insert d1"); + + // The probe's base Read is in effect on every node, B included, before + // the tree narrows it. A denial here is a setup error, not a tree result. + for (index, node) in cluster.nodes.iter().enumerate() { + let (probe, handle) = connect_probe(node.pg_addr).await; + let deadline = Instant::now() + Duration::from_secs(10); + loop { + let read = probe_read(&probe).await; + if read == Read::Rows(vec!["d1".to_string()]) { + break; + } + assert!( + Instant::now() < deadline, + "node {index}: base Read before the tree: last read {read:?}" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + drop(probe); + handle.abort(); + } + + cluster + .exec_ddl_on_any_leader( + "ALTER COLLECTION ptl_docs SET PERMISSION_TREE = '{\ + \"resource_column\":\"id\",\ + \"graph_index\":\"ptl_docs_tree\",\ + \"permission_table\":\"ptl_grants\"\ + }'", + ) + .await + .expect("set permission tree"); + writer + .exec( + "INSERT INTO ptl_grants (resource_id, grantee, level, inherited) \ + VALUES ('d1', 'ptl_role', 'viewer', false)", + ) + .await + .expect("grant d1"); + + let (probe, probe_handle) = connect_probe(cluster.nodes[b].pg_addr).await; + let deadline = Instant::now() + Duration::from_secs(10); + loop { + match probe_read(&probe).await { + Read::Rows(rows) if rows == vec!["d1".to_string()] => break, + Read::Rows(rows) => panic!("node B served {rows:?} after the grant was acknowledged"), + Read::Refused => {} + } + assert!(Instant::now() < deadline, "node B never served the grant"); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + partition(&cluster, b, true); + writer + .exec("DELETE FROM ptl_grants WHERE resource_id = 'd1' AND grantee = 'ptl_role'") + .await + .expect("the revoke is acknowledged once B's lease lapses"); + + // B's lease has lapsed: it refuses rather than plan against the grant. + for _ in 0..5 { + assert_eq!( + probe_read(&probe).await, + Read::Refused, + "the partitioned node planned without a lease" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + partition(&cluster, b, false); + let deadline = Instant::now() + Duration::from_secs(20); + loop { + match probe_read(&probe).await { + Read::Rows(rows) => { + assert!(rows.is_empty(), "node B served the revoked grant: {rows:?}"); + break; + } + Read::Refused => {} + } + assert!( + Instant::now() < deadline, + "node B never renewed after the partition healed" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + drop(probe); + probe_handle.abort(); + cluster.shutdown().await; +} diff --git a/nodedb-cluster/src/calvin/applied_acks.rs b/nodedb-cluster/src/calvin/applied_acks.rs new file mode 100644 index 000000000..ff9ad7893 --- /dev/null +++ b/nodedb-cluster/src/calvin/applied_acks.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Completion acks this node applied from the sequencer log, with their +//! Raft index. +//! +//! A participant's scheduler on the vShard leader proposes a `CompletionAck` +//! once it applied a transaction. Every other replica of that vShard applies +//! the transaction in its own time. A node that serves authorization state +//! from its replicas needs to know, for each ack in the log, whether its own +//! replica applied the transaction too. The sequencer state machine records +//! each ack here as it applies it. The host crate drains the log and settles +//! each ack against its local schedulers. +//! +//! The log records nothing until the host enables it, so a node that never +//! drains it holds no entries. + +use std::collections::VecDeque; +use std::sync::Mutex; + +use super::completion::TxnId; + +/// One `CompletionAck` applied from the sequencer log. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct AppliedCompletionAck { + /// Raft index of the ack in the sequencer group. + pub index: u64, + pub txn: TxnId, + pub vshard_id: u32, +} + +/// Applied acks not yet drained by the host. `None` until enabled. +#[derive(Debug, Default)] +pub struct AppliedAckLog { + acks: Mutex>>, +} + +impl AppliedAckLog { + /// Start recording applied acks. + pub fn enable(&self) { + let mut acks = self.acks.lock().unwrap_or_else(|p| p.into_inner()); + if acks.is_none() { + *acks = Some(VecDeque::new()); + } + } + + /// Record an ack the sequencer state machine applied at `index`. + pub fn record(&self, ack: AppliedCompletionAck) { + if let Some(acks) = self.acks.lock().unwrap_or_else(|p| p.into_inner()).as_mut() { + acks.push_back(ack); + } + } + + /// Take every recorded ack, in log order. + pub fn drain(&self) -> Vec { + self.acks + .lock() + .unwrap_or_else(|p| p.into_inner()) + .as_mut() + .map(|acks| acks.drain(..).collect()) + .unwrap_or_default() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn ack(index: u64) -> AppliedCompletionAck { + AppliedCompletionAck { + index, + txn: TxnId::new(index, 0), + vshard_id: 3, + } + } + + #[test] + fn nothing_is_recorded_until_enabled() { + let log = AppliedAckLog::default(); + log.record(ack(1)); + assert!(log.drain().is_empty()); + log.enable(); + log.record(ack(2)); + log.record(ack(3)); + assert_eq!(log.drain(), vec![ack(2), ack(3)]); + assert!(log.drain().is_empty()); + } +} diff --git a/nodedb-cluster/src/calvin/completion.rs b/nodedb-cluster/src/calvin/completion.rs index f2c3fb1e7..4f27a7532 100644 --- a/nodedb-cluster/src/calvin/completion.rs +++ b/nodedb-cluster/src/calvin/completion.rs @@ -194,6 +194,8 @@ pub struct CalvinCompletionRegistry { /// the signal into a `SequencerEntry::Verdict` proposal. `pub(crate)`: also /// used by `note_vote` in `completion_verdict.rs`. pub(crate) verdict_tx: mpsc::Sender<(TxnId, VerdictOutcome)>, + /// Completion acks this node applied, with their Raft index. + pub applied_acks: super::applied_acks::AppliedAckLog, } impl CalvinCompletionRegistry { @@ -204,6 +206,7 @@ impl CalvinCompletionRegistry { Arc::new(Self { inner: Mutex::new(Inner::default()), verdict_tx, + applied_acks: super::applied_acks::AppliedAckLog::default(), }) } diff --git a/nodedb-cluster/src/calvin/mod.rs b/nodedb-cluster/src/calvin/mod.rs index 3b24f07ad..659690779 100644 --- a/nodedb-cluster/src/calvin/mod.rs +++ b/nodedb-cluster/src/calvin/mod.rs @@ -1,10 +1,12 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod applied_acks; pub mod completion; mod completion_verdict; pub mod sequencer; pub mod types; +pub use applied_acks::{AppliedAckLog, AppliedCompletionAck}; pub use completion::{ AttemptOutcome, CalvinCompletionRegistry, ParticipantProgress, ParticipantVote, TxnId, VerdictOutcome, diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs index b7d239d50..df81c52b1 100644 --- a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs @@ -268,8 +268,15 @@ impl SequencerStateMachine { position, vshard_id, } => { + let txn = crate::calvin::TxnId::new(epoch, position); + self.completion_registry.note_completion_ack(txn, vshard_id); self.completion_registry - .note_completion_ack(crate::calvin::TxnId::new(epoch, position), vshard_id); + .applied_acks + .record(crate::calvin::AppliedCompletionAck { + index, + txn, + vshard_id, + }); } // Broadcast the OLLP predicate-mismatch signal to ALL replicas so the // coordinator's registry fires wherever it lives (including remote nodes). diff --git a/nodedb-cluster/src/lib.rs b/nodedb-cluster/src/lib.rs index e17427a5a..0464926ea 100644 --- a/nodedb-cluster/src/lib.rs +++ b/nodedb-cluster/src/lib.rs @@ -108,8 +108,8 @@ pub use migration_executor::{ }; pub use multi_raft::{GroupStatus, MultiRaft}; pub use raft_loop::{ - AssignRemoteSurrogate, CalvinSubmit, CalvinSubmitInbox, CommitApplier, RaftLoop, - ReleaseReservation, ReserveRead, ShuffleAggregator, ShuffleConsumer, ShuffleProducer, + AssignRemoteSurrogate, AuthLeaseService, CalvinSubmit, CalvinSubmitInbox, CommitApplier, + RaftLoop, ReleaseReservation, ReserveRead, ShuffleAggregator, ShuffleConsumer, ShuffleProducer, ShuffleReceiver, SnapshotApplier, SnapshotBuilder, SnapshotQuarantineHook, VShardEnvelopeHandler, }; @@ -126,13 +126,14 @@ pub use rebalancer::{ pub use routing::RoutingTable; pub use routing_liveness::{NodeIdResolver, RoutingLivenessHook}; pub use rpc_codec::{ - AssignSurrogateRequest, AssignSurrogateResponse, DataPlaneErrorCode, JoinKeyPair, MacKey, - PartNodeEntry, RaftRpc, ReleaseReservationRequest, ReleaseReservationResponse, - ReserveReadRequest, ReserveReadResponse, ShuffleAggregateConsumeRequest, - ShuffleAggregateConsumeResponse, ShuffleConsumeRequest, ShuffleConsumeResponse, - ShuffleProduceRequest, ShuffleProduceResponse, ShufflePushChunk, ShufflePushEnd, - ShufflePushRequest, SortKey, SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, - SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, TypedClusterError, + AssignSurrogateRequest, AssignSurrogateResponse, AuthBarrierOutcome, AuthBarrierRequest, + AuthBarrierResponse, AuthLeaseRenewOutcome, AuthLeaseRenewRequest, AuthLeaseRenewResponse, + DataPlaneErrorCode, GroupCoverage, JoinKeyPair, MacKey, PartNodeEntry, RaftRpc, + ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, + ShuffleAggregateConsumeRequest, ShuffleAggregateConsumeResponse, ShuffleConsumeRequest, + ShuffleConsumeResponse, ShuffleProduceRequest, ShuffleProduceResponse, ShufflePushChunk, + ShufflePushEnd, ShufflePushRequest, SortKey, SubmitCalvinInboxRequest, + SubmitCalvinInboxResponse, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, TypedClusterError, }; pub use topology::{ClusterTopology, NodeInfo, NodeState}; pub use transport::{ diff --git a/nodedb-cluster/src/multi_raft/core.rs b/nodedb-cluster/src/multi_raft/core.rs index bef7d79ae..d8c201b96 100644 --- a/nodedb-cluster/src/multi_raft/core.rs +++ b/nodedb-cluster/src/multi_raft/core.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! `MultiRaft` struct, constructors, group lifecycle, tick, observability. +//! `MultiRaft` struct, constructors, group lifecycle and tick. use std::collections::HashMap; use std::path::PathBuf; @@ -16,40 +16,6 @@ use crate::error::{ClusterError, Result}; use crate::raft_storage::RedbLogStorage; use crate::routing::RoutingTable; -/// Snapshot of a single Raft group's state for observability. -#[derive(Debug, Clone, serde::Serialize)] -pub struct GroupStatus { - pub group_id: u64, - /// Role as a human-readable string ("Leader", "Follower", "Candidate", "Learner"). - pub role: String, - pub leader_id: u64, - pub term: u64, - pub commit_index: u64, - pub last_applied: u64, - pub last_log_index: u64, - /// Highest log index covered by the latest compacted snapshot. - /// Advances when the group's log is compacted past the start (gated - /// by `RaftConfig::log_compaction_threshold`). A non-zero value - /// means entries at or below it are no longer in the log and a - /// lagging peer below this index can only be caught up via - /// `InstallSnapshot`, never `AppendEntries`. - pub snapshot_index: u64, - pub member_count: usize, - pub learner_count: usize, - pub vshard_count: usize, -} - -/// Membership snapshot for a hosted Raft group. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct GroupMembership { - pub group_id: u64, - pub leader_id: u64, - /// Voting members, including this node when it is a voter. - pub voters: Vec, - /// Non-voting learners, including this node when it is a learner. - pub learners: Vec, -} - /// Multi-Raft coordinator managing multiple Raft groups on a single node. /// /// This coordinator: @@ -157,6 +123,17 @@ impl MultiRaft { self } + /// The shortest election timeout a group on this node waits before it + /// campaigns. + pub fn election_timeout_min(&self) -> Duration { + self.election_timeout_min + } + + /// How often a leader on this node sends heartbeats. + pub fn heartbeat_interval(&self) -> Duration { + self.heartbeat_interval + } + /// Configure the auto-compaction threshold for every group created on /// this node. `None` disables auto-compaction (the default). See /// [`RaftConfig::log_compaction_threshold`]. @@ -299,207 +276,10 @@ impl MultiRaft { ids } - /// Snapshot the actual Raft membership rather than the vShard routing view. - pub fn group_membership(&self, group_id: u64) -> Option { - let node = self.groups.get(&group_id)?; - let mut voters = node.voters().to_vec(); - let mut learners = node.learners().to_vec(); - match node.role() { - nodedb_raft::NodeRole::Learner => learners.push(self.node_id), - nodedb_raft::NodeRole::Observer => {} - _ => voters.push(self.node_id), - } - voters.sort_unstable(); - voters.dedup(); - learners.sort_unstable(); - learners.dedup(); - Some(GroupMembership { - group_id, - leader_id: node.leader_id(), - voters, - learners, - }) - } - /// Mutable access to the underlying Raft groups (for testing / bootstrap). pub fn groups_mut(&mut self) -> &mut HashMap> { &mut self.groups } - - /// Snapshot of all Raft group states for observability. - pub fn group_statuses(&self) -> Vec { - let mut statuses = Vec::with_capacity(self.groups.len()); - for (&group_id, node) in &self.groups { - let vshard_count = self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .vshards_for_group(group_id) - .len(); - let self_is_voter = !matches!( - node.role(), - nodedb_raft::NodeRole::Learner | nodedb_raft::NodeRole::Observer - ); - - statuses.push(GroupStatus { - group_id, - role: format!("{:?}", node.role()), - leader_id: node.leader_id(), - term: node.current_term(), - commit_index: node.commit_index(), - last_applied: node.last_applied(), - last_log_index: node.last_log_index(), - snapshot_index: node.log_snapshot_index(), - member_count: node.voters().len() + usize::from(self_is_voter), - learner_count: node.learners().len() - + usize::from(node.role() == nodedb_raft::NodeRole::Learner), - vshard_count, - }); - } - statuses.sort_by_key(|s| s.group_id); - statuses - } - - /// Get the leader for a given vShard (from local group state). - pub fn leader_for_vshard(&self, vshard_id: u32) -> Result> { - let group_id = self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .group_for_vshard(vshard_id)?; - let node = self - .groups - .get(&group_id) - .ok_or(ClusterError::GroupNotFound { group_id })?; - let lid = node.leader_id(); - Ok(if lid == 0 { None } else { Some(lid) }) - } - - /// Whether THIS node is currently the leader of the data-group that owns - /// `vshard_id`. - /// - /// Maps the vshard to its Raft group via the routing table and reuses the - /// existing local leader-role check — no new election. Returns `false` when - /// the vshard has no group mapping or this node is a follower/learner for - /// the owning group. Used by the Calvin scheduler to stamp the per-node, - /// non-replicated `is_group_leader` dispatch flag so the OLLP optimistic-lock - /// verification runs only on the leader while every replica applies the same - /// predicted write-set (determinism). - pub fn vshard_role_is_leader(&self, vshard_id: u32) -> bool { - match self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .group_for_vshard(vshard_id) - { - Ok(group_id) => self.is_group_leader(group_id), - Err(_) => false, - } - } - - /// Propose a command to the Raft group that owns the given vShard. - /// - /// Returns `(group_id, log_index)` on success. - pub fn propose(&mut self, vshard_id: u32, data: Vec) -> Result<(u64, u64)> { - let group_id = self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .group_for_vshard(vshard_id)?; - let node = self - .groups - .get_mut(&group_id) - .ok_or(ClusterError::GroupNotFound { group_id })?; - let log_index = node.propose(data)?; - Ok((group_id, log_index)) - } - - /// Returns `true` if this node is currently the leader of `group_id`. - /// - /// Returns `false` when the group does not exist on this node or when the - /// node is a follower, candidate, or learner in the group. - pub fn is_group_leader(&self, group_id: u64) -> bool { - use nodedb_raft::state::NodeRole; - self.groups - .get(&group_id) - .map(|n| n.role() == NodeRole::Leader) - .unwrap_or(false) - } - - /// Propose a command directly to a specific Raft group (e.g. the - /// metadata group, which has no vShard mapping). - /// - /// Returns the committed log index on success. - pub fn propose_to_group(&mut self, group_id: u64, data: Vec) -> Result { - let node = self - .groups - .get_mut(&group_id) - .ok_or(ClusterError::GroupNotFound { group_id })?; - Ok(node.propose(data)?) - } - - /// Read committed log entries for a Raft group in the inclusive index - /// range `[lo, hi]`. - /// - /// `hi` is clamped to the group's `commit_index` so callers that pass - /// `u64::MAX` never read uncommitted entries. - /// - /// Used by the Calvin scheduler's rebuild path to replay sequenced - /// transactions from the sequencer Raft log after a restart. - /// - /// Returns `Err(ClusterError::Raft(RaftError::LogCompacted))` if `lo` - /// has been compacted into a snapshot (caller must install a snapshot - /// instead of replaying from log). - pub fn read_committed_entries( - &self, - group_id: u64, - lo: u64, - hi: u64, - ) -> Result> { - let node = self - .groups - .get(&group_id) - .ok_or(ClusterError::GroupNotFound { group_id })?; - let entries = node.log_entries_range(lo, hi)?; - Ok(entries.to_vec()) - } - - /// The lowest committed index still available in `group_id`'s retained log - /// (`snapshot_index + 1`), or `None` when the group is absent on this node. - /// - /// Used to arm a Calvin scheduler catch-up from the earliest replayable - /// sequencer index so its drain reads exactly the retained log and never - /// faults on a compacted range. - pub fn first_available_index(&self, group_id: u64) -> Option { - self.groups - .get(&group_id) - .map(|n| n.first_available_index()) - } - - /// Auto-compact a group's log if its configured threshold has been - /// reached, given the DATA-PLANE applied watermark `applied_index`. - /// - /// `applied_index` MUST be the index the data-plane state machine has - /// durably applied to (NOT raft's commit index). Compacting past an - /// unapplied index would let the `SnapshotBuilder` serialize - /// incomplete state and corrupt a lagging follower's snapshot. - /// - /// No-op (returns `Ok(false)`) when the group is absent on this node, - /// the threshold is `None`, or the retained-entry count is below the - /// threshold. Returns `Ok(true)` when a compaction was performed. - pub fn maybe_compact_group(&mut self, group_id: u64, applied_index: u64) -> Result { - // Defer compaction while a snapshot transfer for this group is in - // flight: advancing the snapshot boundary mid-transfer would corrupt - // the catching-up peer. The apply loop retries on the next applied - // entry, so the watermark still advances once the transfer completes. - if self.in_flight_snapshots.is_active(group_id) { - return Ok(false); - } - let Some(node) = self.groups.get_mut(&group_id) else { - return Ok(false); - }; - Ok(node.maybe_compact_log(applied_index)?) - } } // Re-export LogEntry so callers of `read_committed_entries` can name the type. @@ -508,6 +288,7 @@ pub use nodedb_raft::LogEntry; #[cfg(test)] mod tests { use super::*; + use crate::multi_raft::status::GroupMembership; use std::time::Instant; #[test] diff --git a/nodedb-cluster/src/multi_raft/mod.rs b/nodedb-cluster/src/multi_raft/mod.rs index e05a41916..e2c3ddc55 100644 --- a/nodedb-cluster/src/multi_raft/mod.rs +++ b/nodedb-cluster/src/multi_raft/mod.rs @@ -6,8 +6,10 @@ //! //! Split across files: //! - [`core`]: struct, constructors, group lifecycle (`add_group`, -//! `add_group_as_learner`), tick pipeline, routing accessors, -//! observability (`group_statuses`). +//! `add_group_as_learner`), tick pipeline, routing accessors. +//! - [`status`]: observability snapshots (`group_statuses`, +//! `group_membership`). +//! - [`proposals`]: leadership checks, proposals and log access. //! - [`rpc_dispatch`]: inbound RPC routing to the correct group and the //! corresponding response handlers. //! - [`conf_change`]: `propose_conf_change` / `apply_conf_change` with @@ -19,7 +21,10 @@ pub mod conf_change; pub mod core; pub mod membership; +pub mod proposals; pub mod read_index; pub mod rpc_dispatch; +pub mod status; -pub use core::{GroupStatus, MultiRaft, MultiRaftReady}; +pub use core::{MultiRaft, MultiRaftReady}; +pub use status::{GroupMembership, GroupStatus}; diff --git a/nodedb-cluster/src/multi_raft/proposals.rs b/nodedb-cluster/src/multi_raft/proposals.rs new file mode 100644 index 000000000..5dc69b9c0 --- /dev/null +++ b/nodedb-cluster/src/multi_raft/proposals.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Leadership checks, proposals and log access on hosted Raft groups. + +use crate::error::{ClusterError, Result}; +use crate::multi_raft::core::MultiRaft; + +impl MultiRaft { + /// Get the leader for a given vShard (from local group state). + pub fn leader_for_vshard(&self, vshard_id: u32) -> Result> { + let group_id = self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard_id)?; + let node = self + .groups + .get(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + let lid = node.leader_id(); + Ok(if lid == 0 { None } else { Some(lid) }) + } + + /// Whether THIS node is currently the leader of the data-group that owns + /// `vshard_id`. + /// + /// Maps the vshard to its Raft group via the routing table and reuses the + /// existing local leader-role check — no new election. Returns `false` when + /// the vshard has no group mapping or this node is a follower/learner for + /// the owning group. Used by the Calvin scheduler to stamp the per-node, + /// non-replicated `is_group_leader` dispatch flag so the OLLP optimistic-lock + /// verification runs only on the leader while every replica applies the same + /// predicted write-set (determinism). + pub fn vshard_role_is_leader(&self, vshard_id: u32) -> bool { + match self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard_id) + { + Ok(group_id) => self.is_group_leader(group_id), + Err(_) => false, + } + } + + /// Propose a command to the Raft group that owns the given vShard. + /// + /// Returns `(group_id, log_index)` on success. + pub fn propose(&mut self, vshard_id: u32, data: Vec) -> Result<(u64, u64)> { + let group_id = self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard_id)?; + let node = self + .groups + .get_mut(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + let log_index = node.propose(data)?; + Ok((group_id, log_index)) + } + + /// Returns `true` if this node is currently the leader of `group_id`. + /// + /// Returns `false` when the group does not exist on this node or when the + /// node is a follower, candidate, or learner in the group. + pub fn is_group_leader(&self, group_id: u64) -> bool { + use nodedb_raft::state::NodeRole; + self.groups + .get(&group_id) + .map(|n| n.role() == NodeRole::Leader) + .unwrap_or(false) + } + + /// Propose a command directly to a specific Raft group (e.g. the + /// metadata group, which has no vShard mapping). + /// + /// Returns the committed log index on success. + pub fn propose_to_group(&mut self, group_id: u64, data: Vec) -> Result { + let node = self + .groups + .get_mut(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + Ok(node.propose(data)?) + } + + /// Read committed log entries for a Raft group in the inclusive index + /// range `[lo, hi]`. + /// + /// `hi` is clamped to the group's `commit_index` so callers that pass + /// `u64::MAX` never read uncommitted entries. + /// + /// Used by the Calvin scheduler's rebuild path to replay sequenced + /// transactions from the sequencer Raft log after a restart. + /// + /// Returns `Err(ClusterError::Raft(RaftError::LogCompacted))` if `lo` + /// has been compacted into a snapshot (caller must install a snapshot + /// instead of replaying from log). + pub fn read_committed_entries( + &self, + group_id: u64, + lo: u64, + hi: u64, + ) -> Result> { + let node = self + .groups + .get(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + let entries = node.log_entries_range(lo, hi)?; + Ok(entries.to_vec()) + } + + /// The lowest committed index still available in `group_id`'s retained log + /// (`snapshot_index + 1`), or `None` when the group is absent on this node. + /// + /// Used to arm a Calvin scheduler catch-up from the earliest replayable + /// sequencer index so its drain reads exactly the retained log and never + /// faults on a compacted range. + pub fn first_available_index(&self, group_id: u64) -> Option { + self.groups + .get(&group_id) + .map(|n| n.first_available_index()) + } + + /// Auto-compact a group's log if its configured threshold has been + /// reached, given the DATA-PLANE applied watermark `applied_index`. + /// + /// `applied_index` MUST be the index the data-plane state machine has + /// durably applied to (NOT raft's commit index). Compacting past an + /// unapplied index would let the `SnapshotBuilder` serialize + /// incomplete state and corrupt a lagging follower's snapshot. + /// + /// No-op (returns `Ok(false)`) when the group is absent on this node, + /// the threshold is `None`, or the retained-entry count is below the + /// threshold. Returns `Ok(true)` when a compaction was performed. + pub fn maybe_compact_group(&mut self, group_id: u64, applied_index: u64) -> Result { + // Defer compaction while a snapshot transfer for this group is in + // flight: advancing the snapshot boundary mid-transfer would corrupt + // the catching-up peer. The apply loop retries on the next applied + // entry, so the watermark still advances once the transfer completes. + if self.in_flight_snapshots.is_active(group_id) { + return Ok(false); + } + let Some(node) = self.groups.get_mut(&group_id) else { + return Ok(false); + }; + Ok(node.maybe_compact_log(applied_index)?) + } +} diff --git a/nodedb-cluster/src/multi_raft/status.rs b/nodedb-cluster/src/multi_raft/status.rs new file mode 100644 index 000000000..081fbff1e --- /dev/null +++ b/nodedb-cluster/src/multi_raft/status.rs @@ -0,0 +1,97 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Observability snapshots of the Raft groups hosted on this node. + +use crate::multi_raft::core::MultiRaft; + +/// Snapshot of a single Raft group's state for observability. +#[derive(Debug, Clone, serde::Serialize)] +pub struct GroupStatus { + pub group_id: u64, + /// Role as a human-readable string ("Leader", "Follower", "Candidate", "Learner"). + pub role: String, + pub leader_id: u64, + pub term: u64, + pub commit_index: u64, + pub last_applied: u64, + pub last_log_index: u64, + /// Highest log index covered by the latest compacted snapshot. + /// Advances when the group's log is compacted past the start (gated + /// by `RaftConfig::log_compaction_threshold`). A non-zero value + /// means entries at or below it are no longer in the log and a + /// lagging peer below this index can only be caught up via + /// `InstallSnapshot`, never `AppendEntries`. + pub snapshot_index: u64, + pub member_count: usize, + pub learner_count: usize, + pub vshard_count: usize, +} + +/// Membership snapshot for a hosted Raft group. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GroupMembership { + pub group_id: u64, + pub leader_id: u64, + /// Voting members, including this node when it is a voter. + pub voters: Vec, + /// Non-voting learners, including this node when it is a learner. + pub learners: Vec, +} + +impl MultiRaft { + /// Snapshot the actual Raft membership rather than the vShard routing view. + pub fn group_membership(&self, group_id: u64) -> Option { + let node = self.groups.get(&group_id)?; + let mut voters = node.voters().to_vec(); + let mut learners = node.learners().to_vec(); + match node.role() { + nodedb_raft::NodeRole::Learner => learners.push(self.node_id), + nodedb_raft::NodeRole::Observer => {} + _ => voters.push(self.node_id), + } + voters.sort_unstable(); + voters.dedup(); + learners.sort_unstable(); + learners.dedup(); + Some(GroupMembership { + group_id, + leader_id: node.leader_id(), + voters, + learners, + }) + } + + /// Snapshot of all Raft group states for observability. + pub fn group_statuses(&self) -> Vec { + let mut statuses = Vec::with_capacity(self.groups.len()); + for (&group_id, node) in &self.groups { + let vshard_count = self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .vshards_for_group(group_id) + .len(); + let self_is_voter = !matches!( + node.role(), + nodedb_raft::NodeRole::Learner | nodedb_raft::NodeRole::Observer + ); + + statuses.push(GroupStatus { + group_id, + role: format!("{:?}", node.role()), + leader_id: node.leader_id(), + term: node.current_term(), + commit_index: node.commit_index(), + last_applied: node.last_applied(), + last_log_index: node.last_log_index(), + snapshot_index: node.log_snapshot_index(), + member_count: node.voters().len() + usize::from(self_is_voter), + learner_count: node.learners().len() + + usize::from(node.role() == nodedb_raft::NodeRole::Learner), + vshard_count, + }); + } + statuses.sort_by_key(|s| s.group_id); + statuses + } +} diff --git a/nodedb-cluster/src/raft_loop/auth_lease.rs b/nodedb-cluster/src/raft_loop/auth_lease.rs new file mode 100644 index 000000000..be6e87e6f --- /dev/null +++ b/nodedb-cluster/src/raft_loop/auth_lease.rs @@ -0,0 +1,37 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Answer authorization lease renewals and barriers through the host hook. + +use crate::error::Result; +use crate::forward::PlanExecutor; +use crate::rpc_codec::{ + AuthBarrierOutcome, AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewOutcome, + AuthLeaseRenewRequest, AuthLeaseRenewResponse, RaftRpc, +}; + +use super::loop_core::{CommitApplier, RaftLoop}; + +impl RaftLoop { + pub(super) async fn handle_auth_lease_renew_rpc( + &self, + req: AuthLeaseRenewRequest, + ) -> Result { + let response = match &self.auth_lease { + Some(service) => service.renew(req).await, + None => AuthLeaseRenewResponse { + outcome: AuthLeaseRenewOutcome::NotLeader { leader_hint: None }, + }, + }; + Ok(RaftRpc::AuthLeaseRenewResponse(response)) + } + + pub(super) async fn handle_auth_barrier_rpc(&self, req: AuthBarrierRequest) -> Result { + let response = match &self.auth_lease { + Some(service) => service.barrier(req).await, + None => AuthBarrierResponse { + outcome: AuthBarrierOutcome::NotLeader { leader_hint: None }, + }, + }; + Ok(RaftRpc::AuthBarrierResponse(response)) + } +} diff --git a/nodedb-cluster/src/raft_loop/auth_lease_hook.rs b/nodedb-cluster/src/raft_loop/auth_lease_hook.rs new file mode 100644 index 000000000..5772c2de8 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/auth_lease_hook.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Hook for the authorization lease service. +//! +//! `nodedb-cluster` cannot depend on `nodedb` (circular). The lease table, +//! the coverage rules and the barrier wait live in `nodedb` behind this +//! `Send + Sync` hook. The transport calls it when a lease renewal or an +//! authorization barrier reaches this node. Cluster-only tests leave the +//! `RaftLoop` field `None`, and such a request is answered `NotLeader`. + +use crate::rpc_codec::{ + AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewRequest, AuthLeaseRenewResponse, +}; + +#[async_trait::async_trait] +pub trait AuthLeaseService: Send + Sync + 'static { + /// Grant or withhold the sender's lease from its coverage report. + async fn renew(&self, req: AuthLeaseRenewRequest) -> AuthLeaseRenewResponse; + + /// Hold the answer until no lease holder can plan against state older + /// than the request's targets. + async fn barrier(&self, req: AuthBarrierRequest) -> AuthBarrierResponse; +} diff --git a/nodedb-cluster/src/raft_loop/builder.rs b/nodedb-cluster/src/raft_loop/builder.rs index db681ea80..e3729119a 100644 --- a/nodedb-cluster/src/raft_loop/builder.rs +++ b/nodedb-cluster/src/raft_loop/builder.rs @@ -54,6 +54,7 @@ impl RaftLoop { calvin_submit_inbox: self.calvin_submit_inbox, reserve_read: self.reserve_read, release_reservation: self.release_reservation, + auth_lease: self.auth_lease, snapshot_builder: self.snapshot_builder, snapshot_applier: self.snapshot_applier, partial_snapshots: self.partial_snapshots, @@ -152,6 +153,17 @@ impl RaftLoop { self } + /// Attach the authorization lease service (builder chain). Lease + /// renewals and authorization barriers reaching this node are answered + /// through it. + pub fn with_auth_lease( + mut self, + service: Arc, + ) -> Self { + self.auth_lease = Some(service); + self + } + /// Attach the routed Calvin-submit hook (Cv1, builder chain). /// /// The supplied implementation (backed by `nodedb`'s Calvin sequencer inbox diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs index c66d43cc0..775dbe7a1 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs @@ -41,6 +41,11 @@ impl RaftRpcHandler for RaftLoop { RaftRpc::MetadataProposeRequest(req) => self.handle_metadata_propose_rpc(req), // Data-group proposal forwarding. RaftRpc::DataProposeRequest(req) => self.handle_data_propose_rpc(req), + // Read index for a node that does not lead the group. + RaftRpc::ReadIndexRequest(req) => self.handle_read_index_rpc(req).await, + // Authorization lease renewal and barrier, answered by the host hook. + RaftRpc::AuthLeaseRenewRequest(req) => self.handle_auth_lease_renew_rpc(req).await, + RaftRpc::AuthBarrierRequest(req) => self.handle_auth_barrier_rpc(req).await, // VShardEnvelope — dispatch to registered handler (Event Plane, etc.). RaftRpc::VShardEnvelope(bytes) => self.handle_vshard_envelope_rpc(bytes).await, other => Err(ClusterError::Transport { diff --git a/nodedb-cluster/src/raft_loop/loop_core.rs b/nodedb-cluster/src/raft_loop/loop_core.rs index a6ff5b419..6c1bf03b0 100644 --- a/nodedb-cluster/src/raft_loop/loop_core.rs +++ b/nodedb-cluster/src/raft_loop/loop_core.rs @@ -231,6 +231,11 @@ pub struct RaftLoop { /// configured" error. pub(super) release_reservation: Option>, + /// Optional authorization lease service. When set (by the `nodedb` + /// binary via `with_auth_lease`), lease renewals and authorization + /// barriers are answered through it. + pub(super) auth_lease: Option>, + /// Optional per-group snapshot builder for the SEND path. /// /// When set (by the `nodedb` binary via `with_snapshot_builder`), the @@ -334,6 +339,7 @@ impl RaftLoop { calvin_submit_inbox: None, reserve_read: None, release_reservation: None, + auth_lease: None, snapshot_builder: None, snapshot_applier: None, partial_snapshots: Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())), diff --git a/nodedb-cluster/src/raft_loop/mod.rs b/nodedb-cluster/src/raft_loop/mod.rs index b93c669f9..c4c6119b8 100644 --- a/nodedb-cluster/src/raft_loop/mod.rs +++ b/nodedb-cluster/src/raft_loop/mod.rs @@ -15,6 +15,8 @@ //! peer, propose `AddLearner` on every group, wait for commit, //! broadcast topology, persist catalog, build the wire response. +mod auth_lease; +pub mod auth_lease_hook; mod builder; pub mod handle_rpc; pub mod hooks; @@ -26,8 +28,10 @@ pub mod loop_core; mod membership_convergence; mod placement_reconcile; pub mod proposals; +mod read_index; pub mod tick; +pub use auth_lease_hook::AuthLeaseService; pub use hooks::{ AssignRemoteSurrogate, CalvinSubmit, CalvinSubmitInbox, ReleaseReservation, ReserveRead, ShuffleAggregator, ShuffleConsumer, ShuffleProducer, ShuffleReceiver, SnapshotApplier, diff --git a/nodedb-cluster/src/raft_loop/read_index.rs b/nodedb-cluster/src/raft_loop/read_index.rs new file mode 100644 index 000000000..e6d6ca56f --- /dev/null +++ b/nodedb-cluster/src/raft_loop/read_index.rs @@ -0,0 +1,109 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Read index for any node: confirmed locally on the leader, asked of the +//! leader over the transport on every other node. +//! +//! A read index is the leader's commit index at a moment a quorum confirmed +//! its leadership. A node whose state machine has applied a group through +//! that index observes every entry committed before the read index was taken. + +use std::time::Duration; + +use crate::error::{ClusterError, Result}; +use crate::forward::PlanExecutor; +use crate::read_index_wait::confirm_read_index; +use crate::rpc_codec::{RaftRpc, ReadIndexOutcome, ReadIndexRequest, ReadIndexResponse}; + +use super::loop_core::{CommitApplier, RaftLoop}; + +/// Upper bound on the quorum wait a remote node may ask for. +const MAX_REMOTE_TIMEOUT: Duration = Duration::from_secs(10); + +impl RaftLoop { + /// Obtain a read index for `group_id`. + /// + /// On the group leader, confirms leadership against a quorum. On any other + /// node, asks the known leader. Fails with + /// [`ClusterError::ReadIndexNotLeader`] when no leader is known or the + /// asked node no longer leads, and with [`ClusterError::ReadIndexTimeout`] + /// when no quorum answered within `timeout`. + pub async fn read_index_via_leader(&self, group_id: u64, timeout: Duration) -> Result { + let leader = { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if mr.is_group_leader(group_id) { + None + } else { + Some(mr.group_leader(group_id)) + } + }; + let leader_id = match leader { + None => return confirm_read_index(&self.multi_raft, group_id, timeout).await, + Some(id) if id == 0 || id == self.node_id => { + return Err(ClusterError::ReadIndexNotLeader { group_id }); + } + Some(id) => id, + }; + self.register_peer_addr(leader_id)?; + let request = RaftRpc::ReadIndexRequest(ReadIndexRequest { + group_id, + timeout_ms: u64::try_from(timeout.as_millis()).unwrap_or(u64::MAX), + }); + match self.transport.send_rpc(leader_id, request).await? { + RaftRpc::ReadIndexResponse(ReadIndexResponse { outcome }) => match outcome { + ReadIndexOutcome::Confirmed { read_index } => Ok(read_index), + ReadIndexOutcome::NotLeader { .. } => { + Err(ClusterError::ReadIndexNotLeader { group_id }) + } + ReadIndexOutcome::Timeout { waited_ms } => Err(ClusterError::ReadIndexTimeout { + group_id, + waited_ms, + }), + }, + other => Err(ClusterError::Transport { + detail: format!("read index: unexpected response variant {other:?}"), + }), + } + } + + /// Answer a remote node's [`ReadIndexRequest`] for a group this node may + /// lead. + pub(super) async fn handle_read_index_rpc(&self, req: ReadIndexRequest) -> Result { + let timeout = Duration::from_millis(req.timeout_ms).min(MAX_REMOTE_TIMEOUT); + let outcome = match confirm_read_index(&self.multi_raft, req.group_id, timeout).await { + Ok(read_index) => ReadIndexOutcome::Confirmed { read_index }, + Err(ClusterError::ReadIndexTimeout { waited_ms, .. }) => { + ReadIndexOutcome::Timeout { waited_ms } + } + Err(_) => { + let hint = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .group_leader(req.group_id); + ReadIndexOutcome::NotLeader { + leader_hint: (hint != 0).then_some(hint), + } + } + }; + Ok(RaftRpc::ReadIndexResponse(ReadIndexResponse { outcome })) + } + + /// Register `node_id`'s listen address with the transport, from the local + /// topology. + fn register_peer_addr(&self, node_id: u64) -> Result<()> { + let topo = self.topology.read().unwrap_or_else(|p| p.into_inner()); + let node = topo + .get_node(node_id) + .ok_or_else(|| ClusterError::Transport { + detail: format!("read index: leader {node_id} not in local topology"), + })?; + let addr = node.socket_addr().ok_or_else(|| ClusterError::Transport { + detail: format!( + "read index: leader {node_id} has unparseable addr {:?}", + node.addr + ), + })?; + self.transport.register_peer(node_id, addr); + Ok(()) + } +} diff --git a/nodedb-cluster/src/raft_loop/tick/apply_committed.rs b/nodedb-cluster/src/raft_loop/tick/apply_committed.rs index 61028aeb9..123d0fd55 100644 --- a/nodedb-cluster/src/raft_loop/tick/apply_committed.rs +++ b/nodedb-cluster/src/raft_loop/tick/apply_committed.rs @@ -72,13 +72,21 @@ impl RaftLoop { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); if let Err(e) = mr.advance_applied(group_id, last_applied) { warn!(group_id, error = %e, "failed to advance applied index"); - } else if group_id == crate::metadata_group::METADATA_GROUP_ID { + } else if group_id == crate::metadata_group::METADATA_GROUP_ID + || group_id == crate::calvin::SEQUENCER_GROUP_ID + { // Metadata group: the metadata applier // applied entries synchronously to redb // before returning, so the apply // watermark is data-visible at this // point. Bump the watcher. // + // Sequencer group: the host applies each + // entry to the sequencer state machine + // inline, before returning. The watcher + // is the sequencer's applied index that + // authorization lease coverage reports. + // // Data groups are NOT bumped here — for // them `applier.apply_committed` only // enqueues entries onto the diff --git a/nodedb-cluster/src/rpc_codec/auth_lease.rs b/nodedb-cluster/src/rpc_codec/auth_lease.rs new file mode 100644 index 000000000..219f8f9c0 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/auth_lease.rs @@ -0,0 +1,221 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Authorization lease wire types and codecs. +//! +//! Every node plans permission-checked statements from its local +//! authorization state only while it holds a lease from the metadata group +//! leader. A node renews its lease with a coverage report: for each Raft group +//! it names, the index through which its local authorization state holds +//! every change. +//! +//! A writer acknowledges an authorization change only after a barrier on the +//! metadata leader releases it. The barrier releases once every node holding +//! an unexpired lease reported coverage of the change, or its lease expired. + +use super::discriminants::*; +use super::header::write_frame; +use super::raft_rpc::RaftRpc; +use crate::error::{ClusterError, Result}; + +/// A holder's claim on one Raft group: its local authorization state holds +/// every change of `group_id` at or below `through`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct GroupCoverage { + pub group_id: u64, + pub through: u64, +} + +/// Renew the sender's authorization lease. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct AuthLeaseRenewRequest { + pub node_id: u64, + /// Coverage of every group the sender knows. A group it does not + /// replicate is reported at `u64::MAX`: the sender never plans against it. + pub coverage: Vec, +} + +/// The leader's answer to an [`AuthLeaseRenewRequest`]. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum AuthLeaseRenewOutcome { + /// The lease runs `lease_ms` from the moment the sender sent the request. + Granted { lease_ms: u64 }, + /// The report does not cover every acknowledged change. The sender's + /// lease is not extended. + Withheld, + /// The receiver does not lead the metadata group. + NotLeader { leader_hint: Option }, +} + +/// Response to an [`AuthLeaseRenewRequest`]. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct AuthLeaseRenewResponse { + pub outcome: AuthLeaseRenewOutcome, +} + +/// Hold the sender's acknowledgement until every lease holder covers +/// `targets`, or its lease expired. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct AuthBarrierRequest { + pub targets: Vec, + /// How long the leader may hold the request. + pub timeout_ms: u64, +} + +/// The leader's answer to an [`AuthBarrierRequest`]. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum AuthBarrierOutcome { + /// No node can plan against state older than the targets. + Released, + /// The receiver does not lead the metadata group. + NotLeader { leader_hint: Option }, + /// The barrier did not release in time. + Timeout { waited_ms: u64 }, +} + +/// Response to an [`AuthBarrierRequest`]. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct AuthBarrierResponse { + pub outcome: AuthBarrierOutcome, +} + +macro_rules! to_bytes { + ($msg:expr) => { + rkyv::to_bytes::($msg) + .map(|b| b.to_vec()) + .map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize: {e}"), + }) + }; +} + +macro_rules! from_bytes { + ($payload:expr, $T:ty, $name:expr) => {{ + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity($payload.len()); + aligned.extend_from_slice($payload); + rkyv::from_bytes::<$T, rkyv::rancor::Error>(&aligned).map_err(|e| ClusterError::Codec { + detail: format!("rkyv deserialize {}: {e}", $name), + }) + }}; +} + +pub(super) fn encode_renew_req(msg: &AuthLeaseRenewRequest, out: &mut Vec) -> Result<()> { + write_frame(RPC_AUTH_LEASE_RENEW_REQ, &to_bytes!(msg)?, out) +} +pub(super) fn encode_renew_resp(msg: &AuthLeaseRenewResponse, out: &mut Vec) -> Result<()> { + write_frame(RPC_AUTH_LEASE_RENEW_RESP, &to_bytes!(msg)?, out) +} +pub(super) fn encode_barrier_req(msg: &AuthBarrierRequest, out: &mut Vec) -> Result<()> { + write_frame(RPC_AUTH_BARRIER_REQ, &to_bytes!(msg)?, out) +} +pub(super) fn encode_barrier_resp(msg: &AuthBarrierResponse, out: &mut Vec) -> Result<()> { + write_frame(RPC_AUTH_BARRIER_RESP, &to_bytes!(msg)?, out) +} + +pub(super) fn decode_renew_req(payload: &[u8]) -> Result { + Ok(RaftRpc::AuthLeaseRenewRequest(from_bytes!( + payload, + AuthLeaseRenewRequest, + "AuthLeaseRenewRequest" + )?)) +} +pub(super) fn decode_renew_resp(payload: &[u8]) -> Result { + Ok(RaftRpc::AuthLeaseRenewResponse(from_bytes!( + payload, + AuthLeaseRenewResponse, + "AuthLeaseRenewResponse" + )?)) +} +pub(super) fn decode_barrier_req(payload: &[u8]) -> Result { + Ok(RaftRpc::AuthBarrierRequest(from_bytes!( + payload, + AuthBarrierRequest, + "AuthBarrierRequest" + )?)) +} +pub(super) fn decode_barrier_resp(payload: &[u8]) -> Result { + Ok(RaftRpc::AuthBarrierResponse(from_bytes!( + payload, + AuthBarrierResponse, + "AuthBarrierResponse" + )?)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + fn roundtrip(rpc: RaftRpc) -> RaftRpc { + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + decode(&encoded, &epoch).expect("decode") + } + + fn coverage() -> Vec { + vec![ + GroupCoverage { + group_id: 0, + through: 41, + }, + GroupCoverage { + group_id: 3, + through: u64::MAX, + }, + ] + } + + #[test] + fn a_renewal_survives_the_wire() { + match roundtrip(RaftRpc::AuthLeaseRenewRequest(AuthLeaseRenewRequest { + node_id: 2, + coverage: coverage(), + })) { + RaftRpc::AuthLeaseRenewRequest(req) => { + assert_eq!(req.node_id, 2); + assert_eq!(req.coverage, coverage()); + } + other => panic!("decoded the wrong variant: {other:?}"), + } + for outcome in [ + AuthLeaseRenewOutcome::Granted { lease_ms: 150 }, + AuthLeaseRenewOutcome::Withheld, + AuthLeaseRenewOutcome::NotLeader { + leader_hint: Some(1), + }, + ] { + match roundtrip(RaftRpc::AuthLeaseRenewResponse(AuthLeaseRenewResponse { + outcome: outcome.clone(), + })) { + RaftRpc::AuthLeaseRenewResponse(resp) => assert_eq!(resp.outcome, outcome), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + } + + #[test] + fn a_barrier_survives_the_wire() { + match roundtrip(RaftRpc::AuthBarrierRequest(AuthBarrierRequest { + targets: coverage(), + timeout_ms: 5000, + })) { + RaftRpc::AuthBarrierRequest(req) => { + assert_eq!(req.targets, coverage()); + assert_eq!(req.timeout_ms, 5000); + } + other => panic!("decoded the wrong variant: {other:?}"), + } + for outcome in [ + AuthBarrierOutcome::Released, + AuthBarrierOutcome::NotLeader { leader_hint: None }, + AuthBarrierOutcome::Timeout { waited_ms: 5000 }, + ] { + match roundtrip(RaftRpc::AuthBarrierResponse(AuthBarrierResponse { + outcome: outcome.clone(), + })) { + RaftRpc::AuthBarrierResponse(resp) => assert_eq!(resp.outcome, outcome), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/discriminants.rs b/nodedb-cluster/src/rpc_codec/discriminants.rs index 5b706aa20..94db52e5b 100644 --- a/nodedb-cluster/src/rpc_codec/discriminants.rs +++ b/nodedb-cluster/src/rpc_codec/discriminants.rs @@ -134,6 +134,24 @@ pub const RPC_RELEASE_RESERVATION_RESP: u8 = 44; pub const RPC_PRE_VOTE_REQ: u8 = 45; pub const RPC_PRE_VOTE_RESP: u8 = 46; +/// Routed read index. A node that does not lead a Raft group sends a +/// `RPC_READ_INDEX_REQ` to the group leader; the leader confirms its +/// leadership against a quorum and replies with exactly one +/// `RPC_READ_INDEX_RESP` carrying its read index, a leader hint, or a +/// timeout. One-shot request/response — no streaming. +pub const RPC_READ_INDEX_REQ: u8 = 47; +pub const RPC_READ_INDEX_RESP: u8 = 48; +/// Authorization lease renewal: a node reports its authorization coverage to +/// the metadata group leader in `RPC_AUTH_LEASE_RENEW_REQ` and receives one +/// `RPC_AUTH_LEASE_RENEW_RESP` granting or withholding its lease. +pub const RPC_AUTH_LEASE_RENEW_REQ: u8 = 49; +pub const RPC_AUTH_LEASE_RENEW_RESP: u8 = 50; +/// Authorization barrier: a writer holds its acknowledgement until the +/// metadata group leader answers `RPC_AUTH_BARRIER_REQ` with one +/// `RPC_AUTH_BARRIER_RESP`. +pub const RPC_AUTH_BARRIER_REQ: u8 = 51; +pub const RPC_AUTH_BARRIER_RESP: u8 = 52; + // VShardMessageType discriminants for distributed array ops (u16, range 80-89). // These mirror `crate::wire::VShardMessageType` repr values and are declared // here so external code can reference them without importing the full enum. diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index 5efaef1d5..1f8826468 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -9,6 +9,7 @@ //! - All wire types re-exported from their sub-modules. pub mod auth_envelope; +pub mod auth_lease; pub mod calvin_submit; pub mod cluster_mgmt; pub mod data_plane_error; @@ -21,6 +22,7 @@ pub mod metadata; pub mod peer_seq; pub mod raft_msgs; pub mod raft_rpc; +pub mod read_index; pub mod reservation; pub mod shuffle; pub mod surrogate; @@ -29,6 +31,10 @@ pub mod vshard; pub use auth_envelope::{ ENVELOPE_OVERHEAD, ENVELOPE_VERSION, EnvelopeFields, parse_envelope, write_envelope, }; +pub use auth_lease::{ + AuthBarrierOutcome, AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewOutcome, + AuthLeaseRenewRequest, AuthLeaseRenewResponse, GroupCoverage, +}; pub use calvin_submit::{ SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, @@ -48,6 +54,7 @@ pub use mac::{MAC_LEN, MacKey}; pub use metadata::{MetadataProposeRequest, MetadataProposeResponse}; pub use peer_seq::{PeerSeqSender, PeerSeqWindow, REPLAY_WINDOW}; pub use raft_rpc::{RaftRpc, decode, encode, frame_size}; +pub use read_index::{ReadIndexOutcome, ReadIndexRequest, ReadIndexResponse}; pub use reservation::{ ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, }; diff --git a/nodedb-cluster/src/rpc_codec/raft_rpc.rs b/nodedb-cluster/src/rpc_codec/raft_rpc.rs index c0b9c7f2a..40d993067 100644 --- a/nodedb-cluster/src/rpc_codec/raft_rpc.rs +++ b/nodedb-cluster/src/rpc_codec/raft_rpc.rs @@ -7,6 +7,9 @@ use nodedb_raft::message::{ PreVoteRequest, PreVoteResponse, RequestVoteRequest, RequestVoteResponse, TimeoutNowRequest, }; +use super::auth_lease::{ + AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewRequest, AuthLeaseRenewResponse, +}; use super::calvin_submit::{ SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, @@ -19,6 +22,7 @@ use super::discriminants::*; use super::execute::{ExecuteRequest, ExecuteResponse, ExecuteStreamChunk, ExecuteStreamEnd}; use super::header::HEADER_SIZE; use super::metadata::{MetadataProposeRequest, MetadataProposeResponse}; +use super::read_index::{ReadIndexRequest, ReadIndexResponse}; use super::reservation::{ ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, }; @@ -29,8 +33,8 @@ use super::shuffle::{ }; use super::surrogate::{AssignSurrogateRequest, AssignSurrogateResponse}; use super::{ - calvin_submit, cluster_mgmt, data_propose, execute, metadata, raft_msgs, reservation, shuffle, - surrogate, vshard, + auth_lease, calvin_submit, cluster_mgmt, data_propose, execute, metadata, raft_msgs, + read_index, reservation, shuffle, surrogate, vshard, }; use crate::error::{ClusterError, Result}; use crate::wire_version::{unwrap_bytes_versioned, wrap_bytes_versioned}; @@ -143,6 +147,16 @@ pub enum RaftRpc { // Data-group proposal forwarding (groups 1+) DataProposeRequest(DataProposeRequest), DataProposeResponse(DataProposeResponse), + // Routed read index. A node that does not lead a group asks the leader + // for a read index confirmed against a quorum. + ReadIndexRequest(ReadIndexRequest), + ReadIndexResponse(ReadIndexResponse), + // Authorization lease renewal and the writer-side barrier, both answered + // by the metadata group leader. + AuthLeaseRenewRequest(AuthLeaseRenewRequest), + AuthLeaseRenewResponse(AuthLeaseRenewResponse), + AuthBarrierRequest(AuthBarrierRequest), + AuthBarrierResponse(AuthBarrierResponse), } /// Encode a [`RaftRpc`] into a framed binary message stamped with `epoch`. @@ -212,6 +226,12 @@ pub fn encode(rpc: &RaftRpc, epoch: &crate::cluster_epoch::ClusterEpochState) -> } RaftRpc::DataProposeRequest(m) => data_propose::encode_data_propose_req(m, &mut out), RaftRpc::DataProposeResponse(m) => data_propose::encode_data_propose_resp(m, &mut out), + RaftRpc::ReadIndexRequest(m) => read_index::encode_read_index_req(m, &mut out), + RaftRpc::ReadIndexResponse(m) => read_index::encode_read_index_resp(m, &mut out), + RaftRpc::AuthLeaseRenewRequest(m) => auth_lease::encode_renew_req(m, &mut out), + RaftRpc::AuthLeaseRenewResponse(m) => auth_lease::encode_renew_resp(m, &mut out), + RaftRpc::AuthBarrierRequest(m) => auth_lease::encode_barrier_req(m, &mut out), + RaftRpc::AuthBarrierResponse(m) => auth_lease::encode_barrier_resp(m, &mut out), }?; super::header::stamp_epoch(&mut out, epoch)?; Ok(out) @@ -307,6 +327,12 @@ pub fn decode(data: &[u8], epoch: &crate::cluster_epoch::ClusterEpochState) -> R RPC_RELEASE_RESERVATION_RESP => reservation::decode_release_reservation_resp(payload), RPC_DATA_PROPOSE_REQ => data_propose::decode_data_propose_req(payload), RPC_DATA_PROPOSE_RESP => data_propose::decode_data_propose_resp(payload), + RPC_READ_INDEX_REQ => read_index::decode_read_index_req(payload), + RPC_READ_INDEX_RESP => read_index::decode_read_index_resp(payload), + RPC_AUTH_LEASE_RENEW_REQ => auth_lease::decode_renew_req(payload), + RPC_AUTH_LEASE_RENEW_RESP => auth_lease::decode_renew_resp(payload), + RPC_AUTH_BARRIER_REQ => auth_lease::decode_barrier_req(payload), + RPC_AUTH_BARRIER_RESP => auth_lease::decode_barrier_resp(payload), _ => Err(ClusterError::Codec { detail: format!("unknown rpc_type: {rpc_type}"), }), diff --git a/nodedb-cluster/src/rpc_codec/read_index.rs b/nodedb-cluster/src/rpc_codec/read_index.rs new file mode 100644 index 000000000..420fbff29 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/read_index.rs @@ -0,0 +1,130 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! ReadIndexRequest / ReadIndexResponse wire types and codecs. +//! +//! A node that does not lead a Raft group asks the group leader for a read +//! index. The leader confirms its leadership against a quorum and answers +//! with its commit index at the time of the request. Once the asking node has +//! applied the group through that index, its state includes every entry +//! committed before the request. + +use super::discriminants::*; +use super::header::write_frame; +use super::raft_rpc::RaftRpc; +use crate::error::{ClusterError, Result}; + +/// Ask the leader of `group_id` for a confirmed read index. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct ReadIndexRequest { + pub group_id: u64, + /// How long the leader may wait for a quorum to confirm it. + pub timeout_ms: u64, +} + +/// The leader's answer to a [`ReadIndexRequest`]. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum ReadIndexOutcome { + /// A quorum confirmed the leader. Reads may be served at `read_index`. + Confirmed { read_index: u64 }, + /// The receiver does not lead the group. `leader_hint` names the leader + /// it knows of. + NotLeader { leader_hint: Option }, + /// The receiver leads the group, but no quorum answered in time. + Timeout { waited_ms: u64 }, +} + +/// Response to a [`ReadIndexRequest`]. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct ReadIndexResponse { + pub outcome: ReadIndexOutcome, +} + +macro_rules! to_bytes { + ($msg:expr) => { + rkyv::to_bytes::($msg) + .map(|b| b.to_vec()) + .map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize: {e}"), + }) + }; +} + +macro_rules! from_bytes { + ($payload:expr, $T:ty, $name:expr) => {{ + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity($payload.len()); + aligned.extend_from_slice($payload); + rkyv::from_bytes::<$T, rkyv::rancor::Error>(&aligned).map_err(|e| ClusterError::Codec { + detail: format!("rkyv deserialize {}: {e}", $name), + }) + }}; +} + +pub(super) fn encode_read_index_req(msg: &ReadIndexRequest, out: &mut Vec) -> Result<()> { + write_frame(RPC_READ_INDEX_REQ, &to_bytes!(msg)?, out) +} +pub(super) fn encode_read_index_resp(msg: &ReadIndexResponse, out: &mut Vec) -> Result<()> { + write_frame(RPC_READ_INDEX_RESP, &to_bytes!(msg)?, out) +} + +pub(super) fn decode_read_index_req(payload: &[u8]) -> Result { + Ok(RaftRpc::ReadIndexRequest(from_bytes!( + payload, + ReadIndexRequest, + "ReadIndexRequest" + )?)) +} +pub(super) fn decode_read_index_resp(payload: &[u8]) -> Result { + Ok(RaftRpc::ReadIndexResponse(from_bytes!( + payload, + ReadIndexResponse, + "ReadIndexResponse" + )?)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + fn roundtrip(rpc: RaftRpc) -> RaftRpc { + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + decode(&encoded, &epoch).expect("decode") + } + + #[test] + fn a_request_survives_the_wire() { + let rpc = roundtrip(RaftRpc::ReadIndexRequest(ReadIndexRequest { + group_id: 7, + timeout_ms: 750, + })); + match rpc { + RaftRpc::ReadIndexRequest(req) => { + assert_eq!(req.group_id, 7); + assert_eq!(req.timeout_ms, 750); + } + other => panic!("decoded the wrong variant: {other:?}"), + } + } + + #[test] + fn every_outcome_survives_the_wire() { + for outcome in [ + ReadIndexOutcome::Confirmed { read_index: 42 }, + ReadIndexOutcome::NotLeader { + leader_hint: Some(3), + }, + ReadIndexOutcome::NotLeader { leader_hint: None }, + ReadIndexOutcome::Timeout { waited_ms: 750 }, + ] { + let rpc = roundtrip(RaftRpc::ReadIndexResponse(ReadIndexResponse { + outcome: outcome.clone(), + })); + match rpc { + RaftRpc::ReadIndexResponse(resp) => assert_eq!(resp.outcome, outcome), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + } +} diff --git a/nodedb-cluster/src/transport/client/mod.rs b/nodedb-cluster/src/transport/client/mod.rs index 8e1e8abf0..3ddc6d012 100644 --- a/nodedb-cluster/src/transport/client/mod.rs +++ b/nodedb-cluster/src/transport/client/mod.rs @@ -13,6 +13,7 @@ pub mod pool; pub mod raft_impl; pub mod send; pub mod serve; +pub mod sever; pub mod shuffle_push; pub mod transport; diff --git a/nodedb-cluster/src/transport/client/send.rs b/nodedb-cluster/src/transport/client/send.rs index 7fe24d1c5..f662269a8 100644 --- a/nodedb-cluster/src/transport/client/send.rs +++ b/nodedb-cluster/src/transport/client/send.rs @@ -159,6 +159,7 @@ impl NexarTransport { rpc: RaftRpc, read_timeout: Duration, ) -> Result { + self.check_not_severed(target)?; // Encode the inner RPC once (codec errors are not retryable). // Each retry wraps it in a fresh envelope so the seq advances // per attempt — a retry is a new frame, not a replayed frame. diff --git a/nodedb-cluster/src/transport/client/sever.rs b/nodedb-cluster/src/transport/client/sever.rs new file mode 100644 index 000000000..b2bed0598 --- /dev/null +++ b/nodedb-cluster/src/transport/client/sever.rs @@ -0,0 +1,46 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sever this node from chosen peers. +//! +//! A severed peer receives no RPC from this node: every send to it fails at +//! once with a transport error, as if the network dropped it. Severing each +//! side from the other models a network partition between them, which is +//! how tests show that a partitioned node loses its leases and catches up +//! once the partition heals. + +use crate::error::{ClusterError, Result}; + +use super::transport::NexarTransport; + +impl NexarTransport { + /// Stop sending to `peer` until [`Self::heal`] is called. + pub fn sever(&self, peer: u64) { + self.severed + .write() + .unwrap_or_else(|p| p.into_inner()) + .insert(peer); + } + + /// Resume sending to `peer`. + pub fn heal(&self, peer: u64) { + self.severed + .write() + .unwrap_or_else(|p| p.into_inner()) + .remove(&peer); + } + + /// Refuse a send to a severed peer. + pub(super) fn check_not_severed(&self, target: u64) -> Result<()> { + if self + .severed + .read() + .unwrap_or_else(|p| p.into_inner()) + .contains(&target) + { + return Err(ClusterError::Transport { + detail: format!("node {} is severed from node {target}", self.node_id), + }); + } + Ok(()) + } +} diff --git a/nodedb-cluster/src/transport/client/transport.rs b/nodedb-cluster/src/transport/client/transport.rs index b2a11e876..3b5829b86 100644 --- a/nodedb-cluster/src/transport/client/transport.rs +++ b/nodedb-cluster/src/transport/client/transport.rs @@ -60,6 +60,8 @@ pub struct NexarTransport { /// inside the cluster transport and never crosses the SPSC bridge /// into the Data Plane. pub(super) agreed_versions: RwLock>, + /// Peers this node refuses to send to. See [`super::sever`]. + pub(super) severed: RwLock>, } fn default_identity_store(creds: &TransportCredentials) -> Arc { @@ -226,6 +228,7 @@ impl NexarTransport { local_spki_pin, bootstrap_peer_spki, agreed_versions: RwLock::new(HashMap::new()), + severed: RwLock::new(std::collections::HashSet::new()), }) } diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs index d36f1a4eb..273f7b7ea 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs @@ -441,6 +441,18 @@ impl TestClusterNode { let _ = connection.await; }); + // The node plans permission-checked statements only under an + // authorization lease, as a production node opens its gateway only + // once it holds one. + if let Some(timing) = shared.authorization_fence.timing() { + shared + .authorization_fence + .holder() + .await_valid(Duration::from_secs(15), timing.renew_every) + .await + .map_err(|e| format!("node {node_id}: {e}"))?; + } + Ok(Self { node_id, listen_addr, diff --git a/nodedb-test-support/src/pgwire_harness/restart.rs b/nodedb-test-support/src/pgwire_harness/restart.rs index 25cfc6de6..4a0e1f7bf 100644 --- a/nodedb-test-support/src/pgwire_harness/restart.rs +++ b/nodedb-test-support/src/pgwire_harness/restart.rs @@ -322,6 +322,12 @@ impl TestServer { shutdown_bus: shutdown_bus.clone(), }); + // Load grants and hierarchy edges before the listener opens, as the + // production boot does once the data groups replayed. + nodedb::bootstrap::permission_tree_load::load_permission_trees(&shared) + .await + .expect("permission tree load on restart"); + let pg_listener = PgListener::bind("127.0.0.1:0".parse().unwrap()) .await .unwrap(); diff --git a/nodedb/src/bootstrap/cluster_ready.rs b/nodedb/src/bootstrap/cluster_ready.rs index 731445079..46b1fa1c4 100644 --- a/nodedb/src/bootstrap/cluster_ready.rs +++ b/nodedb/src/bootstrap/cluster_ready.rs @@ -159,6 +159,17 @@ pub async fn await_cluster_ready( )); } + // Grants and hierarchy edges live in collections the data groups just + // finished replaying. Load them before the gateway opens, so no statement + // plans against an empty permission cache. + if let Err(error) = crate::bootstrap::permission_tree_load::load_permission_trees(shared).await + { + data_groups_gate.fail(format!("permission tree load failed: {error}")); + return Err(anyhow::anyhow!( + "permission tree load failed during startup: {error}" + )); + } + data_groups_gate.fire(); transport_gate.fire(); @@ -192,6 +203,25 @@ pub async fn await_cluster_ready( } warm_peers_gate.fire(); health_loop_gate.fire(); + + // In a cluster, a node plans permission-checked statements only under + // an authorization lease. The renewal loop runs from Raft start, and + // every input of a grant is live by now: the Raft groups, the replayed + // data groups, the permission cache and the Event Plane. The gateway + // opens once the first lease is granted, so the first statements are + // not refused. + if let Some(timing) = shared.authorization_fence.timing() + && let Err(error) = shared + .authorization_fence + .holder() + .await_valid(RAFT_READY_STALL_TIMEOUT, timing.renew_every) + .await + { + gateway_enable_gate.fail(format!("authorization lease not granted: {error}")); + return Err(anyhow::anyhow!( + "authorization lease not granted during startup: {error}" + )); + } gateway_enable_gate.fire(); Ok(()) diff --git a/nodedb/src/bootstrap/mod.rs b/nodedb/src/bootstrap/mod.rs index 8faa98793..7b0ea34aa 100644 --- a/nodedb/src/bootstrap/mod.rs +++ b/nodedb/src/bootstrap/mod.rs @@ -12,6 +12,7 @@ pub mod diagnostics; pub mod index_registry_seed; pub mod listeners; pub mod panic_hook; +pub mod permission_tree_load; pub mod quota_replay; pub mod schema_rehydrate; pub mod signal; diff --git a/nodedb/src/bootstrap/permission_tree_load.rs b/nodedb/src/bootstrap/permission_tree_load.rs new file mode 100644 index 000000000..4e8272f92 --- /dev/null +++ b/nodedb/src/bootstrap/permission_tree_load.rs @@ -0,0 +1,24 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Load permission-tree state before the gateway serves. +//! +//! The permission cache is in-memory. After a restart it holds the tree +//! definitions from the catalog, but none of the hierarchy edges or grants +//! stored in the source collections. This step reads them, once every data +//! group finished its replay, so the first statement plans against the +//! grants that were in force before the restart. + +use std::sync::Arc; + +use crate::control::security::permission_tree::reload; +use crate::control::state::SharedState; + +/// Apply the tree-definition changes the metadata replay committed, then +/// load every tree's edges and grants from this node's cores. +pub async fn load_permission_trees(shared: &Arc) -> crate::Result<()> { + { + let mut cache = shared.permission_cache.write().await; + shared.authorization_fence.tree_defs().apply_to(&mut cache); + } + reload::reload_all(shared, None).await +} diff --git a/nodedb/src/control/catalog_entry/authorization.rs b/nodedb/src/control/catalog_entry/authorization.rs new file mode 100644 index 000000000..4b9c613cf --- /dev/null +++ b/nodedb/src/control/catalog_entry/authorization.rs @@ -0,0 +1,124 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which catalog entries change authorization state. +//! +//! An entry that changes who may do what, or what a statement may see, is +//! acknowledged only after it binds every node (see the authorization +//! lease). The match is exhaustive, so a new entry kind is classified when +//! it is added. + +use super::entry::CatalogEntry; + +impl CatalogEntry { + /// Whether applying this entry can change what a statement is allowed to + /// read or write. + pub fn bears_authorization(&self) -> bool { + match self { + // A collection carries its owner and its permission tree. Dropping + // or purging it removes the owner and the grants on it. + Self::PutCollection(_) + | Self::PutCollectionIfAbsent(_) + | Self::DeactivateCollection { .. } + | Self::PurgeCollection { .. } => true, + Self::PutUser(_) + | Self::DropUser { .. } + | Self::PutRole(_) + | Self::DeleteRole { .. } + | Self::PutApiKey(_) + | Self::RevokeApiKey { .. } + | Self::PutAuthUser(_) + | Self::PutOidcProvider(_) + | Self::DeleteOidcProvider { .. } => true, + Self::PutTenant(_) | Self::PutTenantWithAdmin { .. } | Self::DeleteTenant { .. } => { + true + } + Self::PutRlsPolicy(_) + | Self::DeleteRlsPolicy { .. } + | Self::PutRedactionPolicy(_) + | Self::DeleteRedactionPolicy { .. } => true, + Self::PutPermission(_) + | Self::DeletePermission { .. } + | Self::PutScopeGrant(_) + | Self::DeleteScopeGrant { .. } + | Self::PutOwner(_) + | Self::DeleteOwner { .. } => true, + Self::PutDatabase(_) + | Self::DeleteDatabase { .. } + | Self::PutDatabaseGrant { .. } + | Self::DeleteDatabaseGrant { .. } + | Self::CloneDatabase { .. } => true, + Self::PutSequence(_) + | Self::DeleteSequence { .. } + | Self::PutSequenceState(_) + | Self::PutTrigger(_) + | Self::DeleteTrigger { .. } + | Self::PutFunction(_) + | Self::DeleteFunction { .. } + | Self::PutProcedure(_) + | Self::DeleteProcedure { .. } + | Self::PutSchedule(_) + | Self::DeleteSchedule { .. } + | Self::PutChangeStream(_) + | Self::DeleteChangeStream { .. } + | Self::PutMaterializedView(_) + | Self::DeleteMaterializedView { .. } + | Self::PutStreamingMaterializedView(_) + | Self::DeleteStreamingMaterializedView { .. } + | Self::PutContinuousAggregate(_) + | Self::DeleteContinuousAggregate { .. } + | Self::PutIndexRecord(_) + | Self::DeleteIndexRecord { .. } + | Self::PutSynonymGroup(_) + | Self::DeleteSynonymGroup { .. } + | Self::PutCustomType(_) + | Self::DeleteCustomType { .. } + | Self::RecordWalTombstone { .. } + | Self::MoveTenantCutover { .. } + | Self::PutDatabaseQuota { .. } + | Self::DeleteDatabaseQuota { .. } + | Self::PutTenantQuota { .. } + | Self::DeleteTenantQuota { .. } + | Self::PutScopeQuota(_) + | Self::DeleteScopeQuota { .. } + | Self::PutRetentionPolicy(_) + | Self::DeleteRetentionPolicy { .. } + | Self::PutAlertRule(_) + | Self::DeleteAlertRule { .. } + | Self::CreateTopicIfAbsent(_) + | Self::DeleteTopicWithConsumerGroups { .. } + | Self::PutConsumerGroupIfAbsent(_) + | Self::DeleteConsumerGroup { .. } + | Self::MigrateConsumerGroupStream { .. } + | Self::PutCheckpoint(_) + | Self::DeleteCheckpoint { .. } + | Self::CompactHistory { .. } + | Self::PutVectorModel(_) + | Self::DeleteVectorModel { .. } + | Self::PutVectorIndexParams(_) + | Self::PutColumnStats(_) + | Self::DeleteVectorIndexParams { .. } => false, + } + } +} + +#[cfg(test)] +mod tests { + use crate::control::catalog_entry::entry::CatalogEntry; + use crate::control::security::catalog::StoredCollection; + + #[test] + fn a_collection_bears_authorization_and_a_sequence_does_not() { + assert!( + CatalogEntry::PutCollection(Box::new(StoredCollection::new(1, "a", "b"))) + .bears_authorization() + ); + assert!( + !CatalogEntry::DeleteSequence { + database_id: 0, + tenant_id: 1, + name: "c".into(), + } + .bears_authorization() + ); + } +} diff --git a/nodedb/src/control/catalog_entry/mod.rs b/nodedb/src/control/catalog_entry/mod.rs index 5a36c9895..5e6f8420f 100644 --- a/nodedb/src/control/catalog_entry/mod.rs +++ b/nodedb/src/control/catalog_entry/mod.rs @@ -31,6 +31,7 @@ //! compile error everywhere a caller needs to handle it. pub mod apply; +pub mod authorization; pub mod codec; pub mod descriptor_stamp; pub mod descriptor_validate; diff --git a/nodedb/src/control/catalog_entry/post_apply/collection.rs b/nodedb/src/control/catalog_entry/post_apply/collection.rs index 4eaabf5d4..a33819699 100644 --- a/nodedb/src/control/catalog_entry/post_apply/collection.rs +++ b/nodedb/src/control/catalog_entry/post_apply/collection.rs @@ -6,8 +6,10 @@ use std::sync::Arc; use tracing::debug; +use crate::control::security::auth_fence::TreeDefChange; use crate::control::security::catalog::{StoredCollection, StoredOwner}; use crate::control::state::SharedState; +use crate::types::DatabaseId; /// Synchronous half of `PutCollection` post-apply: install the owner /// record into the in-memory `PermissionStore`. Called inline by the @@ -42,6 +44,50 @@ pub fn put_owner_sync(stored: &StoredCollection, shared: Arc) { } } +/// Queue the tree-definition change a committed collection descriptor makes. +/// Every node runs this, so each node's permission cache learns the tree +/// defined through any node. Planning and lease coverage move the queue into +/// the cache. +pub fn queue_tree_def_sync(stored: &StoredCollection, shared: &SharedState) { + match TreeDefChange::from_collection(stored) { + Ok(Some(change)) => { + change.note_committed(shared.authorization_fence.sources()); + shared.authorization_fence.tree_defs().push(change); + } + Ok(None) => {} + Err(e) => { + // The DDL commits the serialization of a parsed definition, so + // this JSON always parses. The prior definition stays in place: + // removing it would drop the filter and expose rows. + tracing::error!( + collection = %stored.name, + tenant = stored.tenant_id, + error = %e, + "post_apply: PERMISSION_TREE of a committed collection could not be read" + ); + } + } +} + +/// Queue the removal of a collection's tree definition, for a collection +/// that was dropped or purged. +pub fn queue_tree_def_removal_sync( + database_id: u64, + tenant_id: u64, + name: &str, + shared: &SharedState, +) { + if database_id != DatabaseId::DEFAULT.as_u64() { + return; + } + let change = TreeDefChange::Unregister { + tenant_id, + collection: name.to_owned(), + }; + change.note_committed(shared.authorization_fence.sources()); + shared.authorization_fence.tree_defs().push(change); +} + /// Register-dispatch half: dispatch a `Register` request to this node's /// Data Plane so subsequent `DocumentOp::Scan` calls find the collection /// in `doc_configs` and decode strict (Binary Tuple) documents correctly. diff --git a/nodedb/src/control/catalog_entry/post_apply/sync.rs b/nodedb/src/control/catalog_entry/post_apply/sync.rs index b75ff0a2d..0f9d854f8 100644 --- a/nodedb/src/control/catalog_entry/post_apply/sync.rs +++ b/nodedb/src/control/catalog_entry/post_apply/sync.rs @@ -45,6 +45,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { // Install owner from the CANONICAL catalog collection, not the @@ -60,15 +61,18 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc collection::put_owner_sync(&canonical, Arc::clone(shared)), - None => collection::put_owner_sync(stored, Arc::clone(shared)), - } + let canonical = canonical.as_ref().unwrap_or(&**stored); + collection::put_owner_sync(canonical, Arc::clone(shared)); + collection::queue_tree_def_sync(canonical, shared); } CatalogEntry::DeactivateCollection { - tenant_id, name, .. + database_id, + tenant_id, + name, + .. } => { collection::deactivate(*tenant_id, name.clone(), Arc::clone(shared)); + collection::queue_tree_def_removal_sync(*database_id, *tenant_id, name, shared); } CatalogEntry::PurgeCollection { database_id, @@ -76,6 +80,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { collection::purge_sync(*database_id, *tenant_id, name.clone(), Arc::clone(shared)); + collection::queue_tree_def_removal_sync(*database_id, *tenant_id, name, shared); } CatalogEntry::PutSequence(stored) => { sequence::put((**stored).clone(), Arc::clone(shared)); diff --git a/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs b/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs new file mode 100644 index 000000000..660a93f61 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs @@ -0,0 +1,145 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A shared view of which Calvin positions this node's scheduler for one +//! vShard applied. +//! +//! The scheduler owns its [`super::AppliedGate`] and mutates it on its own +//! task. Other Control-Plane code needs one question answered: did this +//! node's replica of the vShard apply `(epoch, position)`? The scheduler +//! mirrors each applied position and each watermark advance here, so the +//! answer needs no message to the scheduler task. +//! +//! The mirror keeps the same shape as the gate: a fully-applied watermark and +//! the applied positions above it. The tail is pruned as the watermark +//! advances, so it stays as small as the gate's. + +use std::collections::{BTreeSet, HashMap}; +use std::sync::{Arc, Mutex}; + +use super::recovery::NOT_YET_APPLIED_EPOCH; + +#[derive(Debug)] +struct MirrorState { + /// Every position of every epoch at or below this is applied. + /// [`NOT_YET_APPLIED_EPOCH`] means none is. + fully_applied_epoch: u64, + /// Applied positions of epochs above the watermark. + tail: BTreeSet<(u64, u32)>, +} + +/// Applied positions of one vShard's scheduler on this node. +#[derive(Debug)] +pub struct AppliedMirror { + state: Mutex, +} + +impl AppliedMirror { + /// A mirror seeded from the scheduler's recovery scan. + pub fn new(fully_applied_epoch: u64, tail: BTreeSet<(u64, u32)>) -> Self { + Self { + state: Mutex::new(MirrorState { + fully_applied_epoch, + tail, + }), + } + } + + /// Record that `(epoch, position)` applied. + pub fn mark(&self, epoch: u64, position: u32) { + let mut state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + if state.fully_applied_epoch != NOT_YET_APPLIED_EPOCH && epoch <= state.fully_applied_epoch + { + return; + } + state.tail.insert((epoch, position)); + } + + /// Record that every position of every epoch at or below `watermark` + /// applied. + pub fn fold(&self, watermark: u64) { + let mut state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + if state.fully_applied_epoch != NOT_YET_APPLIED_EPOCH + && watermark <= state.fully_applied_epoch + { + return; + } + state.fully_applied_epoch = watermark; + state.tail = state.tail.split_off(&(watermark.saturating_add(1), 0)); + } + + /// Whether this node's replica applied `(epoch, position)`. + pub fn is_applied(&self, epoch: u64, position: u32) -> bool { + let state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + (state.fully_applied_epoch != NOT_YET_APPLIED_EPOCH && epoch <= state.fully_applied_epoch) + || state.tail.contains(&(epoch, position)) + } +} + +/// The applied mirror of every vShard scheduler on this node. +#[derive(Debug, Default)] +pub struct AppliedMirrors { + by_vshard: Mutex>>, +} + +impl AppliedMirrors { + /// Register the mirror of a scheduler starting for `vshard_id`. A + /// restarted scheduler replaces its predecessor's mirror. + pub fn register( + &self, + vshard_id: u32, + fully_applied_epoch: u64, + tail: &BTreeSet<(u64, u32)>, + ) -> Arc { + let mirror = Arc::new(AppliedMirror::new(fully_applied_epoch, tail.clone())); + self.by_vshard + .lock() + .unwrap_or_else(|p| p.into_inner()) + .insert(vshard_id, Arc::clone(&mirror)); + mirror + } + + /// The mirror of `vshard_id`, when this node runs its scheduler. + pub fn get(&self, vshard_id: u32) -> Option> { + self.by_vshard + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&vshard_id) + .cloned() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_mirror_answers_like_the_gate() { + let mirror = AppliedMirror::new(NOT_YET_APPLIED_EPOCH, BTreeSet::new()); + assert!(!mirror.is_applied(0, 0)); + mirror.mark(3, 1); + assert!(mirror.is_applied(3, 1)); + assert!(!mirror.is_applied(3, 0)); + mirror.fold(3); + assert!(mirror.is_applied(3, 0)); + assert!(mirror.is_applied(2, 9)); + assert!(!mirror.is_applied(4, 0)); + // A mark at or below the watermark changes nothing. + mirror.mark(1, 0); + assert!(mirror.is_applied(1, 0)); + // A lower watermark never moves it back. + mirror.fold(1); + assert!(mirror.is_applied(3, 0)); + } + + #[test] + fn a_restarted_scheduler_replaces_its_mirror() { + let mirrors = AppliedMirrors::default(); + let first = mirrors.register(7, NOT_YET_APPLIED_EPOCH, &BTreeSet::new()); + first.mark(1, 0); + let second = mirrors.register(7, 4, &BTreeSet::new()); + let current = mirrors.get(7).expect("mirror"); + assert!(Arc::ptr_eq(¤t, &second)); + assert!(current.is_applied(4, 0)); + assert!(mirrors.get(8).is_none()); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs index a01dc5da2..67d1c9d05 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -184,7 +184,8 @@ impl Scheduler { let key = nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position); let mut results = self .shared - .calvin_apply_results + .calvin + .apply_results .lock() .unwrap_or_else(|p| p.into_inner()); match results.entry(key) { diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs index 4ad2b3468..e396e1261 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs @@ -108,12 +108,14 @@ impl Scheduler { if committed { self.shared - .calvin_counters + .calvin + .counters .commits_flushed .fetch_add(1, Ordering::Relaxed); } else { self.shared - .calvin_counters + .calvin + .counters .commits_dropped .fetch_add(1, Ordering::Relaxed); } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs index 74edb77c3..be5738a4e 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs @@ -70,7 +70,8 @@ impl Scheduler { // participant error never validated a read-set, so it must not count // here. self.shared - .calvin_counters + .calvin + .counters .read_set_validation_failures .fetch_add(1, Ordering::Relaxed); } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs index 3e9733646..9fa99a857 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs @@ -184,7 +184,8 @@ impl Scheduler { // no read-set was checked. if response.read_set_valid == Some(false) { self.shared - .calvin_counters + .calvin + .counters .read_set_validation_failures .fetch_add(1, std::sync::atomic::Ordering::Relaxed); } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs index 1863be958..4173b8b7f 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs @@ -283,7 +283,8 @@ impl Scheduler { let _ = rx.recv().await; }); self.shared - .calvin_counters + .calvin + .counters .write_versions_recorded .fetch_add(1, Ordering::Relaxed); } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs index cde8a5cad..2f1140b16 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs @@ -299,7 +299,9 @@ impl Scheduler { // once ALL of its positions for this vShard have terminally completed, // so any advertised watermark reflects a FULLY-applied epoch — the value // `BEGIN` needs for a torn-free cross-shard snapshot anchor. - if let Some(watermark) = self.applied.mark_applied(txn_id.epoch, txn_id.position) { + let folded = self.applied.mark_applied(txn_id.epoch, txn_id.position); + self.applied_mirror.mark(txn_id.epoch, txn_id.position); + if let Some(watermark) = folded { self.publish_watermark(watermark); } } @@ -406,7 +408,7 @@ mod tests { build_test_scheduler_with_data_side(test_coll_vshard(), registry); let shared = Arc::clone(&scheduler.shared); fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); - let watermark_before = shared.last_applied_calvin_epoch.load(Ordering::Acquire); + let watermark_before = shared.calvin.last_applied_epoch.load(Ordering::Acquire); scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); @@ -415,7 +417,7 @@ mod tests { "a capacity refusal must not mark the position applied" ); assert_eq!( - shared.last_applied_calvin_epoch.load(Ordering::Acquire), + shared.calvin.last_applied_epoch.load(Ordering::Acquire), watermark_before, "a capacity refusal must not publish a watermark for the txn's epoch" ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index eb0e66a74..4da6037d3 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -76,7 +76,7 @@ pub struct Scheduler { Arc>, /// Deterministic lock manager for this vshard. Shared (via `Arc>`) /// with the Control-Plane write-admission gate through - /// `SharedState.calvin_lock_managers`, so a fast-path point write contends + /// `SharedState.calvin.lock_managers`, so a fast-path point write contends /// on the SAME lock table this scheduler validates against. The scheduler /// still runs single-threaded per vShard, so the mutex is uncontended except /// for the brief probe the gate takes. @@ -112,6 +112,9 @@ pub struct Scheduler { /// deterministic threshold below which an orphaned shared reservation is /// released. Purely a function of replicated input order — no wall clock. pub(in crate::control::cluster::calvin::scheduler::driver::core) max_input_epoch: u64, + /// Shared mirror of `applied`, read by authorization coverage. + pub(in crate::control::cluster::calvin::scheduler::driver::core) applied_mirror: + Arc, /// Scheduler configuration. pub(in crate::control::cluster::calvin::scheduler::driver::core) config: SchedulerConfig, /// Metrics. @@ -190,11 +193,11 @@ pub struct SchedulerParams { pub read_result_rx: mpsc::Receiver, /// The shared lock table for this vShard. Constructed by /// `reconcile_vshard_schedulers` and registered in - /// `SharedState.calvin_lock_managers` under the SAME `Arc` passed here. + /// `SharedState.calvin.lock_managers` under the SAME `Arc` passed here. pub lock_manager: Arc>, /// Receiver for gate-side lock promotions. Constructed by /// `reconcile_vshard_schedulers`; its `UnboundedSender` is registered in - /// `SharedState.calvin_promotion_senders` for this same vShard so a fast-path + /// `SharedState.calvin.promotion_senders` for this same vShard so a fast-path /// guard drop can hand promoted waiters back to this scheduler. pub promotion_rx: mpsc::UnboundedReceiver>, /// Shared completion registry for verdict probes on the commit barrier. @@ -232,6 +235,12 @@ impl Scheduler { let completion_cap = config.channel_capacity; let (completion_tx, completion_rx) = mpsc::channel(completion_cap); + let applied_mirror = shared.authorization_fence.calvin_mirrors().register( + vshard_id, + fully_applied_epoch, + &applied_tail, + ); + let capacity_freed = shared .dispatcher .lock() @@ -252,6 +261,7 @@ impl Scheduler { dependent_barrier: BTreeMap::new(), read_result_rx, applied: AppliedGate::new(fully_applied_epoch, applied_tail), + applied_mirror, rebuild_target_epoch, max_input_epoch: 0, config, @@ -301,7 +311,7 @@ impl Scheduler { /// Publish an advanced fully-applied watermark to the metrics gauge and the /// shared cross-shard snapshot anchor. /// - /// `BEGIN` reads `SharedState::last_applied_calvin_epoch` to anchor a + /// `BEGIN` reads `CalvinLocalState::last_applied_epoch` to anchor a /// session's cross-shard snapshot version, so it MUST reflect the /// FULLY-applied epoch — never an epoch that has only some of its positions /// committed, which would let a session anchor on a torn epoch. `fetch_max` @@ -311,8 +321,10 @@ impl Scheduler { watermark: u64, ) { self.metrics.update_last_applied_epoch(watermark); + self.applied_mirror.fold(watermark); self.shared - .last_applied_calvin_epoch + .calvin + .last_applied_epoch .fetch_max(watermark, std::sync::atomic::Ordering::Release); } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs index cc0610ee4..f786560f9 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs @@ -32,7 +32,7 @@ pub(super) struct PendingTxn { /// Whether this vShard's slice carries a primary user data write (a non-edge /// Document/KV/Vector/Timeseries/Columnar/Array write). Only the primary-write /// participant deposits its applied `Response` (affected-count and any - /// RETURNING rows) into `SharedState::calvin_apply_results`. The implicit-edge + /// RETURNING rows) into `CalvinLocalState::apply_results`. The implicit-edge /// cleanup participants that dual-home alongside it carry no primary write and /// so never clobber the entry the coordinator drains. pub has_primary_write: bool, diff --git a/nodedb/src/control/cluster/calvin/scheduler/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/mod.rs index ec86c0cd1..78d9184c7 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 pub mod applied_gate; +pub mod applied_mirror; pub mod driver; pub mod lock; pub mod metrics; @@ -8,6 +9,7 @@ mod metrics_flow; pub mod recovery; pub use applied_gate::AppliedGate; +pub use applied_mirror::{AppliedMirror, AppliedMirrors}; pub use driver::{ CalvinReadResultProposal, RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, SequencerProposer, propose_calvin_read_result, diff --git a/nodedb/src/control/cluster/read_index.rs b/nodedb/src/control/cluster/read_index.rs index 20a97a19d..c4e91dd60 100644 --- a/nodedb/src/control/cluster/read_index.rs +++ b/nodedb/src/control/cluster/read_index.rs @@ -12,7 +12,7 @@ //! table to decide the read belonged here, so it builds the redirect from //! what it knows rather than having a second, staler answer passed back. -use std::sync::{Arc, Mutex}; +use std::sync::{Arc, Mutex, Weak}; use std::time::Duration; use async_trait::async_trait; @@ -43,16 +43,50 @@ pub trait RaftReadGate: Send + Sync { /// Whether this node's replica of `group_id` is within `max_staleness` of /// the leader. Local state only — no quorum round, so this does not block. fn within_staleness_bound(&self, group_id: u64, max_staleness: Duration) -> bool; + + /// A read index for `group_id` on any node: confirmed here when this node + /// leads the group, asked of the leader otherwise. + /// + /// Once this node has applied the group through the returned index, its + /// state holds every entry committed before the call. A gate whose node + /// always leads its groups answers with [`Self::confirm_leader`]. + async fn read_index(&self, group_id: u64, timeout: Duration) -> Result { + self.confirm_leader(group_id, timeout).await + } } +/// The Raft loop type the production gate forwards read-index requests +/// through. +type GateRaftLoop = nodedb_cluster::RaftLoop< + crate::control::cluster::spsc_applier::SpscCommitApplier, + crate::control::LocalPlanExecutor, +>; + /// Production implementation, backed by the Raft loop's coordinator. +/// +/// Holds the loop weakly: the loop keeps `SharedState` alive, and the gate +/// lives on `SharedState`, so a strong reference would pin both. pub struct MultiRaftReadGate { multi_raft: Arc>, + raft_loop: Weak, } impl MultiRaftReadGate { - pub fn new(multi_raft: Arc>) -> Self { - Self { multi_raft } + pub fn new(multi_raft: Arc>, raft_loop: Weak) -> Self { + Self { + multi_raft, + raft_loop, + } + } +} + +/// The refusal a read-index error means to the caller. +fn refusal_of(error: ClusterError) -> ReadIndexRefusal { + match error { + ClusterError::ReadIndexTimeout { waited_ms, .. } => ReadIndexRefusal::Timeout { waited_ms }, + // Not hosted here, not leading, leadership lost mid-probe, or the + // leader unreachable: the caller asks again later. + _ => ReadIndexRefusal::NotLeader, } } @@ -63,16 +97,20 @@ impl RaftReadGate for MultiRaftReadGate { group_id: u64, timeout: Duration, ) -> Result { - match nodedb_cluster::confirm_read_index(&self.multi_raft, group_id, timeout).await { - Ok(index) => Ok(index), - Err(ClusterError::ReadIndexTimeout { waited_ms, .. }) => { - Err(ReadIndexRefusal::Timeout { waited_ms }) - } - // Every other path out of the confirmation — not hosted here, not - // leading, leadership lost mid-probe — means the same thing to the - // caller: ask the leader instead. - Err(_) => Err(ReadIndexRefusal::NotLeader), - } + nodedb_cluster::confirm_read_index(&self.multi_raft, group_id, timeout) + .await + .map_err(refusal_of) + } + + async fn read_index(&self, group_id: u64, timeout: Duration) -> Result { + let Some(raft_loop) = self.raft_loop.upgrade() else { + // The loop is gone: the node is shutting down. + return Err(ReadIndexRefusal::NotLeader); + }; + raft_loop + .read_index_via_leader(group_id, timeout) + .await + .map_err(refusal_of) } fn within_staleness_bound(&self, group_id: u64, max_staleness: Duration) -> bool { diff --git a/nodedb/src/control/cluster/snapshot_applier.rs b/nodedb/src/control/cluster/snapshot_applier.rs index 098836a69..e4e46501c 100644 --- a/nodedb/src/control/cluster/snapshot_applier.rs +++ b/nodedb/src/control/cluster/snapshot_applier.rs @@ -168,6 +168,10 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { } } + // The install emitted no per-row events, so the permission cache + // reloads before this node reports coverage of the group again. + self.shared.authorization_fence.note_snapshot_installed(); + Ok(()) } } diff --git a/nodedb/src/control/cluster/start_raft/loop_build.rs b/nodedb/src/control/cluster/start_raft/loop_build.rs index a1b591409..9607675f3 100644 --- a/nodedb/src/control/cluster/start_raft/loop_build.rs +++ b/nodedb/src/control/cluster/start_raft/loop_build.rs @@ -84,6 +84,34 @@ pub(super) fn build_raft_loop( replication_factor, } = setup; + // The authorization lease runs on the Raft timing of this node. + let lease_timing = crate::control::security::auth_lease::LeaseTiming::from_raft( + multi_raft.election_timeout_min(), + multi_raft.heartbeat_interval(), + )?; + if !shared.authorization_fence.install_timing(lease_timing) { + tracing::warn!( + "authorization lease timing already set — start_raft appears to have run twice" + ); + } + let lease_service = Arc::new( + crate::control::security::auth_lease::LeaderLeaseService::new( + Arc::downgrade(shared), + lease_timing, + ), + ); + if !shared + .authorization_fence + .install_leader(Arc::clone(&lease_service)) + { + tracing::warn!( + "authorization lease service already set — start_raft appears to have run twice" + ); + } + // Authorization coverage of the sequencer group settles each completion + // ack against this node's schedulers. + calvin_completion_registry.applied_acks.enable(); + let raft_loop = Arc::new( nodedb_cluster::RaftLoop::new( multi_raft, @@ -109,6 +137,7 @@ pub(super) fn build_raft_loop( .with_calvin_submit_inbox(hooks.calvin_submit_inbox) .with_reserve_read(hooks.reserve_read) .with_release_reservation(hooks.release_reservation) + .with_auth_lease(lease_service) .with_data_dir(data_dir.to_path_buf()) .with_snapshot_chunk_bytes(snapshot_chunk_bytes) .with_orphan_partial_max_age_secs(orphan_partial_max_age_secs) diff --git a/nodedb/src/control/cluster/start_raft/observability.rs b/nodedb/src/control/cluster/start_raft/observability.rs index 067fdcaee..8ca55225a 100644 --- a/nodedb/src/control/cluster/start_raft/observability.rs +++ b/nodedb/src/control/cluster/start_raft/observability.rs @@ -100,14 +100,36 @@ pub(super) fn finish_observability( // Publish the leadership confirmer so a linearizable read served on this // node proves against a quorum that it is still the leader. Holds the - // same coordinator mutex the loop ticks, not the loop itself. - let gate: Arc = Arc::new( - crate::control::cluster::read_index::MultiRaftReadGate::new(raft_loop.multi_raft_handle()), - ); + // same coordinator mutex the loop ticks, and the loop only weakly, to ask + // a group leader for a read index from a follower. + let gate: Arc = + Arc::new(crate::control::cluster::read_index::MultiRaftReadGate::new( + raft_loop.multi_raft_handle(), + Arc::downgrade(&raft_loop), + )); if shared.raft_read_gate.set(gate).is_err() { tracing::warn!("raft_read_gate already set — start_raft appears to have run twice"); } + // Renew this node's authorization lease with the metadata leader. Its + // confirmed coverage needs the read gate above. + if let Some(timing) = shared.authorization_fence.timing() { + let renew_state = Arc::clone(shared); + crate::control::shutdown::spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "auth_lease_renew", + crate::control::shutdown::ShutdownPhase::DrainingControlPlane, + move |shutdown| { + crate::control::security::auth_lease::renew_loop::run_renew_loop( + renew_state, + timing, + shutdown, + ) + }, + ); + } + // Publish this node's cluster-epoch state so the routing gate can tell // whether this node has missed a topology transition before it coordinates // work on a view of the cluster that may already be superseded. diff --git a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs index 449088cb9..0d94b8d98 100644 --- a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs +++ b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs @@ -147,11 +147,15 @@ pub(super) fn wire_proposers( // Weak for the same cycle-breaking reason as `raft_proposer` above. let raft_loop_async = Arc::downgrade(raft_loop); let tracker_for_proposer = tracker.clone(); + // Held weakly for the same cycle-breaking reason as `raft_proposer` above: + // the proposer lives on `SharedState`. + let state_for_proposer = Arc::downgrade(shared); let deadline_secs = shared.tuning.network.default_deadline_secs; let async_proposer: Arc = Arc::new(move |vshard_id, idempotency_key, data| { let rl_weak = raft_loop_async.clone(); let tk = tracker_for_proposer.clone(); + let state_weak = state_for_proposer.clone(); Box::pin(async move { let rl = rl_weak.upgrade().ok_or_else(|| crate::Error::Internal { detail: "raft propose (async): cluster not running".into(), @@ -171,42 +175,63 @@ pub(super) fn wire_proposers( // `RetryableLeaderChange` instead of leaking a // not-our-payload back to the caller. let rx = tk.register(group_id, log_index, idempotency_key); - tokio::time::timeout(std::time::Duration::from_secs(deadline_secs), rx) - .await - .map_err(|_| crate::Error::Dispatch { - detail: format!( - "raft commit timeout for group {group_id} index {log_index}" - ), - })? - .map_err(|_| crate::Error::Dispatch { - detail: "propose waiter channel closed".into(), - })? - // Preserve `RetryableLeaderChange` so the gateway - // retry loop can re-propose against the new leader - // — wrapping it in `Dispatch` would hide the - // retryable signal and surface as silent INSERT - // success. Only machinery failures stay wrapped for - // diagnostics; a classified apply verdict keeps its - // client-visible classification. - .map_err(|e| { - if crate::error_classify::is_unclassified_failure(&e) { - crate::Error::Dispatch { - detail: format!("apply error: {e}"), + let applied = + tokio::time::timeout(std::time::Duration::from_secs(deadline_secs), rx) + .await + .map_err(|_| crate::Error::Dispatch { + detail: format!( + "raft commit timeout for group {group_id} index {log_index}" + ), + })? + .map_err(|_| crate::Error::Dispatch { + detail: "propose waiter channel closed".into(), + })? + // Preserve `RetryableLeaderChange` so the gateway + // retry loop can re-propose against the new leader + // — wrapping it in `Dispatch` would hide the + // retryable signal and surface as silent INSERT + // success. Only machinery failures stay wrapped for + // diagnostics; a classified apply verdict keeps its + // client-visible classification. + .map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + crate::Error::Dispatch { + detail: format!("apply error: {e}"), + } + } else { + e } - } else { - e - } - }) - // Carry out the write-version the APPLY side stamped, not - // `log_index`. The tracker resolves on the node that applied - // the entry locally, so `write_version` is this replica's own - // post-write `coll_write_lsn` — a WAL LSN, the same domain - // every other feed of that map records in, and the only - // domain the shard-local OCC read validator compares in. The - // raft log index is a per-group counter on a different scale - // entirely; publishing it here made reads validate a WAL LSN - // against a log index. - .map(|applied| (applied.payload, applied.write_version)) + }) + // Carry out the write-version the APPLY side stamped, not + // `log_index`. The tracker resolves on the node that applied + // the entry locally, so `write_version` is this replica's own + // post-write `coll_write_lsn` — a WAL LSN, the same domain + // every other feed of that map records in, and the only + // domain the shard-local OCC read validator compares in. The + // raft log index is a per-group counter on a different scale + // entirely; publishing it here made reads validate a WAL LSN + // against a log index. + .map(|applied| (applied.payload, applied.write_version)); + let applied = applied?; + // A write to a vShard homing a permission-tree source is + // acknowledged only once every lease holder covers it, or its + // lease expired. + if let Some(state) = state_weak.upgrade() + && state + .authorization_fence + .sources() + .is_source_vshard(vshard_id) + { + crate::control::security::auth_lease::authorization_barrier( + &state, + vec![nodedb_cluster::GroupCoverage { + group_id, + through: log_index, + }], + ) + .await?; + } + Ok(applied) }) }); crate::control::vshard_admission::install_async_raft_proposer(shared, async_proposer)?; diff --git a/nodedb/src/control/cluster/start_raft_helpers.rs b/nodedb/src/control/cluster/start_raft_helpers.rs index bc2f2b3f3..bd62ca97c 100644 --- a/nodedb/src/control/cluster/start_raft_helpers.rs +++ b/nodedb/src/control/cluster/start_raft_helpers.rs @@ -201,13 +201,14 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: // The deterministic lock table is shared between this scheduler and the // Control-Plane write-admission gate: build it once and register the - // SAME `Arc` in `calvin_lock_managers` so a fast-path point write and + // SAME `Arc` in `CalvinLocalState::lock_managers` so a fast-path point write and // this scheduler's validation contend on one mutex. let lock_manager = Arc::new(Mutex::new( crate::control::cluster::calvin::scheduler::lock_manager::LockManager::new(), )); shared - .calvin_lock_managers + .calvin + .lock_managers .lock() .unwrap_or_else(|p| p.into_inner()) .insert(vshard_id, Arc::clone(&lock_manager)); @@ -221,7 +222,8 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: // synchronous `Drop` that must not block. let (promotion_tx, promotion_rx) = tokio::sync::mpsc::unbounded_channel(); shared - .calvin_promotion_senders + .calvin + .promotion_senders .lock() .unwrap_or_else(|p| p.into_inner()) .insert(vshard_id, promotion_tx); diff --git a/nodedb/src/control/event_trigger.rs b/nodedb/src/control/event_trigger.rs index b9347ce83..e49e47f9d 100644 --- a/nodedb/src/control/event_trigger.rs +++ b/nodedb/src/control/event_trigger.rs @@ -413,6 +413,7 @@ mod tests { "doc'; DELETE FROM audit; --", )), lsn: Lsn::new(1), + record: None, database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/src/control/gateway/sql_execute.rs b/nodedb/src/control/gateway/sql_execute.rs index 34fa083a2..c3f06d3c9 100644 --- a/nodedb/src/control/gateway/sql_execute.rs +++ b/nodedb/src/control/gateway/sql_execute.rs @@ -54,11 +54,10 @@ impl Gateway { // Read before reverify runs, not derived per-name inside it: both // pseudo-entries share one live tenant snapshot the same way the // real collection entries share one catalog snapshot. - let permission_tree_version = shared - .permission_cache - .read() - .await - .tenant_version(tenant_id); + let permission_tree_version = + crate::control::security::auth_fence::permission_view(&shared, ctx.tenant_id) + .await? + .tenant_version(tenant_id); let rls_version = shared.rls.tenant_version(tenant_id); let ptree_key = permission_tree_version_key(tenant_id); let rls_key = rls_version_key(tenant_id); diff --git a/nodedb/src/control/metadata_proposer.rs b/nodedb/src/control/metadata_proposer.rs deleted file mode 100644 index f5e8eee83..000000000 --- a/nodedb/src/control/metadata_proposer.rs +++ /dev/null @@ -1,710 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Synchronous `propose-and-wait-for-local-apply` helper for -//! replicated catalog DDL. -//! -//! The sole entry point pgwire DDL handlers use to write a -//! [`CatalogEntry`] through the metadata raft group (group 0). It is -//! deliberately sync — pgwire DDL handlers are not async, and -//! `tokio::task::block_in_place`-style wrapping keeps the blocking -//! wait from starving the tokio runtime. -//! -//! Semantics: -//! -//! 1. If no cluster is configured (`shared.metadata_raft` not -//! installed), returns `ProposeOutcome::LocalOnly`. The caller's -//! single-node direct-write path stays authoritative. -//! 2. If this node is the metadata-group leader, proposes the -//! entry, blocks until its local applied watermark reaches the -//! assigned log index (5s default timeout), and returns the -//! log index on success. -//! 3. If this node is NOT the leader, returns -//! `Error::Config { detail: "metadata propose: not leader ..." }`. -//! Gateway-side redirection will make this transparent. - -use std::sync::atomic::Ordering; -use std::sync::{Arc, Weak}; -use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; - -use tokio::runtime::RuntimeFlavor; - -use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, WaitOutcome, encode_entry}; - -#[cfg(test)] -use nodedb_cluster::AppliedIndexWatcher; - -use crate::control::catalog_entry::{self, CatalogEntry}; -use crate::control::propose_outcome::ProposeOutcome; -use crate::control::state::SharedState; -use crate::error::Error; - -/// Default upper bound on how long a single -/// `propose_catalog_entry` call will block before returning an -/// error. -pub const DEFAULT_PROPOSE_TIMEOUT: Duration = Duration::from_secs(5); - -/// Default upper bound on how long a DDL drain will wait for -/// prior-version leases to release before giving up. Must be at -/// least `ClusterTransportTuning::descriptor_lease_duration_secs` -/// so an existing lease gets at least one full lifetime to -/// expire naturally. 35 seconds matches the 300s lease duration -/// plus a 30-second grace minus the typical 5-minute default -/// cut down for test budget — in production -/// `propose_catalog_entry_with_drain_timeout` can pass a longer -/// value if an operator is willing to wait. -pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(35); -const DDL_PREPARE_LEASE: Duration = Duration::from_secs(60); -const DDL_PREPARE_WAIT: Duration = Duration::from_secs(70); - -/// Type-erased handle for proposing to the metadata raft group. -/// -/// The apply watermark for the metadata group lives on -/// [`SharedState::applied_index_watcher`] (keyed by -/// [`nodedb_cluster::METADATA_GROUP_ID`]); callers of [`Self::propose`] -/// look it up there rather than receiving it through this handle. -pub trait MetadataRaftHandle: Send + Sync { - /// Propose a raw encoded `MetadataEntry` to the metadata group. - /// Returns its assigned log index on success. - fn propose(&self, bytes: Vec) -> Result; -} - -/// Concrete impl wrapping `nodedb_cluster::RaftLoop`. -/// -/// Holds the loop weakly: this handle lives on `SharedState`, which is -/// itself kept alive transitively by the `RaftLoop`, so a strong -/// reference here would close a cycle that pins both forever and blocks -/// clean shutdown. The loop is kept alive by its own spawned tasks; -/// `upgrade` therefore succeeds throughout normal operation and only -/// fails once the loop has been dropped on shutdown. -pub struct RaftLoopProposerHandle { - raft_loop: Weak< - nodedb_cluster::RaftLoop< - crate::control::cluster::SpscCommitApplier, - crate::control::LocalPlanExecutor, - >, - >, -} - -impl RaftLoopProposerHandle { - pub fn new( - raft_loop: Arc< - nodedb_cluster::RaftLoop< - crate::control::cluster::SpscCommitApplier, - crate::control::LocalPlanExecutor, - >, - >, - ) -> Self { - Self { - raft_loop: Arc::downgrade(&raft_loop), - } - } -} - -impl MetadataRaftHandle for RaftLoopProposerHandle { - fn propose(&self, bytes: Vec) -> Result { - // The cluster crate's `propose_to_metadata_group_via_leader` - // is async because it may need to forward to the metadata - // leader over QUIC. The trait method is sync because every - // caller (catalog DDL handlers, lease grant/release helpers) - // is itself sync but runs inside a tokio task. Wrap in - // `block_in_place` + the current runtime's `block_on` so the - // forwarding QUIC round-trip drives without starving the - // raft tick that produces the leader_hint. - // `upgrade` fails only once the raft loop has been dropped on - // shutdown; a request racing shutdown then fails cleanly with a - // typed error instead of panicking. - let raft_loop = self.raft_loop.upgrade().ok_or_else(|| Error::Config { - detail: "metadata propose: cluster not running".into(), - })?; - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current() - .block_on(raft_loop.propose_to_metadata_group_via_leader(bytes)) - }) - .map_err(|e| match e { - // An election in progress is transient, not a failure of this - // proposal. Keep it typed rather than flattening it into a generic - // config error, so callers can wait the election out instead of - // failing the statement — a node that has just restarted answers - // every metadata proposal this way for a moment. - nodedb_cluster::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { - leader_hint: None, - }) => Error::MetadataLeaderUnavailable, - other => Error::Config { - detail: format!("metadata propose: {other}"), - }, - }) - } -} - -fn wall_now_ns() -> u64 { - SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap_or_default() - .as_nanos() - .min(u64::MAX as u128) as u64 -} - -fn propose_metadata_and_wait( - shared: &SharedState, - handle: &dyn MetadataRaftHandle, - entry: &MetadataEntry, - timeout: Duration, -) -> Result { - let raw = encode_entry(entry).map_err(|e| Error::Config { - detail: format!("metadata entry encode: {e}"), - })?; - let index = handle.propose(raw)?; - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = tokio::task::block_in_place(|| watcher.wait_for(index, timeout)); - match outcome { - WaitOutcome::Reached => Ok(index), - WaitOutcome::TimedOut => Err(Error::Config { - detail: format!( - "metadata propose timed out after {timeout:?} waiting for log index {index} (current: {})", - watcher.current() - ), - }), - WaitOutcome::GroupGone => Err(Error::Config { - detail: "metadata group no longer hosted on this node".into(), - }), - } -} - -/// RAII ownership of the metadata-Raft-serialized descriptor preparation lease. -/// The matching release is itself replicated, so another node cannot stamp from -/// the same prior catalog version until this guard is dropped and that release -/// has applied. -pub(crate) struct DdlPrepareGuard<'a> { - shared: &'a SharedState, - handle: &'a dyn MetadataRaftHandle, - token: u64, -} - -impl DdlPrepareGuard<'_> { - pub(crate) fn token(&self) -> u64 { - self.token - } -} - -impl Drop for DdlPrepareGuard<'_> { - fn drop(&mut self) { - if let Err(error) = propose_metadata_and_wait( - self.shared, - self.handle, - &MetadataEntry::DdlPrepareRelease { token: self.token }, - DEFAULT_PROPOSE_TIMEOUT, - ) { - tracing::error!(token = self.token, %error, "metadata DDL lease release failed"); - } - } -} - -pub(crate) fn acquire_ddl_prepare_lease<'a>( - shared: &'a SharedState, - handle: &'a dyn MetadataRaftHandle, -) -> Result, Error> { - let sequence = shared - .metadata_ddl_token_seq - .fetch_add(1, Ordering::Relaxed); - let token = shared.node_id.wrapping_mul(0x9e37_79b9_7f4a_7c15) - ^ wall_now_ns().rotate_left(17) - ^ sequence; - let deadline = Instant::now() + DDL_PREPARE_WAIT; - - loop { - propose_metadata_and_wait( - shared, - handle, - &MetadataEntry::DdlPrepareAcquire { token }, - DEFAULT_PROPOSE_TIMEOUT, - )?; - - loop { - let owner = *shared - .metadata_ddl_owner - .lock() - .map_err(|_| Error::Config { - detail: "metadata DDL owner lock poisoned".into(), - })?; - match owner { - Some((current, _)) if current == token => { - return Ok(DdlPrepareGuard { - shared, - handle, - token, - }); - } - Some((current, acquired_at)) - if shared.is_metadata_leader() - && acquired_at.elapsed() >= DDL_PREPARE_LEASE => - { - // Cancel the dead owner's pending record before releasing its - // lease, so it never lingers visible-but-unresolved past the lease. - if shared.pending_ddl.contains(current) { - propose_metadata_and_wait( - shared, - handle, - &MetadataEntry::DdlPendingCancel { token: current }, - DEFAULT_PROPOSE_TIMEOUT, - )?; - } - propose_metadata_and_wait( - shared, - handle, - &MetadataEntry::DdlPrepareRelease { token: current }, - DEFAULT_PROPOSE_TIMEOUT, - )?; - break; - } - None => break, - Some(_) if Instant::now() < deadline => { - // Reached from async tasks (ILP batch flush -> - // `propose_catalog_entry`), so hand the worker back to - // tokio rather than parking it: the lease owner this - // polls for is released by a raft apply that needs a - // worker to make progress. - tokio::task::block_in_place(|| { - std::thread::sleep(Duration::from_millis(10)); - }); - } - Some(_) => { - return Err(Error::Config { - detail: "metadata DDL preparation lease timed out".into(), - }); - } - } - } - } -} - -/// Take the local DDL preparation lock, handing the wait back to tokio when -/// the caller is on a multi-thread worker. -/// -/// The holder keeps this lock across the distributed preparation lease, the -/// descriptor drain and the local apply wait — each already wrapped in -/// `block_in_place`, but that only tells tokio about the waits *inside* the -/// lock, never about the wait *for* it. A bare `lock()` on a worker therefore -/// removes that worker from the runtime silently, including from the raft -/// apply work the current holder needs in order to finish, which turns -/// contention into a self-sustaining stall. -/// -/// `block_in_place` is a passthrough outside a multi-thread worker (plain sync -/// callers, blocking-pool threads) and panics on the current-thread runtime, -/// so it is applied only where it is both legal and meaningful — mirroring -/// `lease::drain_propose::poll_leases_drained`. -fn lock_ddl_preparation(shared: &SharedState) -> Result, Error> { - let acquire = || { - shared.metadata_ddl_lock.lock().map_err(|_| Error::Config { - detail: "metadata DDL preparation lock poisoned".into(), - }) - }; - match tokio::runtime::Handle::try_current() { - Ok(handle) if handle.runtime_flavor() == RuntimeFlavor::MultiThread => { - tokio::task::block_in_place(acquire) - } - _ => acquire(), - } -} - -/// Propose a `CatalogEntry` and block until the local applied-index -/// watcher confirms the entry has been applied on this node. -/// -/// The returned [`ProposeOutcome`] tells the caller whether to write the -/// catalog itself, leave it to the applier, or do nothing because the entry -/// is held for COMMIT. -pub fn propose_catalog_entry( - shared: &SharedState, - entry: &CatalogEntry, -) -> Result { - propose_catalog_entry_with_timeout(shared, entry, DEFAULT_PROPOSE_TIMEOUT) -} - -/// Same as [`propose_catalog_entry`] but with an explicit timeout. -pub fn propose_catalog_entry_with_timeout( - shared: &SharedState, - entry: &CatalogEntry, - timeout: Duration, -) -> Result { - // Buffering is decided first, ahead of every replication-mode gate: an open - // transaction owns the entry regardless of whether this deployment - // replicates DDL, and COMMIT re-runs the mode choice for the whole batch. - // Entries also stay unstamped until then, so repeated mutations of one - // descriptor receive distinct versions in commit order. - if crate::control::server::shared::session::ddl_buffer::try_buffer(entry.clone()) { - return Ok(ProposeOutcome::Buffered); - } - - let Some(handle) = shared.metadata_raft.get() else { - return Ok(ProposeOutcome::LocalOnly); - }; - - // Rolling-upgrade gate: until every node in the cluster reports - // at least `DISTRIBUTED_CATALOG_VERSION`, fall back to the legacy - // direct-write path on the originating node. Mixing the - // replicated and direct paths during a partial upgrade would - // diverge catalog state across nodes — see - // `control/rolling_upgrade.rs`. - if !shared - .cluster_version_view() - .can_activate_feature(crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION) - { - tracing::warn!( - min_version = shared.cluster_version_view().min_version, - required = crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, - "metadata propose: cluster in compat mode (mixed-version), \ - falling back to legacy direct-write path" - ); - return Ok(ProposeOutcome::LocalOnly); - } - - // Serialize preparation through local apply confirmation. Without this, - // concurrent proposers can both observe persisted version N and emit N+1. - let _local_ddl_guard = lock_ddl_preparation(shared)?; - let distributed_ddl_guard = acquire_ddl_prepare_lease(shared, handle.as_ref())?; - - // Drain for Put* variants that carry descriptor_version. - // Leases acquired at plan time are refcounted and held - // through execute; when the last in-flight query using a - // descriptor completes, its `QueryLeaseScope` drops and the - // refcount hits zero, releasing the lease. Drain is what - // makes this an actual barrier: the proposer waits for all - // prior-version leases to release before committing the new - // `Put*`, giving long-running in-flight queries a bounded - // window (DEFAULT_DRAIN_TIMEOUT) to finish. - if let Some((descriptor_id, prior_version)) = - crate::control::lease::descriptor_id_and_prior_version(entry, shared) - && prior_version > 0 - { - crate::control::lease::drain_for_ddl( - shared, - descriptor_id, - prior_version, - DEFAULT_DRAIN_TIMEOUT, - // No transactional lease scope of its own: this is a bare, - // unbuffered DDL statement, not a COMMIT finalizing buffered DDL - // alongside a buffered write to the same descriptor. - 0, - )?; - } - - // Freeze the descriptor_version / constraint_version / - // modification_hlc HERE, at propose time, so the value is computed - // exactly once from this node's local catalog (`prior + 1`) and - // then replicated verbatim inside the entry. Every node applies the - // frozen value without re-deriving it, which makes replay-from-log - // on restart and re-delivery during learner catch-up idempotent — - // the divergence that a per-node apply-time stamp produced is gone. - // - // Gated on the same rolling-upgrade flag the apply path used to - // gate on: only stamp once every node can activate descriptor - // versioning; otherwise leave the entry's sentinel version `0` - // (downstream resolvers treat `0` as `1`). Older nodes in a - // mixed-version cluster lack the stamp logic, so a stamped value - // would not be reproduced symmetrically there. - let stamped_owned; - let entry: &CatalogEntry = if shared - .cluster_version_view() - .can_activate_feature(crate::control::rolling_upgrade::DESCRIPTOR_VERSIONING_VERSION) - { - stamped_owned = catalog_entry::descriptor_stamp::stamp( - entry.clone(), - &shared.hlc_clock, - shared.credentials.catalog(), - ); - &stamped_owned - } else { - entry - }; - - let payload = catalog_entry::encode(entry)?; - - // Attach J.4 audit context when the pgwire statement boundary - // installed one. Internal callers (descriptor lease grant/release, - // drain proposer) run outside that scope and emit the plain - // `CatalogDdl` variant — they have no SQL text to log. - let catalog_entry = match crate::control::server::shared::session::audit_context::current() { - Some(ctx) => MetadataEntry::CatalogDdlAudited { - payload, - auth_user_id: ctx.auth_user_id, - auth_user_name: ctx.auth_user_name, - sql_text: ctx.sql_text, - }, - None => MetadataEntry::CatalogDdl { payload }, - }; - let metadata_entry = MetadataEntry::DdlPrepared { - token: distributed_ddl_guard.token(), - entry: Box::new(catalog_entry), - }; - let raw = encode_entry(&metadata_entry).map_err(|e| Error::Config { - detail: format!("metadata entry encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - // `wait_for` blocks the calling thread on a Condvar. When the - // caller is already inside a tokio task (pgwire handlers always - // are), parking the worker without telling tokio starves every - // other task that lands on it — including the raft tick that - // would otherwise bump the watcher. Wrap the blocking section - // in `block_in_place` so tokio reassigns a fresh worker. - let outcome = tokio::task::block_in_place(|| watcher.wait_for(log_index, timeout)); - match outcome { - WaitOutcome::Reached - if shared.metadata_ddl_applied_token.load(Ordering::Acquire) - == distributed_ddl_guard.token() => - { - Ok(ProposeOutcome::Replicated { log_index }) - } - WaitOutcome::Reached => Err(Error::Config { - detail: "metadata DDL preparation ownership was superseded before apply".into(), - }), - WaitOutcome::TimedOut => Err(Error::Config { - detail: format!( - "metadata propose timed out after {:?} waiting for log index {} (current: {})", - timeout, - log_index, - watcher.current() - ), - }), - WaitOutcome::GroupGone => Err(Error::Config { - detail: "metadata group no longer hosted on this node".into(), - }), - } -} - -/// Propose a surrogate high-watermark advance to the metadata Raft group -/// and wait for it to be applied locally. -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — the WAL-only path on `SharedState` is -/// still sufficient. In cluster mode this is called by the leader-side -/// flush path instead of (or in addition to) the local WAL record, so -/// every follower's `SurrogateRegistry` advances to the same hwm via the -/// Raft commit. -/// -/// `hwm` is the highest surrogate that has been issued so far on this -/// node. Followers apply the entry by calling -/// `SurrogateRegistry::restore_hwm(hwm)` (idempotent, monotonic). -pub fn propose_surrogate_hwm(shared: &SharedState, hwm: u32) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SurrogateAlloc { hwm }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("surrogate_alloc encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!("surrogate_alloc propose timed out waiting for log index {log_index}"), - }); - } - - Ok(log_index) -} - -/// Propose a HiLo surrogate batch reservation to the metadata Raft group -/// and wait for the commit (returns the assigned log index). -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — single-node uses the local `alloc_one` -/// path and never reaches here. Kept as a safety guard only. -/// -/// The carved `[start, end)` range is NOT decided here: it is computed -/// at apply time on every node by advancing the global watermark in -/// identical log order (see `MetadataEntry::SurrogateReserve`). The -/// caller therefore cannot learn the range from this commit-wait alone -/// — `wait_for` returns on COMMIT, before the apply handler runs. The -/// owning node's apply handler fires an explicit completion signal -/// (`SurrogateAssigner::complete_reservation`) that the caller awaits -/// separately to learn the range. -/// -/// `node_id` + `request_id` identify this node's specific in-flight -/// reservation so the apply handler routes the batch + signal back to it. -pub fn propose_surrogate_reserve( - shared: &SharedState, - node_id: u64, - request_id: u64, - batch_size: u32, -) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SurrogateReserve { - node_id, - request_id, - batch_size, - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("surrogate_reserve encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!( - "surrogate_reserve propose timed out waiting for log index {log_index}" - ), - }); - } - - Ok(log_index) -} - -/// Propose a Lite client registration through the metadata Raft group and -/// wait for it to be applied locally. -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — the local registry write already persisted -/// the state. In cluster mode every follower applies the entry via -/// `SyncProducerRegistry::apply_register` so the `(producer_id, epoch)` pair -/// agrees on all nodes and survives leader failover. -pub fn propose_sync_producer_register( - shared: &SharedState, - lite_id: &str, - producer_id: u64, - tenant_id: u64, - user_id: u64, - epoch: u64, - created_ms: i64, -) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SyncProducerRegister { - lite_id: lite_id.to_owned(), - producer_id, - tenant_id, - user_id, - epoch, - created_ms, - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("sync_producer_register encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!( - "sync_producer_register propose timed out waiting for log index {log_index}" - ), - }); - } - - Ok(log_index) -} - -/// Propose a Lite client epoch fence through the metadata Raft group and -/// wait for it to be applied locally. -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — the local registry write already persisted -/// the state. In cluster mode every follower applies the entry via -/// `SyncProducerRegistry::apply_fence` (max-wins) so the epoch advance -/// survives leader failover. -pub fn propose_sync_producer_fence( - shared: &SharedState, - lite_id: &str, - new_epoch: u64, -) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SyncProducerFence { - lite_id: lite_id.to_owned(), - new_epoch, - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("sync_producer_fence encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!( - "sync_producer_fence propose timed out waiting for log index {log_index}" - ), - }); - } - - Ok(log_index) -} - -/// Propose ownership of one Loro peer id through the metadata Raft group and -/// wait for it to be applied locally. -/// -/// In single-node / no-cluster mode (no `metadata_raft` installed), returns -/// `Ok(0)` immediately — the local registry write already persisted the -/// ownership. In cluster mode the caller must re-read the owner after this -/// returns: the apply is lowest-producer-id-wins, so a node that lost a race it -/// did not know it was in learns the real owner only once the entry lands. -pub fn propose_sync_peer_bind( - shared: &SharedState, - binding: &crate::control::security::catalog::sync_producer::PeerBindingKey, - producer_id: u64, - bound_ms: i64, -) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; - - let entry = MetadataEntry::SyncPeerBind { - database_id: binding.database_id, - tenant_id: binding.tenant_id, - collection: binding.collection.clone(), - peer_id: binding.peer_id, - producer_id, - bound_ms, - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { - detail: format!("sync_peer_bind encode: {e}"), - })?; - - let log_index = handle.propose(raw)?; - - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!("sync_peer_bind propose timed out waiting for log index {log_index}"), - }); - } - - Ok(log_index) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn watcher_helper_returns_reached_on_past_target() { - let w = AppliedIndexWatcher::new(); - w.bump(10); - assert!(w.wait_for(5, Duration::from_millis(1)).is_reached()); - } -} diff --git a/nodedb/src/control/metadata_proposer/catalog.rs b/nodedb/src/control/metadata_proposer/catalog.rs new file mode 100644 index 000000000..8f2d7d10d --- /dev/null +++ b/nodedb/src/control/metadata_proposer/catalog.rs @@ -0,0 +1,208 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Propose a catalog entry and wait until this node applied it. + +use std::sync::atomic::Ordering; +use std::time::Duration; + +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, WaitOutcome, encode_entry}; + +use crate::control::catalog_entry::{self, CatalogEntry}; +use crate::control::propose_outcome::ProposeOutcome; +use crate::control::state::SharedState; +use crate::error::Error; + +use super::ddl_prepare::{acquire_ddl_prepare_lease, lock_ddl_preparation}; +use super::timeouts::{DEFAULT_DRAIN_TIMEOUT, DEFAULT_PROPOSE_TIMEOUT}; + +/// Propose a `CatalogEntry` and block until the local applied-index +/// watcher confirms the entry has been applied on this node. +/// +/// The returned [`ProposeOutcome`] tells the caller whether to write the +/// catalog itself, leave it to the applier, or do nothing because the entry +/// is held for COMMIT. +pub fn propose_catalog_entry( + shared: &SharedState, + entry: &CatalogEntry, +) -> Result { + propose_catalog_entry_with_timeout(shared, entry, DEFAULT_PROPOSE_TIMEOUT) +} + +/// Same as [`propose_catalog_entry`] but with an explicit timeout. +/// +/// An entry that changes authorization state returns only once it binds +/// every node: the authorization barrier runs after the local apply, with the +/// DDL preparation lock already released. +pub fn propose_catalog_entry_with_timeout( + shared: &SharedState, + entry: &CatalogEntry, + timeout: Duration, +) -> Result { + let outcome = propose_and_apply_locally(shared, entry, timeout)?; + if let ProposeOutcome::Replicated { log_index } = outcome + && entry.bears_authorization() + { + crate::control::security::auth_lease::block_on_barrier( + shared, + vec![nodedb_cluster::GroupCoverage { + group_id: METADATA_GROUP_ID, + through: log_index, + }], + )?; + } + Ok(outcome) +} + +/// Propose `entry` and wait until this node applied it. +fn propose_and_apply_locally( + shared: &SharedState, + entry: &CatalogEntry, + timeout: Duration, +) -> Result { + // Buffering is decided first, ahead of every replication-mode gate: an open + // transaction owns the entry regardless of whether this deployment + // replicates DDL, and COMMIT re-runs the mode choice for the whole batch. + // Entries also stay unstamped until then, so repeated mutations of one + // descriptor receive distinct versions in commit order. + if crate::control::server::shared::session::ddl_buffer::try_buffer(entry.clone()) { + return Ok(ProposeOutcome::Buffered); + } + + let Some(handle) = shared.metadata_raft.get() else { + return Ok(ProposeOutcome::LocalOnly); + }; + + // Rolling-upgrade gate: until every node in the cluster reports + // at least `DISTRIBUTED_CATALOG_VERSION`, fall back to the legacy + // direct-write path on the originating node. Mixing the + // replicated and direct paths during a partial upgrade would + // diverge catalog state across nodes — see + // `control/rolling_upgrade.rs`. + if !shared + .cluster_version_view() + .can_activate_feature(crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION) + { + tracing::warn!( + min_version = shared.cluster_version_view().min_version, + required = crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, + "metadata propose: cluster in compat mode (mixed-version), \ + falling back to legacy direct-write path" + ); + return Ok(ProposeOutcome::LocalOnly); + } + + // Serialize preparation through local apply confirmation. Without this, + // concurrent proposers can both observe persisted version N and emit N+1. + let _local_ddl_guard = lock_ddl_preparation(shared)?; + let distributed_ddl_guard = acquire_ddl_prepare_lease(shared, handle.as_ref())?; + + // Drain for Put* variants that carry descriptor_version. + // Leases acquired at plan time are refcounted and held + // through execute; when the last in-flight query using a + // descriptor completes, its `QueryLeaseScope` drops and the + // refcount hits zero, releasing the lease. Drain is what + // makes this an actual barrier: the proposer waits for all + // prior-version leases to release before committing the new + // `Put*`, giving long-running in-flight queries a bounded + // window (DEFAULT_DRAIN_TIMEOUT) to finish. + if let Some((descriptor_id, prior_version)) = + crate::control::lease::descriptor_id_and_prior_version(entry, shared) + && prior_version > 0 + { + crate::control::lease::drain_for_ddl( + shared, + descriptor_id, + prior_version, + DEFAULT_DRAIN_TIMEOUT, + // No transactional lease scope of its own: this is a bare, + // unbuffered DDL statement, not a COMMIT finalizing buffered DDL + // alongside a buffered write to the same descriptor. + 0, + )?; + } + + // Freeze the descriptor_version / constraint_version / + // modification_hlc HERE, at propose time, so the value is computed + // exactly once from this node's local catalog (`prior + 1`) and + // then replicated verbatim inside the entry. Every node applies the + // frozen value without re-deriving it, which makes replay-from-log + // on restart and re-delivery during learner catch-up idempotent — + // the divergence that a per-node apply-time stamp produced is gone. + // + // Gated on the same rolling-upgrade flag the apply path used to + // gate on: only stamp once every node can activate descriptor + // versioning; otherwise leave the entry's sentinel version `0` + // (downstream resolvers treat `0` as `1`). Older nodes in a + // mixed-version cluster lack the stamp logic, so a stamped value + // would not be reproduced symmetrically there. + let stamped_owned; + let entry: &CatalogEntry = if shared + .cluster_version_view() + .can_activate_feature(crate::control::rolling_upgrade::DESCRIPTOR_VERSIONING_VERSION) + { + stamped_owned = catalog_entry::descriptor_stamp::stamp( + entry.clone(), + &shared.hlc_clock, + shared.credentials.catalog(), + ); + &stamped_owned + } else { + entry + }; + + let payload = catalog_entry::encode(entry)?; + + // Attach J.4 audit context when the pgwire statement boundary + // installed one. Internal callers (descriptor lease grant/release, + // drain proposer) run outside that scope and emit the plain + // `CatalogDdl` variant — they have no SQL text to log. + let catalog_entry = match crate::control::server::shared::session::audit_context::current() { + Some(ctx) => MetadataEntry::CatalogDdlAudited { + payload, + auth_user_id: ctx.auth_user_id, + auth_user_name: ctx.auth_user_name, + sql_text: ctx.sql_text, + }, + None => MetadataEntry::CatalogDdl { payload }, + }; + let metadata_entry = MetadataEntry::DdlPrepared { + token: distributed_ddl_guard.token(), + entry: Box::new(catalog_entry), + }; + let raw = encode_entry(&metadata_entry).map_err(|e| Error::Config { + detail: format!("metadata entry encode: {e}"), + })?; + + let log_index = handle.propose(raw)?; + + let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); + // `wait_for` blocks the calling thread on a Condvar. When the + // caller is already inside a tokio task (pgwire handlers always + // are), parking the worker without telling tokio starves every + // other task that lands on it — including the raft tick that + // would otherwise bump the watcher. Wrap the blocking section + // in `block_in_place` so tokio reassigns a fresh worker. + let outcome = tokio::task::block_in_place(|| watcher.wait_for(log_index, timeout)); + match outcome { + WaitOutcome::Reached + if shared.metadata_ddl_applied_token.load(Ordering::Acquire) + == distributed_ddl_guard.token() => + { + Ok(ProposeOutcome::Replicated { log_index }) + } + WaitOutcome::Reached => Err(Error::Config { + detail: "metadata DDL preparation ownership was superseded before apply".into(), + }), + WaitOutcome::TimedOut => Err(Error::Config { + detail: format!( + "metadata propose timed out after {:?} waiting for log index {} (current: {})", + timeout, + log_index, + watcher.current() + ), + }), + WaitOutcome::GroupGone => Err(Error::Config { + detail: "metadata group no longer hosted on this node".into(), + }), + } +} diff --git a/nodedb/src/control/metadata_proposer/ddl_prepare.rs b/nodedb/src/control/metadata_proposer/ddl_prepare.rs new file mode 100644 index 000000000..7f5310325 --- /dev/null +++ b/nodedb/src/control/metadata_proposer/ddl_prepare.rs @@ -0,0 +1,192 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The DDL preparation lease: the local lock and the replicated lease that +//! serialize descriptor preparation across the cluster. + +use std::sync::atomic::Ordering; +use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; + +use tokio::runtime::RuntimeFlavor; + +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, WaitOutcome, encode_entry}; + +use crate::control::state::SharedState; +use crate::error::Error; + +use super::handle::MetadataRaftHandle; +use super::timeouts::DEFAULT_PROPOSE_TIMEOUT; + +const DDL_PREPARE_LEASE: Duration = Duration::from_secs(60); +const DDL_PREPARE_WAIT: Duration = Duration::from_secs(70); + +fn wall_now_ns() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_nanos() + .min(u64::MAX as u128) as u64 +} + +fn propose_metadata_and_wait( + shared: &SharedState, + handle: &dyn MetadataRaftHandle, + entry: &MetadataEntry, + timeout: Duration, +) -> Result { + let raw = encode_entry(entry).map_err(|e| Error::Config { + detail: format!("metadata entry encode: {e}"), + })?; + let index = handle.propose(raw)?; + let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); + let outcome = tokio::task::block_in_place(|| watcher.wait_for(index, timeout)); + match outcome { + WaitOutcome::Reached => Ok(index), + WaitOutcome::TimedOut => Err(Error::Config { + detail: format!( + "metadata propose timed out after {timeout:?} waiting for log index {index} (current: {})", + watcher.current() + ), + }), + WaitOutcome::GroupGone => Err(Error::Config { + detail: "metadata group no longer hosted on this node".into(), + }), + } +} + +/// RAII ownership of the metadata-Raft-serialized descriptor preparation lease. +/// The matching release is itself replicated, so another node cannot stamp from +/// the same prior catalog version until this guard is dropped and that release +/// has applied. +pub(crate) struct DdlPrepareGuard<'a> { + shared: &'a SharedState, + handle: &'a dyn MetadataRaftHandle, + token: u64, +} + +impl DdlPrepareGuard<'_> { + pub(crate) fn token(&self) -> u64 { + self.token + } +} + +impl Drop for DdlPrepareGuard<'_> { + fn drop(&mut self) { + if let Err(error) = propose_metadata_and_wait( + self.shared, + self.handle, + &MetadataEntry::DdlPrepareRelease { token: self.token }, + DEFAULT_PROPOSE_TIMEOUT, + ) { + tracing::error!(token = self.token, %error, "metadata DDL lease release failed"); + } + } +} + +pub(crate) fn acquire_ddl_prepare_lease<'a>( + shared: &'a SharedState, + handle: &'a dyn MetadataRaftHandle, +) -> Result, Error> { + let sequence = shared + .metadata_ddl_token_seq + .fetch_add(1, Ordering::Relaxed); + let token = shared.node_id.wrapping_mul(0x9e37_79b9_7f4a_7c15) + ^ wall_now_ns().rotate_left(17) + ^ sequence; + let deadline = Instant::now() + DDL_PREPARE_WAIT; + + loop { + propose_metadata_and_wait( + shared, + handle, + &MetadataEntry::DdlPrepareAcquire { token }, + DEFAULT_PROPOSE_TIMEOUT, + )?; + + loop { + let owner = *shared + .metadata_ddl_owner + .lock() + .map_err(|_| Error::Config { + detail: "metadata DDL owner lock poisoned".into(), + })?; + match owner { + Some((current, _)) if current == token => { + return Ok(DdlPrepareGuard { + shared, + handle, + token, + }); + } + Some((current, acquired_at)) + if shared.is_metadata_leader() + && acquired_at.elapsed() >= DDL_PREPARE_LEASE => + { + // Cancel the dead owner's pending record before releasing its + // lease, so it never lingers visible-but-unresolved past the lease. + if shared.pending_ddl.contains(current) { + propose_metadata_and_wait( + shared, + handle, + &MetadataEntry::DdlPendingCancel { token: current }, + DEFAULT_PROPOSE_TIMEOUT, + )?; + } + propose_metadata_and_wait( + shared, + handle, + &MetadataEntry::DdlPrepareRelease { token: current }, + DEFAULT_PROPOSE_TIMEOUT, + )?; + break; + } + None => break, + Some(_) if Instant::now() < deadline => { + // Reached from async tasks (ILP batch flush -> + // `propose_catalog_entry`), so hand the worker back to + // tokio rather than parking it: the lease owner this + // polls for is released by a raft apply that needs a + // worker to make progress. + tokio::task::block_in_place(|| { + std::thread::sleep(Duration::from_millis(10)); + }); + } + Some(_) => { + return Err(Error::Config { + detail: "metadata DDL preparation lease timed out".into(), + }); + } + } + } + } +} + +/// Take the local DDL preparation lock, handing the wait back to tokio when +/// the caller is on a multi-thread worker. +/// +/// The holder keeps this lock across the distributed preparation lease, the +/// descriptor drain and the local apply wait — each already wrapped in +/// `block_in_place`, but that only tells tokio about the waits *inside* the +/// lock, never about the wait *for* it. A bare `lock()` on a worker therefore +/// removes that worker from the runtime silently, including from the raft +/// apply work the current holder needs in order to finish, which turns +/// contention into a self-sustaining stall. +/// +/// `block_in_place` is a passthrough outside a multi-thread worker (plain sync +/// callers, blocking-pool threads) and panics on the current-thread runtime, +/// so it is applied only where it is both legal and meaningful — mirroring +/// `lease::drain_propose::poll_leases_drained`. +pub(super) fn lock_ddl_preparation( + shared: &SharedState, +) -> Result, Error> { + let acquire = || { + shared.metadata_ddl_lock.lock().map_err(|_| Error::Config { + detail: "metadata DDL preparation lock poisoned".into(), + }) + }; + match tokio::runtime::Handle::try_current() { + Ok(handle) if handle.runtime_flavor() == RuntimeFlavor::MultiThread => { + tokio::task::block_in_place(acquire) + } + _ => acquire(), + } +} diff --git a/nodedb/src/control/metadata_proposer/handle.rs b/nodedb/src/control/metadata_proposer/handle.rs new file mode 100644 index 000000000..608c3d33d --- /dev/null +++ b/nodedb/src/control/metadata_proposer/handle.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The handle DDL proposes metadata entries through. + +use std::sync::{Arc, Weak}; + +use crate::error::Error; + +/// Type-erased handle for proposing to the metadata raft group. +/// +/// The apply watermark for the metadata group lives on +/// [`crate::control::state::SharedState::applied_index_watcher`] (keyed by +/// [`nodedb_cluster::METADATA_GROUP_ID`]); callers of [`Self::propose`] +/// look it up there rather than receiving it through this handle. +pub trait MetadataRaftHandle: Send + Sync { + /// Propose a raw encoded `MetadataEntry` to the metadata group. + /// Returns its assigned log index on success. + fn propose(&self, bytes: Vec) -> Result; +} + +/// Concrete impl wrapping `nodedb_cluster::RaftLoop`. +/// +/// Holds the loop weakly: this handle lives on `SharedState`, which is +/// itself kept alive transitively by the `RaftLoop`, so a strong +/// reference here would close a cycle that pins both forever and blocks +/// clean shutdown. The loop is kept alive by its own spawned tasks; +/// `upgrade` therefore succeeds throughout normal operation and only +/// fails once the loop has been dropped on shutdown. +pub struct RaftLoopProposerHandle { + raft_loop: Weak< + nodedb_cluster::RaftLoop< + crate::control::cluster::SpscCommitApplier, + crate::control::LocalPlanExecutor, + >, + >, +} + +impl RaftLoopProposerHandle { + pub fn new( + raft_loop: Arc< + nodedb_cluster::RaftLoop< + crate::control::cluster::SpscCommitApplier, + crate::control::LocalPlanExecutor, + >, + >, + ) -> Self { + Self { + raft_loop: Arc::downgrade(&raft_loop), + } + } +} + +impl MetadataRaftHandle for RaftLoopProposerHandle { + fn propose(&self, bytes: Vec) -> Result { + // The cluster crate's `propose_to_metadata_group_via_leader` + // is async because it may need to forward to the metadata + // leader over QUIC. The trait method is sync because every + // caller (catalog DDL handlers, lease grant/release helpers) + // is itself sync but runs inside a tokio task. Wrap in + // `block_in_place` + the current runtime's `block_on` so the + // forwarding QUIC round-trip drives without starving the + // raft tick that produces the leader_hint. + // `upgrade` fails only once the raft loop has been dropped on + // shutdown; a request racing shutdown then fails cleanly with a + // typed error instead of panicking. + let raft_loop = self.raft_loop.upgrade().ok_or_else(|| Error::Config { + detail: "metadata propose: cluster not running".into(), + })?; + tokio::task::block_in_place(|| { + tokio::runtime::Handle::current() + .block_on(raft_loop.propose_to_metadata_group_via_leader(bytes)) + }) + .map_err(|e| match e { + // An election in progress is transient, not a failure of this + // proposal. Keep it typed rather than flattening it into a generic + // config error, so callers can wait the election out instead of + // failing the statement — a node that has just restarted answers + // every metadata proposal this way for a moment. + nodedb_cluster::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: None, + }) => Error::MetadataLeaderUnavailable, + other => Error::Config { + detail: format!("metadata propose: {other}"), + }, + }) + } +} diff --git a/nodedb/src/control/metadata_proposer/mod.rs b/nodedb/src/control/metadata_proposer/mod.rs new file mode 100644 index 000000000..6a12b572a --- /dev/null +++ b/nodedb/src/control/metadata_proposer/mod.rs @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Synchronous `propose-and-wait-for-local-apply` helper for +//! replicated catalog DDL. +//! +//! The sole entry point pgwire DDL handlers use to write a +//! [`crate::control::catalog_entry::CatalogEntry`] through the metadata raft group (group 0). It is +//! deliberately sync — pgwire DDL handlers are not async, and +//! `tokio::task::block_in_place`-style wrapping keeps the blocking +//! wait from starving the tokio runtime. +//! +//! Semantics: +//! +//! 1. If no cluster is configured (`shared.metadata_raft` not +//! installed), returns `ProposeOutcome::LocalOnly`. The caller's +//! single-node direct-write path stays authoritative. +//! 2. If this node is the metadata-group leader, proposes the +//! entry, blocks until its local applied watermark reaches the +//! assigned log index (5s default timeout), and returns the +//! log index on success. +//! 3. If this node is NOT the leader, returns +//! `Error::Config { detail: "metadata propose: not leader ..." }`. +//! Gateway-side redirection will make this transparent. + +pub mod catalog; +pub mod ddl_prepare; +pub mod handle; +pub mod replicated_entries; +pub mod timeouts; + +pub use catalog::{propose_catalog_entry, propose_catalog_entry_with_timeout}; +pub(crate) use ddl_prepare::{DdlPrepareGuard, acquire_ddl_prepare_lease}; +pub use handle::{MetadataRaftHandle, RaftLoopProposerHandle}; +pub use replicated_entries::{ + propose_surrogate_hwm, propose_surrogate_reserve, propose_sync_peer_bind, + propose_sync_producer_fence, propose_sync_producer_register, +}; +pub use timeouts::{DEFAULT_DRAIN_TIMEOUT, DEFAULT_PROPOSE_TIMEOUT}; diff --git a/nodedb/src/control/metadata_proposer/replicated_entries.rs b/nodedb/src/control/metadata_proposer/replicated_entries.rs new file mode 100644 index 000000000..f45af8b4e --- /dev/null +++ b/nodedb/src/control/metadata_proposer/replicated_entries.rs @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Metadata entries that replicate node-local registries: surrogate +//! allocation and Lite sync producers. +//! +//! Each one proposes the entry and waits for its commit on this node. In +//! single-node mode (no `metadata_raft` installed) each returns `Ok(0)`: the +//! local write already persisted the state. + +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, encode_entry}; + +use crate::control::state::SharedState; +use crate::error::Error; + +use super::timeouts::DEFAULT_PROPOSE_TIMEOUT; + +/// Propose `entry` and wait until this node reaches its log index. `label` +/// names the entry in the errors. Returns `Ok(0)` when no cluster runs. +fn propose_and_wait( + shared: &SharedState, + entry: &MetadataEntry, + label: &str, +) -> Result { + let Some(handle) = shared.metadata_raft.get() else { + return Ok(0); + }; + let raw = encode_entry(entry).map_err(|e| Error::Config { + detail: format!("{label} encode: {e}"), + })?; + + let log_index = handle.propose(raw)?; + + let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); + let outcome = + tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); + if !outcome.is_reached() { + return Err(Error::Config { + detail: format!("{label} propose timed out waiting for log index {log_index}"), + }); + } + + Ok(log_index) +} + +/// Propose a surrogate high-watermark advance to the metadata Raft group +/// and wait for it to be applied locally. +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), +/// returns `Ok(0)` immediately — the WAL-only path on `SharedState` is +/// still sufficient. In cluster mode this is called by the leader-side +/// flush path instead of (or in addition to) the local WAL record, so +/// every follower's `SurrogateRegistry` advances to the same hwm via the +/// Raft commit. +/// +/// `hwm` is the highest surrogate that has been issued so far on this +/// node. Followers apply the entry by calling +/// `SurrogateRegistry::restore_hwm(hwm)` (idempotent, monotonic). +pub fn propose_surrogate_hwm(shared: &SharedState, hwm: u32) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SurrogateAlloc { hwm }, + "surrogate_alloc", + ) +} + +/// Propose a HiLo surrogate batch reservation to the metadata Raft group +/// and wait for the commit (returns the assigned log index). +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), +/// returns `Ok(0)` immediately — single-node uses the local `alloc_one` +/// path and never reaches here. Kept as a safety guard only. +/// +/// The carved `[start, end)` range is NOT decided here: it is computed +/// at apply time on every node by advancing the global watermark in +/// identical log order (see `MetadataEntry::SurrogateReserve`). The +/// caller therefore cannot learn the range from this commit-wait alone +/// — `wait_for` returns on COMMIT, before the apply handler runs. The +/// owning node's apply handler fires an explicit completion signal +/// (`SurrogateAssigner::complete_reservation`) that the caller awaits +/// separately to learn the range. +/// +/// `node_id` + `request_id` identify this node's specific in-flight +/// reservation so the apply handler routes the batch + signal back to it. +pub fn propose_surrogate_reserve( + shared: &SharedState, + node_id: u64, + request_id: u64, + batch_size: u32, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SurrogateReserve { + node_id, + request_id, + batch_size, + }, + "surrogate_reserve", + ) +} + +/// Propose a Lite client registration through the metadata Raft group and +/// wait for it to be applied locally. +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), +/// returns `Ok(0)` immediately — the local registry write already persisted +/// the state. In cluster mode every follower applies the entry via +/// `SyncProducerRegistry::apply_register` so the `(producer_id, epoch)` pair +/// agrees on all nodes and survives leader failover. +pub fn propose_sync_producer_register( + shared: &SharedState, + lite_id: &str, + producer_id: u64, + tenant_id: u64, + user_id: u64, + epoch: u64, + created_ms: i64, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SyncProducerRegister { + lite_id: lite_id.to_owned(), + producer_id, + tenant_id, + user_id, + epoch, + created_ms, + }, + "sync_producer_register", + ) +} + +/// Propose a Lite client epoch fence through the metadata Raft group and +/// wait for it to be applied locally. +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), +/// returns `Ok(0)` immediately — the local registry write already persisted +/// the state. In cluster mode every follower applies the entry via +/// `SyncProducerRegistry::apply_fence` (max-wins) so the epoch advance +/// survives leader failover. +pub fn propose_sync_producer_fence( + shared: &SharedState, + lite_id: &str, + new_epoch: u64, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SyncProducerFence { + lite_id: lite_id.to_owned(), + new_epoch, + }, + "sync_producer_fence", + ) +} + +/// Propose ownership of one Loro peer id through the metadata Raft group and +/// wait for it to be applied locally. +/// +/// In single-node / no-cluster mode (no `metadata_raft` installed), returns +/// `Ok(0)` immediately — the local registry write already persisted the +/// ownership. In cluster mode the caller must re-read the owner after this +/// returns: the apply is lowest-producer-id-wins, so a node that lost a race it +/// did not know it was in learns the real owner only once the entry lands. +pub fn propose_sync_peer_bind( + shared: &SharedState, + binding: &crate::control::security::catalog::sync_producer::PeerBindingKey, + producer_id: u64, + bound_ms: i64, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::SyncPeerBind { + database_id: binding.database_id, + tenant_id: binding.tenant_id, + collection: binding.collection.clone(), + peer_id: binding.peer_id, + producer_id, + bound_ms, + }, + "sync_peer_bind", + ) +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use nodedb_cluster::AppliedIndexWatcher; + + #[test] + fn watcher_helper_returns_reached_on_past_target() { + let w = AppliedIndexWatcher::new(); + w.bump(10); + assert!(w.wait_for(5, Duration::from_millis(1)).is_reached()); + } +} diff --git a/nodedb/src/control/metadata_proposer/timeouts.rs b/nodedb/src/control/metadata_proposer/timeouts.rs new file mode 100644 index 000000000..c1fe63ddb --- /dev/null +++ b/nodedb/src/control/metadata_proposer/timeouts.rs @@ -0,0 +1,21 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! How long a metadata proposal and a DDL drain wait. + +use std::time::Duration; + +/// Default upper bound on how long a single +/// `propose_catalog_entry` call will block before returning an +/// error. +pub const DEFAULT_PROPOSE_TIMEOUT: Duration = Duration::from_secs(5); + +/// Default upper bound on how long a DDL drain will wait for +/// prior-version leases to release before giving up. Must be at +/// least `ClusterTransportTuning::descriptor_lease_duration_secs` +/// so an existing lease gets at least one full lifetime to +/// expire naturally. 35 seconds matches the 300s lease duration +/// plus a 30-second grace minus the typical 5-minute default +/// cut down for test budget — in production +/// `propose_catalog_entry_with_drain_timeout` can pass a longer +/// value if an operator is willing to wait. +pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(35); diff --git a/nodedb/src/control/planner/calvin/dependent_recon.rs b/nodedb/src/control/planner/calvin/dependent_recon.rs index caed1c7e5..1931bd4f0 100644 --- a/nodedb/src/control/planner/calvin/dependent_recon.rs +++ b/nodedb/src/control/planner/calvin/dependent_recon.rs @@ -400,6 +400,20 @@ async fn dispatch_dependent_edge_recon_inner( } }; + // A write to a permission-tree source is acknowledged only once it binds + // every node. Tree sources live in the default database. + let sources = state.authorization_fence.sources(); + let binds_authorization = database_id == crate::types::DatabaseId::DEFAULT + && tasks.iter().any(|task| { + task.plan + .named_collections() + .iter() + .any(|collection| sources.is_source_collection(collection)) + }); + if binds_authorization { + crate::control::security::auth_lease::calvin_write_barrier(state).await?; + } + // Completion fired: the scheduler deposited the applied Response (with any // RETURNING rows) into the sidecar before proposing the ack that woke the // retry loop, so the entry is present now if this write carried RETURNING. @@ -407,7 +421,8 @@ async fn dispatch_dependent_edge_recon_inner( // `Conflict` (>1 RETURNING participant) fails loudly rather than returning a // partial cross-shard union. let drained = state - .calvin_apply_results + .calvin + .apply_results .lock() .unwrap_or_else(|p| p.into_inner()) .remove(&completed_txn); diff --git a/nodedb/src/control/planner/calvin/submit/local.rs b/nodedb/src/control/planner/calvin/submit/local.rs index 84e8f496a..b209bdc88 100644 --- a/nodedb/src/control/planner/calvin/submit/local.rs +++ b/nodedb/src/control/planner/calvin/submit/local.rs @@ -76,6 +76,20 @@ pub async fn submit_and_await_calvin_with_timeout( .get() .ok_or(Error::SequencerUnavailable)?; + // A write to a permission-tree source is acknowledged only once it binds + // every node. Tree sources live in the default database. + let binds_authorization = tx_class.database_id == crate::types::DatabaseId::DEFAULT + && tx_class + .write_set + .participating_vshards_in_database(tx_class.database_id) + .iter() + .any(|vshard| { + state + .authorization_fence + .sources() + .is_source_vshard(vshard.as_u32()) + }); + let inbox_seq = inbox.submit(tx_class).map_err(|e| Error::BadRequest { detail: format!("Calvin sequencer rejected transaction: {e}"), })?; @@ -141,6 +155,9 @@ pub async fn submit_and_await_calvin_with_timeout( detail: "OLLP mismatch outcome on non-dependent Calvin path".to_owned(), }); } + if binds_authorization { + crate::control::security::auth_lease::calvin_write_barrier(state).await?; + } // Completion fired: the scheduler deposited the applied Response (with any // RETURNING rows) into the sidecar BEFORE proposing the ack that woke this @@ -150,7 +167,8 @@ pub async fn submit_and_await_calvin_with_timeout( // `Conflict` (>1 RETURNING participant) fails loudly rather than returning a // partial cross-shard union. let drained = state - .calvin_apply_results + .calvin + .apply_results .lock() .unwrap_or_else(|p| p.into_inner()) .remove(&TxnId::new(epoch, position)); diff --git a/nodedb/src/control/security/auth_fence/cluster.rs b/nodedb/src/control/security/auth_fence/cluster.rs new file mode 100644 index 000000000..a99dfb790 --- /dev/null +++ b/nodedb/src/control/security/auth_fence/cluster.rs @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Cluster helpers shared by the planning view and the authorization lease: +//! hosting checks, confirmed read indexes and applied-index waits. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::WaitOutcome; + +use crate::control::cluster::read_index::ReadIndexRefusal; +use crate::control::state::SharedState; + +/// Refusal for a statement whose authorization state is not current. +pub(crate) fn behind(detail: impl Into) -> crate::Error { + crate::Error::AuthorizationStateBehind { + detail: detail.into(), + } +} + +/// Whether this node replicates `group_id`, as a voter or a learner. +pub(crate) fn hosts_group(state: &SharedState, group_id: u64) -> bool { + let Some(routing) = state.cluster_routing.as_ref() else { + return false; + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing.group_info(group_id).is_some_and(|info| { + info.members.contains(&state.node_id) || info.learners.contains(&state.node_id) + }) +} + +/// The data group that homes `vshard_id`, from this node's routing table. +pub(crate) fn group_of_vshard(state: &SharedState, vshard_id: u32) -> crate::Result { + let Some(routing) = state.cluster_routing.as_ref() else { + return Err(behind("no routing table on this node")); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing + .group_for_vshard(vshard_id) + .map_err(|e| behind(format!("raft group of vShard {vshard_id}: {e}"))) +} + +/// Every data group in this node's routing table. +pub(crate) fn routed_groups(state: &SharedState) -> Vec { + let Some(routing) = state.cluster_routing.as_ref() else { + return Vec::new(); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing.group_ids() +} + +/// A read index of `group_id` confirmed by its leader against a quorum, +/// taken by a probe that started after this call. +pub(crate) async fn confirmed_read_index( + state: &SharedState, + group_id: u64, + timeout: Duration, +) -> crate::Result { + let Some(gate) = state.raft_read_gate.get() else { + return Err(behind(format!( + "no read index service for raft group {group_id} yet" + ))); + }; + let deadline = Instant::now() + timeout; + state + .authorization_fence + .read_index_coalescer(group_id) + .read_index(deadline, || gate.read_index(group_id, timeout)) + .await + .map_err(|refusal| match refusal { + ReadIndexRefusal::NotLeader => { + behind(format!("raft group {group_id} has no reachable leader")) + } + ReadIndexRefusal::Timeout { waited_ms } => behind(format!( + "raft group {group_id} confirmed no read index within {waited_ms}ms" + )), + }) +} + +/// Wait until this node's applied index of `group_id` reaches `target`. +pub(crate) async fn wait_applied( + state: &SharedState, + group_id: u64, + target: u64, + timeout: Duration, +) -> crate::Result<()> { + let watcher = state.applied_index_watcher(group_id); + if watcher.current() >= target { + return Ok(()); + } + // The watcher parks its caller on a condition variable, so the wait runs + // on the blocking pool. + let outcome = tokio::task::spawn_blocking(move || watcher.wait_for(target, timeout)) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("applied-index wait for raft group {group_id} did not finish: {e}"), + })?; + match outcome { + WaitOutcome::Reached => Ok(()), + WaitOutcome::TimedOut => Err(behind(format!( + "raft group {group_id} was not applied through index {target} in time" + ))), + WaitOutcome::GroupGone => Err(behind(format!( + "raft group {group_id} left this node while it caught up" + ))), + } +} diff --git a/nodedb/src/control/security/auth_fence/mod.rs b/nodedb/src/control/security/auth_fence/mod.rs new file mode 100644 index 000000000..ea63f5a9f --- /dev/null +++ b/nodedb/src/control/security/auth_fence/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod cluster; +pub mod read_index; +pub mod state; +pub mod tree_defs; +pub mod view; + +pub use state::AuthorizationFence; +pub use tree_defs::{PendingTreeDefs, TreeDefChange}; +pub use view::permission_view; diff --git a/nodedb/src/control/security/auth_fence/read_index.rs b/nodedb/src/control/security/auth_fence/read_index.rs new file mode 100644 index 000000000..07cf96fe2 --- /dev/null +++ b/nodedb/src/control/security/auth_fence/read_index.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Coalesce concurrent read-index requests for one Raft group. +//! +//! A read index may answer a request only when the probe that produced it +//! started after the request arrived. A probe that started earlier can carry +//! a commit index below an entry acknowledged just before the request. So a +//! request that finds a probe in flight waits for the next one, and one probe +//! then answers every request that arrived while the previous probe ran. + +use std::future::Future; +use std::sync::Mutex; +use std::time::Instant; + +use tokio::sync::Notify; + +use crate::control::cluster::read_index::ReadIndexRefusal; + +/// Probe numbering and the last answer. +#[derive(Debug, Default)] +struct CoalescerState { + /// The probe running now, if any. + in_flight: Option, + /// The highest probe that finished. + completed: u64, + /// The answer of probe `completed`. + result: Option>, +} + +/// Coalesces read-index probes for one group. +#[derive(Debug, Default)] +pub struct ReadIndexCoalescer { + state: Mutex, + done: Notify, +} + +/// Clears the in-flight probe when its runner stops without an answer, so a +/// cancelled runner never blocks later requests. +struct RunGuard<'a> { + coalescer: &'a ReadIndexCoalescer, + probe: u64, + finished: bool, +} + +impl Drop for RunGuard<'_> { + fn drop(&mut self) { + if self.finished { + return; + } + { + let mut state = self.coalescer.lock(); + if state.in_flight == Some(self.probe) { + state.in_flight = None; + } + } + self.coalescer.done.notify_waiters(); + } +} + +impl ReadIndexCoalescer { + pub fn new() -> Self { + Self::default() + } + + fn lock(&self) -> std::sync::MutexGuard<'_, CoalescerState> { + self.state.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// A read index taken by a probe that started after this call, running + /// `probe` when no such probe is in flight. Refuses with a timeout once + /// `deadline` passes. + pub async fn read_index( + &self, + deadline: Instant, + probe: F, + ) -> Result + where + F: Fn() -> Fut, + Fut: Future>, + { + let started = Instant::now(); + let target = { + let state = self.lock(); + match state.in_flight { + Some(running) => running + 1, + None => state.completed + 1, + } + }; + loop { + let notified = self.done.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + + let run = { + let mut state = self.lock(); + if state.completed >= target + && let Some(result) = state.result + { + return result; + } + if state.in_flight.is_none() { + state.in_flight = Some(target); + true + } else { + false + } + }; + + if run { + let mut guard = RunGuard { + coalescer: self, + probe: target, + finished: false, + }; + let result = probe().await; + { + let mut state = self.lock(); + state.completed = state.completed.max(target); + state.result = Some(result); + state.in_flight = None; + } + guard.finished = true; + self.done.notify_waiters(); + return result; + } + + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { + return Err(ReadIndexRefusal::Timeout { + waited_ms: u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX), + }); + } + let _ = tokio::time::timeout(remaining, notified).await; + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::sync::atomic::{AtomicU64, Ordering}; + use std::time::Duration; + + use super::*; + + fn deadline() -> Instant { + Instant::now() + Duration::from_secs(5) + } + + #[tokio::test] + async fn a_request_with_no_probe_in_flight_runs_one() { + let coalescer = ReadIndexCoalescer::new(); + let probes = AtomicU64::new(0); + let index = coalescer + .read_index(deadline(), || async { + Ok(probes.fetch_add(1, Ordering::SeqCst) + 10) + }) + .await; + assert_eq!(index, Ok(10)); + assert_eq!(probes.load(Ordering::SeqCst), 1); + } + + /// A request that arrives while a probe runs never takes that probe's + /// answer: it waits for a probe that started after it. + #[tokio::test] + async fn a_request_arriving_during_a_probe_waits_for_the_next_one() { + let coalescer = Arc::new(ReadIndexCoalescer::new()); + let release = Arc::new(Notify::new()); + let probes = Arc::new(AtomicU64::new(0)); + + let first = { + let coalescer = Arc::clone(&coalescer); + let release = Arc::clone(&release); + let probes = Arc::clone(&probes); + tokio::spawn(async move { + coalescer + .read_index(deadline(), || { + let release = Arc::clone(&release); + let probes = Arc::clone(&probes); + async move { + let n = probes.fetch_add(1, Ordering::SeqCst) + 1; + if n == 1 { + release.notified().await; + } + Ok(n * 100) + } + }) + .await + }) + }; + while probes.load(Ordering::SeqCst) == 0 { + tokio::task::yield_now().await; + } + + let second = { + let coalescer = Arc::clone(&coalescer); + let probes = Arc::clone(&probes); + tokio::spawn(async move { + coalescer + .read_index(deadline(), || { + let probes = Arc::clone(&probes); + async move { Ok((probes.fetch_add(1, Ordering::SeqCst) + 1) * 100) } + }) + .await + }) + }; + tokio::task::yield_now().await; + release.notify_one(); + + assert_eq!(first.await.expect("first"), Ok(100)); + assert_eq!(second.await.expect("second"), Ok(200)); + assert_eq!(probes.load(Ordering::SeqCst), 2); + } + + /// A runner dropped mid-probe leaves no probe marked in flight. + #[tokio::test] + async fn a_cancelled_runner_does_not_block_later_requests() { + let coalescer = Arc::new(ReadIndexCoalescer::new()); + let stuck = { + let coalescer = Arc::clone(&coalescer); + tokio::spawn( + async move { coalescer.read_index(deadline(), std::future::pending).await }, + ) + }; + tokio::task::yield_now().await; + stuck.abort(); + let _ = stuck.await; + + let index = coalescer.read_index(deadline(), || async { Ok(7) }).await; + assert_eq!(index, Ok(7)); + } + + #[tokio::test] + async fn a_request_past_its_deadline_times_out() { + let coalescer = Arc::new(ReadIndexCoalescer::new()); + let holder = { + let coalescer = Arc::clone(&coalescer); + tokio::spawn( + async move { coalescer.read_index(deadline(), std::future::pending).await }, + ) + }; + tokio::task::yield_now().await; + let refused = coalescer + .read_index(Instant::now() + Duration::from_millis(20), || async { + Ok(1) + }) + .await; + assert!(matches!(refused, Err(ReadIndexRefusal::Timeout { .. }))); + holder.abort(); + } +} diff --git a/nodedb/src/control/security/auth_fence/state.rs b/nodedb/src/control/security/auth_fence/state.rs new file mode 100644 index 000000000..fd010136b --- /dev/null +++ b/nodedb/src/control/security/auth_fence/state.rs @@ -0,0 +1,154 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared state of the authorization fence. + +use std::collections::HashMap; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, OnceLock}; + +use tokio::sync::Notify; + +use crate::control::cluster::calvin::scheduler::AppliedMirrors; +use crate::control::security::auth_lease::{ + CalvinAckCoverage, LeaderLeaseService, LeaseHolder, LeaseTiming, +}; +use crate::control::security::permission_tree::SourceIndex; +use crate::event::progress::CoreEmitProgress; + +use super::read_index::ReadIndexCoalescer; +use super::tree_defs::PendingTreeDefs; + +/// State the authorization fence and the authorization lease share. +#[derive(Debug)] +pub struct AuthorizationFence { + /// One counter per Data Plane core, installed when the Event Plane starts. + emit_progress: OnceLock>>, + /// Woken after the permission step or a reload advances the cache. + permission_applied: Notify, + /// Read-index coalescers, by Raft group. + read_index: Mutex>>, + /// Tree-definition changes the metadata applier committed. + tree_defs: PendingTreeDefs, + /// The permission cache's source collections, readable without its lock. + sources: Arc, + /// Which Calvin positions this node's schedulers applied. + calvin_mirrors: AppliedMirrors, + /// Sequencer completion acks not yet settled against local schedulers. + calvin_acks: CalvinAckCoverage, + /// This node's authorization lease. + holder: LeaseHolder, + /// Lease timing, installed when the node joins a cluster. Absent on a + /// single node, which plans without a lease. + timing: OnceLock, + /// The leader-side lease service, installed with the Raft loop. + leader: OnceLock>, + /// A Raft snapshot was installed since the permission cache last + /// reloaded for one. + snapshot_installed: AtomicBool, +} + +impl AuthorizationFence { + /// State sharing `sources` with the permission cache. + pub fn new(sources: Arc) -> Self { + Self { + emit_progress: OnceLock::new(), + permission_applied: Notify::new(), + read_index: Mutex::new(HashMap::new()), + tree_defs: PendingTreeDefs::default(), + sources, + calvin_mirrors: AppliedMirrors::default(), + calvin_acks: CalvinAckCoverage::default(), + holder: LeaseHolder::default(), + timing: OnceLock::new(), + leader: OnceLock::new(), + snapshot_installed: AtomicBool::new(false), + } + } + + /// Record that a Raft snapshot replaced data-group state. The rows it + /// brought emitted no events. + pub fn note_snapshot_installed(&self) { + self.snapshot_installed.store(true, Ordering::Release); + } + + /// Whether a snapshot was installed since the last call. + pub fn take_snapshot_installed(&self) -> bool { + self.snapshot_installed.swap(false, Ordering::AcqRel) + } + + /// Install the emitted-event counters, one per core in core order. + /// Returns `false` when counters were already installed. + pub fn install_emit_progress(&self, progress: Vec>) -> bool { + self.emit_progress.set(progress).is_ok() + } + + /// The emitted-event counter of every core, read now. `None` before the + /// Event Plane starts: no permission step runs, so the cache cannot track + /// writes and a coverage wait reloads it. + pub fn emitted_snapshot(&self) -> Option> { + self.emit_progress + .get() + .map(|cores| cores.iter().map(|core| core.emitted()).collect()) + } + + /// The wake-up the permission step and a reload fire. + pub fn permission_applied(&self) -> &Notify { + &self.permission_applied + } + + /// The tree-definition changes waiting for the cache. + pub fn tree_defs(&self) -> &PendingTreeDefs { + &self.tree_defs + } + + /// The permission cache's source collections. + pub fn sources(&self) -> &SourceIndex { + &self.sources + } + + /// Which Calvin positions this node's schedulers applied. + pub fn calvin_mirrors(&self) -> &AppliedMirrors { + &self.calvin_mirrors + } + + /// Sequencer completion acks not yet settled against local schedulers. + pub fn calvin_acks(&self) -> &CalvinAckCoverage { + &self.calvin_acks + } + + /// This node's authorization lease. + pub fn holder(&self) -> &LeaseHolder { + &self.holder + } + + /// Install the lease timing. Returns `false` when already installed. + pub fn install_timing(&self, timing: LeaseTiming) -> bool { + self.timing.set(timing).is_ok() + } + + /// The lease timing, when this node runs in a cluster. + pub fn timing(&self) -> Option { + self.timing.get().copied() + } + + /// Install the leader-side lease service. Returns `false` when already + /// installed. + pub fn install_leader(&self, service: Arc) -> bool { + self.leader.set(service).is_ok() + } + + /// The leader-side lease service, once installed. + pub fn leader(&self) -> Option<&Arc> { + self.leader.get() + } + + /// The read-index coalescer of `group_id`. + pub fn read_index_coalescer(&self, group_id: u64) -> Arc { + let mut coalescers = self.read_index.lock().unwrap_or_else(|p| p.into_inner()); + Arc::clone( + coalescers + .entry(group_id) + .or_insert_with(|| Arc::new(ReadIndexCoalescer::new())), + ) + } +} diff --git a/nodedb/src/control/security/auth_fence/tree_defs.rs b/nodedb/src/control/security/auth_fence/tree_defs.rs new file mode 100644 index 000000000..728d35745 --- /dev/null +++ b/nodedb/src/control/security/auth_fence/tree_defs.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Permission-tree definition changes the metadata applier committed but the +//! permission cache has not taken yet. +//! +//! The applier runs synchronously and cannot take the cache's async lock, so +//! it queues each change here. The planning view and the lease coverage +//! apply the queue before they read the cache. Changes apply in commit +//! order. + +use std::sync::Mutex; + +use crate::control::security::catalog::StoredCollection; +use crate::control::security::permission_tree::{PermissionCache, PermissionTreeDef, SourceIndex}; +use crate::types::DatabaseId; + +/// One committed change to a collection's tree definition. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum TreeDefChange { + Register { + tenant_id: u64, + collection: String, + def: PermissionTreeDef, + }, + Unregister { + tenant_id: u64, + collection: String, + }, +} + +impl TreeDefChange { + /// The change a committed collection descriptor makes. Tree definitions + /// live on default-database collections only, as the DDL writes them. + /// An inactive collection governs nothing. + pub fn from_collection(stored: &StoredCollection) -> crate::Result> { + if stored.database_id != DatabaseId::DEFAULT { + return Ok(None); + } + let tenant_id = stored.tenant_id; + let collection = stored.name.clone(); + let def = match (&stored.permission_tree_def, stored.is_active) { + (Some(json), true) => json, + _ => { + return Ok(Some(Self::Unregister { + tenant_id, + collection, + })); + } + }; + let def: PermissionTreeDef = + sonic_rs::from_str(def).map_err(|e| crate::Error::Serialization { + format: "json".into(), + detail: format!("PERMISSION_TREE of collection '{collection}': {e}"), + })?; + Ok(Some(Self::Register { + tenant_id, + collection, + def, + })) + } + + /// Record this committed change in the source index, ahead of the cache. + pub fn note_committed(&self, sources: &SourceIndex) { + match self { + Self::Register { + tenant_id, + collection, + def, + } => sources.note_committed(*tenant_id, collection, Some(def)), + Self::Unregister { + tenant_id, + collection, + } => sources.note_committed(*tenant_id, collection, None), + } + } + + /// Apply this change to `cache`. + pub fn apply(self, cache: &mut PermissionCache) { + match self { + Self::Register { + tenant_id, + collection, + def, + } => cache.register_tree_def(tenant_id, &collection, def), + Self::Unregister { + tenant_id, + collection, + } => cache.unregister_tree_def(tenant_id, &collection), + } + } +} + +/// The queue of committed changes not yet in the cache. +#[derive(Debug, Default)] +pub struct PendingTreeDefs { + changes: Mutex>, +} + +impl PendingTreeDefs { + /// Queue a change the applier committed. + pub fn push(&self, change: TreeDefChange) { + self.changes + .lock() + .unwrap_or_else(|p| p.into_inner()) + .push(change); + } + + /// Whether a change waits. + pub fn is_empty(&self) -> bool { + self.changes + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_empty() + } + + /// Apply every queued change to `cache`. The caller holds the cache's + /// write lock, so two callers cannot apply out of order. + pub fn apply_to(&self, cache: &mut PermissionCache) { + let changes = std::mem::take(&mut *self.changes.lock().unwrap_or_else(|p| p.into_inner())); + for change in changes { + change.apply(cache); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn def() -> PermissionTreeDef { + sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def") + } + + #[test] + fn queued_changes_apply_in_commit_order() { + let pending = PendingTreeDefs::default(); + pending.push(TreeDefChange::Register { + tenant_id: 1, + collection: "docs".into(), + def: def(), + }); + pending.push(TreeDefChange::Unregister { + tenant_id: 1, + collection: "docs".into(), + }); + pending.push(TreeDefChange::Register { + tenant_id: 1, + collection: "notes".into(), + def: def(), + }); + assert!(!pending.is_empty()); + + let mut cache = PermissionCache::new(); + pending.apply_to(&mut cache); + + assert!(pending.is_empty()); + assert!(cache.get_tree_def(1, "docs").is_none()); + assert_eq!(cache.get_tree_def(1, "notes"), Some(&def())); + } +} diff --git a/nodedb/src/control/security/auth_fence/view.rs b/nodedb/src/control/security/auth_fence/view.rs new file mode 100644 index 000000000..4b5c22ecf --- /dev/null +++ b/nodedb/src/control/security/auth_fence/view.rs @@ -0,0 +1,85 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The read of authorization state that every planning path takes. +//! +//! Planning uses local state only, and every check here is local: +//! +//! 1. Tree-definition changes the metadata applier queued move into the +//! permission cache. +//! 2. A stale cache reloads from this node's own cores: no reload covers it +//! yet (startup, or a tree definition changed), or a core lost an event. +//! A writer's acknowledgement never reloads, so a write covered only by a +//! reload binds this statement through this step. +//! 3. A tenant whose tree rows live in a Raft group this node does not +//! replicate is refused: this node never covers that group. +//! 4. In a cluster, the node must hold a valid authorization lease. A writer +//! acknowledges an authorization change only after every lease holder +//! covered it or its lease expired, so a valid lease means the state read +//! here holds every change acknowledged before this point. +//! +//! The lease is checked after the cache guard is taken. The guard fixes the +//! cache for the whole plan, and a change acknowledged after the check was +//! acknowledged after the statement started planning. + +use std::time::Instant; + +use tokio::sync::RwLockReadGuard; + +use crate::control::security::permission_tree::{PermissionCache, reload}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, VShardId}; + +use super::cluster::{behind, group_of_vshard, hosts_group}; + +/// Read the permission cache for planning a statement of `tenant_id`. +pub async fn permission_view( + state: &SharedState, + tenant_id: TenantId, +) -> crate::Result> { + apply_committed_tree_defs(state).await; + reload::reload_if_stale(state).await?; + + let cache = state.permission_cache.read().await; + if state.cluster_routing.is_some() && cache.has_tree_defs_for_tenant(tenant_id.as_u64()) { + for source in cache + .tree_sources() + .into_iter() + .filter(|source| source.tenant_id == tenant_id.as_u64()) + { + let vshard = + VShardId::from_collection_in_database(DatabaseId::DEFAULT, &source.collection); + let group_id = group_of_vshard(state, vshard.as_u32())?; + if !hosts_group(state, group_id) { + return Err(behind(format!( + "this node does not replicate raft group {group_id}, which homes \ + permission source '{}'; run the statement on a node that does", + source.collection + ))); + } + } + } + if state.authorization_fence.timing().is_some() + && !state + .authorization_fence + .holder() + .is_valid_at(Instant::now()) + { + return Err(behind( + "this node holds no valid authorization lease; it has not confirmed the latest \ + authorization changes", + )); + } + Ok(cache) +} + +/// Move the tree-definition changes the metadata applier committed into the +/// cache. The applier queues each change before it advances the applied +/// index, so the queue holds every change this node applied. +pub(crate) async fn apply_committed_tree_defs(state: &SharedState) { + let pending = state.authorization_fence.tree_defs(); + if pending.is_empty() { + return; + } + let mut cache = state.permission_cache.write().await; + pending.apply_to(&mut cache); +} diff --git a/nodedb/src/control/security/auth_lease/barrier.rs b/nodedb/src/control/security/auth_lease/barrier.rs new file mode 100644 index 000000000..50712374b --- /dev/null +++ b/nodedb/src/control/security/auth_lease/barrier.rs @@ -0,0 +1,166 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The writer's side: hold an authorization change's acknowledgement until +//! no node can plan against the state before it. +//! +//! - **In a cluster** the metadata leader holds the barrier until every node +//! with an unexpired lease covered the targets, or its lease expired. The +//! writing node is a lease holder too, so its own Event Plane lag closes +//! the same way. +//! - **On a single node** there is no lease. The barrier waits until the +//! local permission cache reflects every event the cores emitted, which +//! covers the change just applied. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::{AuthBarrierOutcome, AuthBarrierRequest, GroupCoverage, RaftRpc}; +use tokio::runtime::RuntimeFlavor; + +use crate::control::security::auth_fence::cluster::behind; +use crate::control::state::SharedState; + +use super::coverage::permission_step_covers_now; +use super::leadership::{metadata_leader, send_to_leader}; + +/// Hold the acknowledgement of a change until it binds every node. +/// +/// `targets` name the change: per group, the index it committed at. The +/// change itself is committed; an error means only that the barrier did not +/// release within the request deadline. +/// +/// Latency: in a cluster, every authorization-bearing write waits about one +/// renewal interval (the Raft heartbeat) before it is acknowledged. Each +/// holder reports its coverage only with its next renewal. This covers every +/// DDL that bears authorization, `CREATE COLLECTION` included. A holder that +/// cannot renew adds up to one lease duration (the election timeout), until +/// its lease expires. On a single node the wait is the permission step's +/// lag only. +pub async fn authorization_barrier( + state: &SharedState, + targets: Vec, +) -> crate::Result<()> { + let deadline_secs = state.tuning.network.default_deadline_secs; + let deadline = Instant::now() + Duration::from_secs(deadline_secs); + let Some(timing) = state.authorization_fence.timing() else { + return await_local_coverage(state, deadline).await; + }; + loop { + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { + return Err(committed_but_pending(format!( + "the barrier did not release within {deadline_secs}s" + ))); + } + let request = AuthBarrierRequest { + targets: targets.clone(), + timeout_ms: u64::try_from(remaining.as_millis()).unwrap_or(u64::MAX), + }; + let outcome = match metadata_leader(state).filter(|(leader, _)| *leader != 0) { + None => None, + Some((leader_id, _)) if leader_id == state.node_id => { + match state.authorization_fence.leader() { + Some(service) => Some(service.hold_barrier(request).await.outcome), + None => None, + } + } + Some((leader_id, _)) => match send_to_leader( + state, + leader_id, + RaftRpc::AuthBarrierRequest(request), + remaining + timing.lease, + ) + .await + { + Ok(RaftRpc::AuthBarrierResponse(response)) => Some(response.outcome), + Ok(other) => { + tracing::warn!( + leader_id, + "authorization barrier: unexpected reply {other:?}" + ); + None + } + Err(error) => { + tracing::debug!(%error, "authorization barrier: not delivered"); + None + } + }, + }; + match outcome { + Some(AuthBarrierOutcome::Released) => return Ok(()), + // No leader known, a leader change, or a lost message: ask the + // leader again. A new leader holds the barrier to its own floors. + Some(AuthBarrierOutcome::NotLeader { .. }) + | Some(AuthBarrierOutcome::Timeout { .. }) + | None => tokio::time::sleep(timing.renew_every).await, + } + } +} + +/// Hold the acknowledgement of a Calvin transaction that wrote a +/// permission-tree source. +/// +/// Its completion acks sit in the sequencer log, at or below the sequencer +/// group's commit index now. A node covers that index only once its own +/// replicas applied every acknowledged transaction below it. +pub async fn calvin_write_barrier(state: &SharedState) -> crate::Result<()> { + if state.authorization_fence.timing().is_none() { + let deadline = + Instant::now() + Duration::from_secs(state.tuning.network.default_deadline_secs); + return await_local_coverage(state, deadline).await; + } + let commit_index = state + .raft_status_fn + .get() + .and_then(|status| { + status() + .into_iter() + .find(|group| group.group_id == SEQUENCER_GROUP_ID) + .map(|group| group.commit_index) + }) + .ok_or_else(|| committed_but_pending("this node does not replicate the sequencer group"))?; + authorization_barrier( + state, + vec![GroupCoverage { + group_id: SEQUENCER_GROUP_ID, + through: commit_index, + }], + ) + .await +} + +/// Run [`authorization_barrier`] from synchronous code on a Tokio worker. +pub fn block_on_barrier(state: &SharedState, targets: Vec) -> crate::Result<()> { + let handle = tokio::runtime::Handle::try_current().map_err(|_| crate::Error::Internal { + detail: "authorization barrier: called outside a Tokio runtime".into(), + })?; + if handle.runtime_flavor() != RuntimeFlavor::MultiThread { + return Err(crate::Error::Internal { + detail: "authorization barrier: synchronous callers need a multi-thread runtime".into(), + }); + } + tokio::task::block_in_place(|| handle.block_on(authorization_barrier(state, targets))) +} + +/// Wait until the permission step covers every event the cores emitted +/// before this call, which includes the write just applied. +/// +/// This is a writer's acknowledgement wait: it only waits, and never reloads +/// or dispatches. A cache that needs a reload is reloaded by the next +/// statement's planning, before it reads the cache. +pub async fn await_local_coverage(state: &SharedState, deadline: Instant) -> crate::Result<()> { + let remaining = deadline.saturating_duration_since(Instant::now()); + if permission_step_covers_now(state, remaining).await { + return Ok(()); + } + Err(committed_but_pending( + "the permission cache did not catch up with the change", + )) +} + +fn committed_but_pending(detail: impl std::fmt::Display) -> crate::Error { + behind(format!( + "the authorization change is committed, but {detail}; nodes still planning against \ + the previous state refuse until they catch up" + )) +} diff --git a/nodedb/src/control/security/auth_lease/calvin_acks.rs b/nodedb/src/control/security/auth_lease/calvin_acks.rs new file mode 100644 index 000000000..206465ab3 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/calvin_acks.rs @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Coverage of the sequencer group. +//! +//! A Calvin transaction is acknowledged once every participant vShard's +//! leader proposed a `CompletionAck` to the sequencer log. Every other +//! replica of the vShard applies the transaction in its own time. This node +//! covers sequencer index `i` when it applied the sequencer log through `i` +//! and, for every ack at or below `i` naming a vShard this node replicates, +//! its own scheduler applied that transaction too. +//! +//! The sequencer state machine records each ack it applies. This tracker +//! takes them in log order and settles each against the local scheduler's +//! applied mirror. The first unsettled ack bounds the coverage. + +use std::collections::VecDeque; +use std::sync::Mutex; + +use nodedb_cluster::calvin::{AppliedCompletionAck, CalvinCompletionRegistry}; + +use crate::control::cluster::calvin::scheduler::AppliedMirrors; + +/// Acks taken from the sequencer log and not yet settled. +#[derive(Debug, Default)] +pub struct CalvinAckCoverage { + pending: Mutex>, +} + +impl CalvinAckCoverage { + /// The sequencer index this node covers, given that it applied the + /// sequencer log through `applied`, read before this call. + /// + /// `replicates` answers whether this node replicates a vShard. An ack for + /// a vShard it does not replicate settles at once: this node never plans + /// against that vShard's rows. + pub fn covered_through( + &self, + registry: &CalvinCompletionRegistry, + mirrors: &AppliedMirrors, + applied: u64, + replicates: impl Fn(u32) -> bool, + ) -> u64 { + let mut pending = self.pending.lock().unwrap_or_else(|p| p.into_inner()); + pending.extend(registry.applied_acks.drain()); + while let Some(ack) = pending.front() { + let settled = !replicates(ack.vshard_id) + || mirrors + .get(ack.vshard_id) + .is_some_and(|mirror| mirror.is_applied(ack.txn.epoch, ack.txn.position)); + if !settled { + break; + } + pending.pop_front(); + } + match pending.front() { + Some(ack) => applied.min(ack.index.saturating_sub(1)), + None => applied, + } + } +} + +#[cfg(test)] +mod tests { + use std::collections::BTreeSet; + + use nodedb_cluster::calvin::TxnId; + + use super::*; + use crate::control::cluster::calvin::scheduler::NOT_YET_APPLIED_EPOCH; + + fn ack(index: u64, epoch: u64, vshard_id: u32) -> AppliedCompletionAck { + AppliedCompletionAck { + index, + txn: TxnId::new(epoch, 0), + vshard_id, + } + } + + #[test] + fn an_ack_the_local_replica_has_not_applied_bounds_the_coverage() { + let registry = CalvinCompletionRegistry::new_detached(); + registry.applied_acks.enable(); + let mirrors = AppliedMirrors::default(); + let mirror = mirrors.register(7, NOT_YET_APPLIED_EPOCH, &BTreeSet::new()); + let coverage = CalvinAckCoverage::default(); + + registry.applied_acks.record(ack(4, 1, 7)); + registry.applied_acks.record(ack(6, 2, 9)); + // vShard 7 is replicated here and its scheduler has not applied epoch 1. + let replicates = |vshard: u32| vshard == 7; + assert_eq!( + coverage.covered_through(®istry, &mirrors, 8, replicates), + 3 + ); + + mirror.mark(1, 0); + // The ack for vShard 9 settles at once: it is not replicated here. + assert_eq!( + coverage.covered_through(®istry, &mirrors, 8, replicates), + 8 + ); + } + + #[test] + fn a_replicated_vshard_without_a_scheduler_yet_is_unsettled() { + let registry = CalvinCompletionRegistry::new_detached(); + registry.applied_acks.enable(); + let mirrors = AppliedMirrors::default(); + let coverage = CalvinAckCoverage::default(); + registry.applied_acks.record(ack(2, 1, 5)); + assert_eq!( + coverage.covered_through(®istry, &mirrors, 3, |_| true), + 1 + ); + } +} diff --git a/nodedb/src/control/security/auth_lease/coverage.rs b/nodedb/src/control/security/auth_lease/coverage.rs new file mode 100644 index 000000000..0bf20c5c4 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/coverage.rs @@ -0,0 +1,196 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What this node's authorization state covers, per Raft group. +//! +//! - **Metadata group:** roles, grants, RLS policies, scope grants and tree +//! definitions. The metadata applier updates each store before it advances +//! the applied index, and queues tree-definition changes, which this step +//! moves into the permission cache. The applied index is covered as read. +//! - **Data groups:** the permission cache holds tree rows through the +//! Event Plane's permission step. A group's applied index read before the +//! emitted-event counters is covered once the step reaches those counters. +//! - **Sequencer group:** Calvin writes. Covered as [`super::calvin_acks`] +//! settles, then through the permission step like a data group. +//! - **A group this node does not replicate** is reported at `u64::MAX`: +//! planning here refuses any tenant whose tree rows live in it. +//! +//! Indexes are read first, then the counters, then the cache is checked. A +//! group's writes at or below the index read emitted their events before the +//! counters were read, so a cache that reached the counters holds them. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::{GroupCoverage, METADATA_GROUP_ID}; + +use crate::control::security::auth_fence::cluster::{group_of_vshard, hosts_group, routed_groups}; +use crate::control::security::auth_fence::view::apply_committed_tree_defs; +use crate::control::security::permission_tree::reload; +use crate::control::state::SharedState; + +/// Raw indexes, read before the emitted-event counters. +struct RawCoverage { + metadata: u64, + /// Data groups and the sequencer group, which the permission step covers. + event_backed: Vec, +} + +fn read_raw(state: &SharedState) -> RawCoverage { + let metadata = state.applied_index_watcher(METADATA_GROUP_ID).current(); + let mut event_backed: Vec = routed_groups(state) + .into_iter() + .filter(|group_id| *group_id != METADATA_GROUP_ID && *group_id != SEQUENCER_GROUP_ID) + .map(|group_id| GroupCoverage { + group_id, + through: if hosts_group(state, group_id) { + state.applied_index_watcher(group_id).current() + } else { + u64::MAX + }, + }) + .collect(); + event_backed.push(GroupCoverage { + group_id: SEQUENCER_GROUP_ID, + through: sequencer_coverage(state), + }); + RawCoverage { + metadata, + event_backed, + } +} + +/// The sequencer index this node's Calvin replicas cover. +fn sequencer_coverage(state: &SharedState) -> u64 { + let hosts_sequencer = state.raft_status_fn.get().is_some_and(|status| { + status() + .iter() + .any(|group| group.group_id == SEQUENCER_GROUP_ID) + }); + let Some(registry) = state.calvin_completion_registry.get() else { + return u64::MAX; + }; + if !hosts_sequencer { + // No sequencer replica runs here, so no local scheduler applies a + // Calvin write and none can be planned against. + return u64::MAX; + } + let applied = state.applied_index_watcher(SEQUENCER_GROUP_ID).current(); + let fence = &state.authorization_fence; + fence + .calvin_acks() + .covered_through(registry, fence.calvin_mirrors(), applied, |vshard_id| { + group_of_vshard(state, vshard_id).is_ok_and(|g| hosts_group(state, g)) + }) +} + +/// This node's coverage, confirmed through the permission step. +/// +/// The metadata group is always current. The event-backed groups are taken +/// from this snapshot when the permission step reaches it within `wait`; +/// otherwise `previous` stands for them, which an earlier call confirmed. +pub async fn confirmed_coverage( + state: &SharedState, + previous: &[GroupCoverage], + wait: Duration, +) -> crate::Result> { + let raw = read_raw(state); + apply_committed_tree_defs(state).await; + if state.authorization_fence.take_snapshot_installed() { + // The snapshot rows emitted no events. A reload after the indexes + // were read holds them. + reload::reload_all(state, None).await?; + } + reload::reload_if_stale(state).await?; + + let caught_up = permission_step_reaches_now(state, wait).await?; + let mut coverage = vec![GroupCoverage { + group_id: METADATA_GROUP_ID, + through: raw.metadata, + }]; + if caught_up { + coverage.extend(raw.event_backed); + } else { + coverage.extend( + previous + .iter() + .filter(|report| report.group_id != METADATA_GROUP_ID) + .copied(), + ); + } + Ok(coverage) +} + +/// Whether the permission cache reflects every event the cores emitted +/// before this call, within `wait`. A core that lost an event is reloaded. +pub(crate) async fn permission_step_reaches_now( + state: &SharedState, + wait: Duration, +) -> crate::Result { + let fence = &state.authorization_fence; + let Some(targets) = fence.emitted_snapshot() else { + // No permission step runs, so only a reload reflects the writes. + reload::reload_all(state, None).await?; + return Ok(true); + }; + let until = Instant::now() + wait; + loop { + let notified = fence.permission_applied().notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + let needs_reload = { + let cache = state.permission_cache.read().await; + if cache.progress().caught_up(&targets) { + return Ok(true); + } + cache.progress().needs_reload_for(&targets) + }; + if needs_reload { + reload::reload_all(state, Some(&targets)).await?; + continue; + } + let now = Instant::now(); + if now >= until { + return Ok(false); + } + let _ = tokio::time::timeout(until - now, notified).await; + } +} + +/// Whether the permission step covers every event the cores emitted before +/// this call, within `wait`. A writer holding its acknowledgement calls this. +/// +/// It never reloads. A cache that only a reload can bring to the targets is +/// stale, and [`reload::reload_if_stale`] reloads it before the next +/// statement plans. That reload reads each core after the write applied, so +/// the write already binds every later plan. Before the Event Plane starts no +/// permission step counts writes, so the cache is marked for a reload. +pub(crate) async fn permission_step_covers_now(state: &SharedState, wait: Duration) -> bool { + let fence = &state.authorization_fence; + let Some(targets) = fence.emitted_snapshot() else { + state + .permission_cache + .write() + .await + .progress_mut() + .mark_reload_needed(); + return true; + }; + let until = Instant::now() + wait; + loop { + let notified = fence.permission_applied().notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + { + let cache = state.permission_cache.read().await; + let progress = cache.progress(); + if progress.caught_up(&targets) || progress.needs_reload_for(&targets) { + return true; + } + } + let now = Instant::now(); + if now >= until { + return false; + } + let _ = tokio::time::timeout(until - now, notified).await; + } +} diff --git a/nodedb/src/control/security/auth_lease/holder.rs b/nodedb/src/control/security/auth_lease/holder.rs new file mode 100644 index 000000000..b1f6c7f26 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/holder.rs @@ -0,0 +1,92 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The holder side of this node's authorization lease. +//! +//! A statement is planned against local authorization state only while the +//! lease is valid. The lease ends on this node's clock before it ends on the +//! leader's (see [`super::timing`]), so once the leader treats it as expired +//! no statement here can still plan under it. + +use std::sync::Mutex; +use std::time::Instant; + +/// The end of this node's lease, if it holds one. +#[derive(Debug, Default)] +pub struct LeaseHolder { + valid_until: Mutex>, +} + +impl LeaseHolder { + /// Extend the lease to `until`. A grant never shortens a lease already + /// held: the leader granted each one against the state it covers. + pub fn install(&self, until: Instant) { + let mut valid_until = self.valid_until.lock().unwrap_or_else(|p| p.into_inner()); + if valid_until.is_none_or(|current| current < until) { + *valid_until = Some(until); + } + } + + /// Whether the lease is valid at `now`. + pub fn is_valid_at(&self, now: Instant) -> bool { + self.valid_until + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_some_and(|until| now < until) + } + + /// When the lease ends, if one was granted. + pub fn valid_until(&self) -> Option { + *self.valid_until.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Wait until a lease is valid, polling every `poll`, or refuse once + /// `timeout` passes. + pub async fn await_valid( + &self, + timeout: std::time::Duration, + poll: std::time::Duration, + ) -> crate::Result<()> { + let deadline = Instant::now() + timeout; + while !self.is_valid_at(Instant::now()) { + if Instant::now() >= deadline { + return Err(crate::Error::AuthorizationStateBehind { + detail: format!("no authorization lease was granted within {timeout:?}"), + }); + } + tokio::time::sleep(poll).await; + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + use crate::control::security::auth_lease::LeaseTiming; + + #[test] + fn a_lease_is_valid_until_its_margin_adjusted_end() { + let timing = LeaseTiming::from_raft(Duration::from_millis(150), Duration::from_millis(50)) + .expect("timing"); + let holder = LeaseHolder::default(); + let sent_at = Instant::now(); + assert!(!holder.is_valid_at(sent_at)); + + holder.install(timing.holder_expiry(sent_at, timing.lease)); + assert!(holder.is_valid_at(sent_at + Duration::from_millis(99))); + // The leader's lease still runs at 100ms, but the holder's has ended. + assert!(!holder.is_valid_at(sent_at + Duration::from_millis(100))); + assert!(!holder.is_valid_at(sent_at + timing.lease)); + } + + #[test] + fn an_older_grant_never_shortens_the_lease() { + let holder = LeaseHolder::default(); + let now = Instant::now(); + holder.install(now + Duration::from_millis(200)); + holder.install(now + Duration::from_millis(100)); + assert!(holder.is_valid_at(now + Duration::from_millis(150))); + } +} diff --git a/nodedb/src/control/security/auth_lease/leadership.rs b/nodedb/src/control/security/auth_lease/leadership.rs new file mode 100644 index 000000000..c9189e281 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/leadership.rs @@ -0,0 +1,58 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Who leads the metadata group, from this node's Raft status. + +use std::collections::BTreeSet; +use std::time::Duration; + +use nodedb_cluster::{METADATA_GROUP_ID, RaftRpc}; + +use crate::control::server::exchange::resolve::register_peers_from_topology; +use crate::control::state::SharedState; + +/// The metadata group's leader and term, as this node sees them. A leader +/// id of `0` means none is known. +pub(crate) fn metadata_leader(state: &SharedState) -> Option<(u64, u64)> { + let status = state.raft_status_fn.get()?; + status() + .into_iter() + .find(|group| group.group_id == METADATA_GROUP_ID) + .map(|group| (group.leader_id, group.term)) +} + +/// The term this node leads the metadata group in, if it does. +pub(crate) fn leading_term(state: &SharedState) -> Option { + metadata_leader(state) + .filter(|(leader_id, _)| *leader_id == state.node_id) + .map(|(_, term)| term) +} + +/// The leader hint to send back with a refusal. +pub(crate) fn leader_hint(state: &SharedState) -> Option { + metadata_leader(state) + .map(|(leader_id, _)| leader_id) + .filter(|leader_id| *leader_id != 0) +} + +/// Send `rpc` to the metadata leader `leader_id` and return its answer. +pub(crate) async fn send_to_leader( + state: &SharedState, + leader_id: u64, + rpc: RaftRpc, + timeout: Duration, +) -> crate::Result { + let Some(transport) = state.cluster_transport.as_ref() else { + return Err(crate::Error::Internal { + detail: "authorization lease: no cluster transport on this node".into(), + }); + }; + let mut targets = BTreeSet::new(); + targets.insert(leader_id); + register_peers_from_topology(state, transport, &targets); + transport + .send_rpc_with_read_timeout(leader_id, rpc, timeout) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("authorization lease: rpc to metadata leader {leader_id}: {e}"), + }) +} diff --git a/nodedb/src/control/security/auth_lease/mod.rs b/nodedb/src/control/security/auth_lease/mod.rs new file mode 100644 index 000000000..f3667f9b4 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/mod.rs @@ -0,0 +1,21 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod barrier; +pub mod calvin_acks; +pub mod coverage; +pub mod holder; +pub mod leadership; +pub mod renew_loop; +pub mod service; +pub mod status; +pub mod table; +pub mod timing; + +pub use barrier::{ + authorization_barrier, await_local_coverage, block_on_barrier, calvin_write_barrier, +}; +pub use calvin_acks::CalvinAckCoverage; +pub use holder::LeaseHolder; +pub use service::LeaderLeaseService; +pub use status::{LeaseStatus, lease_status}; +pub use timing::LeaseTiming; diff --git a/nodedb/src/control/security/auth_lease/renew_loop.rs b/nodedb/src/control/security/auth_lease/renew_loop.rs new file mode 100644 index 000000000..6c55cb39f --- /dev/null +++ b/nodedb/src/control/security/auth_lease/renew_loop.rs @@ -0,0 +1,113 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The holder's renewal loop. +//! +//! Every renewal interval the node computes its confirmed coverage (see +//! [`super::coverage`]) and sends it to the metadata leader. A grant extends +//! the lease from the moment the request left. A withheld or failed renewal +//! extends nothing, so the lease lapses unless a later renewal succeeds. + +use std::sync::Arc; +use std::time::Instant; + +use nodedb_cluster::{ + AuthLeaseRenewOutcome, AuthLeaseRenewRequest, AuthLeaseRenewResponse, GroupCoverage, RaftRpc, +}; + +use crate::control::shutdown::ShutdownReceiver; +use crate::control::state::SharedState; + +use super::coverage::confirmed_coverage; +use super::leadership::{metadata_leader, send_to_leader}; +use super::timing::LeaseTiming; + +/// Renew this node's lease until shutdown. +pub async fn run_renew_loop( + state: Arc, + timing: LeaseTiming, + mut shutdown: ShutdownReceiver, +) { + let mut confirmed: Vec = Vec::new(); + loop { + match confirmed_coverage(&state, &confirmed, timing.renew_every).await { + Ok(coverage) => { + confirmed = coverage; + renew_once(&state, timing, &confirmed).await; + } + Err(error) => { + tracing::warn!(%error, "authorization lease: coverage could not be computed"); + } + } + tokio::select! { + _ = tokio::time::sleep(timing.renew_every) => {} + _ = shutdown.wait_cancelled() => return, + } + } +} + +/// Send one renewal and install a granted lease. +async fn renew_once(state: &SharedState, timing: LeaseTiming, coverage: &[GroupCoverage]) { + let Some((leader_id, _)) = metadata_leader(state).filter(|(leader, _)| *leader != 0) else { + return; + }; + let request = AuthLeaseRenewRequest { + node_id: state.node_id, + coverage: coverage.to_vec(), + }; + let sent_at = Instant::now(); + let response = if leader_id == state.node_id { + match state.authorization_fence.leader() { + Some(service) => service.renew_lease(request).await, + None => return, + } + } else { + match send_to_leader( + state, + leader_id, + RaftRpc::AuthLeaseRenewRequest(request), + timing.lease, + ) + .await + { + Ok(RaftRpc::AuthLeaseRenewResponse(response)) => response, + Ok(other) => { + tracing::warn!( + leader_id, + "authorization lease: unexpected renewal reply {other:?}" + ); + return; + } + Err(error) => { + tracing::debug!(%error, "authorization lease: renewal not delivered"); + return; + } + } + }; + install(state, timing, sent_at, response); +} + +fn install( + state: &SharedState, + timing: LeaseTiming, + sent_at: Instant, + response: AuthLeaseRenewResponse, +) { + match response.outcome { + AuthLeaseRenewOutcome::Granted { lease_ms } => { + let granted = std::time::Duration::from_millis(lease_ms); + state + .authorization_fence + .holder() + .install(timing.holder_expiry(sent_at, granted)); + } + AuthLeaseRenewOutcome::Withheld => { + tracing::debug!("authorization lease: renewal withheld until coverage catches up"); + } + AuthLeaseRenewOutcome::NotLeader { leader_hint } => { + tracing::debug!( + ?leader_hint, + "authorization lease: renewal reached a non-leader" + ); + } + } +} diff --git a/nodedb/src/control/security/auth_lease/service.rs b/nodedb/src/control/security/auth_lease/service.rs new file mode 100644 index 000000000..3a39085e9 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/service.rs @@ -0,0 +1,275 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The metadata leader's side of the authorization lease. +//! +//! Only the current metadata leader answers renewals and barriers. Each +//! leadership term starts a fresh [`LeaseTable`], whose floors are loaded +//! before anything is granted or released: +//! +//! - **Metadata group:** the leader's own confirmed read index. +//! - **Sequencer group and every group homing a tree source:** a read index +//! from each group's leader. +//! +//! A read index is at or above every entry committed before it was taken, so +//! the floors cover every change acknowledged in an earlier term. +//! +//! A grant and a release are answered only after the leader confirms its +//! leadership against a quorum, taken after the decision. A leader deposed +//! meanwhile answers `NotLeader`, so no lease it grants and no barrier it +//! releases outlives its term unseen. + +use std::collections::HashSet; +use std::sync::{Mutex, Weak}; +use std::time::{Duration, Instant}; + +use futures::future::try_join_all; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::{ + AuthBarrierOutcome, AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewOutcome, + AuthLeaseRenewRequest, AuthLeaseRenewResponse, GroupCoverage, METADATA_GROUP_ID, +}; +use tokio::sync::Notify; + +use crate::control::security::auth_fence::cluster::{ + confirmed_read_index, group_of_vshard, wait_applied, +}; +use crate::control::security::auth_fence::view::apply_committed_tree_defs; +use crate::control::state::SharedState; + +use super::leadership::{leader_hint, leading_term}; +use super::table::{BarrierState, LeaseTable, RenewDecision}; +use super::timing::LeaseTiming; + +/// Answers lease renewals and barriers while this node leads the metadata +/// group. +pub struct LeaderLeaseService { + /// Held weakly: the service lives on `SharedState`. + state: Weak, + timing: LeaseTiming, + table: Mutex>, + /// Woken on every renewal and floor load, for waiting barriers. + changed: Notify, + /// One floor load at a time. + floors_loading: tokio::sync::Mutex<()>, +} + +impl std::fmt::Debug for LeaderLeaseService { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("LeaderLeaseService") + .field("timing", &self.timing) + .finish_non_exhaustive() + } +} + +impl LeaderLeaseService { + pub fn new(state: Weak, timing: LeaseTiming) -> Self { + Self { + state, + timing, + table: Mutex::new(None), + changed: Notify::new(), + floors_loading: tokio::sync::Mutex::new(()), + } + } + + fn table(&self) -> std::sync::MutexGuard<'_, Option> { + self.table.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Make the table of `term` current and load its floors. + async fn table_ready(&self, state: &SharedState, term: u64) -> crate::Result<()> { + { + let mut table = self.table(); + if table.as_ref().is_none_or(|t| t.term() != term) { + *table = Some(LeaseTable::new(term, Instant::now())); + } + if table.as_ref().is_some_and(LeaseTable::floors_ready) { + return Ok(()); + } + } + let _loading = self.floors_loading.lock().await; + if self + .table() + .as_ref() + .is_some_and(|t| t.term() == term && t.floors_ready()) + { + return Ok(()); + } + let floors = self.load_floors(state).await?; + if let Some(table) = self.table().as_mut().filter(|t| t.term() == term) { + table.load_floors(&floors); + } + self.changed.notify_waiters(); + Ok(()) + } + + /// The floors a new term starts from. + async fn load_floors(&self, state: &SharedState) -> crate::Result> { + let timeout = self.timing.lease; + let metadata = confirmed_read_index(state, METADATA_GROUP_ID, timeout).await?; + // The source set comes from tree definitions, which live in the + // metadata group. Apply it through the read index first. + wait_applied(state, METADATA_GROUP_ID, metadata, timeout).await?; + apply_committed_tree_defs(state).await; + + let mut groups: HashSet = HashSet::new(); + for vshard_id in state.authorization_fence.sources().source_vshards() { + groups.insert(group_of_vshard(state, vshard_id)?); + } + groups.insert(SEQUENCER_GROUP_ID); + let group_floors = try_join_all(groups.into_iter().map(|group_id| async move { + confirmed_read_index(state, group_id, timeout) + .await + .map(|through| GroupCoverage { group_id, through }) + })) + .await?; + + let mut floors = vec![GroupCoverage { + group_id: METADATA_GROUP_ID, + through: metadata, + }]; + floors.extend(group_floors); + Ok(floors) + } + + /// Whether this node still leads the metadata group in `term`, confirmed + /// against a quorum now. + async fn confirm_leadership(&self, state: &SharedState, term: u64) -> bool { + confirmed_read_index(state, METADATA_GROUP_ID, self.timing.lease) + .await + .is_ok() + && leading_term(state) == Some(term) + } + + fn not_leader_renewal(state: Option<&SharedState>) -> AuthLeaseRenewResponse { + AuthLeaseRenewResponse { + outcome: AuthLeaseRenewOutcome::NotLeader { + leader_hint: state.and_then(leader_hint), + }, + } + } + + fn not_leader_barrier(state: Option<&SharedState>) -> AuthBarrierResponse { + AuthBarrierResponse { + outcome: AuthBarrierOutcome::NotLeader { + leader_hint: state.and_then(leader_hint), + }, + } + } + + /// Grant or withhold the lease of the node that sent `req`. + pub async fn renew_lease(&self, req: AuthLeaseRenewRequest) -> AuthLeaseRenewResponse { + let Some(state) = self.state.upgrade() else { + return Self::not_leader_renewal(None); + }; + let Some(term) = leading_term(&state) else { + return Self::not_leader_renewal(Some(&state)); + }; + if let Err(error) = self.table_ready(&state, term).await { + tracing::debug!(%error, "authorization lease: floors not loaded; renewal withheld"); + return AuthLeaseRenewResponse { + outcome: AuthLeaseRenewOutcome::Withheld, + }; + } + let decision = { + let mut table = self.table(); + match table.as_mut().filter(|t| t.term() == term) { + Some(table) => table.renew( + req.node_id, + &req.coverage, + Instant::now(), + self.timing.lease, + ), + None => return Self::not_leader_renewal(Some(&state)), + } + }; + self.changed.notify_waiters(); + let outcome = match decision { + RenewDecision::Withheld => AuthLeaseRenewOutcome::Withheld, + RenewDecision::Granted if self.confirm_leadership(&state, term).await => { + AuthLeaseRenewOutcome::Granted { + lease_ms: u64::try_from(self.timing.lease.as_millis()).unwrap_or(u64::MAX), + } + } + RenewDecision::Granted => { + return Self::not_leader_renewal(Some(&state)); + } + }; + AuthLeaseRenewResponse { outcome } + } + + /// Answer once no lease holder can plan against state older than the + /// request's targets. + pub async fn hold_barrier(&self, req: AuthBarrierRequest) -> AuthBarrierResponse { + let started = Instant::now(); + let deadline = started + Duration::from_millis(req.timeout_ms); + let timed_out = || AuthBarrierResponse { + outcome: AuthBarrierOutcome::Timeout { + waited_ms: u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX), + }, + }; + loop { + let Some(state) = self.state.upgrade() else { + return Self::not_leader_barrier(None); + }; + let Some(term) = leading_term(&state) else { + return Self::not_leader_barrier(Some(&state)); + }; + if let Err(error) = self.table_ready(&state, term).await { + tracing::debug!(%error, "authorization barrier: floors not loaded yet"); + if Instant::now() >= deadline { + return timed_out(); + } + tokio::time::sleep(self.timing.renew_every).await; + continue; + } + let notified = self.changed.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + let status = { + let mut table = self.table(); + match table.as_mut().filter(|t| t.term() == term) { + Some(table) => { + table.raise_floors(&req.targets); + table.barrier(&req.targets, Instant::now(), self.timing.lease) + } + None => continue, + } + }; + match status { + BarrierState::Released => { + return if self.confirm_leadership(&state, term).await { + AuthBarrierResponse { + outcome: AuthBarrierOutcome::Released, + } + } else { + Self::not_leader_barrier(Some(&state)) + }; + } + BarrierState::NotReady => tokio::time::sleep(self.timing.renew_every).await, + BarrierState::Waiting { until } => { + let wake = tokio::time::Instant::from_std(until.min(deadline)); + drop(state); + tokio::select! { + _ = notified => {} + _ = tokio::time::sleep_until(wake) => {} + } + } + } + if Instant::now() >= deadline { + return timed_out(); + } + } + } +} + +#[async_trait::async_trait] +impl nodedb_cluster::AuthLeaseService for LeaderLeaseService { + async fn renew(&self, req: AuthLeaseRenewRequest) -> AuthLeaseRenewResponse { + self.renew_lease(req).await + } + + async fn barrier(&self, req: AuthBarrierRequest) -> AuthBarrierResponse { + self.hold_barrier(req).await + } +} diff --git a/nodedb/src/control/security/auth_lease/status.rs b/nodedb/src/control/security/auth_lease/status.rs new file mode 100644 index 000000000..f4a182ce0 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/status.rs @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! This node's lease state, as `/healthz` and the metrics endpoint report it. +//! +//! Boot waits for the first lease before the gates open. A node that later +//! loses its lease refuses every permission-checked statement until it +//! renews. It still serves every other path, so readiness reports it as +//! degraded, with the reason, for as long as it holds no valid lease. + +use std::fmt::Write as _; +use std::time::{Duration, Instant}; + +use crate::control::security::auth_fence::AuthorizationFence; + +/// Whether this node can plan permission-checked statements now. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum LeaseStatus { + /// A single node: no lease exists and none is needed. + NotRequired, + /// The lease is valid for `remaining`. + Valid { remaining: Duration }, + /// No valid lease. `expired_for` is how long ago the last one ended, or + /// `None` when no lease was ever granted. + Invalid { expired_for: Option }, +} + +/// The lease state at `now`. +pub fn lease_status(fence: &AuthorizationFence, now: Instant) -> LeaseStatus { + if fence.timing().is_none() { + return LeaseStatus::NotRequired; + } + match fence.holder().valid_until() { + Some(until) if now < until => LeaseStatus::Valid { + remaining: until - now, + }, + Some(until) => LeaseStatus::Invalid { + expired_for: Some(now.saturating_duration_since(until)), + }, + None => LeaseStatus::Invalid { expired_for: None }, + } +} + +/// Append the lease gauges. A single node holds no lease and emits none. +/// +/// - `nodedb_authorization_lease_valid`: 1 while the lease is valid, else 0. +/// - `nodedb_authorization_lease_remaining_seconds`: time left on the lease, +/// 0 without one. +pub fn render_prometheus(fence: &AuthorizationFence, out: &mut String) { + let (valid, remaining) = match lease_status(fence, Instant::now()) { + LeaseStatus::NotRequired => return, + LeaseStatus::Valid { remaining } => (1, remaining), + LeaseStatus::Invalid { .. } => (0, Duration::ZERO), + }; + let _ = writeln!( + out, + "# HELP nodedb_authorization_lease_valid Whether this node holds a valid \ + authorization lease and can plan permission-checked statements\n\ + # TYPE nodedb_authorization_lease_valid gauge\n\ + nodedb_authorization_lease_valid {valid}\n\ + # HELP nodedb_authorization_lease_remaining_seconds Time left on this \ + node's authorization lease\n\ + # TYPE nodedb_authorization_lease_remaining_seconds gauge\n\ + nodedb_authorization_lease_remaining_seconds {}", + remaining.as_secs_f64() + ); +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::security::auth_lease::LeaseTiming; + use crate::control::security::permission_tree::SourceIndex; + + fn timing() -> LeaseTiming { + LeaseTiming::from_raft(Duration::from_millis(1000), Duration::from_millis(100)) + .expect("timing") + } + + #[test] + fn a_node_without_timing_needs_no_lease() { + let fence = AuthorizationFence::new(std::sync::Arc::new(SourceIndex::default())); + assert_eq!( + lease_status(&fence, Instant::now()), + LeaseStatus::NotRequired + ); + let mut out = String::new(); + render_prometheus(&fence, &mut out); + assert!(out.is_empty()); + } + + #[test] + fn a_lapsed_lease_is_invalid_and_reports_zero() { + let fence = AuthorizationFence::new(std::sync::Arc::new(SourceIndex::default())); + assert!(fence.install_timing(timing())); + let now = Instant::now(); + assert_eq!( + lease_status(&fence, now), + LeaseStatus::Invalid { expired_for: None } + ); + + fence.holder().install(now + Duration::from_secs(2)); + assert_eq!( + lease_status(&fence, now), + LeaseStatus::Valid { + remaining: Duration::from_secs(2) + } + ); + let mut out = String::new(); + render_prometheus(&fence, &mut out); + assert!(out.contains("nodedb_authorization_lease_valid 1")); + + let later = now + Duration::from_secs(3); + assert_eq!( + lease_status(&fence, later), + LeaseStatus::Invalid { + expired_for: Some(Duration::from_secs(1)) + } + ); + } +} diff --git a/nodedb/src/control/security/auth_lease/table.rs b/nodedb/src/control/security/auth_lease/table.rs new file mode 100644 index 000000000..332372978 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/table.rs @@ -0,0 +1,274 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The metadata leader's lease table. +//! +//! One table exists per leadership term. It records, for each node that +//! renewed with this leader, when its lease ends and what its last report +//! covered. It also records the floors: per group, the highest index of any +//! authorization change a barrier registered. A lease is granted only to a +//! report that covers every floor. +//! +//! A barrier releases once, for every target, each node holding an unexpired +//! lease reported coverage of it. Leases granted by an earlier leader are not +//! in the table. They end within one lease duration of this leader taking +//! over, so a barrier also waits until then. +//! +//! The table is pure: callers pass the clock, so every rule is testable. + +use std::collections::HashMap; +use std::time::{Duration, Instant}; + +use nodedb_cluster::GroupCoverage; + +/// What a node's last renewal reported and when its lease ends. +#[derive(Debug, Default)] +struct HolderRecord { + /// End of the lease this leader granted, if any. + expires_at: Option, + /// Coverage by group, from the last renewal. + coverage: HashMap, +} + +impl HolderRecord { + fn covers(&self, group_id: u64, index: u64) -> bool { + self.coverage + .get(&group_id) + .is_some_and(|through| *through >= index) + } +} + +/// The answer to a renewal. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RenewDecision { + Granted, + Withheld, +} + +/// Where a barrier stands. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BarrierState { + /// No node can plan against state older than the targets. + Released, + /// Waiting for a report or an expiry. Nothing changes on its own before + /// the instant named, except a renewal. + Waiting { until: Instant }, + /// The floors of this term are not loaded yet. + NotReady, +} + +/// The lease table of one leadership term. +#[derive(Debug)] +pub struct LeaseTable { + term: u64, + leader_since: Instant, + floors_ready: bool, + floors: HashMap, + holders: HashMap, +} + +impl LeaseTable { + /// A table for `term`, whose leadership this node observed at `now`. + pub fn new(term: u64, now: Instant) -> Self { + Self { + term, + leader_since: now, + floors_ready: false, + floors: HashMap::new(), + holders: HashMap::new(), + } + } + + pub fn term(&self) -> u64 { + self.term + } + + pub fn floors_ready(&self) -> bool { + self.floors_ready + } + + /// Load the floors this term starts from: an index per group at or above + /// every change acknowledged before the term. + pub fn load_floors(&mut self, floors: &[GroupCoverage]) { + self.raise_floors(floors); + self.floors_ready = true; + } + + /// Raise the floors to cover `targets`. + pub fn raise_floors(&mut self, targets: &[GroupCoverage]) { + for target in targets { + let floor = self.floors.entry(target.group_id).or_insert(0); + *floor = (*floor).max(target.through); + } + } + + /// Record a renewal from `node_id` and decide on its lease. + pub fn renew( + &mut self, + node_id: u64, + coverage: &[GroupCoverage], + now: Instant, + lease: Duration, + ) -> RenewDecision { + let record = self.holders.entry(node_id).or_default(); + record.coverage = coverage + .iter() + .map(|report| (report.group_id, report.through)) + .collect(); + let covered = self + .floors + .iter() + .all(|(group_id, floor)| record.covers(*group_id, *floor)); + if !self.floors_ready || !covered { + return RenewDecision::Withheld; + } + record.expires_at = Some(now + lease); + RenewDecision::Granted + } + + /// Where a barrier on `targets` stands at `now`. + pub fn barrier( + &self, + targets: &[GroupCoverage], + now: Instant, + lease: Duration, + ) -> BarrierState { + if !self.floors_ready { + return BarrierState::NotReady; + } + let mut until: Option = None; + let mut wait_for = |instant: Instant| { + until = Some(until.map_or(instant, |current: Instant| current.min(instant))); + }; + let earlier_leases_end = self.leader_since + lease; + if now < earlier_leases_end { + wait_for(earlier_leases_end); + } + for record in self.holders.values() { + let Some(expires_at) = record.expires_at.filter(|end| *end > now) else { + continue; + }; + let covered = targets + .iter() + .all(|target| record.covers(target.group_id, target.through)); + if !covered { + wait_for(expires_at); + } + } + match until { + Some(until) => BarrierState::Waiting { until }, + None => BarrierState::Released, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const LEASE: Duration = Duration::from_millis(150); + + fn cover(group_id: u64, through: u64) -> GroupCoverage { + GroupCoverage { group_id, through } + } + + /// A table past the window of earlier leaders' leases. + fn settled_table(start: Instant) -> LeaseTable { + let mut table = LeaseTable::new(4, start); + table.load_floors(&[cover(0, 10)]); + table + } + + #[test] + fn nothing_is_granted_or_released_before_the_floors_load() { + let now = Instant::now(); + let mut table = LeaseTable::new(1, now); + assert_eq!( + table.renew(2, &[cover(0, 99)], now, LEASE), + RenewDecision::Withheld + ); + assert_eq!(table.barrier(&[], now, LEASE), BarrierState::NotReady); + } + + #[test] + fn a_report_below_a_floor_is_withheld() { + let start = Instant::now(); + let mut table = settled_table(start); + assert_eq!( + table.renew(2, &[cover(0, 9)], start, LEASE), + RenewDecision::Withheld + ); + assert_eq!( + table.renew(2, &[cover(0, 10)], start, LEASE), + RenewDecision::Granted + ); + // A group the report omits counts as uncovered. + table.raise_floors(&[cover(5, 1)]); + assert_eq!( + table.renew(2, &[cover(0, 10)], start, LEASE), + RenewDecision::Withheld + ); + } + + #[test] + fn a_barrier_waits_out_the_leases_of_earlier_leaders() { + let start = Instant::now(); + let table = settled_table(start); + assert_eq!( + table.barrier(&[cover(0, 5)], start, LEASE), + BarrierState::Waiting { + until: start + LEASE + } + ); + assert_eq!( + table.barrier(&[cover(0, 5)], start + LEASE, LEASE), + BarrierState::Released + ); + } + + #[test] + fn a_barrier_releases_on_coverage_or_expiry() { + let start = Instant::now(); + let mut table = settled_table(start); + let now = start + LEASE; + assert_eq!( + table.renew(2, &[cover(0, 10)], now, LEASE), + RenewDecision::Granted + ); + let target = [cover(0, 12)]; + table.raise_floors(&target); + // Node 2 holds a lease and has not covered index 12. + assert_eq!( + table.barrier(&target, now, LEASE), + BarrierState::Waiting { until: now + LEASE } + ); + // Its renewal below the new floor is withheld, and its lease is not + // extended. + assert_eq!( + table.renew(2, &[cover(0, 11)], now, LEASE), + RenewDecision::Withheld + ); + // Covering the target releases the barrier at once. + assert_eq!( + table.renew(2, &[cover(0, 12)], now, LEASE), + RenewDecision::Granted + ); + assert_eq!(table.barrier(&target, now, LEASE), BarrierState::Released); + + // A node that never covers releases the barrier when its lease ends. + let later = [cover(0, 20)]; + table.raise_floors(&later); + assert_eq!( + table.barrier(&later, now, LEASE), + BarrierState::Waiting { until: now + LEASE } + ); + assert_eq!( + table.barrier(&later, now + LEASE, LEASE), + BarrierState::Released + ); + // Once expired, it gets no lease back without covering the floor. + assert_eq!( + table.renew(2, &[cover(0, 12)], now + LEASE, LEASE), + RenewDecision::Withheld + ); + } +} diff --git a/nodedb/src/control/security/auth_lease/timing.rs b/nodedb/src/control/security/auth_lease/timing.rs new file mode 100644 index 000000000..5cebf13ad --- /dev/null +++ b/nodedb/src/control/security/auth_lease/timing.rs @@ -0,0 +1,97 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Lease durations derived from the Raft timing. +//! +//! - **Lease:** the minimum election timeout. A leader that loses its +//! quorum is replaced no sooner than that, the same bound Raft leader +//! leases rely on. +//! - **Skew margin:** the heartbeat interval. The holder ends its lease this +//! much before the leader does, so a holder clock that runs slow by up to +//! the heartbeat-to-election ratio never outlives the leader's view. +//! - **Renewal:** every heartbeat interval. A holder that covers a change +//! renews within one heartbeat, so an acknowledgement waits about one +//! heartbeat when every node is healthy. + +use std::time::{Duration, Instant}; + +/// Lease durations of one cluster. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct LeaseTiming { + /// How long a granted lease runs on the leader's clock. + pub lease: Duration, + /// How much earlier a holder ends its lease than the leader. + pub skew_margin: Duration, + /// How often a holder renews. + pub renew_every: Duration, +} + +impl LeaseTiming { + /// Derive the timing from the Raft election timeout and heartbeat. + pub fn from_raft( + election_timeout_min: Duration, + heartbeat_interval: Duration, + ) -> crate::Result { + if heartbeat_interval.is_zero() || heartbeat_interval >= election_timeout_min { + return Err(crate::Error::Config { + detail: format!( + "authorization lease: heartbeat interval {heartbeat_interval:?} must be \ + non-zero and below the minimum election timeout {election_timeout_min:?}" + ), + }); + } + Ok(Self { + lease: election_timeout_min, + skew_margin: heartbeat_interval, + renew_every: heartbeat_interval, + }) + } + + /// When a lease the leader granted for `granted`, requested at `sent_at` + /// on the holder's clock, ends on the holder's clock. + /// + /// The leader starts its lease no earlier than it received the request, + /// which is after `sent_at`. Ending at `sent_at + granted - skew_margin` + /// therefore ends first on any holder clock that runs no slower than the + /// margin allows. + pub fn holder_expiry(&self, sent_at: Instant, granted: Duration) -> Instant { + sent_at + granted.saturating_sub(self.skew_margin) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn timing_follows_the_raft_configuration() { + let timing = LeaseTiming::from_raft(Duration::from_millis(150), Duration::from_millis(50)) + .expect("timing"); + assert_eq!(timing.lease, Duration::from_millis(150)); + assert_eq!(timing.skew_margin, Duration::from_millis(50)); + assert_eq!(timing.renew_every, Duration::from_millis(50)); + } + + #[test] + fn a_heartbeat_at_or_above_the_election_timeout_is_refused() { + assert!( + LeaseTiming::from_raft(Duration::from_millis(50), Duration::from_millis(50)).is_err() + ); + assert!(LeaseTiming::from_raft(Duration::from_millis(50), Duration::ZERO).is_err()); + } + + /// The holder ends its lease a skew margin before the leader does, even + /// when the leader granted at the very moment the request left. + #[test] + fn the_holder_expires_early_by_the_skew_margin() { + let timing = LeaseTiming::from_raft(Duration::from_millis(150), Duration::from_millis(50)) + .expect("timing"); + let sent_at = Instant::now(); + let holder_end = timing.holder_expiry(sent_at, timing.lease); + let leader_end = sent_at + timing.lease; + assert_eq!(leader_end - holder_end, timing.skew_margin); + // A holder clock slow by a third of the lease still ends first: 100ms + // of holder time is at most 133ms of real time, inside 150ms. + let slow_real_elapsed = (holder_end - sent_at).mul_f64(4.0 / 3.0); + assert!(sent_at + slow_real_elapsed <= leader_end); + } +} diff --git a/nodedb/src/control/security/mod.rs b/nodedb/src/control/security/mod.rs index 8f73759b8..7ae4ea187 100644 --- a/nodedb/src/control/security/mod.rs +++ b/nodedb/src/control/security/mod.rs @@ -4,6 +4,8 @@ pub mod apikey; pub mod audit; pub mod auth_apikey; pub mod auth_context; +pub mod auth_fence; +pub mod auth_lease; pub mod blacklist; pub mod buses; pub mod catalog; diff --git a/nodedb/src/control/security/permission_tree/cache.rs b/nodedb/src/control/security/permission_tree/cache.rs index f092da0ab..cb798f133 100644 --- a/nodedb/src/control/security/permission_tree/cache.rs +++ b/nodedb/src/control/security/permission_tree/cache.rs @@ -2,14 +2,18 @@ //! In-memory permission cache: parent hierarchy + grant lookups. //! -//! Loaded from the resource graph and permission collection on startup. -//! Maintained via CDC events for real-time invalidation. -//! Lives entirely in the Control Plane (Send + Sync). +//! Loaded from the governed collections and permission tables by a reload, +//! and kept current by the Event Plane's permission step. `progress` records +//! how far the cache reflects each core's writes. Lives entirely in the +//! Control Plane (Send + Sync). use std::collections::{HashMap, HashSet}; +use std::sync::Arc; use tracing::{debug, info}; +use super::sources::SourceIndex; +use super::sync_state::ApplyProgress; use super::types::{PermissionGrant, PermissionTreeDef}; /// Per-tenant permission state: resource hierarchy + permission grants. @@ -37,7 +41,7 @@ struct TenantPermissions { /// Central permission cache shared across all sessions. /// /// Thread-safe: wrapped in `Arc>` by SharedState. -#[derive(Default, Debug)] +#[derive(Debug)] pub struct PermissionCache { /// Per-tenant permission state. tenants: HashMap, @@ -45,15 +49,111 @@ pub struct PermissionCache { /// Per-collection permission tree definitions. /// Key: `(tenant_id, collection_name)`. tree_defs: HashMap<(u64, String), PermissionTreeDef>, + + /// How far the cache reflects each core's writes. + progress: ApplyProgress, + + /// The source collections of `tree_defs`, readable without this cache's + /// lock. Rebuilt on every tree-definition change. + sources: Arc, +} + +impl Default for PermissionCache { + fn default() -> Self { + Self::new() + } +} + +/// One collection a reload scans, and what its rows carry. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct TreeSource { + pub tenant_id: u64, + pub collection: String, + pub kind: TreeSourceKind, +} + +/// What a source collection's rows carry. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum TreeSourceKind { + /// A governed collection: each row's `id` and `parent_id` form an edge. + Hierarchy, + /// A permission table: each row is a grant. + Grants, } impl PermissionCache { + /// An empty cache. It needs a reload before planning may use it. pub fn new() -> Self { - Self::default() + Self { + tenants: HashMap::new(), + tree_defs: HashMap::new(), + progress: ApplyProgress::new(), + sources: Arc::new(SourceIndex::default()), + } + } + + /// The lock-free index of this cache's source collections. + pub fn sources(&self) -> Arc { + Arc::clone(&self.sources) + } + + /// How far the cache reflects each core's writes. + pub fn progress(&self) -> &ApplyProgress { + &self.progress } - /// Register a permission tree definition for a collection. + /// Mutable access to the apply progress, for the permission step and a + /// reload. + pub fn progress_mut(&mut self) -> &mut ApplyProgress { + &mut self.progress + } + + /// Every collection a reload scans, deduplicated. + pub fn tree_sources(&self) -> Vec { + let mut sources: HashSet = HashSet::new(); + for ((tenant_id, collection), def) in &self.tree_defs { + sources.insert(TreeSource { + tenant_id: *tenant_id, + collection: collection.clone(), + kind: TreeSourceKind::Hierarchy, + }); + sources.insert(TreeSource { + tenant_id: *tenant_id, + collection: def.permission_table.clone(), + kind: TreeSourceKind::Grants, + }); + } + sources.into_iter().collect() + } + + /// Replace a tenant's hierarchy and grants with a reload's result, and + /// bump its version so a cached plan built from the old state goes stale. + pub fn replace_tenant_state( + &mut self, + tenant_id: u64, + edges: &[(String, String)], + grants: &[PermissionGrant], + ) { + let version = self.tenant_version(tenant_id); + self.tenants.insert( + tenant_id, + TenantPermissions { + version, + ..TenantPermissions::default() + }, + ); + self.load_edges(tenant_id, edges); + self.load_grants(tenant_id, grants); + self.bump_tenant_version(tenant_id); + } + + /// Register a permission tree definition for a collection. Registering + /// the definition already held changes nothing, so the DDL node and the + /// metadata applier can both apply one change. pub fn register_tree_def(&mut self, tenant_id: u64, collection: &str, def: PermissionTreeDef) { + if self.get_tree_def(tenant_id, collection) == Some(&def) { + return; + } info!( tenant_id, collection, @@ -62,11 +162,25 @@ impl PermissionCache { ); self.tree_defs .insert((tenant_id, collection.to_owned()), def); + self.sources.rebuild(&self.tree_defs); + // The new sources may already hold rows no reload has read. + self.progress.mark_reload_needed(); + self.bump_tenant_version(tenant_id); } - /// Remove a permission tree definition for a collection. + /// Remove a permission tree definition for a collection. Removing an + /// absent definition changes nothing. pub fn unregister_tree_def(&mut self, tenant_id: u64, collection: &str) { - self.tree_defs.remove(&(tenant_id, collection.to_owned())); + if self + .tree_defs + .remove(&(tenant_id, collection.to_owned())) + .is_none() + { + return; + } + self.sources.rebuild(&self.tree_defs); + self.progress.mark_reload_needed(); + self.bump_tenant_version(tenant_id); info!(tenant_id, collection, "permission_tree: unregistered"); } @@ -386,7 +500,66 @@ mod tests { assert!(cache.get_tree_def(1, "other").is_none()); assert!(cache.get_tree_def(2, "documents").is_none()); + // The DDL node and the metadata applier both apply one change. + let version = cache.tenant_version(1); + cache.register_tree_def(1, "documents", def); + assert_eq!(cache.tenant_version(1), version); + cache.unregister_tree_def(1, "documents"); assert!(cache.get_tree_def(1, "documents").is_none()); + let version = cache.tenant_version(1); + cache.unregister_tree_def(1, "documents"); + assert_eq!(cache.tenant_version(1), version); + } + + #[test] + fn a_reload_replaces_the_tenant_state_and_bumps_its_version() { + let mut cache = PermissionCache::new(); + cache.put_edge(1, "doc-1", "folder-1"); + cache.put_grant( + 1, + &PermissionGrant { + resource_id: "doc-1".into(), + grantee: "user-1".into(), + level: "viewer".into(), + inherited: false, + }, + ); + let before = cache.bump_tenant_version(1); + + cache.replace_tenant_state(1, &[("doc-2".into(), "folder-2".into())], &[]); + + assert_eq!(cache.get_parent(1, "doc-1"), None); + assert_eq!(cache.get_parent(1, "doc-2"), Some("folder-2")); + assert!(cache.get_grant(1, "doc-1", "user-1").is_none()); + assert!(cache.tenant_version(1) > before); + } + + #[test] + fn tree_sources_name_the_governed_collection_and_the_permission_table() { + let mut cache = PermissionCache::new(); + let def: PermissionTreeDef = sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def"); + cache.register_tree_def(1, "docs", def); + let mut sources = cache.tree_sources(); + sources.sort_by(|a, b| a.collection.cmp(&b.collection)); + assert_eq!( + sources, + vec![ + TreeSource { + tenant_id: 1, + collection: "docs".into(), + kind: TreeSourceKind::Hierarchy, + }, + TreeSource { + tenant_id: 1, + collection: "grants".into(), + kind: TreeSourceKind::Grants, + }, + ] + ); + assert!(cache.progress().needs_reload_for(&[])); } } diff --git a/nodedb/src/control/security/permission_tree/event_handler.rs b/nodedb/src/control/security/permission_tree/event_handler.rs index 9c227c3db..828a27b4f 100644 --- a/nodedb/src/control/security/permission_tree/event_handler.rs +++ b/nodedb/src/control/security/permission_tree/event_handler.rs @@ -1,9 +1,16 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Event Plane integration: process WriteEvents that affect permission trees. +//! Event Plane integration: the permission step. //! -//! When a row is written to a collection that serves as a permission table -//! or resource hierarchy for some permission tree, update the in-memory cache. +//! The Event Plane passes every event it takes off a core's ring through +//! [`apply_ring_events`], in ring order. An event for a permission table or a +//! governed collection updates the cache. Every event advances the core's +//! apply progress, so lease coverage can tell how far the cache reflects the +//! core's writes (see [`super::sync_state`]). +//! +//! Events rebuilt from the WAL during catch-up never pass through here. Their +//! ring copies do, or, when the ring dropped them, a reload covers their +//! writes. use std::sync::Arc; @@ -15,28 +22,47 @@ use super::cache::PermissionCache; use super::invalidation; use super::types::PermissionGrant; -/// Process a WriteEvent and update the permission cache if relevant. -/// -/// Called from the Event Plane consumer for every data event, on both the -/// Normal-mode and WAL-catchup paths. Checks if the event's collection is a -/// permission table or resource graph for any registered permission tree, -/// and updates the cache accordingly. +/// Apply the permission effect of `events`, taken off core `core_id`'s ring +/// in ring order, and advance that core's apply progress. /// -/// Waits for the write lock. Each grant or edge event is the only update for -/// that row, so a skipped event leaves the cache wrong with no later repair. -/// This function holds the lock across no await. The wait ends when the -/// planners that hold a read lock finish. -pub async fn handle_permission_event( - event: &WriteEvent, +/// Holds the write lock for the whole batch and across no await. `notify` +/// wakes the coverage waits for this core. +pub async fn apply_ring_events( + core_id: usize, + events: &[WriteEvent], cache: &Arc>, + notify: &tokio::sync::Notify, ) { + if events.is_empty() { + return; + } + // A test parks the permission step here to prove a permission-row write + // is not acknowledged before the step applies it. The gate sits before + // the lock, so a reload can still run. + #[cfg(feature = "failpoints")] + crate::control::fail_gate::wait("permission_tree::before_apply").await; + { + let mut guard = cache.write().await; + for event in events { + if event.op.is_data_event() + && !guard.progress().covered_by_reload(core_id, event.sequence) + { + apply_event(&mut guard, event); + } + guard.progress_mut().note_consumed(core_id, event.sequence); + } + } + notify.notify_waiters(); +} + +/// Apply one data event to the cache when its collection is a permission +/// table or a governed collection of a registered tree. +fn apply_event(cache: &mut PermissionCache, event: &WriteEvent) { let collection = event.collection.as_ref(); let tenant_id = event.tenant_id.as_u64(); - let mut guard = cache.write().await; - - let is_permission_table = guard.tree_defs_using_permission_table(tenant_id, collection); - let is_resource_graph = guard.tree_defs_using_graph(tenant_id, collection); + let is_permission_table = cache.tree_defs_using_permission_table(tenant_id, collection); + let is_resource_graph = cache.tree_defs_using_graph(tenant_id, collection); if !is_permission_table && !is_resource_graph { return; @@ -53,16 +79,13 @@ pub async fn handle_permission_event( && let Some(ref val) = new_val && let Some(grant) = extract_grant(val) { - invalidation::on_grant_upsert(&mut guard, tenant_id, &grant); + invalidation::on_grant_upsert(cache, tenant_id, &grant); } if is_resource_graph && let Some(ref val) = new_val - && let (Some(child_id), Some(parent_id)) = ( - val.get("id").and_then(|v| v.as_str()), - val.get("parent_id").and_then(|v| v.as_str()), - ) + && let Some((child_id, parent_id)) = extract_edge(val) { - invalidation::on_edge_upsert(&mut guard, tenant_id, child_id, parent_id); + invalidation::on_edge_upsert(cache, tenant_id, child_id, parent_id); } } crate::event::types::WriteOp::Delete => { @@ -78,26 +101,33 @@ pub async fn handle_permission_event( val.get("grantee").and_then(|v| v.as_str()), ) { - invalidation::on_grant_delete(&mut guard, tenant_id, resource_id, grantee); + invalidation::on_grant_delete(cache, tenant_id, resource_id, grantee); } if is_resource_graph && let Some(ref val) = old_val && let Some(child_id) = val.get("id").and_then(|v| v.as_str()) { - invalidation::on_edge_delete(&mut guard, tenant_id, child_id); + invalidation::on_edge_delete(cache, tenant_id, child_id); } } - _ => {} + crate::event::types::WriteOp::BulkInsert { .. } + | crate::event::types::WriteOp::BulkDelete { .. } + | crate::event::types::WriteOp::Heartbeat => {} } debug!( tenant_id, - collection, "permission_tree: cache updated from CDC event" + collection, "permission_tree: cache updated from a write event" ); } +/// Extract a `(child_id, parent_id)` edge from a governed collection's row. +pub(super) fn extract_edge(val: &serde_json::Value) -> Option<(&str, &str)> { + Some((val.get("id")?.as_str()?, val.get("parent_id")?.as_str()?)) +} + /// Extract a PermissionGrant from a JSON value (permission table row). -fn extract_grant(val: &serde_json::Value) -> Option { +pub(super) fn extract_grant(val: &serde_json::Value) -> Option { Some(PermissionGrant { resource_id: val.get("resource_id")?.as_str()?.to_owned(), grantee: val.get("grantee")?.as_str()?.to_owned(), @@ -126,30 +156,37 @@ mod tests { .expect("tree def"); let mut cache = PermissionCache::new(); cache.register_tree_def(TENANT, "docs", def); + cache.progress_mut().install_reload(&[0]); Arc::new(tokio::sync::RwLock::new(cache)) } - fn grant_insert() -> WriteEvent { + fn grant_event(sequence: u64, op: WriteOp) -> WriteEvent { let row = serde_json::json!({ "resource_id": "d1", "grantee": "role_a", "level": "viewer", "inherited": false, }); + let body: Option> = Some(Arc::from( + nodedb_types::json_to_msgpack(&row).expect("encode grant row"), + )); + let (new_value, old_value) = match op { + WriteOp::Delete => (None, body), + _ => (body, None), + }; WriteEvent { - sequence: 1, + sequence, collection: Arc::from("grants"), - op: WriteOp::Insert, + op, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("g1")), - lsn: Lsn::new(1), + lsn: Lsn::new(sequence), + record: None, database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(TENANT), vshard_id: VShardId::new(0), source: EventSource::User, - new_value: Some(Arc::from( - nodedb_types::json_to_msgpack(&row).expect("encode grant row"), - )), - old_value: None, + new_value, + old_value, system_time_ms: None, valid_time_ms: None, user_id: None, @@ -160,11 +197,16 @@ mod tests { #[tokio::test] async fn grant_event_applies_after_a_held_read_lock_is_released() { let cache = cache_with_tree(); + let notify = Arc::new(tokio::sync::Notify::new()); + let version_before = cache.read().await.tenant_version(TENANT); let reader = cache.read().await; let apply = { let cache = Arc::clone(&cache); - tokio::spawn(async move { handle_permission_event(&grant_insert(), &cache).await }) + let notify = Arc::clone(¬ify); + tokio::spawn(async move { + apply_ring_events(0, &[grant_event(1, WriteOp::Insert)], &cache, ¬ify).await + }) }; tokio::task::yield_now().await; assert!(!apply.is_finished(), "the update waits for the reader"); @@ -176,6 +218,52 @@ mod tests { guard.get_grant(TENANT, "d1", "role_a"), Some(("viewer", false)) ); - assert_eq!(guard.tenant_version(TENANT), 1); + assert_eq!(guard.tenant_version(TENANT), version_before + 1); + assert!(guard.progress().caught_up(&[1])); + } + + /// An event a reload already covers is not applied again: the ring may + /// have dropped a later write to the same row. + #[tokio::test] + async fn an_event_covered_by_a_reload_is_not_applied_again() { + let cache = cache_with_tree(); + let notify = tokio::sync::Notify::new(); + cache.write().await.progress_mut().install_reload(&[2]); + + apply_ring_events(0, &[grant_event(2, WriteOp::Insert)], &cache, ¬ify).await; + assert!( + cache + .read() + .await + .get_grant(TENANT, "d1", "role_a") + .is_none() + ); + + apply_ring_events(0, &[grant_event(3, WriteOp::Insert)], &cache, ¬ify).await; + let guard = cache.read().await; + assert!(guard.get_grant(TENANT, "d1", "role_a").is_some()); + assert!(guard.progress().caught_up(&[3])); + } + + /// A gap in the ring leaves the core behind even though later events + /// apply. + #[tokio::test] + async fn a_dropped_event_leaves_the_core_behind() { + let cache = cache_with_tree(); + let notify = tokio::sync::Notify::new(); + apply_ring_events( + 0, + &[ + grant_event(1, WriteOp::Insert), + grant_event(3, WriteOp::Delete), + ], + &cache, + ¬ify, + ) + .await; + let guard = cache.read().await; + assert!(guard.get_grant(TENANT, "d1", "role_a").is_none()); + assert!(!guard.progress().caught_up(&[3])); + assert!(guard.progress().needs_reload_for(&[3])); } } diff --git a/nodedb/src/control/security/permission_tree/mod.rs b/nodedb/src/control/security/permission_tree/mod.rs index a9d4eefef..8baa47db6 100644 --- a/nodedb/src/control/security/permission_tree/mod.rs +++ b/nodedb/src/control/security/permission_tree/mod.rs @@ -3,8 +3,12 @@ pub mod cache; pub mod event_handler; pub mod invalidation; +pub mod reload; pub mod resolver; +pub mod sources; +pub mod sync_state; pub mod types; -pub use cache::PermissionCache; +pub use cache::{PermissionCache, TreeSource, TreeSourceKind}; +pub use sources::SourceIndex; pub use types::PermissionTreeDef; diff --git a/nodedb/src/control/security/permission_tree/reload.rs b/nodedb/src/control/security/permission_tree/reload.rs new file mode 100644 index 000000000..f552da0df --- /dev/null +++ b/nodedb/src/control/security/permission_tree/reload.rs @@ -0,0 +1,143 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Reload the permission cache from its source collections. +//! +//! A reload reads the emitted-event counter of every core, then scans every +//! governed collection and permission table on this node's own cores. A scan +//! runs on its core after every write whose event the counter already +//! counted, so the reloaded state covers each of those events. The reload +//! holds the cache's write lock throughout: the permission step waits, so no +//! event it applies falls between the counter read and the install. +//! +//! The scans read local state, never a leader's. The cache must match this +//! node's own apply position. The scans dispatch through the read-only local +//! path, never the write funnel. + +use std::collections::HashMap; + +use nodedb_types::{QualifiedCollection, TenantId}; + +use crate::control::server::dispatch_utils::{ + LocalRead, dispatch_local_read, reject_data_plane_error, +}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, VShardId}; + +use super::cache::{PermissionCache, TreeSource, TreeSourceKind}; +use super::event_handler::{extract_edge, extract_grant}; +use super::types::PermissionGrant; + +/// Edges and grants a reload read for one tenant. +#[derive(Default)] +struct TenantRows { + edges: Vec<(String, String)>, + grants: Vec, +} + +/// Reload every registered tree's sources, unless the cache already reflects +/// every event at or below `targets` (another reload got there first). +/// +/// `targets` of `None` reloads unconditionally: no permission step runs, so +/// only a reload reflects writes. +pub async fn reload_all(state: &SharedState, targets: Option<&[u64]>) -> crate::Result<()> { + let mut cache = state.permission_cache.write().await; + if let Some(targets) = targets + && cache.progress().caught_up(targets) + { + return Ok(()); + } + reload_locked(state, &mut cache).await +} + +/// Reload when the cache is stale: no reload covers the registered trees yet +/// (startup, or a tree definition changed), or a core lost an event. +/// +/// Planning and the lease renewal call this. A writer waiting for its +/// acknowledgement never does: a stale cache is reloaded before the next +/// statement plans, and that reload reads the write. +pub async fn reload_if_stale(state: &SharedState) -> crate::Result<()> { + // Checked under the read lock first: planning calls this for every + // statement, and the write lock is needed only when the cache is stale. + if !state.permission_cache.read().await.progress().is_stale() { + return Ok(()); + } + let mut cache = state.permission_cache.write().await; + if !cache.progress().is_stale() { + return Ok(()); + } + reload_locked(state, &mut cache).await +} + +/// Reload every registered tree's sources into `cache`, whose write lock the +/// caller holds. +async fn reload_locked(state: &SharedState, cache: &mut PermissionCache) -> crate::Result<()> { + let fence = &state.authorization_fence; + // Read under the write lock: every event the permission step applied is + // counted, and none it applies later can be. + let emitted = fence.emitted_snapshot().unwrap_or_default(); + + let mut rows: HashMap = HashMap::new(); + for source in cache.tree_sources() { + let docs = scan_source(state, &source).await?; + let tenant = rows.entry(source.tenant_id).or_default(); + match source.kind { + TreeSourceKind::Hierarchy => tenant.edges.extend(docs.iter().filter_map(|doc| { + extract_edge(doc).map(|(child, parent)| (child.to_owned(), parent.to_owned())) + })), + TreeSourceKind::Grants => tenant.grants.extend(docs.iter().filter_map(extract_grant)), + } + } + for (tenant_id, tenant) in rows { + cache.replace_tenant_state(tenant_id, &tenant.edges, &tenant.grants); + } + cache.progress_mut().install_reload(&emitted); + fence.permission_applied().notify_waiters(); + Ok(()) +} + +/// Every row of one source collection, read on this node's own core. +/// +/// A source that no longer exists holds no rows. A source that is not a +/// document collection cannot carry edges or grants, and refuses the reload: +/// planning with it would silently grant or deny nothing. +async fn scan_source( + state: &SharedState, + source: &TreeSource, +) -> crate::Result> { + let database_id = DatabaseId::DEFAULT; + let catalog = state.credentials.catalog(); + let stored = catalog.get_collection(database_id, source.tenant_id, &source.collection)?; + let Some(stored) = stored.filter(|collection| collection.is_active) else { + return Ok(Vec::new()); + }; + if !stored.collection_type.is_document() { + return Err(crate::Error::FeatureNotSupported { + detail: format!( + "permission tree source '{}' is a {} collection; the hierarchy and the \ + permission table must be document collections", + source.collection, stored.collection_type + ), + }); + } + + // A read-only dispatch: a reload can run while a write waits for its + // acknowledgement, and it must never issue a request through the write + // path. + let response = dispatch_local_read( + state, + TenantId::new(source.tenant_id), + database_id, + VShardId::from_collection_in_database(database_id, &source.collection), + LocalRead::DocumentScan { + collection: QualifiedCollection::new(database_id, &source.collection), + }, + ) + .await?; + reject_data_plane_error(&response)?; + Ok( + crate::data::executor::response_codec::decode_raw_scan_to_docs(response.payload.as_bytes()) + .into_iter() + .filter_map(|(_, body)| nodedb_types::json_from_msgpack(&body).ok()) + .collect(), + ) +} diff --git a/nodedb/src/control/security/permission_tree/sources.rs b/nodedb/src/control/security/permission_tree/sources.rs new file mode 100644 index 000000000..be05d6eba --- /dev/null +++ b/nodedb/src/control/security/permission_tree/sources.rs @@ -0,0 +1,175 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which collections feed a permission tree, readable without the cache lock. +//! +//! A write path decides on every write whether the write changes +//! authorization state: it does when it writes a governed collection or a +//! permission table of a registered tree. That check runs on hot write paths, +//! so it reads this index under a plain read lock rather than the cache's +//! async lock. +//! +//! Two sets feed the answer, and a collection in either counts: +//! +//! - **Cached:** rebuilt by the cache whenever a tree definition changes. +//! - **Committed:** updated by the metadata applier as it commits a change, +//! before the cache takes it from the queue. +//! +//! A write therefore counts as an authorization change from the moment its +//! tree definition committed on this node. A removed tree may count a little +//! longer, until both sets drop it, which only adds a barrier. +//! +//! Tree sources live in the default database, as the permission-tree DDL +//! writes them, so each source collection homes on one vShard. + +use std::collections::{HashMap, HashSet}; +use std::sync::RwLock; + +use crate::types::{DatabaseId, VShardId}; + +use super::types::PermissionTreeDef; + +#[derive(Debug, Default)] +struct SourceSet { + collections: HashSet, + vshards: HashSet, +} + +impl SourceSet { + fn from_defs<'a>( + defs: impl IntoIterator, + ) -> Self { + let mut set = Self::default(); + for ((_, governed), def) in defs { + for collection in [governed.as_str(), def.permission_table.as_str()] { + set.collections.insert(collection.to_owned()); + set.vshards.insert( + VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection).as_u32(), + ); + } + } + set + } +} + +#[derive(Debug, Default)] +struct Committed { + defs: HashMap<(u64, String), PermissionTreeDef>, + set: SourceSet, +} + +/// The source collections of every registered or committed tree. +#[derive(Debug, Default)] +pub struct SourceIndex { + cached: RwLock, + committed: RwLock, +} + +impl SourceIndex { + /// Rebuild the cached set from the cache's tree definitions. + pub(super) fn rebuild(&self, tree_defs: &HashMap<(u64, String), PermissionTreeDef>) { + *self.cached.write().unwrap_or_else(|p| p.into_inner()) = SourceSet::from_defs(tree_defs); + } + + /// Record a tree definition the metadata applier committed. `None` + /// removes the tree of `(tenant_id, collection)`. + pub fn note_committed( + &self, + tenant_id: u64, + collection: &str, + def: Option<&PermissionTreeDef>, + ) { + let mut committed = self.committed.write().unwrap_or_else(|p| p.into_inner()); + let key = (tenant_id, collection.to_owned()); + match def { + Some(def) => { + committed.defs.insert(key, def.clone()); + } + None => { + committed.defs.remove(&key); + } + } + committed.set = SourceSet::from_defs(&committed.defs); + } + + fn any(&self, test: impl Fn(&SourceSet) -> bool) -> bool { + test(&self.cached.read().unwrap_or_else(|p| p.into_inner())) + || test(&self.committed.read().unwrap_or_else(|p| p.into_inner()).set) + } + + /// Whether no tree is registered or committed. + pub fn is_empty(&self) -> bool { + !self.any(|set| !set.collections.is_empty()) + } + + /// Whether `collection` feeds a tree. + pub fn is_source_collection(&self, collection: &str) -> bool { + self.any(|set| set.collections.contains(collection)) + } + + /// Whether `vshard_id` homes a collection that feeds a tree. + pub fn is_source_vshard(&self, vshard_id: u32) -> bool { + self.any(|set| set.vshards.contains(&vshard_id)) + } + + /// Every vShard that homes a source collection. + pub fn source_vshards(&self) -> Vec { + let mut vshards: HashSet = self + .cached + .read() + .unwrap_or_else(|p| p.into_inner()) + .vshards + .clone(); + vshards.extend( + self.committed + .read() + .unwrap_or_else(|p| p.into_inner()) + .set + .vshards + .iter() + .copied(), + ); + vshards.into_iter().collect() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_index_names_governed_collections_and_permission_tables() { + let def: PermissionTreeDef = sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def"); + let mut defs = HashMap::new(); + defs.insert((1, "docs".to_owned()), def); + let index = SourceIndex::default(); + assert!(index.is_empty()); + index.rebuild(&defs); + assert!(index.is_source_collection("docs")); + assert!(index.is_source_collection("grants")); + assert!(!index.is_source_collection("other")); + let grants_vshard = + VShardId::from_collection_in_database(DatabaseId::DEFAULT, "grants").as_u32(); + assert!(index.is_source_vshard(grants_vshard)); + defs.clear(); + index.rebuild(&defs); + assert!(index.is_empty()); + assert!(!index.is_source_vshard(grants_vshard)); + } + + #[test] + fn a_committed_tree_counts_before_the_cache_takes_it() { + let def: PermissionTreeDef = sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def"); + let index = SourceIndex::default(); + index.note_committed(1, "docs", Some(&def)); + assert!(index.is_source_collection("grants")); + assert!(index.is_source_collection("docs")); + index.note_committed(1, "docs", None); + assert!(index.is_empty()); + } +} diff --git a/nodedb/src/control/security/permission_tree/sync_state.rs b/nodedb/src/control/security/permission_tree/sync_state.rs new file mode 100644 index 000000000..8fc34fdda --- /dev/null +++ b/nodedb/src/control/security/permission_tree/sync_state.rs @@ -0,0 +1,198 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! How far the permission cache reflects each core's writes. +//! +//! Every Data Plane core numbers the events it emits with a contiguous +//! sequence. The Event Plane passes every event it takes off a core's ring +//! through the permission step, in ring order. For each core the cache keeps +//! `applied`: every event numbered at or below it was either applied to the +//! cache, or its write is covered by a reload. +//! +//! `applied` only advances by one. A dropped event leaves a gap the step can +//! never close, so the core stays behind until a reload covers it. A reload +//! reads the emitted counters first, then scans the source collections. Each +//! scan runs on its core after every write whose event was counted, so the +//! reload covers every event numbered at or below the counter it read. +//! +//! A reload also names, per core, the highest event it covers. An event at or +//! below that number reaching the step later is not applied again. Its write +//! is in the reload, and a later write the ring dropped may have overwritten +//! it. + +/// Apply progress of one core. +#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] +struct CoreApply { + /// Every event at or below this number is reflected in the cache. + applied: u64, + /// Every event at or below this number is covered by the last reload. + reloaded_through: u64, + /// An event above `applied + 1` passed the step: one was dropped. + gap: bool, +} + +/// Apply progress of the permission cache, per core. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ApplyProgress { + cores: Vec, + /// No reload has covered the registered trees yet. Set at startup, before + /// the durable grants are loaded, and whenever a tree definition changes. + needs_reload: bool, +} + +impl Default for ApplyProgress { + fn default() -> Self { + Self::new() + } +} + +impl ApplyProgress { + /// Progress of a cache that holds no source data yet. + pub fn new() -> Self { + Self { + cores: Vec::new(), + needs_reload: true, + } + } + + fn core_mut(&mut self, core_id: usize) -> &mut CoreApply { + if self.cores.len() <= core_id { + self.cores.resize(core_id + 1, CoreApply::default()); + } + &mut self.cores[core_id] + } + + fn core(&self, core_id: usize) -> CoreApply { + self.cores.get(core_id).copied().unwrap_or_default() + } + + /// Whether the last reload covers event `sequence` of `core_id`. Such an + /// event is never applied again. + pub fn covered_by_reload(&self, core_id: usize, sequence: u64) -> bool { + sequence <= self.core(core_id).reloaded_through + } + + /// Record that event `sequence` of `core_id` passed the permission step. + pub fn note_consumed(&mut self, core_id: usize, sequence: u64) { + let core = self.core_mut(core_id); + if sequence <= core.applied { + return; + } + if sequence == core.applied + 1 { + core.applied = sequence; + } else { + core.gap = true; + } + } + + /// Whether planning must reload before it reads the cache: no reload + /// covers the registered trees, or a core lost an event the step can + /// never apply. + pub fn is_stale(&self) -> bool { + self.needs_reload || self.cores.iter().any(|core| core.gap) + } + + /// Record that the source set changed: a tree definition was registered + /// or removed. + pub fn mark_reload_needed(&mut self) { + self.needs_reload = true; + } + + /// Whether the cache reflects every event numbered at or below + /// `targets[core]` on each core. + pub fn caught_up(&self, targets: &[u64]) -> bool { + !self.needs_reload + && targets + .iter() + .enumerate() + .all(|(core_id, target)| self.core(core_id).applied >= *target) + } + + /// Whether waiting cannot reach `targets`: no reload covered the sources + /// yet, or a core that is behind its target lost an event. + pub fn needs_reload_for(&self, targets: &[u64]) -> bool { + self.needs_reload + || targets.iter().enumerate().any(|(core_id, target)| { + let core = self.core(core_id); + core.gap && core.applied < *target + }) + } + + /// Record a reload that covers every event at or below `emitted[core]`. + pub fn install_reload(&mut self, emitted: &[u64]) { + for (core_id, through) in emitted.iter().enumerate() { + let core = self.core_mut(core_id); + core.applied = core.applied.max(*through); + core.reloaded_through = core.reloaded_through.max(*through); + // Every event the step saw was emitted before the counter was + // read, so the reload covers any gap among them. + core.gap = false; + } + self.needs_reload = false; + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn reloaded(emitted: &[u64]) -> ApplyProgress { + let mut progress = ApplyProgress::new(); + progress.install_reload(emitted); + progress + } + + #[test] + fn a_fresh_cache_needs_a_reload_before_it_is_caught_up() { + let progress = ApplyProgress::new(); + assert!(!progress.caught_up(&[])); + assert!(progress.needs_reload_for(&[])); + assert!(reloaded(&[0]).caught_up(&[0])); + } + + #[test] + fn contiguous_events_advance_the_core() { + let mut progress = reloaded(&[0, 0]); + progress.note_consumed(1, 1); + progress.note_consumed(1, 2); + assert!(progress.caught_up(&[0, 2])); + assert!(!progress.caught_up(&[0, 3])); + assert!(!progress.needs_reload_for(&[0, 3])); + } + + #[test] + fn a_dropped_event_holds_the_core_until_a_reload() { + let mut progress = reloaded(&[0]); + progress.note_consumed(0, 1); + progress.note_consumed(0, 3); + progress.note_consumed(0, 4); + assert!(!progress.caught_up(&[4])); + assert!(progress.needs_reload_for(&[4])); + // A target the core reached before the gap needs nothing. + assert!(!progress.needs_reload_for(&[1])); + + progress.install_reload(&[4]); + assert!(progress.caught_up(&[4])); + assert!(progress.covered_by_reload(0, 3)); + assert!(!progress.covered_by_reload(0, 5)); + progress.note_consumed(0, 5); + assert!(progress.caught_up(&[5])); + } + + #[test] + fn a_tree_change_requires_a_new_reload() { + let mut progress = reloaded(&[2]); + progress.mark_reload_needed(); + assert!(!progress.caught_up(&[2])); + assert!(progress.needs_reload_for(&[2])); + } + + #[test] + fn a_lost_event_makes_the_cache_stale_until_a_reload() { + let mut progress = reloaded(&[0]); + assert!(!progress.is_stale()); + progress.note_consumed(0, 2); + assert!(progress.is_stale()); + progress.install_reload(&[2]); + assert!(!progress.is_stale()); + } +} diff --git a/nodedb/src/control/security/permission_tree/types.rs b/nodedb/src/control/security/permission_tree/types.rs index c51c76dc8..8766807bf 100644 --- a/nodedb/src/control/security/permission_tree/types.rs +++ b/nodedb/src/control/security/permission_tree/types.rs @@ -25,7 +25,7 @@ pub const DEFAULT_DELETE_LEVEL: &str = "owner"; /// /// Stored as JSON in `StoredCollection.permission_tree_def`. /// Binds the collection to a resource hierarchy graph and a permission table. -#[derive(Debug, Clone, Serialize, Deserialize)] +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] pub struct PermissionTreeDef { /// Column in this collection that serves as the resource identifier. /// Used to look up the resource in the permission graph. diff --git a/nodedb/src/control/server/dispatch_utils/local_read.rs b/nodedb/src/control/server/dispatch_utils/local_read.rs new file mode 100644 index 000000000..b751f1d38 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/local_read.rs @@ -0,0 +1,111 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Read-only dispatch of an internal scan to this node's own Data Plane. +//! +//! A caller that must never re-enter the write funnel reads through here. +//! [`LocalRead`] can express only reads, so no write plan can reach a core +//! through this path, and no function here calls `submit_write`. Holding a +//! write's acknowledgement open can therefore never wait on a request that +//! passes through the write path again. +//! +//! The request carries `Admission::Exempt(Read)`: a read never takes the +//! write fence. It reads local state on the core that homes the vShard. + +use std::time::{Duration, Instant}; + +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; +use nodedb_types::{QualifiedCollection, SystemTimeScope}; + +use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request, Response}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; + +use super::collect::{DeadlineCollect, collect_under_deadline}; + +/// A read an internal caller issues against its own node. +/// +/// Every variant is a read. A write is not representable. +pub(crate) enum LocalRead { + /// Every current row of one document collection. + DocumentScan { collection: QualifiedCollection }, +} + +impl LocalRead { + fn into_plan(self) -> PhysicalPlan { + match self { + LocalRead::DocumentScan { collection } => PhysicalPlan::Document(DocumentOp::Scan { + collection, + filters: Vec::new(), + limit: usize::MAX, + offset: 0, + sort_keys: Vec::new(), + distinct: false, + projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), + system_time: SystemTimeScope::Current, + valid_at_ms: None, + prefilter: None, + }), + } + } +} + +/// Dispatch `read` to the core that homes `vshard_id` and collect its +/// bounded response before the request deadline. +pub(crate) async fn dispatch_local_read( + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + read: LocalRead, +) -> crate::Result { + let deadline = + Instant::now() + Duration::from_secs(shared.tuning.network.default_deadline_secs); + let request_id = shared.next_request_id(); + let request = Request { + request_id, + tenant_id, + database_id, + vshard_id, + plan: read.into_plan(), + deadline, + priority: Priority::Normal, + trace_id: TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: crate::event::EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + admission: Admission::Exempt(ExemptReason::Read), + }; + + let mut rx = shared.tracker.register(request_id); + let dispatched = match shared.dispatcher.lock() { + Ok(mut dispatcher) => dispatcher.dispatch(request), + Err(poisoned) => poisoned.into_inner().dispatch(request), + }; + if let Err(error) = dispatched { + // No response will ever arrive for a refused request. + shared.tracker.cancel(&request_id); + return Err(error); + } + let collected = collect_under_deadline( + &mut rx, + DeadlineCollect { + request_id, + deadline, + max_result_bytes: shared.tuning.network.max_query_result_bytes as usize, + context: "internal local read", + }, + ) + .await; + if collected.is_err() { + shared.tracker.cancel(&request_id); + } + collected +} diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index 6260f3edd..1f08b6354 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -8,6 +8,7 @@ mod dispatch; mod durability_barrier; mod durable_write; mod error_status; +mod local_read; mod minted; mod submit_write; mod types; @@ -32,6 +33,7 @@ pub(crate) use durable_write::{ dispatch_durable_autocommit_write, }; pub(crate) use error_status::reject_data_plane_error; +pub(crate) use local_read::{LocalRead, dispatch_local_read}; pub(crate) use minted::{ Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, }; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index 5c9bbf659..396f5a2bd 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -42,6 +42,16 @@ pub(crate) async fn submit_write( database_id, vshard_id, }; + // On a single node no lease exists, so a write to a permission-tree source + // is acknowledged only once the local permission cache holds it. In a + // cluster the lease barrier on the Raft proposal path covers it instead. + let binds_authorization = shared.authorization_fence.timing().is_none() + && plan.named_collections().iter().any(|collection| { + shared + .authorization_fence + .sources() + .is_source_collection(collection) + }); // Records the caller appended for this write, under their outcome-floor // window. Every path below closes the window. let caller_minted = durability.take_minted(); @@ -206,7 +216,7 @@ pub(crate) async fn submit_write( // Collect response(s), classify the outcome, and run the post-apply steps // a successful write still owes. let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; - collect_classify_and_finish( + let outcome = collect_classify_and_finish( shared, max_result_bytes, ResponsePhaseInput { @@ -229,5 +239,14 @@ pub(crate) async fn submit_write( minted, }, ) - .await + .await?; + if binds_authorization { + crate::control::security::auth_lease::await_local_coverage( + shared, + std::time::Instant::now() + + std::time::Duration::from_secs(shared.tuning.network.default_deadline_secs), + ) + .await?; + } + Ok(outcome) } diff --git a/nodedb/src/control/server/http/routes/health.rs b/nodedb/src/control/server/http/routes/health.rs index 811282d4d..abafab8b5 100644 --- a/nodedb/src/control/server/http/routes/health.rs +++ b/nodedb/src/control/server/http/routes/health.rs @@ -168,6 +168,15 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { // A held window keeps the outcome floor below it by design, so it never // degrades readiness. The count shows how many restart replay will reach. body["held_windows"] = json!(state.shared.outcome_floor.held_windows()); + // A node that lost its authorization lease refuses permission-checked + // statements until it renews, and serves everything else. Checked once + // the startup gate is green: boot holds the gateway until the first lease. + if status == StatusCode::OK + && let Some(body) = lease_invalid_body(&state) + { + return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)); + } + body["authorization_lease"] = json!(lease_label(&state)); // Checked only once the startup gate is otherwise green, so a node still // advancing through phases keeps reporting the phase it is stuck in. if status == StatusCode::OK @@ -183,6 +192,35 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { (status, axum::Json(body)) } +/// The lease state `/healthz` reports: `valid`, `invalid`, or `not_required` +/// on a single node without a cluster. +fn lease_label(state: &AppState) -> &'static str { + use crate::control::security::auth_lease::{LeaseStatus, lease_status}; + match lease_status(&state.shared.authorization_fence, std::time::Instant::now()) { + LeaseStatus::NotRequired => "not_required", + LeaseStatus::Valid { .. } => "valid", + LeaseStatus::Invalid { .. } => "invalid", + } +} + +/// The degraded body for a node that holds no valid authorization lease, or +/// `None` when it holds one or needs none. +fn lease_invalid_body(state: &AppState) -> Option { + use crate::control::security::auth_lease::{LeaseStatus, lease_status}; + match lease_status(&state.shared.authorization_fence, std::time::Instant::now()) { + LeaseStatus::NotRequired | LeaseStatus::Valid { .. } => None, + LeaseStatus::Invalid { expired_for } => Some(json!({ + "status": "degraded", + "reason": "authorization_lease_invalid", + "detail": "this node holds no valid authorization lease; it refuses \ + permission-checked statements until it renews", + "node_id": state.shared.node_id, + "lease_expired_ms_ago": expired_for + .map(|expired| u64::try_from(expired.as_millis()).unwrap_or(u64::MAX)), + })), + } +} + /// The degraded body for an outcome floor held past `bound` by a window that /// is not held, or `None` when no such window exists. fn outcome_floor_stuck_body( @@ -405,6 +443,34 @@ mod tests { assert_eq!(body["step"], "flush"); } + #[tokio::test] + async fn a_node_without_a_valid_lease_reports_degraded_with_the_reason() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = app_state(&dir); + assert!( + lease_invalid_body(&state).is_none(), + "a node without lease timing needs no lease" + ); + let timing = crate::control::security::auth_lease::LeaseTiming::from_raft( + std::time::Duration::from_millis(1000), + std::time::Duration::from_millis(100), + ) + .expect("timing"); + assert!(state.shared.authorization_fence.install_timing(timing)); + + let body = lease_invalid_body(&state).expect("no lease was granted"); + assert_eq!(body["status"], "degraded"); + assert_eq!(body["reason"], "authorization_lease_invalid"); + assert!(body["lease_expired_ms_ago"].is_null()); + + state + .shared + .authorization_fence + .holder() + .install(std::time::Instant::now() + std::time::Duration::from_secs(60)); + assert!(lease_invalid_body(&state).is_none()); + } + #[tokio::test] async fn healthz_without_a_calvin_halt_names_no_calvin_halt() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/nodedb/src/control/server/http/routes/metrics.rs b/nodedb/src/control/server/http/routes/metrics.rs index 8ee8d5c61..b80157944 100644 --- a/nodedb/src/control/server/http/routes/metrics.rs +++ b/nodedb/src/control/server/http/routes/metrics.rs @@ -274,6 +274,13 @@ pub async fn metrics( // Auth observability: method-specific counters, duration histograms, anomaly detection. output.push_str(&state.shared.auth_metrics.to_prometheus()); + // Authorization lease validity. A node without a valid lease refuses + // permission-checked statements. + crate::control::security::auth_lease::status::render_prometheus( + &state.shared.authorization_fence, + &mut output, + ); + // Metering capacity: dropped-entry counters, so a refused (i.e. never // billed) usage record is observable without reading server logs. crate::control::security::metering::metrics::render_prometheus( diff --git a/nodedb/src/control/server/native/dispatch/sql_admin.rs b/nodedb/src/control/server/native/dispatch/sql_admin.rs index f290c8c13..731bbf6b9 100644 --- a/nodedb/src/control/server/native/dispatch/sql_admin.rs +++ b/nodedb/src/control/server/native/dispatch/sql_admin.rs @@ -101,7 +101,13 @@ pub(super) async fn handle_explain(ctx: &DispatchCtx<'_>, seq: u64, sql: &str) - }; } - let perm_cache = ctx.state.permission_cache.read().await; + let perm_cache = + match crate::control::security::auth_fence::permission_view(ctx.state, ctx.tenant_id()) + .await + { + Ok(view) => view, + Err(e) => return error_to_native(seq, &e), + }; let sec = crate::control::planner::context::PlanSecurityContext { identity: ctx.identity, auth: ctx.auth_context(), diff --git a/nodedb/src/control/server/pgwire/handler/cursor_query.rs b/nodedb/src/control/server/pgwire/handler/cursor_query.rs index 7d0310141..ce4ed5225 100644 --- a/nodedb/src/control/server/pgwire/handler/cursor_query.rs +++ b/nodedb/src/control/server/pgwire/handler/cursor_query.rs @@ -57,7 +57,10 @@ impl NodeDbPgHandler { // rejected cursor declaration consumes no descriptor lease. The scope // remains live while every cursor-materialization task is dispatched. let (tasks, _lease_scope) = retry_on_schema_change(move || async move { - let perm_cache = self.state.permission_cache.read().await; + let perm_cache = + crate::control::security::auth_fence::permission_view(&self.state, tenant_id) + .await + .map_err(StatementSetupError::from)?; let sec = crate::control::planner::context::PlanSecurityContext { identity, auth: auth_ctx, diff --git a/nodedb/src/control/server/pgwire/handler/prepared/parser.rs b/nodedb/src/control/server/pgwire/handler/prepared/parser.rs index 11c044fba..e53bb7318 100644 --- a/nodedb/src/control/server/pgwire/handler/prepared/parser.rs +++ b/nodedb/src/control/server/pgwire/handler/prepared/parser.rs @@ -209,7 +209,12 @@ impl NodeDbQueryParser { self.state.auth_stores(), database_id, ); - let permission_cache = self.state.permission_cache.read().await; + // Parse plans against the same authorization state as every other + // planning path. A refusal is an error, not "not plannable". + let permission_cache = + crate::control::security::auth_fence::permission_view(&self.state, identity.tenant_id) + .await + .map_err(|e| crate::control::server::pgwire::types::error_map::error_to_pg(&e))?; let security = crate::control::planner::context::PlanSecurityContext { identity, auth: scope.auth(), diff --git a/nodedb/src/control/server/pgwire/handler/routing/planning.rs b/nodedb/src/control/server/pgwire/handler/routing/planning.rs index 769a92c39..c2c6584fc 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/planning.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/planning.rs @@ -205,10 +205,15 @@ impl NodeDbPgHandler { // bumped on every mutation; a cache hit re-validates the stamped // versions against these live values so a revoked grant or dropped // policy evicts the entry instead of replaying a frozen filter. - let current_permission_tree_version = { - let perm_cache = self.state.permission_cache.read().await; - perm_cache.tenant_version(tenant_id.as_u64()) - }; + // + // The checked read refuses unless this node holds every + // authorization change acknowledged before this statement. The plain + // reads below see that state or newer. + let current_permission_tree_version = + crate::control::security::auth_fence::permission_view(&self.state, tenant_id) + .await + .map_err(StatementSetupError::from)? + .tenant_version(tenant_id.as_u64()); let current_rls_version = self.state.rls.tenant_version(tenant_id.as_u64()); let cached_tasks = if bypass_cache { diff --git a/nodedb/src/control/server/pgwire/handler/session_explain.rs b/nodedb/src/control/server/pgwire/handler/session_explain.rs index 842900300..0eee63bdf 100644 --- a/nodedb/src/control/server/pgwire/handler/session_explain.rs +++ b/nodedb/src/control/server/pgwire/handler/session_explain.rs @@ -75,7 +75,10 @@ impl NodeDbPgHandler { self.state.auth_stores(), database_id, ); - let perm_cache = self.state.permission_cache.read().await; + let perm_cache = + crate::control::security::auth_fence::permission_view(&self.state, tenant_id) + .await + .map_err(|e| crate::control::server::pgwire::types::error_map::error_to_pg(&e))?; let sec = crate::control::planner::context::PlanSecurityContext { identity, auth: scope.auth(), diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 0a9784b26..055211fd5 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -132,6 +132,10 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::DeadlineExceeded { .. } => { ("ERROR", sqlstate::QUERY_CANCELED, err.to_string()) } + // Nothing ran, and a retry plans against caught-up state. + crate::Error::AuthorizationStateBehind { .. } => { + ("ERROR", sqlstate::STALE_READ_NOT_LEADER, err.to_string()) + } crate::Error::ConflictRetry { .. } => { ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index 68c717ee9..85d9a919f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -150,7 +150,10 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a // un-injected copies. let (mut tasks, output_schema, versions) = { let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); - let permission_cache = state.permission_cache.read().await; + let permission_cache = + crate::control::security::auth_fence::permission_view(state, tenant_id) + .await + .map_err(|error| DdlError::from_error(&error))?; let sec = PlanSecurityContext { identity, auth: scope.auth(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs index 3710d381b..c75e7183d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs @@ -96,6 +96,8 @@ pub async fn set_permission_tree( persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) .map_err(|e| err("XX000", e.to_string()))?; + let sources = [collection.clone(), def.permission_table.clone()]; + // Update in-memory cache. state .permission_cache @@ -103,6 +105,13 @@ pub async fn set_permission_tree( .await .register_tree_def(tenant_id.as_u64(), &collection, def); + // Rows already in the sources are grants and edges too. Hold the + // acknowledgement until every lease holder covers each source group + // through its current commit. + source_group_barrier(state, &sources) + .await + .map_err(|e| DdlError::from_error(&e))?; + // Audit. state .audit @@ -118,6 +127,32 @@ pub async fn set_permission_tree( Ok(status("ALTER COLLECTION")) } +/// Barrier on every Raft group homing one of `sources`, at a read index +/// taken now. A single node has no groups; its planning reloads the cache. +async fn source_group_barrier(state: &SharedState, sources: &[String]) -> crate::Result<()> { + let Some(timing) = state.authorization_fence.timing() else { + return Ok(()); + }; + let mut targets: Vec = Vec::new(); + for source in sources { + let vshard = + crate::types::VShardId::from_collection_in_database(DatabaseId::DEFAULT, source); + let group_id = + crate::control::security::auth_fence::cluster::group_of_vshard(state, vshard.as_u32())?; + if targets.iter().any(|target| target.group_id == group_id) { + continue; + } + let through = crate::control::security::auth_fence::cluster::confirmed_read_index( + state, + group_id, + timing.lease, + ) + .await?; + targets.push(nodedb_cluster::GroupCoverage { group_id, through }); + } + crate::control::security::auth_lease::authorization_barrier(state, targets).await +} + /// ALTER COLLECTION DROP PERMISSION_TREE pub async fn drop_permission_tree( state: &SharedState, diff --git a/nodedb/src/control/server/shared/ddl/neutral/planning.rs b/nodedb/src/control/server/shared/ddl/neutral/planning.rs index acf185a58..039a7aeee 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/planning.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/planning.rs @@ -36,7 +36,10 @@ pub async fn plan_authorized_sql( ) -> Result<(Vec, OutputSchema, QueryLeaseScope), DdlError> { // Internal DDL scans still plan in the caller-selected database context. let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); - let permission_cache = state.permission_cache.read().await; + let permission_cache = + crate::control::security::auth_fence::permission_view(state, identity.tenant_id) + .await + .map_err(|error| DdlError::from_error(&error))?; let sec = PlanSecurityContext { identity, auth: scope.auth(), diff --git a/nodedb/src/control/server/shared/ddl/result.rs b/nodedb/src/control/server/shared/ddl/result.rs index 09a85d0c1..460a9512a 100644 --- a/nodedb/src/control/server/shared/ddl/result.rs +++ b/nodedb/src/control/server/shared/ddl/result.rs @@ -77,6 +77,15 @@ impl DdlError { } } + /// Build a `DdlError` from an internal error: the SQLSTATE and message + /// the pgwire table gives it, with its classified code and details. + pub fn from_error(error: &crate::Error) -> Self { + let (_, sqlstate, message) = + crate::control::server::pgwire::types::error_to_sqlstate(error); + let public = crate::error_classify::classify(error); + Self::from_public(sqlstate, message, &public) + } + /// Build a `DdlError` with an explicit code, bypassing derivation. /// Used by the named constructors below for ambiguous SQLSTATEs. fn with_code(sqlstate: &'static str, code: ErrorCode, message: impl Into) -> Self { diff --git a/nodedb/src/control/server/shared/plan_admission.rs b/nodedb/src/control/server/shared/plan_admission.rs index a64306050..1c4112fe7 100644 --- a/nodedb/src/control/server/shared/plan_admission.rs +++ b/nodedb/src/control/server/shared/plan_admission.rs @@ -95,7 +95,8 @@ async fn plan_authorize_and_admit_once( // Re-read per attempt: a retry must plan against the catalog and permission // state as they are NOW, not as they were when the drained attempt started. let (mut tasks, output_schema, versions) = { - let permission_cache = state.permission_cache.read().await; + let permission_cache = + crate::control::security::auth_fence::permission_view(state, tenant_id).await?; let security = PlanSecurityContext { identity, auth: auth_ctx, diff --git a/nodedb/src/control/server/shared/session/ddl_authorization.rs b/nodedb/src/control/server/shared/session/ddl_authorization.rs new file mode 100644 index 000000000..d21a5454b --- /dev/null +++ b/nodedb/src/control/server/shared/session/ddl_authorization.rs @@ -0,0 +1,45 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Hold a transactional DDL commit until its authorization changes bind +//! every node. + +use nodedb_cluster::{GroupCoverage, METADATA_GROUP_ID, MetadataEntry, PendingDdlObject}; + +use crate::control::catalog_entry; +use crate::control::state::SharedState; + +/// Whether any of `objects` changes authorization state. +pub(super) fn objects_bear_authorization(objects: &[PendingDdlObject]) -> crate::Result { + for object in objects { + let entry = match object { + PendingDdlObject::Create { entry } | PendingDdlObject::Alter { entry, .. } => entry, + }; + let payload = match entry.as_ref() { + MetadataEntry::CatalogDdl { payload } + | MetadataEntry::CatalogDdlAudited { payload, .. } => payload, + other => { + return Err(crate::Error::Internal { + detail: format!( + "transactional DDL: pending object wire shape is not CatalogDdl: {other:?}" + ), + }); + } + }; + if catalog_entry::decode(payload)?.bears_authorization() { + return Ok(true); + } + } + Ok(false) +} + +/// Run the authorization barrier on the metadata entry committed at +/// `log_index`. +pub(super) fn barrier_at(state: &SharedState, log_index: u64) -> crate::Result<()> { + crate::control::security::auth_lease::block_on_barrier( + state, + vec![GroupCoverage { + group_id: METADATA_GROUP_ID, + through: log_index, + }], + ) +} diff --git a/nodedb/src/control/server/shared/session/ddl_flush.rs b/nodedb/src/control/server/shared/session/ddl_flush.rs index 5eae637e5..3a022ad3c 100644 --- a/nodedb/src/control/server/shared/session/ddl_flush.rs +++ b/nodedb/src/control/server/shared/session/ddl_flush.rs @@ -343,13 +343,18 @@ pub(super) fn finalize_pending( log_index = handle.log_index(), "finalizing pending DDL" ); - propose_and_await( + let bears_authorization = + super::ddl_authorization::objects_bear_authorization(handle.objects())?; + let log_index = propose_and_await( state, raft_handle.as_ref(), &MetadataEntry::DdlPendingFinalize { token: handle.token(), }, )?; + if bears_authorization { + super::ddl_authorization::barrier_at(state, log_index)?; + } Ok(()) } @@ -379,7 +384,12 @@ pub(super) fn compensate_finalized( }; entries.push(MetadataEntry::CatalogDdl { payload }); } - propose_and_await(state, handle.as_ref(), &MetadataEntry::Batch { entries })?; + let log_index = propose_and_await(state, handle.as_ref(), &MetadataEntry::Batch { entries })?; + // The compensation restores prior authorization state, which binds every + // node like any other authorization change. + if super::ddl_authorization::objects_bear_authorization(objects)? { + super::ddl_authorization::barrier_at(state, log_index)?; + } Ok(()) } diff --git a/nodedb/src/control/server/shared/session/hot_key.rs b/nodedb/src/control/server/shared/session/hot_key.rs index bb0e3265f..0b2d9d146 100644 --- a/nodedb/src/control/server/shared/session/hot_key.rs +++ b/nodedb/src/control/server/shared/session/hot_key.rs @@ -19,6 +19,7 @@ use super::read_set::{ReadSetEntry, lock_key_of_read}; pub(super) fn record_read_set_aborts(state: &SharedState, read_set: &[ReadSetEntry]) { let now = std::time::Instant::now(); let mut table = state + .calvin .hot_key_table .lock() .unwrap_or_else(|p| p.into_inner()); diff --git a/nodedb/src/control/server/shared/session/lifecycle.rs b/nodedb/src/control/server/shared/session/lifecycle.rs index 8f7889897..335227e7f 100644 --- a/nodedb/src/control/server/shared/session/lifecycle.rs +++ b/nodedb/src/control/server/shared/session/lifecycle.rs @@ -31,7 +31,8 @@ pub fn run_begin( // Last globally-applied Calvin epoch as the cross-shard snapshot anchor. // 0 in single-node / no-Calvin deployments (the atomic is never advanced). let snapshot_epoch = state - .last_applied_calvin_epoch + .calvin + .last_applied_epoch .load(std::sync::atomic::Ordering::Acquire); ddl_buffer::activate(); sessions @@ -141,7 +142,7 @@ mod tests { use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; /// `run_begin` anchors the session's cross-shard snapshot to the last - /// globally-applied Calvin epoch from `SharedState::last_applied_calvin_epoch`. + /// globally-applied Calvin epoch from `CalvinLocalState::last_applied_epoch`. #[tokio::test] async fn run_begin_anchors_snapshot_epoch() { use std::sync::atomic::Ordering; @@ -161,14 +162,14 @@ mod tests { store.ensure_session(addr); // Seed the applied epoch to 7 and BEGIN — the session anchors to 7. - state.last_applied_calvin_epoch.store(7, Ordering::Release); + state.calvin.last_applied_epoch.store(7, Ordering::Release); run_begin(&store, SessionId::from(&addr), &state).unwrap(); assert_eq!(store.snapshot_epoch(addr), Some(7)); store.commit(addr).unwrap(); assert_eq!(store.snapshot_epoch(addr), None); // Unset (single-node / no-Calvin): BEGIN anchors to 0. - state.last_applied_calvin_epoch.store(0, Ordering::Release); + state.calvin.last_applied_epoch.store(0, Ordering::Release); run_begin(&store, SessionId::from(&addr), &state).unwrap(); assert_eq!(store.snapshot_epoch(addr), Some(0)); } diff --git a/nodedb/src/control/server/shared/session/mod.rs b/nodedb/src/control/server/shared/session/mod.rs index 7f59ca0ed..eed33f7b6 100644 --- a/nodedb/src/control/server/shared/session/mod.rs +++ b/nodedb/src/control/server/shared/session/mod.rs @@ -10,6 +10,7 @@ pub mod connection; pub mod cross_shard_mode; mod cursor; pub mod cursor_spill; +mod ddl_authorization; pub mod ddl_buffer; pub mod ddl_effect; mod ddl_flush; diff --git a/nodedb/src/control/server/shared/session/read_set.rs b/nodedb/src/control/server/shared/session/read_set.rs index cf80fc7fa..b119d5d7d 100644 --- a/nodedb/src/control/server/shared/session/read_set.rs +++ b/nodedb/src/control/server/shared/session/read_set.rs @@ -254,6 +254,7 @@ pub async fn record_read_set( let now = std::time::Instant::now(); let hot = { let table = state + .calvin .hot_key_table .lock() .unwrap_or_else(|p| p.into_inner()); diff --git a/nodedb/src/control/server/shared/session/state.rs b/nodedb/src/control/server/shared/session/state.rs index a7714f35f..71571c058 100644 --- a/nodedb/src/control/server/shared/session/state.rs +++ b/nodedb/src/control/server/shared/session/state.rs @@ -153,7 +153,7 @@ pub struct ConnSession { /// Concurrent writes after this point are invisible to the transaction. pub tx_snapshot_lsn: Option, /// Snapshot epoch captured at BEGIN: the last globally-applied Calvin epoch, - /// read from `SharedState::last_applied_calvin_epoch`. The cross-shard-valid + /// read from `CalvinLocalState::last_applied_epoch`. The cross-shard-valid /// version anchor for the transaction (0 in single-node / no-Calvin). `None` /// outside a transaction block. pub tx_snapshot_epoch: Option, diff --git a/nodedb/src/control/server/shared/write_admission/gate.rs b/nodedb/src/control/server/shared/write_admission/gate.rs index bfc1eb92e..161751787 100644 --- a/nodedb/src/control/server/shared/write_admission/gate.rs +++ b/nodedb/src/control/server/shared/write_admission/gate.rs @@ -22,12 +22,12 @@ //! Calvin-scheduled apply that already holds its locks. //! //! The fence holds because the fast path and the scheduler share the SAME -//! `Arc>` (via [`SharedState::calvin_lock_managers`]): a +//! `Arc>` (via [`CalvinLocalState::lock_managers`]): a //! commit's lock validation calls `acquire` on the same key, is `Blocked`, and //! waits; whoever takes the OS mutex first wins, with no time-of-check / //! time-of-use gap. //! -//! [`SharedState::calvin_lock_managers`]: crate::control::state::SharedState::calvin_lock_managers +//! [`CalvinLocalState::lock_managers`]: crate::control::state::CalvinLocalState::lock_managers use std::collections::BTreeSet; use std::sync::atomic::{AtomicU64, Ordering}; @@ -213,7 +213,8 @@ pub fn admit(shared: &SharedState, target: &WriteTarget<'_>) -> WriteAdmission { // no single static point key here, so it stays unordered on the fast path // (widening that coverage is a later unit). let Some(lock_manager) = shared - .calvin_lock_managers + .calvin + .lock_managers .lock() .unwrap_or_else(|p| p.into_inner()) .get(&vshard.as_u32()) @@ -242,7 +243,10 @@ pub fn admit(shared: &SharedState, target: &WriteTarget<'_>) -> WriteAdmission { // could promote to an unowned (never-released) lock. let txn = TxnId::new( TxnId::AUTOCOMMIT_EPOCH, - shared.autocommit_lock_seq.fetch_add(1, Ordering::Relaxed), + shared + .calvin + .autocommit_lock_seq + .fetch_add(1, Ordering::Relaxed), ); let acquired = { let mut lm = lock_manager.lock().unwrap_or_else(|p| p.into_inner()); @@ -255,7 +259,8 @@ pub fn admit(shared: &SharedState, target: &WriteTarget<'_>) -> WriteAdmission { // lock manager without a promotion sender should not happen, since both // are inserted together per vShard. let promotion_sender = shared - .calvin_promotion_senders + .calvin + .promotion_senders .lock() .unwrap_or_else(|p| p.into_inner()) .get(&vshard.as_u32()) diff --git a/nodedb/src/control/state/calvin_apply.rs b/nodedb/src/control/state/calvin_apply.rs index 413e39533..34f1a358f 100644 --- a/nodedb/src/control/state/calvin_apply.rs +++ b/nodedb/src/control/state/calvin_apply.rs @@ -5,7 +5,7 @@ use crate::bridge::envelope::Response; /// The applied Data-Plane result for one completed Calvin transaction, carried -/// via [`SharedState::calvin_apply_results`] from the per-vShard scheduler to +/// via [`CalvinLocalState::apply_results`] from the per-vShard scheduler to /// the coordinator's completion path. /// /// A cross-shard COMMIT may legitimately have MANY primary-write participants — @@ -21,7 +21,7 @@ use crate::bridge::envelope::Response; /// unsupported. The coordinator then fails the statement loudly rather than /// returning one shard's partial rows. /// -/// [`SharedState::calvin_apply_results`]: crate::control::state::SharedState::calvin_apply_results +/// [`CalvinLocalState::apply_results`]: crate::control::state::CalvinLocalState::apply_results pub enum CalvinApplyResult { /// A participant's applied response. `has_returning` is true iff this /// participant's slice carried RETURNING rows (a plain affected-count diff --git a/nodedb/src/control/state/calvin_local.rs b/nodedb/src/control/state/calvin_local.rs new file mode 100644 index 000000000..2e442b91e --- /dev/null +++ b/nodedb/src/control/state/calvin_local.rs @@ -0,0 +1,96 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! This node's in-process Calvin state, shared between the per-vShard +//! schedulers and the Control-Plane paths that read or fence against them. + +use std::collections::{BTreeMap, HashMap}; +use std::sync::atomic::{AtomicU32, AtomicU64}; +use std::sync::{Arc, Mutex}; + +use crate::control::cluster::calvin::scheduler::lock::HotKeyTable; +use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; + +use super::calvin_apply::CalvinApplyResult; +use super::calvin_counters::CalvinCounters; + +/// Per-vShard promotion senders, keyed by vShard id. +pub type PromotionSenders = BTreeMap>>; + +/// In-process Calvin state of this node. Empty in single-node deployments +/// that run no Calvin scheduler. +pub struct CalvinLocalState { + /// Last globally-applied Calvin epoch, advanced by the per-vShard + /// deterministic schedulers as they apply epochs. Read at `BEGIN` to anchor + /// a session's cross-shard snapshot version (`tx_snapshot_epoch`). `Arc` so + /// schedulers (holding `Arc`) advance the same counter the + /// session reads. 0 in single-node / no-Calvin deployments. + pub last_applied_epoch: Arc, + /// Node-global Calvin observability counters (write versions recorded, + /// read-set validation failures, commits flushed/dropped). + pub counters: CalvinCounters, + /// Local, in-process sidecar carrying the applied Data-Plane + /// [`Response`](crate::bridge::envelope::Response) (affected-count and any + /// RETURNING rows) of a completed Calvin transaction, keyed by its + /// sequencer-assigned `TxnId`. + /// + /// RETURNING rows are a QUERY RESULT, not replicated state, so they MUST NOT + /// ride the sequencer Raft log. The per-vShard scheduler deposits the applied + /// `Response` here BEFORE proposing the replicated `CompletionAck`; the + /// coordinator's completion path (static: `submit_and_await_calvin`; + /// dependent: `dispatch_dependent_edge_recon`) drains it once completion + /// fires. Every primary-write participant deposits (not RETURNING-only) — a + /// multi-collection cross-shard COMMIT can have several plain-write + /// participants, and those coalesce without conflict. Cross-node, the rows + /// travel via the non-Raft routed-submit RPC response instead. + /// + /// Value is [`CalvinApplyResult`]: `Single` for a deposited (possibly + /// coalesced) participant, or `Conflict` only when two RETURNING-bearing + /// participants deposit for the same `TxnId` — a cross-shard RETURNING + /// union, drained as a loud error, never a silent partial. + pub apply_results: Arc>>, + /// Per-vShard deterministic lock managers, lifted out of each Calvin + /// `Scheduler` so the Control-Plane write-admission gate shares the SAME + /// `Arc>` the scheduler holds — a fast-path point write and + /// a Calvin txn's lock validation contend on one OS mutex, no TOCTOU gap. + /// Keyed by vShard id; empty in single-node / no-Calvin deployments. + pub lock_managers: Arc>>>>, + /// Global hot-key detector for Calvin read reservations (CP-local heuristic). + pub hot_key_table: Arc>, + /// Per-vShard promotion channels, parallel to `lock_managers`. When a + /// fast-path write guard releases an uncontended key on drop, `LockManager` + /// may promote a scheduler txn queued behind it; the guard (Control-Plane, + /// not in the scheduler task) forwards the promoted `TxnId`s here for the + /// scheduler to dispatch. Unbounded (low-volume, sent from a non-blocking + /// `Drop`). Keyed by vShard id; empty in single-node / no-Calvin deployments. + pub promotion_senders: Arc>, + /// Monotonic `position` source for autocommit fast-path lock holders. Paired + /// with [`TxnId::AUTOCOMMIT_EPOCH`] to mint holder identities that never + /// collide with a real Calvin `(epoch, position)` schedule position. + pub autocommit_lock_seq: AtomicU32, +} + +impl CalvinLocalState { + /// State of a node whose schedulers have applied nothing yet. + pub fn new() -> Self { + Self { + last_applied_epoch: Arc::new(AtomicU64::new(0)), + counters: CalvinCounters { + write_versions_recorded: Arc::new(AtomicU64::new(0)), + read_set_validation_failures: Arc::new(AtomicU64::new(0)), + commits_flushed: Arc::new(AtomicU64::new(0)), + commits_dropped: Arc::new(AtomicU64::new(0)), + }, + apply_results: Arc::new(Mutex::new(HashMap::new())), + lock_managers: Arc::new(Mutex::new(BTreeMap::new())), + hot_key_table: Arc::new(Mutex::new(HotKeyTable::new())), + promotion_senders: Arc::new(Mutex::new(BTreeMap::new())), + autocommit_lock_seq: AtomicU32::new(0), + } + } +} + +impl Default for CalvinLocalState { + fn default() -> Self { + Self::new() + } +} diff --git a/nodedb/src/control/state/fields.rs b/nodedb/src/control/state/fields.rs index d40afb94b..db4535c6b 100644 --- a/nodedb/src/control/state/fields.rs +++ b/nodedb/src/control/state/fields.rs @@ -16,7 +16,6 @@ use crate::control::security::rls::RlsPolicyStore; use crate::control::security::role::RoleStore; use crate::control::security::tenant::TenantIsolation; use crate::control::server::sync::dlq::SyncDlq; -use crate::control::state::calvin_counters::CalvinCounters; use crate::wal::WalManager; /// Atomically installed Raft proposal handles. @@ -405,6 +404,9 @@ pub struct SharedState { /// them. Set by the Event Plane, which knows how many consumers it spawned; /// absent on a node that runs none. pub action_requeue: OnceLock>, + /// Durable ledgers of the Event Plane's non-idempotent sinks. Set by the + /// Event Plane before its consumers start. + pub sink_ledgers: OnceLock>, /// Test-only drop guard: owns the auto-cleaning temp directory the test /// constructor roots its CDC-offset / job-history / MV-persistence stores /// under, so they're removed on drop instead of leaking `/tmp/nodedb-test-*` @@ -424,77 +426,9 @@ pub struct SharedState { crate::control::server::shared::ddl::neutral::maintenance::auto_analyze::DmlCounter, /// Highest WAL LSN confirmed delivered to Data Plane for timeseries catch-up. pub wal_catchup_lsn: AtomicU64, - /// Last globally-applied Calvin epoch, advanced by the per-vShard - /// deterministic schedulers as they apply epochs. Read at `BEGIN` to anchor - /// a session's cross-shard snapshot version (`tx_snapshot_epoch`). `Arc` so - /// schedulers (holding `Arc`) advance the same counter the - /// session reads. 0 in single-node / no-Calvin deployments. - pub last_applied_calvin_epoch: Arc, - /// Node-global Calvin observability counters (write versions recorded, - /// read-set validation failures, commits flushed/dropped). - pub calvin_counters: CalvinCounters, - /// Local, in-process sidecar carrying the applied Data-Plane [`Response`] - /// (affected-count and any RETURNING rows) of a completed Calvin - /// transaction, keyed by its sequencer-assigned `TxnId`. - /// - /// RETURNING rows are a QUERY RESULT, not replicated state, so they MUST NOT - /// ride the sequencer Raft log. The per-vShard scheduler deposits the applied - /// `Response` here BEFORE proposing the replicated `CompletionAck`; the - /// coordinator's completion path (static: `submit_and_await_calvin`; - /// dependent: `dispatch_dependent_edge_recon`) drains it once completion - /// fires. Every primary-write participant deposits (not RETURNING-only) — a - /// multi-collection cross-shard COMMIT can have several plain-write - /// participants, and those coalesce without conflict. Cross-node, the rows - /// travel via the non-Raft routed-submit RPC response instead. - /// - /// Value is [`CalvinApplyResult`](super::CalvinApplyResult): `Single` for a - /// deposited (possibly coalesced) participant, or `Conflict` only when two - /// RETURNING-bearing participants deposit for the same `TxnId` — a - /// cross-shard RETURNING union, drained as a loud error, never a silent - /// partial. [`Response`](crate::bridge::envelope::Response) is Control-Plane - /// `Send + Sync`; it never touches Raft. - pub calvin_apply_results: Arc< - Mutex>, - >, - /// Per-vShard deterministic lock managers, lifted out of each Calvin - /// `Scheduler` so the Control-Plane write-admission gate shares the SAME - /// `Arc>` the scheduler holds — a fast-path point write and - /// a Calvin txn's lock validation contend on one OS mutex, no TOCTOU gap. - /// Keyed by vShard id; empty in single-node / no-Calvin deployments. - pub calvin_lock_managers: Arc< - Mutex< - std::collections::BTreeMap< - u32, - Arc>, - >, - >, - >, - /// Global hot-key detector for Calvin read reservations (CP-local heuristic). - pub hot_key_table: std::sync::Arc< - std::sync::Mutex, - >, - /// Per-vShard promotion channels, parallel to `calvin_lock_managers`. When a - /// fast-path write guard releases an uncontended key on drop, `LockManager` - /// may promote a scheduler txn queued behind it; the guard (Control-Plane, - /// not in the scheduler task) forwards the promoted `TxnId`s here for the - /// scheduler to dispatch. Unbounded (low-volume, sent from a non-blocking - /// `Drop`). Keyed by vShard id; empty in single-node / no-Calvin deployments. - pub calvin_promotion_senders: Arc< - Mutex< - std::collections::BTreeMap< - u32, - tokio::sync::mpsc::UnboundedSender< - Vec, - >, - >, - >, - >, - /// Monotonic `position` source for autocommit fast-path lock holders. Paired - /// with [`TxnId::AUTOCOMMIT_EPOCH`] to mint holder identities that never - /// collide with a real Calvin `(epoch, position)` schedule position. - /// - /// [`TxnId::AUTOCOMMIT_EPOCH`]: crate::control::cluster::calvin::scheduler::lock_manager::TxnId::AUTOCOMMIT_EPOCH - pub autocommit_lock_seq: std::sync::atomic::AtomicU32, + /// In-process Calvin state: applied epoch, counters, apply results, lock + /// managers and their promotion channels. + pub calvin: super::calvin_local::CalvinLocalState, /// Single-node per-key write-ordering lock. When NO Calvin scheduler is /// registered for a write's vShard there is no lock table to fence against, /// yet concurrent same-key autocommit writes must still serialize so @@ -509,6 +443,8 @@ pub struct SharedState { /// Permission tree cache: in-memory resource hierarchy + permission grants. pub permission_cache: Arc>, + /// Brings authorization state to date before a statement is planned. + pub authorization_fence: Arc, /// Gateway plan-cache invalidator; called after every DDL commit. None until `Gateway::new`. pub gateway_invalidator: std::sync::OnceLock>, diff --git a/nodedb/src/control/state/init.rs b/nodedb/src/control/state/init.rs index 5d7d49db6..5c39a5dae 100644 --- a/nodedb/src/control/state/init.rs +++ b/nodedb/src/control/state/init.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! SharedState constructors: new (test) and new_with_credentials (test+catalog). +//! The shared test constructor of `SharedState`. The public test +//! constructors built on it live in [`super::init_variants`]. use std::sync::atomic::AtomicU64; use std::sync::{Arc, Mutex}; @@ -33,134 +34,12 @@ impl SharedState { COUNTER.fetch_add(1, std::sync::atomic::Ordering::Relaxed) } - /// Create shared state with a pre-built credential store (for tests that need catalog). - /// - /// `is_cluster` is the static, deployment-time surrogate-registry mode - /// choice — same predicate as `SharedState::open`'s `is_cluster` - /// (whether this node's caller is about to wire it into a real Raft - /// cluster), not a property of the credential store. Almost every - /// caller is a single-process fixture and passes `false`; the cluster - /// test harness passes `true`. - pub fn new_with_credentials( + /// Construct the test state every constructor in [`super::init_variants`] + /// builds on. + pub(super) fn new_inner( dispatcher: Dispatcher, wal: Arc, - credentials: Arc, - is_cluster: bool, ) -> crate::Result> { - let wal_for_assigner = Arc::clone(&wal); - let mut state = Self::new_inner(dispatcher, wal)?; - if let Some(s) = Arc::get_mut(&mut state) { - // Rebuild the surrogate assigner against the supplied - // credential store. `new_inner` constructs the assigner - // from a fresh in-memory `CredentialStore` with its own - // in-memory catalog; the supplied store carries the durable - // catalog whose surrogate watermark this fixture must resume. - let registry = Arc::clone(&s.surrogate_registry); - // Seed the registry's high-watermark AND applied-reserve cursor - // from the catalog so restarts in a re-opened test fixture pick up - // where the previous session left off — and so cluster-mode - // metadata-log replay skips already-applied reservations rather - // than double-counting `G`. - let catalog = credentials.catalog(); - // The catalog-derived floor mirrors the production bootstrap: the - // singleton is flushed lazily, so the highest surrogate any live - // binding refers to is the value the allocator can never start - // below. Mode selection mirrors `init_prod/bootstrap.rs::run`: - // `is_cluster` (never seed-list length) picks `Cluster` vs - // `Local`, seeding the applied-reserve cursor only in the - // former. - if let Ok(hwm) = catalog.get_surrogate_hwm() - && let Ok(bound_floor) = catalog.max_bound_surrogate() - && let Ok(mut reg) = registry.write() - { - let floor = hwm.max(bound_floor.as_u32()); - *reg = if is_cluster { - let reserve_index = catalog.get_surrogate_reserve_index().unwrap_or(0); - crate::control::surrogate::SurrogateRegistry::from_persisted_cluster( - floor, - reserve_index, - ) - } else { - crate::control::surrogate::SurrogateRegistry::from_persisted_hwm(floor) - }; - } - let wal_appender: Arc = Arc::new( - crate::control::surrogate::WalSurrogateAppender::new(wal_for_assigner), - ); - s.surrogate_assigner = Arc::new(crate::control::surrogate::SurrogateAssigner::new( - Arc::clone(®istry), - Arc::clone(&credentials), - wal_appender, - )); - // Catalog-backed security stores, rebuilt for the same reason the - // surrogate watermark above is: this constructor's whole purpose - // is to resume a durable catalog, and a memory-only store here - // silently drops every auth-user status and scope grant the - // previous session persisted — so a restart fixture would report - // a clean slate rather than what was actually saved. - s.auth_users = - crate::control::security::jit::auth_user::AuthUserStore::open(catalog.clone())?; - s.scope_grants = - crate::control::security::scope::grant::ScopeGrantStore::open(catalog)?; - // Same reasoning as the grants above: a quota definition is a - // durable catalog object, and a memory-only manager here would - // report every cap as absent after a restart. - s.quota_manager = QuotaManager::open( - s.metering_config.max_tracked_quota_grantees, - credentials.catalog(), - )?; - s.credentials = credentials; - s.ep_topic_registry - .load_from_catalog(s.credentials.catalog())?; - crate::event::topic::hydrate_topic_buffers(s)?; - } - Ok(state) - } - - /// Create shared state with in-memory credential store (for tests). - pub fn new(dispatcher: Dispatcher, wal: Arc) -> crate::Result> { - Self::new_inner(dispatcher, wal) - } - - /// Create shared state whose risk scorer is built from `risk_config` - /// instead of the disabled default (for tests that exercise the risk - /// gate). Production wires the same configuration from `[auth.risk]`. - pub fn new_with_risk_config( - dispatcher: Dispatcher, - wal: Arc, - risk_config: crate::control::security::risk::RiskConfig, - ) -> crate::Result> { - let mut state = Self::new_inner(dispatcher, wal)?; - let s = Arc::get_mut(&mut state).ok_or_else(|| crate::Error::Internal { - detail: "shared state was already shared before the risk scorer could be installed" - .into(), - })?; - s.risk_scorer = crate::control::security::risk::RiskScorer::new(risk_config); - Ok(state) - } - - /// Create shared state whose TLS policy is built from `tls_policy_config` - /// instead of the disabled default (for tests that exercise transport - /// enforcement). Production wires the same configuration from - /// `[auth.tls_policy]`, through the same fallible parse: an unparseable - /// `min_tls_version` is an error here exactly as it is at startup. - pub fn new_with_tls_policy_config( - dispatcher: Dispatcher, - wal: Arc, - tls_policy_config: crate::control::security::tls_policy::TlsPolicyConfig, - ) -> crate::Result> { - let policy = - crate::control::security::tls_policy::TlsPolicy::from_config(&tls_policy_config)?; - let mut state = Self::new_inner(dispatcher, wal)?; - let s = Arc::get_mut(&mut state).ok_or_else(|| crate::Error::Internal { - detail: "shared state was already shared before the TLS policy could be installed" - .into(), - })?; - s.tls_policy = policy; - Ok(state) - } - - fn new_inner(dispatcher: Dispatcher, wal: Arc) -> crate::Result> { let shutdown = Arc::new(crate::control::shutdown::ShutdownWatch::new()); let loop_registry = Arc::new(crate::control::shutdown::LoopRegistry::new()); // Test helpers get a pre-fired gate so listeners start accepting @@ -183,6 +62,12 @@ impl SharedState { Arc::clone(&test_credentials), Arc::new(crate::control::surrogate::NoopWalAppender), )); + let permission_cache = crate::control::security::permission_tree::PermissionCache::new(); + let authorization_fence = Arc::new( + crate::control::security::auth_fence::AuthorizationFence::new( + permission_cache.sources(), + ), + ); let shared_audit = Arc::new(Mutex::new(AuditLog::new(10_000))); let test_session_registry = Arc::new(crate::control::security::sessions::SessionRegistry::new()); @@ -458,6 +343,7 @@ impl SharedState { data_dir: std::path::PathBuf::new(), trigger_dlq: std::sync::OnceLock::new(), action_requeue: std::sync::OnceLock::new(), + sink_ledgers: std::sync::OnceLock::new(), schema_version: crate::control::server::shared::session::plan_cache::SchemaVersion::new( ), materialized_sum_index: @@ -466,31 +352,17 @@ impl SharedState { dml_counter: crate::control::server::shared::ddl::neutral::maintenance::auto_analyze::DmlCounter::new(), wal_catchup_lsn: AtomicU64::new(0), - last_applied_calvin_epoch: Arc::new(AtomicU64::new(0)), - calvin_counters: crate::control::state::CalvinCounters { - write_versions_recorded: Arc::new(AtomicU64::new(0)), - read_set_validation_failures: Arc::new(AtomicU64::new(0)), - commits_flushed: Arc::new(AtomicU64::new(0)), - commits_dropped: Arc::new(AtomicU64::new(0)), - }, - calvin_apply_results: Arc::new(Mutex::new(std::collections::HashMap::new())), - calvin_lock_managers: Arc::new(Mutex::new(std::collections::BTreeMap::new())), - hot_key_table: Arc::new(Mutex::new( - crate::control::cluster::calvin::scheduler::lock::HotKeyTable::new(), - )), - calvin_promotion_senders: Arc::new(Mutex::new(std::collections::BTreeMap::new())), + calvin: super::calvin_local::CalvinLocalState::new(), write_order_locks: Arc::new( crate::control::server::shared::write_admission::KeyedWriteOrderLock::new(), ), - autocommit_lock_seq: std::sync::atomic::AtomicU32::new(0), presence: Arc::new(tokio::sync::RwLock::new( crate::control::server::sync::presence::PresenceManager::new( crate::control::server::sync::presence::PresenceConfig::default(), ), )), - permission_cache: Arc::new(tokio::sync::RwLock::new( - crate::control::security::permission_tree::PermissionCache::new(), - )), + permission_cache: Arc::new(tokio::sync::RwLock::new(permission_cache)), + authorization_fence, gateway_invalidator: std::sync::OnceLock::new(), gateway: std::sync::OnceLock::new(), backup_kek: None, @@ -524,19 +396,4 @@ impl SharedState { Self::wire_session_handle_audit(&state); Ok(state) } - - /// Point the session-handle store's audit hook at this state's - /// `AuditLog`, so `SessionHandleFingerprintMismatch` and - /// `SessionHandleResolveMissSpike` are hash-chained with - /// the rest of the auth-plane event stream. Captures the audit Arc - /// directly — a `Weak` would block the cluster wire-up phase's - /// `Arc::get_mut` on `SharedState`. - pub(super) fn wire_session_handle_audit(state: &Arc) { - let audit = Arc::clone(&state.audit); - state.session_handles.set_audit_hook(move |event| { - if let Ok(mut log) = audit.lock() { - let _ = log.record(event, None, "session_handle", ""); - } - }); - } } diff --git a/nodedb/src/control/state/init_prod/auth_parts.rs b/nodedb/src/control/state/init_prod/auth_parts.rs new file mode 100644 index 000000000..c6c488279 --- /dev/null +++ b/nodedb/src/control/state/init_prod/auth_parts.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The auth-configured parts of `SharedState::open`: metering, rate limits, +//! SIEM export, risk scoring, escalation, TLS policy, and the catalog-backed +//! auth-user, scope-grant and quota stores. + +use std::sync::Arc; + +use crate::control::security::credential::CredentialStore; +use crate::control::security::metering::config::MeteringConfig; +use crate::control::security::metering::quota::QuotaManager; +use crate::control::security::ratelimit::config::RateLimitConfig; + +/// Everything `open` builds from `[auth]` configuration. +pub(super) struct AuthParts { + pub(super) metering_config: MeteringConfig, + pub(super) rate_limit_config: RateLimitConfig, + pub(super) http_client: Arc, + pub(super) siem: crate::control::security::siem::SiemExporter, + pub(super) risk_scorer: crate::control::security::risk::RiskScorer, + pub(super) escalation: crate::control::security::escalation::EscalationEngine, + pub(super) tls_policy: crate::control::security::tls_policy::TlsPolicy, + pub(super) auth_users: crate::control::security::jit::auth_user::AuthUserStore, + pub(super) scope_grants: crate::control::security::scope::grant::ScopeGrantStore, + pub(super) quota_manager: QuotaManager, +} + +/// Build the auth-configured parts from `auth_config` and the catalog. +pub(super) fn build( + auth_config: &crate::config::auth::AuthConfig, + credentials: &CredentialStore, +) -> crate::Result { + // `auth_config.metering` is `None` unless the operator configured a + // `[metering]` section; fall back to `MeteringConfig::default()` so + // the effective bounds always match a real `MeteringConfig` value + // (same source `init.rs`'s test constructor pins to) instead of the + // separately-hardcoded `UsageStore`/`QuotaManager` `Default` impls. + let metering_defaults = MeteringConfig::default(); + let metering_config = auth_config.metering.clone().unwrap_or(metering_defaults); + + // `auth_config.rate_limit` is `None` unless the operator configured a + // `[auth.rate_limit]` section; fall back to `RateLimitConfig::default()` + // (same source `init.rs`'s test constructor pins to) so the limiter's + // effective config always matches a real `RateLimitConfig` value + // instead of the separately-hardcoded `RateLimiter::default()` impl. + let rate_limit_defaults = RateLimitConfig::default(); + let rate_limit_config = auth_config + .rate_limit + .clone() + .unwrap_or(rate_limit_defaults); + + // `auth_config.siem` is `None` unless the operator configured an + // `[auth.siem]` section; the default leaves `destinations` empty and + // `webhook_url` blank, so `is_configured()` is false and the export + // path stays dormant. When it *is* configured the exporter shares the + // process-wide HTTP client rather than building its own pool. + let siem_config = auth_config.siem.clone().unwrap_or_default(); + let http_client = Arc::new(reqwest::Client::new()); + let siem = crate::control::security::siem::SiemExporter::with_client( + siem_config, + Arc::clone(&http_client), + ); + + // `auth_config.risk` is `None` unless the operator configured an + // `[auth.risk]` section, and `RiskConfig::default()` has + // `enabled = false`, so scoring stays dormant either way. When it is + // configured the operator's weights and thresholds reach the scorer + // here — the one place they can, since `RiskScorer` reads its config + // only at construction. + let risk_scorer = crate::control::security::risk::RiskScorer::new( + auth_config.risk.clone().unwrap_or_default(), + ); + + // `auth_config.escalation` is `None` unless the operator configured an + // `[auth.escalation]` section, and `EscalationConfig::default()` has + // `enabled = false`, so no account is auto-suspended either way. When + // it is configured the operator's thresholds reach the engine here — + // the one place they can, since `EscalationEngine` reads its config + // only at construction. + let escalation = crate::control::security::escalation::EscalationEngine::new( + auth_config.escalation.clone().unwrap_or_default(), + ); + + // `auth_config.tls_policy` is `None` unless the operator configured an + // `[auth.tls_policy]` section, and `TlsPolicyConfig::default()` has + // `enabled = false`, so no connection is refused on transport grounds + // either way. When it *is* configured the operator's minimum version + // is parsed here — the one place it can be — and an unparseable value + // fails startup rather than being silently replaced by a default that + // enforces something else. + let tls_policy = crate::control::security::tls_policy::TlsPolicy::from_config( + &auth_config.tls_policy.clone().unwrap_or_default(), + )?; + + // Auth users are catalog-backed in production: an escalation verdict + // written to a record has to still be there after a restart, and a + // memory-only store would drop it. + let auth_users = crate::control::security::jit::auth_user::AuthUserStore::open( + credentials.catalog().clone(), + )?; + // Restore the suspend → ban ladder from the persisted records before + // any request is served. + for user in auth_users.list(false) { + escalation.hydrate_suspensions(&user.id, user.escalation_suspensions); + } + + // Scope grants are catalog-backed for the same reason: a grant — and + // the `WHEN` / `REQUIRE` conditions restricting it — has to survive a + // restart, and a memory-only store silently drops every grant the + // operator issued. + let scope_grants = + crate::control::security::scope::grant::ScopeGrantStore::open(credentials.catalog())?; + + // Quota definitions are catalog objects for the same reason grants + // are: a cap that lived only in memory would be lifted by every + // restart, and a rolling deploy would quietly forgive every ceiling + // the operator set. + let quota_manager = QuotaManager::open( + metering_config.max_tracked_quota_grantees, + credentials.catalog(), + )?; + + Ok(AuthParts { + metering_config, + rate_limit_config, + http_client, + siem, + risk_scorer, + escalation, + tls_policy, + auth_users, + scope_grants, + quota_manager, + }) +} diff --git a/nodedb/src/control/state/init_prod/bootstrap.rs b/nodedb/src/control/state/init_prod/bootstrap.rs index 8d130755d..72009f8b7 100644 --- a/nodedb/src/control/state/init_prod/bootstrap.rs +++ b/nodedb/src/control/state/init_prod/bootstrap.rs @@ -266,18 +266,15 @@ pub(super) fn run( )); // Pre-load permission tree definitions before wrapping in RwLock - // (avoids blocking_write() which panics inside async runtimes). + // (avoids blocking_write() which panics inside async runtimes). The + // edges and grants load at boot once the data groups replayed. let mut permission_cache = PermissionCache::new(); let catalog = credentials.catalog(); - if let Ok(collections) = catalog.load_all_collections(DatabaseId::DEFAULT) { - for coll in &collections { - if let Some(ref def_json) = coll.permission_tree_def - && let Ok(def) = sonic_rs::from_str::< - crate::control::security::permission_tree::PermissionTreeDef, - >(def_json) - { - permission_cache.register_tree_def(coll.tenant_id, &coll.name, def); - } + for coll in &catalog.load_all_collections(DatabaseId::DEFAULT)? { + if let Some(change) = + crate::control::security::auth_fence::TreeDefChange::from_collection(coll)? + { + change.apply(&mut permission_cache); } } diff --git a/nodedb/src/control/state/init_prod/mod.rs b/nodedb/src/control/state/init_prod/mod.rs index 707b25d2f..18db751ee 100644 --- a/nodedb/src/control/state/init_prod/mod.rs +++ b/nodedb/src/control/state/init_prod/mod.rs @@ -2,6 +2,7 @@ //! SharedState::open — production constructor loading from disk. +mod auth_parts; mod bootstrap; mod handles; mod open; diff --git a/nodedb/src/control/state/init_prod/open.rs b/nodedb/src/control/state/init_prod/open.rs index 6aad3157c..e01ad5f7d 100644 --- a/nodedb/src/control/state/init_prod/open.rs +++ b/nodedb/src/control/state/init_prod/open.rs @@ -8,10 +8,7 @@ use std::sync::{Arc, Mutex}; use nodedb_types::config::TuningConfig; use crate::control::request_tracker::RequestTracker; -use crate::control::security::metering::config::MeteringConfig; -use crate::control::security::metering::quota::QuotaManager; use crate::control::security::metering::store::UsageStore; -use crate::control::security::ratelimit::config::RateLimitConfig; use crate::control::security::ratelimit::limiter::RateLimiter; use crate::control::security::tenant::{TenantIsolation, TenantQuota}; use crate::control::server::sync::dlq::{DlqConfig, SyncDlq}; @@ -77,95 +74,18 @@ impl SharedState { bus_consumer_handle, } = super::bootstrap::run(&wal, catalog_path, auth_config, is_cluster)?; - // `auth_config.metering` is `None` unless the operator configured a - // `[metering]` section; fall back to `MeteringConfig::default()` so - // the effective bounds always match a real `MeteringConfig` value - // (same source `init.rs`'s test constructor pins to) instead of the - // separately-hardcoded `UsageStore`/`QuotaManager` `Default` impls. - let metering_defaults = MeteringConfig::default(); - let metering_config = auth_config.metering.as_ref().unwrap_or(&metering_defaults); - - // `auth_config.rate_limit` is `None` unless the operator configured a - // `[auth.rate_limit]` section; fall back to `RateLimitConfig::default()` - // (same source `init.rs`'s test constructor pins to) so the limiter's - // effective config always matches a real `RateLimitConfig` value - // instead of the separately-hardcoded `RateLimiter::default()` impl. - let rate_limit_defaults = RateLimitConfig::default(); - let rate_limit_config = auth_config - .rate_limit - .as_ref() - .unwrap_or(&rate_limit_defaults); - - // `auth_config.siem` is `None` unless the operator configured an - // `[auth.siem]` section; the default leaves `destinations` empty and - // `webhook_url` blank, so `is_configured()` is false and the export - // path stays dormant. When it *is* configured the exporter shares the - // process-wide HTTP client rather than building its own pool. - let siem_config = auth_config.siem.clone().unwrap_or_default(); - let http_client = Arc::new(reqwest::Client::new()); - let siem = crate::control::security::siem::SiemExporter::with_client( - siem_config, - Arc::clone(&http_client), - ); - - // `auth_config.risk` is `None` unless the operator configured an - // `[auth.risk]` section, and `RiskConfig::default()` has - // `enabled = false`, so scoring stays dormant either way. When it is - // configured the operator's weights and thresholds reach the scorer - // here — the one place they can, since `RiskScorer` reads its config - // only at construction. - let risk_scorer = crate::control::security::risk::RiskScorer::new( - auth_config.risk.clone().unwrap_or_default(), - ); - - // `auth_config.escalation` is `None` unless the operator configured an - // `[auth.escalation]` section, and `EscalationConfig::default()` has - // `enabled = false`, so no account is auto-suspended either way. When - // it is configured the operator's thresholds reach the engine here — - // the one place they can, since `EscalationEngine` reads its config - // only at construction. - let escalation = crate::control::security::escalation::EscalationEngine::new( - auth_config.escalation.clone().unwrap_or_default(), - ); - - // `auth_config.tls_policy` is `None` unless the operator configured an - // `[auth.tls_policy]` section, and `TlsPolicyConfig::default()` has - // `enabled = false`, so no connection is refused on transport grounds - // either way. When it *is* configured the operator's minimum version - // is parsed here — the one place it can be — and an unparseable value - // fails startup rather than being silently replaced by a default that - // enforces something else. - let tls_policy = crate::control::security::tls_policy::TlsPolicy::from_config( - &auth_config.tls_policy.clone().unwrap_or_default(), - )?; - - // Auth users are catalog-backed in production: an escalation verdict - // written to a record has to still be there after a restart, and a - // memory-only store would drop it. - let auth_users = crate::control::security::jit::auth_user::AuthUserStore::open( - credentials.catalog().clone(), - )?; - // Restore the suspend → ban ladder from the persisted records before - // any request is served. - for user in auth_users.list(false) { - escalation.hydrate_suspensions(&user.id, user.escalation_suspensions); - } - - // Scope grants are catalog-backed for the same reason: a grant — and - // the `WHEN` / `REQUIRE` conditions restricting it — has to survive a - // restart, and a memory-only store silently drops every grant the - // operator issued. - let scope_grants = - crate::control::security::scope::grant::ScopeGrantStore::open(credentials.catalog())?; - - // Quota definitions are catalog objects for the same reason grants - // are: a cap that lived only in memory would be lifted by every - // restart, and a rolling deploy would quietly forgive every ceiling - // the operator set. - let quota_manager = QuotaManager::open( - metering_config.max_tracked_quota_grantees, - credentials.catalog(), - )?; + let super::auth_parts::AuthParts { + metering_config, + rate_limit_config, + http_client, + siem, + risk_scorer, + escalation, + tls_policy, + auth_users, + scope_grants, + quota_manager, + } = super::auth_parts::build(auth_config, &credentials)?; let state = Arc::new(Self { outcome_floor: dispatcher.outcome_floor(), @@ -417,6 +337,7 @@ impl SharedState { data_dir: std::path::PathBuf::new(), trigger_dlq: std::sync::OnceLock::new(), action_requeue: std::sync::OnceLock::new(), + sink_ledgers: std::sync::OnceLock::new(), // Production stores live under real on-disk paths, not a temp dir. _test_state_dir: None, schema_version: crate::control::server::shared::session::plan_cache::SchemaVersion::new( @@ -427,28 +348,20 @@ impl SharedState { dml_counter: crate::control::server::shared::ddl::neutral::maintenance::auto_analyze::DmlCounter::new(), wal_catchup_lsn: AtomicU64::new(0), - last_applied_calvin_epoch: Arc::new(AtomicU64::new(0)), - calvin_apply_results: Arc::new(Mutex::new(std::collections::HashMap::new())), - calvin_lock_managers: Arc::new(Mutex::new(std::collections::BTreeMap::new())), - hot_key_table: Arc::new(Mutex::new( - crate::control::cluster::calvin::scheduler::lock::HotKeyTable::new(), - )), - calvin_promotion_senders: Arc::new(Mutex::new(std::collections::BTreeMap::new())), + calvin: crate::control::state::calvin_local::CalvinLocalState::new(), write_order_locks: Arc::new( crate::control::server::shared::write_admission::KeyedWriteOrderLock::new(), ), - autocommit_lock_seq: std::sync::atomic::AtomicU32::new(0), - calvin_counters: crate::control::state::CalvinCounters { - write_versions_recorded: Arc::new(AtomicU64::new(0)), - read_set_validation_failures: Arc::new(AtomicU64::new(0)), - commits_flushed: Arc::new(AtomicU64::new(0)), - commits_dropped: Arc::new(AtomicU64::new(0)), - }, presence: Arc::new(tokio::sync::RwLock::new( crate::control::server::sync::presence::PresenceManager::new( crate::control::server::sync::presence::PresenceConfig::default(), ), )), + authorization_fence: Arc::new( + crate::control::security::auth_fence::AuthorizationFence::new( + permission_cache.sources(), + ), + ), permission_cache: Arc::new(tokio::sync::RwLock::new(permission_cache)), gateway_invalidator: std::sync::OnceLock::new(), gateway: std::sync::OnceLock::new(), @@ -491,6 +404,7 @@ impl SharedState { #[cfg(test)] mod tests { use super::*; + use crate::control::security::ratelimit::config::RateLimitConfig; use crate::control::security::ratelimit::limiter::LoginRateLimitOutcome; /// Build a `SharedState` via the production `open()` path with a diff --git a/nodedb/src/control/state/init_variants.rs b/nodedb/src/control/state/init_variants.rs new file mode 100644 index 000000000..06bdc5b35 --- /dev/null +++ b/nodedb/src/control/state/init_variants.rs @@ -0,0 +1,157 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Test constructors of `SharedState`, each built on `new_inner`, and the +//! session-handle audit wiring production shares. + +use std::sync::Arc; + +use crate::bridge::dispatch::Dispatcher; +use crate::control::security::credential::CredentialStore; +use crate::control::security::metering::quota::QuotaManager; +use crate::wal::WalManager; + +use super::SharedState; + +impl SharedState { + /// Create shared state with a pre-built credential store (for tests that need catalog). + /// + /// `is_cluster` is the static, deployment-time surrogate-registry mode + /// choice — same predicate as `SharedState::open`'s `is_cluster` + /// (whether this node's caller is about to wire it into a real Raft + /// cluster), not a property of the credential store. Almost every + /// caller is a single-process fixture and passes `false`; the cluster + /// test harness passes `true`. + pub fn new_with_credentials( + dispatcher: Dispatcher, + wal: Arc, + credentials: Arc, + is_cluster: bool, + ) -> crate::Result> { + let wal_for_assigner = Arc::clone(&wal); + let mut state = Self::new_inner(dispatcher, wal)?; + if let Some(s) = Arc::get_mut(&mut state) { + // Rebuild the surrogate assigner against the supplied + // credential store. `new_inner` constructs the assigner + // from a fresh in-memory `CredentialStore` with its own + // in-memory catalog; the supplied store carries the durable + // catalog whose surrogate watermark this fixture must resume. + let registry = Arc::clone(&s.surrogate_registry); + // Seed the registry's high-watermark AND applied-reserve cursor + // from the catalog so restarts in a re-opened test fixture pick up + // where the previous session left off — and so cluster-mode + // metadata-log replay skips already-applied reservations rather + // than double-counting `G`. + let catalog = credentials.catalog(); + // The catalog-derived floor mirrors the production bootstrap: the + // singleton is flushed lazily, so the highest surrogate any live + // binding refers to is the value the allocator can never start + // below. Mode selection mirrors `init_prod/bootstrap.rs::run`: + // `is_cluster` (never seed-list length) picks `Cluster` vs + // `Local`, seeding the applied-reserve cursor only in the + // former. + if let Ok(hwm) = catalog.get_surrogate_hwm() + && let Ok(bound_floor) = catalog.max_bound_surrogate() + && let Ok(mut reg) = registry.write() + { + let floor = hwm.max(bound_floor.as_u32()); + *reg = if is_cluster { + let reserve_index = catalog.get_surrogate_reserve_index().unwrap_or(0); + crate::control::surrogate::SurrogateRegistry::from_persisted_cluster( + floor, + reserve_index, + ) + } else { + crate::control::surrogate::SurrogateRegistry::from_persisted_hwm(floor) + }; + } + let wal_appender: Arc = Arc::new( + crate::control::surrogate::WalSurrogateAppender::new(wal_for_assigner), + ); + s.surrogate_assigner = Arc::new(crate::control::surrogate::SurrogateAssigner::new( + Arc::clone(®istry), + Arc::clone(&credentials), + wal_appender, + )); + // Catalog-backed security stores, rebuilt for the same reason the + // surrogate watermark above is: this constructor's whole purpose + // is to resume a durable catalog, and a memory-only store here + // silently drops every auth-user status and scope grant the + // previous session persisted — so a restart fixture would report + // a clean slate rather than what was actually saved. + s.auth_users = + crate::control::security::jit::auth_user::AuthUserStore::open(catalog.clone())?; + s.scope_grants = + crate::control::security::scope::grant::ScopeGrantStore::open(catalog)?; + // Same reasoning as the grants above: a quota definition is a + // durable catalog object, and a memory-only manager here would + // report every cap as absent after a restart. + s.quota_manager = QuotaManager::open( + s.metering_config.max_tracked_quota_grantees, + credentials.catalog(), + )?; + s.credentials = credentials; + s.ep_topic_registry + .load_from_catalog(s.credentials.catalog())?; + crate::event::topic::hydrate_topic_buffers(s)?; + } + Ok(state) + } + + /// Create shared state with in-memory credential store (for tests). + pub fn new(dispatcher: Dispatcher, wal: Arc) -> crate::Result> { + Self::new_inner(dispatcher, wal) + } + + /// Create shared state whose risk scorer is built from `risk_config` + /// instead of the disabled default (for tests that exercise the risk + /// gate). Production wires the same configuration from `[auth.risk]`. + pub fn new_with_risk_config( + dispatcher: Dispatcher, + wal: Arc, + risk_config: crate::control::security::risk::RiskConfig, + ) -> crate::Result> { + let mut state = Self::new_inner(dispatcher, wal)?; + let s = Arc::get_mut(&mut state).ok_or_else(|| crate::Error::Internal { + detail: "shared state was already shared before the risk scorer could be installed" + .into(), + })?; + s.risk_scorer = crate::control::security::risk::RiskScorer::new(risk_config); + Ok(state) + } + + /// Create shared state whose TLS policy is built from `tls_policy_config` + /// instead of the disabled default (for tests that exercise transport + /// enforcement). Production wires the same configuration from + /// `[auth.tls_policy]`, through the same fallible parse: an unparseable + /// `min_tls_version` is an error here exactly as it is at startup. + pub fn new_with_tls_policy_config( + dispatcher: Dispatcher, + wal: Arc, + tls_policy_config: crate::control::security::tls_policy::TlsPolicyConfig, + ) -> crate::Result> { + let policy = + crate::control::security::tls_policy::TlsPolicy::from_config(&tls_policy_config)?; + let mut state = Self::new_inner(dispatcher, wal)?; + let s = Arc::get_mut(&mut state).ok_or_else(|| crate::Error::Internal { + detail: "shared state was already shared before the TLS policy could be installed" + .into(), + })?; + s.tls_policy = policy; + Ok(state) + } + + /// Point the session-handle store's audit hook at this state's + /// `AuditLog`, so `SessionHandleFingerprintMismatch` and + /// `SessionHandleResolveMissSpike` are hash-chained with + /// the rest of the auth-plane event stream. Captures the audit Arc + /// directly — a `Weak` would block the cluster wire-up phase's + /// `Arc::get_mut` on `SharedState`. + pub(super) fn wire_session_handle_audit(state: &Arc) { + let audit = Arc::clone(&state.audit); + state.session_handles.set_audit_hook(move |event| { + if let Ok(mut log) = audit.lock() { + let _ = log.record(event, None, "session_handle", ""); + } + }); + } +} diff --git a/nodedb/src/control/state/mod.rs b/nodedb/src/control/state/mod.rs index c9c145d30..d8b465379 100644 --- a/nodedb/src/control/state/mod.rs +++ b/nodedb/src/control/state/mod.rs @@ -3,9 +3,11 @@ mod buses_init; mod calvin_apply; mod calvin_counters; +mod calvin_local; mod fields; mod init; mod init_prod; +mod init_variants; mod methods; mod methods_audit; mod methods_lease; @@ -17,6 +19,7 @@ pub mod idle_timeout_cache; pub use self::calvin_apply::CalvinApplyResult; pub use self::calvin_counters::CalvinCounters; +pub use self::calvin_local::CalvinLocalState; pub use self::fields::SharedState; pub use self::init_prod::DataPlaneHandles; pub use self::tenant_request::TenantRequestGuard; diff --git a/nodedb/src/control/system_txn/scope.rs b/nodedb/src/control/system_txn/scope.rs index a17bb1e17..1c74caeb7 100644 --- a/nodedb/src/control/system_txn/scope.rs +++ b/nodedb/src/control/system_txn/scope.rs @@ -30,7 +30,8 @@ impl SystemTxnScope { crate::types::Lsn::new(next.as_u64().saturating_sub(1)) }; let snapshot_epoch = state - .last_applied_calvin_epoch + .calvin + .last_applied_epoch .load(std::sync::atomic::Ordering::Acquire); // Deliberately no `ddl_buffer::activate()`, unlike the client BEGIN diff --git a/nodedb/src/control/trigger/batch/collector.rs b/nodedb/src/control/trigger/batch/collector.rs index ae87e3d4e..7f3484a91 100644 --- a/nodedb/src/control/trigger/batch/collector.rs +++ b/nodedb/src/control/trigger/batch/collector.rs @@ -4,7 +4,7 @@ //! //! Not currently wired into the Normal-mode consumer loop — per-event //! dispatch is the sole production path for AFTER-ROW trigger firing (see -//! `event::consumer::process_normal_batch`). This collector batches +//! `event::consumer::pipeline::deliver_events`). This collector batches //! consecutive WriteEvents targeting the same collection before dispatching //! triggers, yielding batches of up to `batch_size` rows, and remains //! available for a future WHEN-clause-batched throughput optimization diff --git a/nodedb/src/data/executor/core_loop/deferred.rs b/nodedb/src/data/executor/core_loop/deferred.rs index 2e0a21aeb..be71017e4 100644 --- a/nodedb/src/data/executor/core_loop/deferred.rs +++ b/nodedb/src/data/executor/core_loop/deferred.rs @@ -52,6 +52,9 @@ impl CoreLoop { op: write.op, row_id: RowId::row(write.identity), lsn: self.watermark, + // A deferred trigger event repeats a write whose own event + // already carries the record; catch-up never rebuilds it. + record: None, database_id, tenant_id, vshard_id, diff --git a/nodedb/src/data/executor/core_loop/event_emit.rs b/nodedb/src/data/executor/core_loop/event_emit.rs index 3f60ca90f..fe39e111f 100644 --- a/nodedb/src/data/executor/core_loop/event_emit.rs +++ b/nodedb/src/data/executor/core_loop/event_emit.rs @@ -291,6 +291,9 @@ impl CoreLoop { op, row_id, lsn: self.watermark, + record: task + .wal_lsn() + .map(crate::event::types::RecordPosition::first), database_id: task.request.database_id, tenant_id: task.request.tenant_id, vshard_id: task.request.vshard_id, @@ -349,6 +352,7 @@ impl CoreLoop { // watermark = last committed LSN. Correct for heartbeats: uncommitted // writes should NOT advance the Event Plane's watermark. lsn: self.watermark, + record: None, // Heartbeats are synthetic core-liveness markers rather than data writes, // so they have no database owner and are excluded from CDC routing. database_id: crate::types::DatabaseId::DEFAULT, diff --git a/nodedb/src/error/types.rs b/nodedb/src/error/types.rs index 724aa3157..e133cfc41 100644 --- a/nodedb/src/error/types.rs +++ b/nodedb/src/error/types.rs @@ -363,6 +363,12 @@ pub enum Error { #[error("metadata raft group has no elected leader yet; retry needed")] MetadataLeaderUnavailable, + /// A statement cannot be planned yet: this node's roles, grants, policies + /// or permission trees may be behind a change already acknowledged, and + /// did not catch up before the deadline. Nothing ran; the client retries. + #[error("authorization state is not current on this node: {detail}; retry")] + AuthorizationStateBehind { detail: String }, + #[error("execution limit exceeded: {detail}")] ExecutionLimitExceeded { detail: String }, diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index 4d0d5439d..0a27a56ef 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -185,6 +185,7 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { Error::MetadataLeaderUnavailable => NodeDbError::dispatch( "metadata raft group has no elected leader yet; retry exhausted".to_string(), ), + Error::AuthorizationStateBehind { .. } => NodeDbError::cluster(e.to_string()), Error::ExecutionLimitExceeded { detail } => NodeDbError::bad_request(detail), Error::LimitExceeded { limit_name, diff --git a/nodedb/src/event/audit_dml/consumer.rs b/nodedb/src/event/audit_dml/consumer.rs index 6024b95c6..d4e767014 100644 --- a/nodedb/src/event/audit_dml/consumer.rs +++ b/nodedb/src/event/audit_dml/consumer.rs @@ -2,7 +2,7 @@ //! Event Plane DML audit consumer. //! -//! Called once per `WriteEvent` inside `process_normal_batch`. Records a +//! Called once per delivered `WriteEvent` by `consumer::pipeline`. Records a //! `DmlAudit` entry in the Control Plane audit log when: //! //! 1. The event source is `User` (not Trigger / RaftFollower / CrdtSync / @@ -33,7 +33,15 @@ use crate::event::types::{EventSource, WriteEvent, WriteOp}; /// Silently skips on any miss (unknown collection → unknown database → /// mode is `None`) — the fail-open default keeps DML unblocked when the /// cache is cold at startup. -pub fn audit_dml_event(event: &WriteEvent, state: &Arc) { +/// +/// `key` names the event across a restart. An event the durable audit log +/// already records under its key is not audited again, and a new row carries +/// the key in its detail. +pub fn audit_dml_event( + event: &WriteEvent, + state: &Arc, + key: Option<&crate::event::sink_ledger::SinkEventKey>, +) { // Only User-sourced writes are subject to DML auditing. match event.source { EventSource::User => {} @@ -65,14 +73,26 @@ pub fn audit_dml_event(event: &WriteEvent, state: &Arc) { AuditDmlMode::Writes | AuditDmlMode::All => {} } - // Build a compact detail string: op collection:row_id - let detail = format!( + if let Some(key) = key + && state + .sink_ledgers + .get() + .is_some_and(|ledgers| ledgers.audited.contains(key)) + { + return; + } + + // Build a compact detail string: op collection:row_id, then the key. + let mut detail = format!( "{} {}:{} lsn={}", event.op, event.collection, event.row_id, event.lsn.as_u64(), ); + if let Some(key) = key { + detail.push_str(&crate::event::sink_ledger::audit::detail_suffix(key)); + } let source = event.user_id.as_deref().unwrap_or("unknown").to_string(); @@ -101,6 +121,7 @@ mod tests { op, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("o-1")), lsn: Lsn::new(100), + record: None, database_id: DatabaseId::new(42), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/src/event/bus.rs b/nodedb/src/event/bus.rs index cb69f0417..63faea9b9 100644 --- a/nodedb/src/event/bus.rs +++ b/nodedb/src/event/bus.rs @@ -20,6 +20,8 @@ use nodedb_bridge::backpressure::{BackpressureConfig, BackpressureController, Pr use nodedb_bridge::buffer::{Consumer, Producer, RingBuffer}; use nodedb_bridge::error::BridgeError; +use super::progress::CoreEmitProgress; +use super::record_numbering::RecordNumbering; use super::types::WriteEvent; /// Default ring buffer capacity per core (must be power of two). @@ -45,6 +47,10 @@ pub struct EventProducer { inner: Producer, core_id: usize, backpressure: Arc, + /// The highest sequence this core has emitted, read by the Control Plane. + progress: Arc, + /// Numbers each record's events per row, as WAL catch-up does. + numbering: RecordNumbering, /// Latched once the consumer half is dropped, so we log the /// disconnect exactly once per producer instead of spamming /// a warning for every dropped event. @@ -61,7 +67,18 @@ impl EventProducer { /// Updates backpressure state after each emit. When Suspended (>95%), /// events are dropped more aggressively (the Event Plane will enter /// WAL Catchup Mode to recover). - pub fn emit(&mut self, event: WriteEvent) -> bool { + pub fn emit(&mut self, mut event: WriteEvent) -> bool { + self.numbering.stamp(&mut event); + let sequence = event.sequence; + let pushed = self.push(event); + // After the push: a reader that sees `sequence` finds the event on + // the ring, or knows it was dropped. + self.progress.note_emitted(sequence); + pushed + } + + /// Update the backpressure state, then push `event` onto the ring. + fn push(&mut self, event: WriteEvent) -> bool { let util = self.inner.utilization(); // Update backpressure state. @@ -155,6 +172,7 @@ pub struct EventConsumerRx { inner: Consumer, core_id: usize, backpressure: Arc, + progress: Arc, } impl EventConsumerRx { @@ -171,6 +189,11 @@ impl EventConsumerRx { pub fn pressure_state(&self) -> PressureState { self.backpressure.state() } + + /// The emitted-event counter of this ring's core. + pub fn progress(&self) -> Arc { + Arc::clone(&self.progress) + } } /// Creates the event bus: one ring buffer pair per Data Plane core. @@ -192,11 +215,14 @@ pub fn create_event_bus_with_capacity( for core_id in 0..num_cores { let (producer, consumer) = RingBuffer::channel::(capacity); let backpressure = Arc::new(BackpressureController::new(BackpressureConfig::default())); + let progress = Arc::new(CoreEmitProgress::new()); producers.push(EventProducer { inner: producer, core_id, backpressure: Arc::clone(&backpressure), + progress: Arc::clone(&progress), + numbering: RecordNumbering::new(), disconnect_logged: AtomicBool::new(false), }); @@ -204,6 +230,7 @@ pub fn create_event_bus_with_capacity( inner: consumer, core_id, backpressure, + progress, }); } @@ -224,6 +251,7 @@ mod tests { op: WriteOp::Insert, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(seq), + record: None, database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), @@ -287,6 +315,18 @@ mod tests { assert!(!producer.emit(make_event(99))); } + /// A dropped event still counts as emitted, so a barrier waits for it. + #[test] + fn a_dropped_event_counts_as_emitted() { + let (mut producers, consumers) = create_event_bus_with_capacity(1, 4); + let producer = &mut producers[0]; + for i in 1..=4 { + assert!(producer.emit(make_event(i))); + } + assert!(!producer.emit(make_event(5))); + assert_eq!(consumers[0].progress().emitted(), 5); + } + #[test] fn core_id_propagated() { let (producers, consumers) = create_event_bus(2); diff --git a/nodedb/src/event/cdc/buffer.rs b/nodedb/src/event/cdc/buffer.rs index df1140d3f..0cabe7304 100644 --- a/nodedb/src/event/cdc/buffer.rs +++ b/nodedb/src/event/cdc/buffer.rs @@ -24,6 +24,9 @@ pub struct StreamBuffer { retention: RetentionConfig, total_pushed: std::sync::atomic::AtomicU64, total_evicted: std::sync::atomic::AtomicU64, + /// Set when the retained events changed since the CDC ledger last + /// persisted them: an event was inserted, evicted or compacted away. + changed: std::sync::atomic::AtomicBool, } impl StreamBuffer { @@ -37,6 +40,7 @@ impl StreamBuffer { retention, total_pushed: std::sync::atomic::AtomicU64::new(0), total_evicted: std::sync::atomic::AtomicU64::new(0), + changed: std::sync::atomic::AtomicBool::new(false), } } @@ -91,6 +95,8 @@ impl StreamBuffer { .position(|current| current.position() > position) .unwrap_or(events.len()); events.insert(insertion_index, event); + self.changed + .store(true, std::sync::atomic::Ordering::Release); // Evict after ordered insertion. Hydrating an older committed event // must not displace a newer one merely because it arrived later. @@ -224,6 +230,8 @@ impl StreamBuffer { *events = kept; let removed = (before - events.len()) as u32; if removed > 0 { + self.changed + .store(true, std::sync::atomic::Ordering::Release); self.total_evicted .fetch_add(removed as u64, std::sync::atomic::Ordering::Relaxed); } @@ -271,6 +279,31 @@ impl StreamBuffer { pub fn name(&self) -> &str { &self.name } + + /// Clear the changed flag, and return whether it was set. The CDC ledger + /// calls this before it reads [`Self::snapshot`], so a change made after + /// the read sets the flag again. + pub fn take_changed(&self) -> bool { + self.changed + .swap(false, std::sync::atomic::Ordering::AcqRel) + } + + /// Mark the retained events changed again, after a flush that took the + /// flag did not persist them. + pub fn mark_changed(&self) { + self.changed + .store(true, std::sync::atomic::Ordering::Release); + } + + /// Every retained event, oldest first. + pub fn snapshot(&self) -> Vec> { + self.events + .read() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .iter() + .cloned() + .collect() + } } fn extract_key_value(event: &CdcEvent, key_field: &str) -> String { diff --git a/nodedb/src/event/cdc/router.rs b/nodedb/src/event/cdc/router.rs index edea07f20..6e360d6fe 100644 --- a/nodedb/src/event/cdc/router.rs +++ b/nodedb/src/event/cdc/router.rs @@ -23,12 +23,18 @@ use crate::event::types::WriteEvent; use crate::event::watermark_tracker::WatermarkTracker; use crate::types::DatabaseId; +/// A buffer's identity: `(database_id, tenant_id, stream_name)`. +pub type BufferKey = (DatabaseId, u64, String); + /// Manages per-stream buffers and routes events to matching streams. pub struct CdcRouter { /// Stream registry (shared with DDL handlers). registry: Arc, /// Per-stream retention buffers, keyed by `(database_id, tenant_id, stream_name)`. - buffers: std::sync::RwLock>>, + buffers: std::sync::RwLock>>, + /// Buffers removed since the CDC ledger last persisted: their persisted + /// events are deleted by the next flush. + removed: std::sync::Mutex>, /// Per-stream drop rate tracker — emits `warn!` when threshold is crossed. lag_warner: CdcLagWarner, /// System metrics for per-stream Prometheus counters. `None` in unit tests @@ -41,6 +47,7 @@ impl CdcRouter { Self { registry, buffers: std::sync::RwLock::new(HashMap::new()), + removed: std::sync::Mutex::new(Vec::new()), lag_warner: CdcLagWarner::new(DEFAULT_THRESHOLD), metrics: None, } @@ -276,10 +283,46 @@ impl CdcRouter { pub fn remove_buffer(&self, database_id: DatabaseId, tenant_id: u64, stream_name: &str) { let key = (database_id, tenant_id, stream_name.to_string()); let mut buffers = self.buffers.write().unwrap_or_else(|p| p.into_inner()); - buffers.remove(&key); + if buffers.remove(&key).is_some() { + self.removed + .lock() + .unwrap_or_else(|p| p.into_inner()) + .push(key); + } self.lag_warner.remove_stream(tenant_id, stream_name); } + /// The change streams this router routes to. + pub fn registry(&self) -> &StreamRegistry { + &self.registry + } + + /// The buffer of every registered change stream. Topic buffers are left + /// out: a durable topic hydrates its own buffer. + pub fn stream_buffers(&self) -> Vec<(BufferKey, Arc)> { + let buffers = self.buffers.read().unwrap_or_else(|p| p.into_inner()); + buffers + .iter() + .filter(|((database_id, tenant_id, name), _)| { + self.registry.get(*database_id, *tenant_id, name).is_some() + }) + .map(|(key, buffer)| (key.clone(), Arc::clone(buffer))) + .collect() + } + + /// Take the keys of the buffers removed since the last call. + pub fn take_removed(&self) -> Vec { + std::mem::take(&mut *self.removed.lock().unwrap_or_else(|p| p.into_inner())) + } + + /// Return removed buffer keys a flush took but did not persist. + pub fn requeue_removed(&self, keys: Vec) { + self.removed + .lock() + .unwrap_or_else(|p| p.into_inner()) + .extend(keys); + } + /// Snapshot of all buffer stats (for SHOW CHANGE STREAMS). pub fn buffer_stats(&self) -> Vec { let buffers = self.buffers.read().unwrap_or_else(|p| p.into_inner()); @@ -340,6 +383,7 @@ mod tests { "row-{seq}" ))), lsn: Lsn::new(seq * 10), + record: None, database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/src/event/consumer.rs b/nodedb/src/event/consumer.rs deleted file mode 100644 index ff6268c04..000000000 --- a/nodedb/src/event/consumer.rs +++ /dev/null @@ -1,852 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Event Plane consumer: one Tokio task per Data Plane core ring buffer. -//! -//! Each consumer operates in one of two modes: -//! -//! ```text -//! (boot: resume from persisted watermark) -//! │ -//! ▼ -//! WalCatchup ──[caught up to WAL head]──► Normal -//! ▲ │ -//! └────────[sequence gap detected]────────┘ -//! ``` -//! -//! - **Normal**: polls ring buffer, processes events, persists watermark. -//! - **WalCatchup**: pauses ring buffer entirely, reads events exclusively -//! from WAL on disk until caught up, then switches back. Ring buffer and -//! WAL are NEVER read simultaneously (prevents "thundering WAL" spiral). -//! The consumer also boots directly into this mode (see `consumer_loop`) -//! to replay any WAL suffix past the persisted watermark before serving -//! the ring buffer, closing the restart delivery gap. If that suffix is no -//! longer in the WAL — truncated away, or unreachable because the watermark -//! itself was lost — the node fail-stops rather than resume from a position -//! it cannot prove it reached. - -use std::sync::Arc; -use std::time::Duration; - -use nodedb_bridge::backpressure::PressureState; -use tokio::sync::watch; -use tracing::{debug, info, trace, warn}; - -use super::action::ActionRetryQueue; -use super::bus::EventConsumerRx; -use super::metrics::CoreMetrics; -use super::trigger::dlq::TriggerDlq; -use super::watermark::WatermarkStore; -use crate::control::state::SharedState; -use crate::types::Lsn; -use crate::wal::WalManager; - -use super::consumer_helpers::{ - RingDrainOutcome, accumulate_data_event, dispatch_event, dispatch_event_actions, - drain_and_skip_stale, drain_ring_buffer, flush_watermark, maybe_flush_watermark, record_event, -}; - -/// Initial sleep when the ring buffer is empty. Adaptive backoff ramps -/// up to `EMPTY_POLL_MAX` after `EMPTY_POLL_RAMP` consecutive empty polls -/// so an idle Event Plane consumer does not wake every 1ms forever. -const EMPTY_POLL_MIN: Duration = Duration::from_millis(1); -/// Cap on the empty-poll sleep. 50ms keeps trigger / CDC dispatch latency -/// bounded for the first event after an idle period while limiting idle -/// CPU to ~20 wakes/sec per core. -const EMPTY_POLL_MAX: Duration = Duration::from_millis(50); -/// After this many consecutive empty polls (~32ms of idleness at 1ms), -/// switch to the long sleep. -const EMPTY_POLL_RAMP: u32 = 32; - -/// How often to process the retry queue (check for due retries). -const RETRY_POLL_INTERVAL: Duration = Duration::from_millis(200); - -/// Consumer mode state machine. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum ConsumerMode { - /// Reading from ring buffer. - Normal, - /// Ring buffer paused; reading from WAL on disk. - WalCatchup, -} - -/// Select the only safe post-drain mode. A gap event has already been -/// consumed from the SPSC ring, so Normal mode may not observe or dispatch it. -fn normal_drain_next_mode(outcome: &RingDrainOutcome) -> ConsumerMode { - match outcome { - RingDrainOutcome::Contiguous { .. } => ConsumerMode::Normal, - RingDrainOutcome::Gap { .. } => ConsumerMode::WalCatchup, - } -} - -/// Configuration for spawning a consumer. -pub struct ConsumerConfig { - pub rx: EventConsumerRx, - pub shutdown: watch::Receiver, - /// The node-wide shutdown coordinator. WAL recovery failure is unsafe to - /// continue through, so the consumer initiates this canonical bus. - pub shutdown_bus: crate::control::shutdown::ShutdownBus, - pub wal: Arc, - pub watermark_store: Arc, - pub shared_state: Arc, - pub trigger_dlq: Arc>, - pub cdc_router: Arc, - pub num_cores: usize, - /// Per-consumer slab-pin accounting for WAL memory budget enforcement. - pub slab_account: Arc, -} - -/// Handle to a running consumer task. -pub struct ConsumerHandle { - pub core_id: usize, - pub metrics: Arc, - join_handle: tokio::task::JoinHandle<()>, -} - -impl ConsumerHandle { - pub fn abort(&self) { - self.join_handle.abort(); - } - - /// Await natural task termination without taking ownership of the handle. - /// This permits the shutdown supervisor to retain abort ownership until - /// the configured deadline expires. - pub async fn wait_for_exit(&mut self) { - let _ = (&mut self.join_handle).await; - } - - /// Abort the task and await its termination, consuming the handle so the - /// task future (and every `Arc` it held) is definitely dropped by the - /// time this returns. Used in shutdown paths that must observe `Drop` - /// side effects before reopening resources (e.g. redb file locks). - pub async fn abort_and_join(mut self) { - self.join_handle.abort(); - let _ = (&mut self.join_handle).await; - } - - pub fn events_processed(&self) -> u64 { - use std::sync::atomic::Ordering; - self.metrics.events_processed.load(Ordering::Relaxed) - } -} - -/// Spawn a consumer Tokio task for one Data Plane core's event ring buffer. -pub fn spawn_consumer(config: ConsumerConfig) -> ConsumerHandle { - let core_id = config.rx.core_id(); - let metrics = Arc::new(CoreMetrics::new()); - let metrics_clone = Arc::clone(&metrics); - - let join_handle = tokio::spawn(async move { - consumer_loop(config, metrics_clone).await; - }); - - ConsumerHandle { - core_id, - metrics, - join_handle, - } -} - -/// The main consumer loop. -async fn consumer_loop(config: ConsumerConfig, metrics: Arc) { - let ConsumerConfig { - mut rx, - mut shutdown, - shutdown_bus, - wal, - watermark_store, - shared_state, - trigger_dlq, - cdc_router, - num_cores, - slab_account, - } = config; - - let core_id = rx.core_id(); - let mut last_sequence: u64 = 0; - let mut last_lsn = Lsn::ZERO; - let mut dirty_watermark = false; - let mut last_watermark_flush = tokio::time::Instant::now(); - let mut retry_queue = ActionRetryQueue::for_core(&shared_state.data_dir, core_id); - let mut last_retry_poll = tokio::time::Instant::now(); - - // Load persisted watermark. - match watermark_store.load(core_id) { - Ok(lsn) => { - last_lsn = lsn; - debug!(core_id, lsn = lsn.as_u64(), "loaded watermark"); - } - Err(e) => { - warn!(core_id, error = %e, "failed to load watermark, starting from ZERO"); - } - } - - // Boot directly into WalCatchup: the ring buffer is always empty on a - // fresh process, so any events committed to the WAL but not yet - // dispatched+watermarked before a crash/restart would otherwise be - // silently dropped. Replaying [watermark+1, WAL head] here closes that - // gap; the WalCatchup arm below self-transitions to Normal once caught - // up (or immediately, on a fresh DB with an empty WAL). - let mut mode = ConsumerMode::WalCatchup; - - debug!(core_id, "event plane consumer started"); - let mut wal_retry_count: u32 = 0; - let mut empty_polls: u32 = 0; - - loop { - if *shutdown.borrow() { - if dirty_watermark { - flush_watermark(&watermark_store, core_id, last_lsn); - } - debug!(core_id, "event plane consumer shutting down"); - break; - } - - match mode { - ConsumerMode::Normal => { - let drain = drain_ring_buffer( - &mut rx, - &metrics, - core_id, - &mut last_sequence, - &mut last_lsn, - ); - let next_mode = normal_drain_next_mode(&drain); - let (events, gap_event) = match drain { - RingDrainOutcome::Contiguous { events } => (events, None), - RingDrainOutcome::Gap { - events, - first_gap_event, - .. - } => (events, Some(first_gap_event)), - }; - let batch_count = events.len(); - - if batch_count > 0 { - empty_polls = 0; - dirty_watermark = true; - - process_normal_batch(&events, &shared_state, &mut retry_queue, &cdc_router) - .await; - - let batch_payload_bytes: u64 = events - .iter() - .map(|e| { - e.new_value.as_ref().map_or(0, |v| v.len() as u64) - + e.old_value.as_ref().map_or(0, |v| v.len() as u64) - }) - .sum(); - slab_account.add_pinned(batch_payload_bytes); - drop(events); - slab_account.release_pinned(batch_payload_bytes); - - trace!(core_id, batch_count, "event batch processed"); - } - - if let Some(first_gap_event) = gap_event { - // This event has been removed from the SPSC ring but is not - // safe to record or dispatch. WAL catchup starts from the - // prior complete LSN and must reconstruct the withheld - // preceding record as well as the gap event. - warn!( - core_id, - gap_sequence = first_gap_event.sequence, - gap_lsn = first_gap_event.lsn.as_u64(), - safe_sequence = last_sequence, - safe_lsn = last_lsn.as_u64(), - "ring sequence gap consumed; entering WAL catchup before event side effects" - ); - mode = next_mode; - debug_assert_eq!(mode, ConsumerMode::WalCatchup); - metrics.record_wal_catchup_enter(); - continue; - } - - if batch_count > 0 { - if slab_account.is_shed() { - info!(core_id, "slab budget shed — entering WAL catchup mode"); - slab_account.reset(); - slab_account.clear_shed(); - mode = ConsumerMode::WalCatchup; - metrics.record_wal_catchup_enter(); - continue; - } - - if rx.pressure_state() == PressureState::Suspended { - info!( - core_id, - "backpressure SUSPENDED — entering WAL catchup mode" - ); - mode = ConsumerMode::WalCatchup; - metrics.record_wal_catchup_enter(); - continue; - } - - tokio::task::yield_now().await; - continue; - } - - // No new events — process retry queue if due. An empty queue - // still polls when an operator has requeued something from the - // DLQ, since that action arrives out-of-band and would - // otherwise wait for an unrelated failure to wake the poll. - let requeued_waiting = shared_state - .action_requeue - .get() - .is_some_and(|inbox| inbox.has_work_for(core_id)); - if (!retry_queue.is_empty() || requeued_waiting) - && last_retry_poll.elapsed() >= RETRY_POLL_INTERVAL - { - process_retry_queue(&mut retry_queue, &trigger_dlq, &shared_state, core_id) - .await; - last_retry_poll = tokio::time::Instant::now(); - } - - maybe_flush_watermark( - &watermark_store, - core_id, - last_lsn, - &mut dirty_watermark, - &mut last_watermark_flush, - ); - - empty_polls = empty_polls.saturating_add(1); - let poll_sleep = if empty_polls < EMPTY_POLL_RAMP { - EMPTY_POLL_MIN - } else { - EMPTY_POLL_MAX - }; - - tokio::select! { - _ = tokio::time::sleep(poll_sleep) => {} - _ = shutdown.changed() => { - if dirty_watermark { - flush_watermark(&watermark_store, core_id, last_lsn); - } - debug!(core_id, "event plane consumer received shutdown"); - break; - } - } - } - - ConsumerMode::WalCatchup => { - const MAX_WAL_RETRIES: u32 = 10; - - info!( - core_id, - from_lsn = last_lsn.as_u64(), - "WAL catchup: replaying from WAL" - ); - - match super::wal_replay::replay_wal_mmap( - &wal, - last_lsn.next(), - core_id, - num_cores, - last_sequence, - ) - .or_else(|e| { - // The sequential arm is a fallback for readers that cannot - // see the bytes (mmap misses O_DIRECT writes to the active - // segment), not for a log that no longer holds the records. - // Both arms read the same directory, so retrying a deleted - // suffix only re-derives the same answer. - if is_retained_floor_violation(&e) { - return Err(e); - } - super::wal_replay::replay_wal_to_events( - &wal, - last_lsn.next(), - core_id, - num_cores, - last_sequence, - ) - }) { - Ok(events) => { - wal_retry_count = 0; - let count = events.len() as u64; - for event in &events { - record_event(core_id, event, &metrics); - dispatch_event(event, &shared_state, &mut retry_queue, &cdc_router) - .await; - last_sequence = event.sequence; - if event.lsn.is_ahead_of(last_lsn) { - last_lsn = event.lsn; - } - } - if count > 0 { - metrics.record_wal_replay(count); - info!( - core_id, - events_replayed = count, - new_lsn = last_lsn.as_u64(), - "WAL catchup complete" - ); - } else { - debug!(core_id, "WAL catchup: no new events"); - } - } - Err(e) if is_retained_floor_violation(&e) => { - // The WAL no longer holds the suffix this consumer has - // to replay, so the events between the watermark and - // the retained floor were never dispatched and can - // never be reconstructed: their CDC rows, trigger - // fires, and streaming-MV updates are already lost. - // Retrying cannot heal that, and returning to Normal - // mode would resume from a watermark that claims - // delivery which never happened — divergence that - // spreads silently into every downstream consumer. - // Stop the node instead, while an operator can still - // see where the log begins and restore from a snapshot. - fail_stop_wal_catchup( - core_id, - &e, - "the WAL no longer retains the suffix this consumer must replay", - wal_retry_count, - last_lsn, - &shared_state, - &shutdown_bus, - ); - break; - } - Err(e) => { - wal_retry_count += 1; - if wal_retry_count >= MAX_WAL_RETRIES { - fail_stop_wal_catchup( - core_id, - &e, - "WAL catchup replay kept failing", - wal_retry_count, - last_lsn, - &shared_state, - &shutdown_bus, - ); - // Do not return to Normal or flush an uncertain - // watermark: no later event side effects may run. - break; - } - warn!( - core_id, - error = %e, - retry = wal_retry_count, - max_retries = MAX_WAL_RETRIES, - "WAL catchup replay failed, retrying after delay" - ); - tokio::time::sleep(Duration::from_millis(100)).await; - continue; - } - } - - // Discard ring-buffer events already re-dispatched by replay - // (`lsn <= last_lsn`); the first event past the replay point is - // returned and processed here rather than dropped (the ring is - // SPSC and cannot un-receive it). It is a live ring event, so it - // takes the same Normal-mode path as the events after it, which - // the Normal-mode drain serves on the next iteration. - if let Some(event) = drain_and_skip_stale(&mut rx, last_lsn) { - record_event(core_id, &event, &metrics); - process_normal_batch( - std::slice::from_ref(&event), - &shared_state, - &mut retry_queue, - &cdc_router, - ) - .await; - last_sequence = event.sequence; - if event.lsn.is_ahead_of(last_lsn) { - last_lsn = event.lsn; - } - } - - flush_watermark(&watermark_store, core_id, last_lsn); - dirty_watermark = false; - last_watermark_flush = tokio::time::Instant::now(); - - mode = ConsumerMode::Normal; - info!(core_id, "returned to Normal mode"); - } - } - } - - let processed = { - use std::sync::atomic::Ordering; - metrics.events_processed.load(Ordering::Relaxed) - }; - debug!( - core_id, - total_processed = processed, - "event plane consumer stopped" - ); -} - -/// Is this the WAL reporting that the requested replay suffix was already -/// truncated away? -/// -/// Every other replay failure is potentially transient (a partially written -/// active segment, a reader that cannot see O_DIRECT bytes); this one is not, -/// so the consumer routes it past the retry loop straight to fail-stop. -fn is_retained_floor_violation(error: &crate::Error) -> bool { - matches!( - error, - crate::Error::Wal(nodedb_wal::WalError::ReplayBelowRetainedFloor { .. }) - ) -} - -/// Fail-stop on an unrecoverable WAL catchup failure. -/// -/// `reason` states which failure class stopped the node; `attempts` is how many -/// replay attempts preceded it (zero for a failure that is unrecoverable on the -/// first observation and never retried). -/// -/// `last_safe_lsn` is deliberately only observed for audit/logging; this path -/// never mutates or flushes it. Continuing without a recoverable WAL prefix -/// would permit later event side effects to overtake missing writes. -fn fail_stop_wal_catchup( - core_id: usize, - error: &impl std::fmt::Display, - reason: &'static str, - attempts: u32, - last_safe_lsn: Lsn, - shared_state: &SharedState, - shutdown_bus: &crate::control::shutdown::ShutdownBus, -) { - tracing::error!( - core_id, - error = %error, - reason, - attempts, - last_safe_lsn = last_safe_lsn.as_u64(), - "WAL catchup cannot complete; initiating fail-stop shutdown" - ); - shared_state.audit_record( - crate::control::security::audit::AuditEvent::AdminAction, - None, - "event_plane", - &format!( - "event consumer core {core_id} WAL catchup stopped after {attempts} attempts at safe LSN {} ({reason}): {error}", - last_safe_lsn.as_u64() - ), - ); - drop(shutdown_bus.initiate()); -} - -/// Process a batch of Normal-mode events: CDC, permission cache, streaming MVs, -/// CRDT sync, and awaited AFTER/DEFINE EVENT trigger actions. -/// -/// Row/statement and DEFINE EVENT processing is per-event via -/// [`dispatch_event_actions`]. The normal batch and WAL-catchup -/// (`dispatch_event` → `dispatch_event_actions`) paths therefore execute the -/// same action processor exactly once for every data `WriteEvent`. -async fn process_normal_batch( - events: &[super::types::WriteEvent], - shared_state: &Arc, - retry_queue: &mut ActionRetryQueue, - cdc_router: &Arc, -) { - for event in events { - if !event.op.is_data_event() { - shared_state - .watermark_tracker - .advance_lsn_only(event.vshard_id.as_u32(), event.lsn.as_u64()); - continue; - } - - // DML audit: record to audit log before dispatching triggers. - super::audit_dml::audit_dml_event(event, shared_state); - - // Await every AFTER/DEFINE EVENT action before advancing the Event - // Plane's data watermark or publishing non-trigger side effects. - dispatch_event_actions(event, shared_state, retry_queue).await; - - // Non-trigger side effects (watermark, CDC, permission cache, MVs, CRDT). - accumulate_data_event(event, shared_state, cdc_router).await; - } -} - -/// Process the retry queue: DLQ exhausted entries and retry ready ones. -async fn process_retry_queue( - retry_queue: &mut ActionRetryQueue, - trigger_dlq: &Arc>, - shared_state: &Arc, - core_id: usize, -) { - // Collect anything an operator sent back from the DLQ before draining, so - // a requeued action joins this round instead of waiting for the next poll. - if let Some(inbox) = shared_state.action_requeue.get() { - for action in inbox.take_for_core(core_id) { - retry_queue.enqueue(action); - } - } - - let (ready, exhausted) = retry_queue.drain_due(); - if !exhausted.is_empty() { - let mut dlq = trigger_dlq.lock().unwrap_or_else(|p| p.into_inner()); - for action in exhausted { - let _ = dlq.enqueue(action); - } - // dlq MutexGuard dropped before any await. - } - - for action in ready { - super::trigger::dispatcher::retry_action(&action, shared_state, retry_queue).await; - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::event::bus::create_event_bus_with_capacity; - use crate::event::consumer_helpers::{ - RingDrainOutcome, detect_sequence_gap, drain_ring_buffer, - }; - use crate::event::types::{EventSource, RowId, WriteOp}; - use crate::types::{DatabaseId, TenantId, VShardId}; - - fn make_event(seq: u64) -> super::super::types::WriteEvent { - super::super::types::WriteEvent { - sequence: seq, - collection: Arc::from("test"), - op: WriteOp::Insert, - row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), - lsn: Lsn::new(seq * 10), - database_id: DatabaseId::new(7), - tenant_id: TenantId::new(1), - vshard_id: VShardId::new(0), - source: EventSource::User, - new_value: Some(Arc::from(b"data".as_slice())), - old_value: None, - system_time_ms: None, - valid_time_ms: None, - user_id: None, - statement_digest: None, - } - } - - #[test] - fn gap_detection() { - let metrics = CoreMetrics::new(); - let e1 = make_event(1); - let e5 = make_event(5); - - record_event(0, &e1, &metrics); - detect_sequence_gap(0, &e5, 1, &metrics); - record_event(0, &e5, &metrics); - - use std::sync::atomic::Ordering; - assert_eq!(metrics.events_processed.load(Ordering::Relaxed), 2); - assert_eq!(metrics.events_dropped.load(Ordering::Relaxed), 3); - } - - #[test] - fn gap_event_is_withheld_for_wal_catchup_and_normal_transitions() { - let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 16); - producers[0].emit(make_event(1)); - producers[0].emit(make_event(2)); - producers[0].emit(make_event(4)); - - let metrics = CoreMetrics::new(); - let mut rx = consumers.remove(0); - let mut last_sequence = 0; - let mut last_lsn = Lsn::ZERO; - let outcome = drain_ring_buffer(&mut rx, &metrics, 0, &mut last_sequence, &mut last_lsn); - - assert_eq!(normal_drain_next_mode(&outcome), ConsumerMode::WalCatchup); - let RingDrainOutcome::Gap { - events, - first_gap_event, - initial_safe_lsn, - initial_safe_sequence, - } = outcome - else { - panic!("a sequence gap must force WAL catchup"); - }; - assert_eq!( - events - .iter() - .map(|event| event.sequence) - .collect::>(), - [1] - ); - // The last observed record (sequence 2) is withheld with the gap; - // the earlier completed record remains safe for Normal-mode dispatch. - assert_eq!(first_gap_event.sequence, 4); - assert_eq!(initial_safe_sequence, 0); - assert_eq!(initial_safe_lsn, Lsn::ZERO); - assert_eq!(last_sequence, 1); - assert_eq!(last_lsn, Lsn::new(10)); - - use std::sync::atomic::Ordering; - assert_eq!(metrics.events_processed.load(Ordering::Relaxed), 1); - assert_eq!(metrics.last_processed_lsn.load(Ordering::Relaxed), 10); - assert_eq!(metrics.events_dropped.load(Ordering::Relaxed), 1); - assert_eq!(rx.try_recv().map(|event| event.sequence), None); - } - - #[test] - fn gap_within_wal_record_withholds_its_contiguous_sibling() { - let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 16); - let mut first_sibling = make_event(2); - first_sibling.lsn = Lsn::new(100); - let mut gap_sibling = make_event(4); - gap_sibling.lsn = Lsn::new(100); - producers[0].emit(first_sibling); - producers[0].emit(gap_sibling); - - let metrics = CoreMetrics::new(); - let mut rx = consumers.remove(0); - let mut last_sequence = 1; - let mut last_lsn = Lsn::new(90); - let outcome = drain_ring_buffer(&mut rx, &metrics, 0, &mut last_sequence, &mut last_lsn); - - let RingDrainOutcome::Gap { - events, - first_gap_event, - initial_safe_lsn, - initial_safe_sequence, - } = outcome - else { - panic!("a sequence gap must force WAL catchup"); - }; - assert_eq!(initial_safe_lsn, Lsn::new(90)); - assert_eq!(initial_safe_sequence, 1); - assert!(events.is_empty()); - assert_eq!(first_gap_event.sequence, 4); - assert_eq!(first_gap_event.lsn, Lsn::new(100)); - // The first sibling at LSN 100 is withheld with the gap sibling. The - // replay start must therefore not skip the whole record. - assert_eq!(last_sequence, 1); - assert_eq!(last_lsn, Lsn::new(90)); - let replay_start = last_lsn.next(); - assert_eq!(replay_start, Lsn::new(91)); - assert!( - replay_start <= first_gap_event.lsn, - "replay must include the withheld record" - ); - - use std::sync::atomic::Ordering; - assert_eq!(metrics.events_processed.load(Ordering::Relaxed), 0); - } - - #[test] - fn gap_at_newer_lsn_withholds_preceding_observed_record() { - let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 16); - let mut preceding_record_sibling = make_event(2); - preceding_record_sibling.lsn = Lsn::new(100); - let mut first_post_gap_event = make_event(4); - first_post_gap_event.lsn = Lsn::new(101); - producers[0].emit(preceding_record_sibling); - producers[0].emit(first_post_gap_event); - - let metrics = CoreMetrics::new(); - let mut rx = consumers.remove(0); - let mut last_sequence = 1; - let mut last_lsn = Lsn::new(90); - let outcome = drain_ring_buffer(&mut rx, &metrics, 0, &mut last_sequence, &mut last_lsn); - - let RingDrainOutcome::Gap { - events, - first_gap_event, - initial_safe_lsn, - initial_safe_sequence, - } = outcome - else { - panic!("a sequence gap must force WAL catchup"); - }; - assert_eq!(initial_safe_lsn, Lsn::new(90)); - assert_eq!(initial_safe_sequence, 1); - // Sequence 3 at LSN 100 is missing, so the observed sequence-2 - // sibling cannot be dispatched as a complete durable record. - assert!(events.is_empty()); - assert_eq!(first_gap_event.sequence, 4); - assert_eq!(first_gap_event.lsn, Lsn::new(101)); - // Roll back to the preceding safe durable LSN: replay begins before - // LSN 100 and recovers its missing sibling as well as the gap event. - assert_eq!(last_sequence, 1); - assert_eq!(last_lsn, Lsn::new(90)); - assert!(last_lsn.next() <= Lsn::new(100)); - - use std::sync::atomic::Ordering; - assert_eq!(metrics.events_processed.load(Ordering::Relaxed), 0); - } - - #[tokio::test] - async fn max_wal_replay_failure_initiates_shutdown_without_advancing_watermark() { - let dir = tempfile::tempdir().unwrap(); - let (_wal, watermark_store, shared_state, _trigger_dlq, _cdc_router) = - crate::event::test_utils::event_test_deps(&dir); - watermark_store.save(0, Lsn::new(10)).unwrap(); - let (shutdown_bus, _) = - crate::control::shutdown::ShutdownBus::new(Arc::clone(&shared_state.shutdown)); - - fail_stop_wal_catchup( - 0, - &"replay unavailable", - "WAL catchup replay kept failing", - 10, - Lsn::new(10), - &shared_state, - &shutdown_bus, - ); - - assert!(shared_state.shutdown.is_shutdown()); - assert_eq!(watermark_store.load(0).unwrap(), Lsn::new(10)); - } - - /// A truncated-away suffix is routed past the retry loop: it is not - /// transient, and continuing would advance the watermark past events that - /// were never dispatched. - #[test] - fn truncated_suffix_is_recognised_as_unrecoverable() { - assert!(is_retained_floor_violation(&crate::Error::Wal( - nodedb_wal::WalError::ReplayBelowRetainedFloor { - from_lsn: 10, - retained_floor_lsn: 4096, - earliest_segment: "wal-00000000000000004096.seg".to_string(), - } - ))); - assert!(!is_retained_floor_violation(&crate::Error::Wal( - nodedb_wal::WalError::Sealed - ))); - } - - #[tokio::test] - async fn consumer_processes_and_persists_watermark() { - let (mut producers, consumers) = create_event_bus_with_capacity(1, 64); - let dir = tempfile::tempdir().unwrap(); - let (wal, watermark_store, shared_state, trigger_dlq, cdc_router) = - crate::event::test_utils::event_test_deps(&dir); - - let (shutdown_tx, shutdown_rx) = watch::channel(false); - let shutdown_watch = Arc::new(crate::control::shutdown::ShutdownWatch::new()); - let (shutdown_bus, _) = crate::control::shutdown::ShutdownBus::new(shutdown_watch); - - // Emit events. - for i in 1..=5 { - producers[0].emit(make_event(i)); - } - - let handle = spawn_consumer(ConsumerConfig { - rx: consumers.into_iter().next().unwrap(), - shutdown: shutdown_rx, - shutdown_bus, - wal, - watermark_store: Arc::clone(&watermark_store), - shared_state, - trigger_dlq, - cdc_router, - num_cores: 1, - slab_account: Arc::new(crate::event::slab_budget::ConsumerSlabAccount::new(0)), - }); - - // Let consumer process. - tokio::time::sleep(Duration::from_millis(50)).await; - assert_eq!(handle.events_processed(), 5); - - // Shutdown (triggers final watermark flush). - shutdown_tx.send(true).ok(); - tokio::time::sleep(Duration::from_millis(50)).await; - - // Verify watermark was persisted. - let wm = watermark_store.load(0).unwrap(); - assert_eq!(wm, Lsn::new(50)); // seq 5 → lsn = 5*10 = 50 - } -} diff --git a/nodedb/src/event/consumer/delivery.rs b/nodedb/src/event/consumer/delivery.rs new file mode 100644 index 000000000..5fd2189fd --- /dev/null +++ b/nodedb/src/event/consumer/delivery.rs @@ -0,0 +1,157 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Deliver each event to the side effects once. +//! +//! An event can reach the consumer twice: from the ring, and rebuilt from the +//! WAL by catch-up. An event that reproduces a WAL record is named by its +//! record position, collection, row and write kind. Both paths name the same +//! event the same way. +//! +//! `safe` is a prefix of the WAL: every event of every record at or below it +//! was delivered. The guard remembers the events it delivered above it, and +//! forgets them once the prefix covers them. An event that reproduces no +//! record never reaches catch-up, so only its ring copy arrives, and the +//! guard delivers it as it comes. + +use std::collections::{BTreeMap, HashSet}; +use std::sync::Arc; + +use crate::event::record_numbering::is_delete; +use crate::event::types::{RowId, WriteEvent}; +use crate::types::Lsn; + +/// An event's name within its record. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +struct EventKey { + occurrence: u32, + collection: Arc, + row: RowId, + delete: bool, +} + +/// The consumer's record of what it delivered. +#[derive(Debug)] +pub struct DeliveryGuard { + safe: Lsn, + /// Events delivered above `safe`, by record LSN. + delivered: BTreeMap>, +} + +impl DeliveryGuard { + /// A guard whose prefix is `safe`, the persisted watermark. + pub fn new(safe: Lsn) -> Self { + Self { + safe, + delivered: BTreeMap::new(), + } + } + + /// Every event of every record at or below this LSN was delivered. + pub fn safe(&self) -> Lsn { + self.safe + } + + /// Number of delivered events remembered above the prefix. + pub fn remembered(&self) -> usize { + self.delivered.values().map(HashSet::len).sum() + } + + /// Whether `event` still needs delivery. Records it as delivered when it + /// does, so a second copy is refused. + pub fn admit(&mut self, event: &WriteEvent) -> bool { + let Some(position) = event.record else { + return true; + }; + if !position.lsn.is_ahead_of(self.safe) { + return false; + } + self.delivered + .entry(position.lsn.as_u64()) + .or_default() + .insert(EventKey { + occurrence: position.occurrence, + collection: Arc::clone(&event.collection), + row: event.row_id.clone(), + delete: is_delete(event.op), + }) + } + + /// Raise the prefix to `lsn` and forget the events it now covers. Returns + /// whether the prefix moved. + pub fn advance_safe(&mut self, lsn: Lsn) -> bool { + if !lsn.is_ahead_of(self.safe) { + return false; + } + self.safe = lsn; + self.delivered = self.delivered.split_off(&lsn.as_u64().saturating_add(1)); + true + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::event::types::{EventSource, RecordPosition, WriteOp}; + use crate::types::{DatabaseId, TenantId, VShardId}; + + fn event(record: Option<(u64, u32)>, row: &str, op: WriteOp) -> WriteEvent { + WriteEvent { + sequence: 0, + collection: Arc::from("orders"), + op, + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key(row)), + lsn: Lsn::new(record.map(|(lsn, _)| lsn).unwrap_or(1)), + record: record.map(|(lsn, occurrence)| RecordPosition { + lsn: Lsn::new(lsn), + occurrence, + }), + database_id: DatabaseId::DEFAULT, + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + source: EventSource::User, + new_value: None, + old_value: None, + system_time_ms: None, + valid_time_ms: None, + user_id: None, + statement_digest: None, + } + } + + #[test] + fn a_second_copy_of_an_event_is_refused() { + let mut guard = DeliveryGuard::new(Lsn::new(5)); + assert!(guard.admit(&event(Some((6, 0)), "a", WriteOp::Insert))); + // Catch-up rebuilds the same write as an update: same event. + assert!(!guard.admit(&event(Some((6, 0)), "a", WriteOp::Update))); + // A later write of the same row in the same record is another event. + assert!(guard.admit(&event(Some((6, 1)), "a", WriteOp::Update))); + assert!(guard.admit(&event(Some((6, 0)), "a", WriteOp::Delete))); + } + + #[test] + fn an_event_at_or_below_the_prefix_was_delivered() { + let mut guard = DeliveryGuard::new(Lsn::new(5)); + assert!(!guard.admit(&event(Some((5, 0)), "a", WriteOp::Insert))); + assert!(!guard.admit(&event(Some((3, 0)), "a", WriteOp::Insert))); + } + + #[test] + fn an_event_with_no_record_is_always_delivered() { + let mut guard = DeliveryGuard::new(Lsn::new(5)); + assert!(guard.admit(&event(None, "a", WriteOp::Update))); + assert!(guard.admit(&event(None, "a", WriteOp::Update))); + } + + #[test] + fn advancing_the_prefix_forgets_what_it_covers() { + let mut guard = DeliveryGuard::new(Lsn::ZERO); + assert!(guard.admit(&event(Some((3, 0)), "a", WriteOp::Insert))); + assert!(guard.admit(&event(Some((7, 0)), "b", WriteOp::Insert))); + assert!(guard.advance_safe(Lsn::new(5))); + assert_eq!(guard.remembered(), 1); + assert!(!guard.admit(&event(Some((7, 0)), "b", WriteOp::Insert))); + assert!(!guard.advance_safe(Lsn::new(4))); + assert_eq!(guard.safe(), Lsn::new(5)); + } +} diff --git a/nodedb/src/event/consumer/drain.rs b/nodedb/src/event/consumer/drain.rs new file mode 100644 index 000000000..5e2ee8ef1 --- /dev/null +++ b/nodedb/src/event/consumer/drain.rs @@ -0,0 +1,262 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Take events off one core's ring, and learn when the ring dropped some. +//! +//! The core numbers its events contiguously. An event that does not follow +//! the last one taken means the ring dropped the numbers in between. So does +//! a ring that ran empty while the core's emitted counter, read before the +//! drain, is above the last number taken: the core emitted events the ring +//! no longer holds. +//! +//! A tail drop hands the dropped numbers to WAL catch-up: the cursor counts +//! them as taken, since each one's record is in the WAL recovery replays. +//! +//! The cursor also advances the consumer's safe prefix. Before a drain it +//! reads the final-outcome bound, then the emitted counter. Every record at +//! or below the bound was applied and emitted its events before the bound +//! was read, so once the cursor has taken every event up to the counter, and +//! the ring dropped none of them, every event of those records was taken. + +use crate::event::bus::EventConsumerRx; +use crate::event::consumer_helpers::{DRAIN_BATCH_LIMIT, detect_sequence_gap, record_event}; +use crate::event::metrics::CoreMetrics; +use crate::event::types::WriteEvent; +use crate::types::Lsn; + +/// The events one drain took, in ring order. +#[derive(Debug)] +pub struct Drained { + pub events: Vec, + /// The ring dropped at least one event before or during this drain. + pub dropped: bool, +} + +/// A final-outcome bound and the emitted counter read after it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct SafeSnapshot { + final_bound: Lsn, + emitted: u64, +} + +/// Position of the consumer in one core's event numbers. +#[derive(Debug, Default)] +pub struct RingCursor { + last_sequence: u64, + pending: Option, +} + +impl RingCursor { + pub fn new() -> Self { + Self::default() + } + + /// Remember a snapshot to advance the safe prefix by, unless one is + /// waiting. Read `final_bound` before `emitted`. + pub fn take_snapshot(&mut self, final_bound: Lsn, emitted: u64) { + if self.pending.is_none() { + self.pending = Some(SafeSnapshot { + final_bound, + emitted, + }); + } + } + + /// The bound of the waiting snapshot once every event it counted was + /// taken. Clears the snapshot. + pub fn settled_bound(&mut self) -> Option { + match self.pending { + Some(snapshot) if self.last_sequence >= snapshot.emitted => { + self.pending = None; + Some(snapshot.final_bound) + } + _ => None, + } + } + + /// Take up to [`DRAIN_BATCH_LIMIT`] events. `emitted_before` is the core's + /// emitted counter, read before this call. + pub fn drain( + &mut self, + rx: &mut EventConsumerRx, + metrics: &CoreMetrics, + core_id: usize, + emitted_before: u64, + ) -> Drained { + let mut events = Vec::new(); + let mut dropped = false; + let mut emptied = true; + while let Some(event) = rx.try_recv() { + if event.sequence != self.last_sequence.saturating_add(1) { + detect_sequence_gap(core_id, &event, self.last_sequence, metrics); + dropped = true; + } + self.last_sequence = self.last_sequence.max(event.sequence); + events.push(event); + if events.len() >= DRAIN_BATCH_LIMIT as usize { + emptied = false; + break; + } + } + if emptied && self.last_sequence < emitted_before { + metrics.record_drop(emitted_before - self.last_sequence); + tracing::warn!( + core_id, + last_taken = self.last_sequence, + emitted = emitted_before, + "event ring dropped its newest events; WAL catch-up recovers them" + ); + dropped = true; + // The dropped numbers belong to WAL catch-up from here on: each + // one's record is in the WAL the recovery replays. The cursor + // counts them as taken, so a later snapshot can settle and the + // next event on the ring is not a gap. + self.last_sequence = emitted_before; + } + if dropped { + // The waiting snapshot counted events the ring dropped. + self.pending = None; + } + for event in &events { + record_event(core_id, event, metrics); + } + Drained { events, dropped } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::event::bus::create_event_bus_with_capacity; + use crate::event::types::{EventSource, RowId, WriteOp}; + use crate::types::{DatabaseId, TenantId, VShardId}; + + fn make_event(seq: u64) -> WriteEvent { + WriteEvent { + sequence: seq, + collection: Arc::from("test"), + op: WriteOp::Insert, + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), + lsn: Lsn::new(seq * 10), + record: None, + database_id: DatabaseId::new(7), + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + source: EventSource::User, + new_value: Some(Arc::from(b"data".as_slice())), + old_value: None, + system_time_ms: None, + valid_time_ms: None, + user_id: None, + statement_digest: None, + } + } + + #[test] + fn contiguous_events_are_taken_without_a_drop() { + let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 16); + for seq in 1..=3 { + producers[0].emit(make_event(seq)); + } + let metrics = CoreMetrics::new(); + let mut cursor = RingCursor::new(); + let drained = cursor.drain(&mut consumers[0], &metrics, 0, 3); + assert_eq!(drained.events.len(), 3); + assert!(!drained.dropped); + } + + /// Every event taken off the ring is returned, including the one after a + /// gap: each is a real event, and the guard decides whether to deliver + /// it. + #[test] + fn a_gap_is_reported_and_every_event_is_returned() { + let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 16); + producers[0].emit(make_event(1)); + producers[0].emit(make_event(2)); + producers[0].emit(make_event(4)); + let metrics = CoreMetrics::new(); + let mut cursor = RingCursor::new(); + let drained = cursor.drain(&mut consumers[0], &metrics, 0, 4); + let sequences: Vec = drained.events.iter().map(|e| e.sequence).collect(); + assert_eq!(sequences, vec![1, 2, 4]); + assert!(drained.dropped); + } + + /// A ring that ran empty below the emitted counter dropped its newest + /// events, even though no later event shows a gap. + #[test] + fn a_tail_drop_is_reported() { + let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 4); + for seq in 1..=4 { + assert!(producers[0].emit(make_event(seq))); + } + assert!(!producers[0].emit(make_event(5))); + let emitted = consumers[0].progress().emitted(); + let metrics = CoreMetrics::new(); + let mut cursor = RingCursor::new(); + let drained = cursor.drain(&mut consumers[0], &metrics, 0, emitted); + assert_eq!(drained.events.len(), 4); + assert!(drained.dropped); + } + + /// After a tail drop the cursor stops reporting it: the dropped numbers + /// are recovery's, so the next empty drain reports nothing and a new + /// snapshot settles. + #[test] + fn a_tail_drop_is_reported_once_and_later_snapshots_settle() { + let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 4); + for seq in 1..=4 { + assert!(producers[0].emit(make_event(seq))); + } + assert!(!producers[0].emit(make_event(5))); + let emitted = consumers[0].progress().emitted(); + let metrics = CoreMetrics::new(); + let mut cursor = RingCursor::new(); + assert!( + cursor + .drain(&mut consumers[0], &metrics, 0, emitted) + .dropped + ); + + cursor.take_snapshot(Lsn::new(50), emitted); + let drained = cursor.drain(&mut consumers[0], &metrics, 0, emitted); + assert!(!drained.dropped, "a tail drop is reported once"); + assert_eq!(cursor.settled_bound(), Some(Lsn::new(50))); + + producers[0].emit(make_event(emitted + 1)); + let drained = cursor.drain(&mut consumers[0], &metrics, 0, emitted + 1); + assert!( + !drained.dropped, + "the next event after a tail drop is not a gap" + ); + } + + #[test] + fn the_snapshot_settles_once_its_events_are_taken() { + let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 16); + let metrics = CoreMetrics::new(); + let mut cursor = RingCursor::new(); + producers[0].emit(make_event(1)); + cursor.take_snapshot(Lsn::new(50), 2); + cursor.drain(&mut consumers[0], &metrics, 0, 1); + assert_eq!(cursor.settled_bound(), None); + producers[0].emit(make_event(2)); + cursor.drain(&mut consumers[0], &metrics, 0, 2); + assert_eq!(cursor.settled_bound(), Some(Lsn::new(50))); + assert_eq!(cursor.settled_bound(), None); + } + + #[test] + fn a_drop_discards_the_waiting_snapshot() { + let (mut producers, mut consumers) = create_event_bus_with_capacity(1, 16); + let metrics = CoreMetrics::new(); + let mut cursor = RingCursor::new(); + cursor.take_snapshot(Lsn::new(50), 3); + producers[0].emit(make_event(1)); + producers[0].emit(make_event(3)); + let drained = cursor.drain(&mut consumers[0], &metrics, 0, 3); + assert!(drained.dropped); + assert_eq!(cursor.settled_bound(), None); + } +} diff --git a/nodedb/src/event/consumer/fail_stop.rs b/nodedb/src/event/consumer/fail_stop.rs new file mode 100644 index 000000000..e11c38dbc --- /dev/null +++ b/nodedb/src/event/consumer/fail_stop.rs @@ -0,0 +1,104 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Stop the node when WAL catch-up cannot complete. + +use crate::control::state::SharedState; +use crate::types::Lsn; + +/// Is this the WAL reporting that the requested replay suffix was already +/// truncated away? +/// +/// Every other replay failure is potentially transient (a partially written +/// active segment, a reader that cannot see O_DIRECT bytes); this one is not, +/// so the consumer routes it past the retry loop straight to fail-stop. +pub fn is_retained_floor_violation(error: &crate::Error) -> bool { + matches!( + error, + crate::Error::Wal(nodedb_wal::WalError::ReplayBelowRetainedFloor { .. }) + ) +} + +/// Fail-stop on an unrecoverable WAL catch-up failure. +/// +/// `reason` states which failure class stopped the node; `attempts` is how many +/// replay attempts preceded it (zero for a failure that is unrecoverable on the +/// first observation and never retried). +/// +/// `last_safe_lsn` is only observed for audit and logging; this path never +/// mutates or flushes it. Continuing without a recoverable WAL prefix would let +/// later event side effects overtake missing writes. +pub fn fail_stop_wal_catchup( + core_id: usize, + error: &impl std::fmt::Display, + reason: &'static str, + attempts: u32, + last_safe_lsn: Lsn, + shared_state: &SharedState, + shutdown_bus: &crate::control::shutdown::ShutdownBus, +) { + tracing::error!( + core_id, + error = %error, + reason, + attempts, + last_safe_lsn = last_safe_lsn.as_u64(), + "WAL catchup cannot complete; initiating fail-stop shutdown" + ); + shared_state.audit_record( + crate::control::security::audit::AuditEvent::AdminAction, + None, + "event_plane", + &format!( + "event consumer core {core_id} WAL catchup stopped after {attempts} attempts at safe LSN {} ({reason}): {error}", + last_safe_lsn.as_u64() + ), + ); + drop(shutdown_bus.initiate()); +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + #[tokio::test] + async fn max_wal_replay_failure_initiates_shutdown_without_advancing_watermark() { + let dir = tempfile::tempdir().unwrap(); + let (_wal, watermark_store, shared_state, _trigger_dlq, _cdc_router) = + crate::event::test_utils::event_test_deps(&dir); + watermark_store.save(0, Lsn::new(10)).unwrap(); + let (shutdown_bus, _) = + crate::control::shutdown::ShutdownBus::new(Arc::clone(&shared_state.shutdown)); + + fail_stop_wal_catchup( + 0, + &"replay unavailable", + "WAL catchup replay kept failing", + 10, + Lsn::new(10), + &shared_state, + &shutdown_bus, + ); + + assert!(shared_state.shutdown.is_shutdown()); + assert_eq!(watermark_store.load(0).unwrap(), Lsn::new(10)); + } + + /// A truncated-away suffix is routed past the retry loop: it is not + /// transient, and continuing would advance the watermark past events that + /// were never dispatched. + #[test] + fn truncated_suffix_is_recognised_as_unrecoverable() { + assert!(is_retained_floor_violation(&crate::Error::Wal( + nodedb_wal::WalError::ReplayBelowRetainedFloor { + from_lsn: 10, + retained_floor_lsn: 4096, + earliest_segment: "wal-00000000000000004096.seg".to_string(), + } + ))); + assert!(!is_retained_floor_violation(&crate::Error::Wal( + nodedb_wal::WalError::Sealed + ))); + } +} diff --git a/nodedb/src/event/consumer/handle.rs b/nodedb/src/event/consumer/handle.rs new file mode 100644 index 000000000..087538ef7 --- /dev/null +++ b/nodedb/src/event/consumer/handle.rs @@ -0,0 +1,84 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Spawn a consumer task for one core's ring, and the handle to it. + +use std::sync::Arc; + +use tokio::sync::watch; + +use crate::control::state::SharedState; +use crate::event::bus::EventConsumerRx; +use crate::event::metrics::CoreMetrics; +use crate::event::trigger::dlq::TriggerDlq; +use crate::event::watermark::WatermarkStore; +use crate::wal::WalManager; + +use super::run::consumer_loop; + +/// Configuration for spawning a consumer. +pub struct ConsumerConfig { + pub rx: EventConsumerRx, + pub shutdown: watch::Receiver, + /// The node-wide shutdown coordinator. WAL recovery failure is unsafe to + /// continue through, so the consumer initiates this canonical bus. + pub shutdown_bus: crate::control::shutdown::ShutdownBus, + pub wal: Arc, + pub watermark_store: Arc, + pub shared_state: Arc, + pub trigger_dlq: Arc>, + pub cdc_router: Arc, + pub num_cores: usize, + /// Per-consumer slab-pin accounting for WAL memory budget enforcement. + pub slab_account: Arc, +} + +/// Handle to a running consumer task. +pub struct ConsumerHandle { + pub core_id: usize, + pub metrics: Arc, + join_handle: tokio::task::JoinHandle<()>, +} + +impl ConsumerHandle { + pub fn abort(&self) { + self.join_handle.abort(); + } + + /// Await natural task termination without taking ownership of the handle. + /// This permits the shutdown supervisor to retain abort ownership until + /// the configured deadline expires. + pub async fn wait_for_exit(&mut self) { + let _ = (&mut self.join_handle).await; + } + + /// Abort the task and await its termination, consuming the handle so the + /// task future (and every `Arc` it held) is definitely dropped by the + /// time this returns. Used in shutdown paths that must observe `Drop` + /// side effects before reopening resources (e.g. redb file locks). + pub async fn abort_and_join(mut self) { + self.join_handle.abort(); + let _ = (&mut self.join_handle).await; + } + + pub fn events_processed(&self) -> u64 { + use std::sync::atomic::Ordering; + self.metrics.events_processed.load(Ordering::Relaxed) + } +} + +/// Spawn a consumer Tokio task for one Data Plane core's event ring buffer. +pub fn spawn_consumer(config: ConsumerConfig) -> ConsumerHandle { + let core_id = config.rx.core_id(); + let metrics = Arc::new(CoreMetrics::new()); + let metrics_clone = Arc::clone(&metrics); + + let join_handle = tokio::spawn(async move { + consumer_loop(config, metrics_clone).await; + }); + + ConsumerHandle { + core_id, + metrics, + join_handle, + } +} diff --git a/nodedb/src/event/consumer/mod.rs b/nodedb/src/event/consumer/mod.rs new file mode 100644 index 000000000..7a0b27761 --- /dev/null +++ b/nodedb/src/event/consumer/mod.rs @@ -0,0 +1,12 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod delivery; +pub mod drain; +pub mod fail_stop; +pub mod handle; +pub mod pipeline; +pub mod recovery; +pub mod replay; +mod run; + +pub use handle::{ConsumerConfig, ConsumerHandle, spawn_consumer}; diff --git a/nodedb/src/event/consumer/pipeline.rs b/nodedb/src/event/consumer/pipeline.rs new file mode 100644 index 000000000..773e62b67 --- /dev/null +++ b/nodedb/src/event/consumer/pipeline.rs @@ -0,0 +1,182 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Deliver an event to every side effect. +//! +//! Events from the ring and events WAL catch-up rebuilt take the same path: +//! DML audit, the awaited AFTER and DEFINE EVENT actions, then the watermark, +//! CDC, streaming materialized views and CRDT sync. The guard admits each +//! event once, so an event both paths carry is delivered once. +//! +//! The permission cache is not a side effect here. The permission step runs +//! on ring events by their core numbers, before delivery (see +//! `control::security::permission_tree::event_handler`). + +use std::sync::Arc; + +use crate::control::state::SharedState; +use crate::event::action::ActionRetryQueue; +use crate::event::cdc::CdcRouter; +use crate::event::sink_ledger::SinkEventKey; +use crate::event::types::WriteEvent; + +use super::delivery::DeliveryGuard; + +/// Deliver every event of `events` the guard admits, in order. Returns how +/// many were delivered. +pub async fn deliver_events<'a>( + core_id: usize, + events: impl IntoIterator, + guard: &mut DeliveryGuard, + shared_state: &Arc, + retry_queue: &mut ActionRetryQueue, + cdc_router: &Arc, +) -> u64 { + let mut delivered = 0u64; + for event in events { + if !guard.admit(event) { + continue; + } + deliver_event(core_id, event, shared_state, retry_queue, cdc_router).await; + delivered += 1; + } + delivered +} + +/// Deliver one admitted event. +async fn deliver_event( + core_id: usize, + event: &WriteEvent, + shared_state: &Arc, + retry_queue: &mut ActionRetryQueue, + cdc_router: &Arc, +) { + if !event.op.is_data_event() { + shared_state + .watermark_tracker + .advance_lsn_only(event.vshard_id.as_u32(), event.lsn.as_u64()); + return; + } + // The key the non-idempotent sinks remember the event by, so an event + // delivered again after a restart reaches each of them once. + let key = SinkEventKey::of(core_id, event); + // Recorded before the actions run, as the statement's audit precedes its + // triggers. + crate::event::audit_dml::audit_dml_event(event, shared_state, key.as_ref()); + // Every action finishes before the watermark or any other side effect + // moves past the event. + dispatch_event_actions(event, shared_state, retry_queue).await; + accumulate_data_event(event, key.as_ref(), shared_state, cdc_router); +} + +/// Run the AFTER-ROW triggers and DEFINE EVENT actions of a data event. +pub async fn dispatch_event_actions( + event: &WriteEvent, + shared_state: &Arc, + retry_queue: &mut ActionRetryQueue, +) { + if !event_actions_required(event) { + return; + } + crate::event::trigger::dispatcher::dispatch_triggers(event, shared_state, retry_queue).await; + crate::control::event_trigger::process_write_event( + Arc::clone(shared_state), + event, + retry_queue, + ) + .await; +} + +fn event_actions_required(event: &WriteEvent) -> bool { + event.op.is_data_event() +} + +/// Apply the side effects after the actions: the wall-time watermark, CDC +/// routing, streaming materialized views and CRDT sync packaging. +fn accumulate_data_event( + event: &WriteEvent, + key: Option<&SinkEventKey>, + shared_state: &Arc, + cdc_router: &Arc, +) { + let event_time_ms = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or_default() + .as_millis() as u64; + shared_state.watermark_tracker.advance( + event.vshard_id.as_u32(), + event.lsn.as_u64(), + event_time_ms, + ); + + match shared_state.sink_ledgers.get() { + Some(ledgers) => { + ledgers + .cdc + .route_once(key, cdc_router, event, &shared_state.watermark_tracker) + } + // Without its ledger (the sink state did not load, and the node is + // stopping) a change stream is not exactly-once across a restart. + None => cdc_router.route_event(event, &shared_state.watermark_tracker), + } + let matching_streams = shared_state.stream_registry.find_matching( + event.database_id, + event.tenant_id.as_u64(), + &event.collection, + ); + if !matching_streams.is_empty() { + shared_state.mv_registry.applied().apply_once(key, || { + for stream_def in &matching_streams { + crate::event::streaming_mv::processor::process_write_event_for_mvs( + event, + &shared_state.mv_registry, + &stream_def.name, + ); + } + }); + } + shared_state.delta_packager.package_and_enqueue( + event, + key, + shared_state.sink_ledgers.get().map(|ledgers| &ledgers.crdt), + &shared_state.crdt_sync_delivery, + ); +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::event_actions_required; + use crate::event::types::{EventSource, RowId, WriteEvent, WriteOp}; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + + fn event(op: WriteOp) -> WriteEvent { + WriteEvent { + sequence: 1, + collection: Arc::from("events"), + op, + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), + lsn: Lsn::new(1), + record: None, + database_id: DatabaseId::DEFAULT, + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + source: EventSource::User, + new_value: None, + old_value: None, + system_time_ms: None, + valid_time_ms: None, + user_id: None, + statement_digest: None, + } + } + + #[test] + fn every_data_event_runs_its_actions() { + assert!(event_actions_required(&event(WriteOp::Insert))); + assert!(event_actions_required(&event(WriteOp::BulkDelete { + count: 2 + }))); + assert!(!event_actions_required(&event(WriteOp::Heartbeat))); + } +} diff --git a/nodedb/src/event/consumer/recovery.rs b/nodedb/src/event/consumer/recovery.rs new file mode 100644 index 000000000..1e21e17ef --- /dev/null +++ b/nodedb/src/event/consumer/recovery.rs @@ -0,0 +1,112 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Rebuild dropped events from the WAL. +//! +//! The ring drops an event when it is full. The consumer learns of the drop +//! from a gap in the core's event numbers, or from the core's emitted counter +//! running past the last event it took. Every dropped event belongs to a +//! record already in the WAL: a write appends its record before its core +//! applies it and emits. So recovery replays the WAL up to the head it read +//! when it learned of the drop. +//! +//! Recovery replays only records whose outcome is final. A record above the +//! outcome floor may still be applying, or may yet be refused; its events +//! arrive from the ring once it applies. At boot every record in the WAL is +//! final: the cores replayed it before the Event Plane started. +//! +//! The guard refuses what was already delivered, so recovery may replay a +//! record the ring also delivers. + +use crate::types::Lsn; + +/// A recovery in progress. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Recovery { + /// Every dropped event belongs to a record at or below this LSN. + target: Lsn, + /// Every record at or below this LSN was replayed. + replayed_through: Lsn, +} + +impl Recovery { + /// Recover every record above `safe` up to `target`. + pub fn new(safe: Lsn, target: Lsn) -> Self { + Self { + target, + replayed_through: safe, + } + } + + /// Widen the recovery to `target`, for a drop learned of while it runs. + pub fn extend(&mut self, target: Lsn) { + self.target = self.target.max(target); + } + + /// Every record at or below this LSN was replayed. + pub fn replayed_through(&self) -> Lsn { + self.replayed_through + } + + /// Whether every record a dropped event could belong to was replayed. + pub fn is_done(&self) -> bool { + self.replayed_through >= self.target + } + + /// The records the next pass replays, `(from, upto)` inclusive, when the + /// final-outcome bound lets it make progress. + pub fn next_range(&self, final_bound: Lsn) -> Option<(Lsn, Lsn)> { + let upto = self.target.min(final_bound); + upto.is_ahead_of(self.replayed_through) + .then(|| (self.replayed_through.next(), upto)) + } + + /// Record that a pass replayed every record up to `upto`. + pub fn note_replayed(&mut self, upto: Lsn) { + self.replayed_through = self.replayed_through.max(upto); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_pass_stops_at_the_final_outcome_bound() { + let mut recovery = Recovery::new(Lsn::new(10), Lsn::new(30)); + assert_eq!( + recovery.next_range(Lsn::new(20)), + Some((Lsn::new(11), Lsn::new(20))) + ); + recovery.note_replayed(Lsn::new(20)); + assert!(!recovery.is_done()); + // The bound has not moved: nothing more is final yet. + assert_eq!(recovery.next_range(Lsn::new(20)), None); + assert_eq!( + recovery.next_range(Lsn::new(99)), + Some((Lsn::new(21), Lsn::new(30))) + ); + recovery.note_replayed(Lsn::new(30)); + assert!(recovery.is_done()); + } + + #[test] + fn a_later_drop_widens_the_target() { + let mut recovery = Recovery::new(Lsn::new(10), Lsn::new(15)); + recovery.extend(Lsn::new(12)); + recovery.note_replayed(Lsn::new(15)); + assert!(recovery.is_done()); + recovery.extend(Lsn::new(40)); + assert!(!recovery.is_done()); + assert_eq!( + recovery.next_range(Lsn::new(40)), + Some((Lsn::new(16), Lsn::new(40))) + ); + } + + #[test] + fn a_target_at_or_below_the_prefix_is_already_done() { + let recovery = Recovery::new(Lsn::new(10), Lsn::new(8)); + assert!(recovery.is_done()); + assert_eq!(recovery.next_range(Lsn::new(50)), None); + } +} diff --git a/nodedb/src/event/consumer/replay.rs b/nodedb/src/event/consumer/replay.rs new file mode 100644 index 000000000..ba52cf30d --- /dev/null +++ b/nodedb/src/event/consumer/replay.rs @@ -0,0 +1,35 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Rebuild the events of a range of WAL records for one core. + +use crate::event::types::WriteEvent; +use crate::event::wal_replay::{replay_wal_mmap, replay_wal_to_events}; +use crate::types::Lsn; +use crate::wal::WalManager; + +use super::fail_stop::is_retained_floor_violation; + +/// The events of every record from `from` through `upto` routed to +/// `core_id`, in LSN order. +pub fn replay_range( + wal: &WalManager, + from: Lsn, + upto: Lsn, + core_id: usize, + num_cores: usize, +) -> crate::Result> { + // Rebuilt events carry no ring number; the guard names them by record. + let events = replay_wal_mmap(wal, from, core_id, num_cores, 0).or_else(|e| { + // The sequential reader is a fallback for readers that cannot see the + // bytes (mmap misses O_DIRECT writes to the active segment), not for a + // log that no longer holds the records: both read the same directory. + if is_retained_floor_violation(&e) { + return Err(e); + } + replay_wal_to_events(wal, from, core_id, num_cores, 0) + })?; + Ok(events + .into_iter() + .filter(|event| event.lsn <= upto) + .collect()) +} diff --git a/nodedb/src/event/consumer/run.rs b/nodedb/src/event/consumer/run.rs new file mode 100644 index 000000000..3c66dd93c --- /dev/null +++ b/nodedb/src/event/consumer/run.rs @@ -0,0 +1,840 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The consumer loop of one core's ring. +//! +//! Each pass: +//! +//! 1. Replays what a recovery still owes, up to the records whose outcome is +//! final (see [`super::recovery`]). At boot the recovery covers the WAL +//! above the persisted watermark. +//! 2. Takes events off the ring (see [`super::drain`]), runs the permission +//! step on every one of them, and delivers the ones the guard admits (see +//! [`super::delivery`]). A drop starts or widens a recovery. +//! 3. Advances the safe prefix, which is the persisted watermark: restart +//! replays the WAL above it. +//! +//! Ring and WAL are read in turn by this one task, never at once. + +use std::sync::Arc; +use std::time::Duration; + +use tracing::{debug, info, trace, warn}; + +use crate::control::state::SharedState; +use crate::event::action::ActionRetryQueue; +use crate::event::consumer_helpers::{flush_watermark, maybe_flush_watermark, record_event}; +use crate::event::metrics::CoreMetrics; +use crate::event::trigger::dlq::TriggerDlq; +use crate::types::Lsn; +use crate::wal::WalManager; + +use super::delivery::DeliveryGuard; +use super::drain::RingCursor; +use super::fail_stop::{fail_stop_wal_catchup, is_retained_floor_violation}; +use super::handle::ConsumerConfig; +use super::pipeline::deliver_events; +use super::recovery::Recovery; + +/// Initial sleep when the ring buffer is empty. Adaptive backoff ramps +/// up to `EMPTY_POLL_MAX` after `EMPTY_POLL_RAMP` consecutive empty polls +/// so an idle Event Plane consumer does not wake every 1ms forever. +const EMPTY_POLL_MIN: Duration = Duration::from_millis(1); +/// Cap on the empty-poll sleep. 50ms keeps trigger / CDC dispatch latency +/// bounded for the first event after an idle period while limiting idle +/// CPU to ~20 wakes/sec per core. +const EMPTY_POLL_MAX: Duration = Duration::from_millis(50); +/// After this many consecutive empty polls (~32ms of idleness at 1ms), +/// switch to the long sleep. +const EMPTY_POLL_RAMP: u32 = 32; + +/// How often to process the retry queue (check for due retries). +const RETRY_POLL_INTERVAL: Duration = Duration::from_millis(200); + +/// Replay attempts before a failing recovery stops the node. +const MAX_WAL_RETRIES: u32 = 10; + +/// The highest LSN in the WAL now. +fn last_wal_lsn(wal: &WalManager) -> Lsn { + Lsn::new(wal.next_lsn().as_u64().saturating_sub(1)) +} + +/// Every record at or below this LSN has its final outcome. Records the cores +/// replayed at boot are final, whatever the live outcome floor says. +fn final_outcome_bound(shared_state: &SharedState, boot_end: Lsn) -> Lsn { + shared_state.outcome_floor.floor().max(boot_end) +} + +/// The main consumer loop. +pub(super) async fn consumer_loop(config: ConsumerConfig, metrics: Arc) { + let ConsumerConfig { + mut rx, + mut shutdown, + shutdown_bus, + wal, + watermark_store, + shared_state, + trigger_dlq, + cdc_router, + num_cores, + slab_account, + } = config; + + let core_id = rx.core_id(); + let emit_progress = rx.progress(); + let mut retry_queue = ActionRetryQueue::for_core(&shared_state.data_dir, core_id); + let mut last_retry_poll = tokio::time::Instant::now(); + + let persisted = match watermark_store.load(core_id) { + Ok(lsn) => { + debug!(core_id, lsn = lsn.as_u64(), "loaded watermark"); + lsn + } + Err(e) => { + warn!(core_id, error = %e, "failed to load watermark, starting from ZERO"); + Lsn::ZERO + } + }; + let mut guard = DeliveryGuard::new(persisted); + // Events committed to the WAL but not delivered before a restart are + // recovered first. If that suffix is no longer in the WAL, the node + // fail-stops rather than resume from a position it cannot prove it reached. + let boot_end = last_wal_lsn(&wal); + let mut recovery = Some(Recovery::new(persisted, boot_end)); + let mut cursor = RingCursor::new(); + + let mut dirty_watermark = false; + let mut last_watermark_flush = tokio::time::Instant::now(); + let mut wal_retry_count: u32 = 0; + let mut empty_polls: u32 = 0; + + debug!(core_id, "event plane consumer started"); + + loop { + if *shutdown.borrow() { + if dirty_watermark { + flush_watermark(&shared_state, &watermark_store, core_id, guard.safe()); + } + debug!(core_id, "event plane consumer shutting down"); + break; + } + + let mut replayed = 0u64; + if let Some(active) = recovery.as_mut() { + let final_bound = final_outcome_bound(&shared_state, boot_end); + if let Some((from, upto)) = active.next_range(final_bound) { + match super::replay::replay_range(&wal, from, upto, core_id, num_cores) { + Ok(events) => { + wal_retry_count = 0; + for event in &events { + record_event(core_id, event, &metrics); + } + replayed = deliver_events( + core_id, + &events, + &mut guard, + &shared_state, + &mut retry_queue, + &cdc_router, + ) + .await; + active.note_replayed(upto); + metrics.record_wal_replay(events.len() as u64); + info!( + core_id, + from = from.as_u64(), + upto = upto.as_u64(), + delivered = replayed, + "WAL catchup pass complete" + ); + } + Err(e) if is_retained_floor_violation(&e) => { + // The events between the watermark and the retained + // floor were never delivered and cannot be rebuilt. + // Stop while an operator can still see where the log + // begins and restore from a snapshot. + fail_stop_wal_catchup( + core_id, + &e, + "the WAL no longer retains the suffix this consumer must replay", + wal_retry_count, + guard.safe(), + &shared_state, + &shutdown_bus, + ); + break; + } + Err(e) => { + wal_retry_count += 1; + if wal_retry_count >= MAX_WAL_RETRIES { + fail_stop_wal_catchup( + core_id, + &e, + "WAL catchup replay kept failing", + wal_retry_count, + guard.safe(), + &shared_state, + &shutdown_bus, + ); + break; + } + warn!( + core_id, + error = %e, + retry = wal_retry_count, + max_retries = MAX_WAL_RETRIES, + "WAL catchup replay failed, retrying after delay" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + continue; + } + } + } + if active.is_done() { + recovery = None; + debug!(core_id, "WAL catchup complete"); + } + } + + // Bound first, then the counter: every event of a record at or below + // the bound is counted. + let final_bound = final_outcome_bound(&shared_state, boot_end); + cursor.take_snapshot(final_bound, emit_progress.emitted()); + let emitted_before = emit_progress.emitted(); + let drained = cursor.drain(&mut rx, &metrics, core_id, emitted_before); + let batch_count = drained.events.len(); + + crate::control::security::permission_tree::event_handler::apply_ring_events( + core_id, + &drained.events, + &shared_state.permission_cache, + shared_state.authorization_fence.permission_applied(), + ) + .await; + + if batch_count > 0 { + empty_polls = 0; + let batch_payload_bytes: u64 = drained + .events + .iter() + .map(|e| { + e.new_value.as_ref().map_or(0, |v| v.len() as u64) + + e.old_value.as_ref().map_or(0, |v| v.len() as u64) + }) + .sum(); + slab_account.add_pinned(batch_payload_bytes); + deliver_events( + core_id, + &drained.events, + &mut guard, + &shared_state, + &mut retry_queue, + &cdc_router, + ) + .await; + slab_account.release_pinned(batch_payload_bytes); + trace!(core_id, batch_count, "event batch processed"); + } + + if drained.dropped { + let target = last_wal_lsn(&wal); + match recovery.as_mut() { + Some(active) => active.extend(target), + None => { + recovery = Some(Recovery::new(guard.safe(), target)); + metrics.record_wal_catchup_enter(); + warn!( + core_id, + target = target.as_u64(), + "event ring dropped events; recovering them from the WAL" + ); + } + } + } + + if let Some(bound) = cursor.settled_bound() { + // A recovery still owes records above what it replayed. + let candidate = recovery + .as_ref() + .map_or(bound, |active| bound.min(active.replayed_through())); + if guard.advance_safe(candidate) { + dirty_watermark = true; + } + } + + if batch_count > 0 || replayed > 0 { + if slab_account.is_shed() { + info!(core_id, "slab budget shed; pinned event payloads released"); + slab_account.reset(); + slab_account.clear_shed(); + } + tokio::task::yield_now().await; + continue; + } + + // No new events — process retry queue if due. An empty queue + // still polls when an operator has requeued something from the + // DLQ, since that action arrives out-of-band and would + // otherwise wait for an unrelated failure to wake the poll. + let requeued_waiting = shared_state + .action_requeue + .get() + .is_some_and(|inbox| inbox.has_work_for(core_id)); + if (!retry_queue.is_empty() || requeued_waiting) + && last_retry_poll.elapsed() >= RETRY_POLL_INTERVAL + { + process_retry_queue(&mut retry_queue, &trigger_dlq, &shared_state, core_id).await; + last_retry_poll = tokio::time::Instant::now(); + } + + maybe_flush_watermark( + &shared_state, + &watermark_store, + core_id, + guard.safe(), + &mut dirty_watermark, + &mut last_watermark_flush, + ); + + empty_polls = empty_polls.saturating_add(1); + let poll_sleep = if empty_polls < EMPTY_POLL_RAMP { + EMPTY_POLL_MIN + } else { + EMPTY_POLL_MAX + }; + + tokio::select! { + _ = tokio::time::sleep(poll_sleep) => {} + _ = shutdown.changed() => { + if dirty_watermark { + flush_watermark(&shared_state, &watermark_store, core_id, guard.safe()); + } + debug!(core_id, "event plane consumer received shutdown"); + break; + } + } + } + + let processed = { + use std::sync::atomic::Ordering; + metrics.events_processed.load(Ordering::Relaxed) + }; + debug!( + core_id, + total_processed = processed, + "event plane consumer stopped" + ); +} + +/// Process the retry queue: DLQ exhausted entries and retry ready ones. +async fn process_retry_queue( + retry_queue: &mut ActionRetryQueue, + trigger_dlq: &Arc>, + shared_state: &Arc, + core_id: usize, +) { + // Collect anything an operator sent back from the DLQ before draining, so + // a requeued action joins this round instead of waiting for the next poll. + if let Some(inbox) = shared_state.action_requeue.get() { + for action in inbox.take_for_core(core_id) { + retry_queue.enqueue(action); + } + } + + let (ready, exhausted) = retry_queue.drain_due(); + if !exhausted.is_empty() { + let mut dlq = trigger_dlq.lock().unwrap_or_else(|p| p.into_inner()); + for action in exhausted { + let _ = dlq.enqueue(action); + } + // dlq MutexGuard dropped before any await. + } + + for action in ready { + crate::event::trigger::dispatcher::retry_action(&action, shared_state, retry_queue).await; + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::sync::atomic::Ordering; + use std::time::Duration; + + use nodedb_types::AuditDmlMode; + use nodedb_types::sync::wire::SyncProvenance; + + use super::super::handle::{ConsumerConfig, spawn_consumer}; + use crate::control::security::audit::AuditEvent; + use crate::control::state::SharedState; + use crate::event::bus::create_event_bus_with_capacity; + use crate::event::cdc::ChangeStreamDef; + use crate::event::cdc::stream_def::{ + CompactionConfig, LateDataPolicy, OpFilter, RetentionConfig, StreamFormat, + }; + use crate::event::streaming_mv::StreamingMvDef; + use crate::event::streaming_mv::types::{AggDef, AggFunction}; + use crate::event::types::{EventSource, RecordPosition, RowId, WriteEvent, WriteOp}; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + use crate::wal::manager::NO_APPLY_KEY; + + const WRITES: u64 = 64; + const TENANT: u64 = 1; + const COLLECTION: &str = "orders"; + + /// What every side effect saw of one run. + #[derive(Debug, PartialEq)] + struct Totals { + audit_rows: usize, + mv_count: f64, + mv_sum: f64, + crdt_events: u64, + } + + fn register_sinks(shared: &SharedState) { + shared + .audit_dml_cache + .set(DatabaseId::DEFAULT, AuditDmlMode::Writes); + shared.stream_registry.register(ChangeStreamDef { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + name: "orders_stream".into(), + collection: COLLECTION.into(), + op_filter: OpFilter::all(), + format: StreamFormat::Json, + retention: RetentionConfig { + max_events: 10_000, + max_age_secs: 3600, + }, + compaction: CompactionConfig::default(), + webhook: crate::event::webhook::WebhookConfig::default(), + late_data: LateDataPolicy::default(), + kafka: crate::event::kafka::KafkaDeliveryConfig::default(), + owner: "admin".into(), + created_at: 0, + subscriber_roles: Vec::new(), + }); + shared.mv_registry.register(StreamingMvDef { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + name: "orders_totals".into(), + source_stream: "orders_stream".into(), + group_by_columns: Vec::new(), + aggregates: vec![ + AggDef { + output_name: "cnt".into(), + function: AggFunction::Count, + input_expr: String::new(), + }, + AggDef { + output_name: "total_sum".into(), + function: AggFunction::Sum, + input_expr: "total".into(), + }, + ], + filter_expr: None, + owner: "admin".into(), + created_at: 0, + }); + } + + fn totals(shared: &SharedState) -> Totals { + let audit_rows = shared + .audit + .lock() + .unwrap_or_else(|p| p.into_inner()) + .query_by_event(&AuditEvent::DmlAudit) + .len(); + let (mv_count, mv_sum) = shared + .mv_registry + .get_state(DatabaseId::DEFAULT, TENANT, "orders_totals") + .and_then(|state| state.read_results().into_iter().next()) + .map_or((0.0, 0.0), |(_, row)| (row[0].1, row[1].1)); + let crdt_events = shared.delta_packager.deltas_skipped.load(Ordering::Relaxed) + + shared + .delta_packager + .deltas_packaged + .load(Ordering::Relaxed); + Totals { + audit_rows, + mv_count, + mv_sum, + crdt_events, + } + } + + /// The document value of write `i`. + fn value(i: u64) -> Vec { + nodedb_types::json_to_msgpack(&serde_json::json!({ "total": i })).expect("value") + } + + /// The ring event the Data Plane emits for write `i`, logged at `lsn`. + fn ring_event(sequence: u64, i: u64, lsn: Lsn) -> WriteEvent { + WriteEvent { + sequence, + collection: Arc::from(COLLECTION), + op: WriteOp::Insert, + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key(format!("o-{i}"))), + lsn, + record: Some(RecordPosition::first(lsn)), + database_id: DatabaseId::DEFAULT, + tenant_id: TenantId::new(TENANT), + vshard_id: VShardId::new(0), + source: EventSource::User, + new_value: Some(Arc::from(value(i).as_slice())), + old_value: None, + system_time_ms: None, + valid_time_ms: None, + user_id: None, + statement_digest: None, + } + } + + /// Run `WRITES` logged writes through a consumer whose ring holds + /// `ring_capacity` events, and return what the side effects saw, the + /// number of ring drops, and the persisted watermark after shutdown. + async fn run(ring_capacity: usize) -> (Totals, u64, Lsn, Lsn) { + let dir = tempfile::tempdir().expect("tempdir"); + let (wal, watermark_store, shared, trigger_dlq, cdc_router) = + crate::event::test_utils::event_test_deps(&dir); + register_sinks(&shared); + crate::event::sink_ledger::load_sink_state(&shared, &wal, &watermark_store, 1) + .expect("load sink state"); + + let (mut producers, mut consumers) = create_event_bus_with_capacity(1, ring_capacity); + let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); + let (shutdown_bus, _) = + crate::control::shutdown::ShutdownBus::new(Arc::clone(&shared.shutdown)); + let mut handle = spawn_consumer(ConsumerConfig { + rx: consumers.remove(0), + shutdown: shutdown_rx, + shutdown_bus, + wal: Arc::clone(&wal), + watermark_store: Arc::clone(&watermark_store), + shared_state: Arc::clone(&shared), + trigger_dlq, + cdc_router, + num_cores: 1, + slab_account: Arc::new(crate::event::slab_budget::ConsumerSlabAccount::new(0)), + }); + + // Every write is logged, durable and final before its event leaves, + // as on the write path. The burst outruns a small ring, which drops events the + // consumer then rebuilds from the WAL while the ring still carries + // the rest: some events reach the consumer on both paths. + // Let the idle consumer ramp to its long poll sleep, so the burst + // below lands while it sleeps. + tokio::time::sleep(Duration::from_millis(200)).await; + let mut dropped = 0u64; + let mut last_lsn = Lsn::ZERO; + for i in 1..=WRITES { + let window = shared.outcome_floor.open_write(); + let provenance: Option = None; + let payload = zerompk::to_msgpack_vec(&( + COLLECTION, + format!("o-{i}"), + value(i), + provenance, + 0u32, + )) + .expect("put payload"); + let lsn = wal + .appender(NO_APPLY_KEY) + .append_put( + TenantId::new(TENANT), + VShardId::new(0), + DatabaseId::DEFAULT, + &payload, + ) + .expect("append put"); + // Durable before final, as group commit orders it: a record the + // floor calls final is readable by catch-up. + wal.sync().expect("wal sync"); + window.note_minted(lsn); + window.settle(); + last_lsn = lsn; + if !producers[0].emit(ring_event(i, i, lsn)) { + dropped += 1; + } + } + + let expected_crdt = WRITES; + tokio::time::timeout(Duration::from_secs(10), async { + while totals(&shared).crdt_events < expected_crdt { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("every write reaches the side effects"); + // A second delivery of any event would land within this window. + tokio::time::sleep(Duration::from_millis(300)).await; + let seen = totals(&shared); + + shutdown_tx.send(true).expect("signal consumer shutdown"); + handle.wait_for_exit().await; + let watermark = watermark_store.load(0).expect("load watermark"); + (seen, dropped, watermark, last_lsn) + } + + /// An event carried by both the ring and WAL catch-up reaches every side + /// effect once, and a dropped event reaches each of them all the same. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn catch_up_mid_run_delivers_every_event_exactly_once() { + let expected = Totals { + audit_rows: WRITES as usize, + mv_count: WRITES as f64, + mv_sum: (1..=WRITES).sum::() as f64, + crdt_events: WRITES, + }; + + let (clean, clean_drops, clean_watermark, clean_last) = run(1024).await; + assert_eq!(clean_drops, 0, "the reference run must not drop"); + assert_eq!(clean, expected); + assert_eq!(clean_watermark, clean_last); + + let (caught_up, drops, watermark, last) = run(4).await; + assert!( + drops > 0, + "the burst must overflow the ring to force catch-up" + ); + assert_eq!( + caught_up, clean, + "catch-up changed what the side effects saw" + ); + assert_eq!( + watermark, last, + "the persisted watermark must cover every delivered write" + ); + } + + /// One process of the restart test: its state, its consumer and the + /// deltas its Lite session received. + struct Node { + wal: Arc, + watermark_store: Arc, + shared: Arc, + producers: Vec, + handle: super::super::handle::ConsumerHandle, + shutdown_tx: tokio::sync::watch::Sender, + deltas: tokio::sync::mpsc::Receiver, + _control: tokio::sync::mpsc::Receiver, + } + + /// Open the node's state in `dir`, load the sink state, subscribe a Lite + /// session to the collection, and start the consumer. + fn open_node(dir: &tempfile::TempDir) -> Node { + let (wal, watermark_store, shared, trigger_dlq, cdc_router) = + crate::event::test_utils::event_test_deps(dir); + register_sinks(&shared); + crate::event::sink_ledger::load_sink_state(&shared, &wal, &watermark_store, 1) + .expect("load sink state"); + let (deltas, control) = shared.crdt_sync_delivery.register( + "lite-1".into(), + 1, + TENANT, + DatabaseId::DEFAULT, + vec![COLLECTION.into()], + &crate::event::crdt_sync::types::DeliveryConfig::default(), + ); + let (producers, mut consumers) = create_event_bus_with_capacity(1, 1024); + let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); + let (shutdown_bus, _) = + crate::control::shutdown::ShutdownBus::new(Arc::clone(&shared.shutdown)); + let handle = spawn_consumer(ConsumerConfig { + rx: consumers.remove(0), + shutdown: shutdown_rx, + shutdown_bus, + wal: Arc::clone(&wal), + watermark_store: Arc::clone(&watermark_store), + shared_state: Arc::clone(&shared), + trigger_dlq, + cdc_router, + num_cores: 1, + slab_account: Arc::new(crate::event::slab_budget::ConsumerSlabAccount::new(0)), + }); + Node { + wal, + watermark_store, + shared, + producers, + handle, + shutdown_tx, + deltas, + _control: control, + } + } + + /// Log and emit writes `range`, as the write path does. + fn write(node: &mut Node, range: std::ops::RangeInclusive) { + for i in range { + let window = node.shared.outcome_floor.open_write(); + let provenance: Option = None; + let payload = zerompk::to_msgpack_vec(&( + COLLECTION, + format!("o-{i}"), + value(i), + provenance, + 0u32, + )) + .expect("put payload"); + let lsn = node + .wal + .appender(NO_APPLY_KEY) + .append_put( + TenantId::new(TENANT), + VShardId::new(0), + DatabaseId::DEFAULT, + &payload, + ) + .expect("append put"); + node.wal.sync().expect("wal sync"); + window.note_minted(lsn); + window.settle(); + node.producers[0].emit(ring_event(i, i, lsn)); + } + } + + async fn wait_for_mv_count(shared: &SharedState, count: u64) { + tokio::time::timeout(Duration::from_secs(10), async { + while totals(shared).mv_count < count as f64 { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("the views reach the expected count"); + } + + /// DML audit rows in the durable audit log. + fn durable_audit_rows(wal: &crate::wal::WalManager) -> usize { + wal.recover_audit_entries() + .expect("recover audit entries") + .iter() + .filter(|(_, bytes)| { + zerompk::from_msgpack::(bytes) + .is_ok_and(|entry| entry.event == AuditEvent::DmlAudit) + }) + .count() + } + + /// The row of every event the change stream retains, in stream order. + fn stream_rows(shared: &SharedState) -> Vec { + let buffer = shared + .cdc_router + .get_buffer(DatabaseId::DEFAULT, TENANT, "orders_stream") + .expect("the change stream buffer"); + let mut rows: Vec<(u64, String)> = buffer + .snapshot() + .iter() + .map(|event| { + let index = event + .row_id + .trim_start_matches("o-") + .parse::() + .expect("row index"); + (index, event.row_id.clone()) + }) + .collect(); + rows.sort(); + rows.into_iter().map(|(_, row)| row).collect() + } + + fn sequences( + deltas: &mut tokio::sync::mpsc::Receiver, + ) -> Vec { + let mut out = Vec::new(); + while let Ok(delta) = deltas.try_recv() { + out.push(delta.sequence); + } + out + } + + /// A process that delivered every event but died before its watermark + /// persisted replays them all after a restart. Each sink applies each + /// event once: the views and the change streams restore the state and + /// keys they persisted, the audit log and the CRDT ledger remember what + /// they recorded. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_restart_before_the_watermark_persists_applies_every_event_once() { + const HALF: u64 = WRITES / 2; + let dir = tempfile::tempdir().expect("tempdir"); + + // First process: deliver everything, persist the views halfway, then + // die without persisting the watermark. + let mut node = open_node(&dir); + write(&mut node, 1..=HALF); + wait_for_mv_count(&node.shared, HALF).await; + node.shared + .mv_persistence + .flush_all(&node.shared.mv_registry) + .expect("persist the views"); + node.shared + .sink_ledgers + .get() + .expect("sink ledgers") + .cdc + .flush(&node.shared.cdc_router) + .expect("persist the change streams"); + write(&mut node, HALF + 1..=WRITES); + wait_for_mv_count(&node.shared, WRITES).await; + tokio::time::sleep(Duration::from_millis(200)).await; + node.handle.abort_and_join().await; + assert_eq!( + node.watermark_store.load(0).expect("load watermark"), + Lsn::ZERO, + "the first process must die before its watermark persists" + ); + // The process dies: its background tasks stop and release the state + // they hold open, as they do when a real process exits. The array GC + // task holds the array-sync op log until shutdown is signalled. + node.shared.shutdown.signal(); + tokio::time::timeout(Duration::from_secs(10), async { + while Arc::strong_count(&node.shared.array_sync_op_log) > 1 { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("the first process's background tasks exit on shutdown"); + let mut delivered = sequences(&mut node.deltas); + let shared = node.shared; + let wal = node.wal; + let watermark_store = node.watermark_store; + assert_eq!( + Arc::strong_count(&shared), + 1, + "the first process still runs" + ); + drop((wal, watermark_store, shared)); + + // Second process: the consumer replays the whole WAL above the + // watermark it never persisted. + let mut node = open_node(&dir); + wait_for_mv_count(&node.shared, WRITES).await; + tokio::time::sleep(Duration::from_millis(300)).await; + let seen = totals(&node.shared); + assert_eq!( + seen.mv_count, WRITES as f64, + "a view counted an event twice" + ); + assert_eq!(seen.mv_sum, (1..=WRITES).sum::() as f64); + assert_eq!( + durable_audit_rows(&node.wal), + WRITES as usize, + "an event was audited twice or not at all" + ); + assert_eq!( + stream_rows(&node.shared), + (1..=WRITES).map(|i| format!("o-{i}")).collect::>(), + "the change stream must hold every event once across the restart" + ); + delivered.extend(sequences(&mut node.deltas)); + assert_eq!( + delivered, + (1..=WRITES).collect::>(), + "every event must be packaged once, with sequences continuing across the restart" + ); + + node.shutdown_tx + .send(true) + .expect("signal consumer shutdown"); + node.handle.wait_for_exit().await; + } +} diff --git a/nodedb/src/event/consumer_helpers.rs b/nodedb/src/event/consumer_helpers.rs index 0c7b46a1d..56e16cce4 100644 --- a/nodedb/src/event/consumer_helpers.rs +++ b/nodedb/src/event/consumer_helpers.rs @@ -1,16 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Utility helpers for the Event Plane consumer loop. -//! -//! Extracted from `consumer.rs` to keep the main consumer module focused -//! on the state machine and dispatch orchestration. - -use std::sync::Arc; +//! Utility helpers for the Event Plane consumer loop: event metrics and +//! watermark persistence. use tracing::{trace, warn}; -use super::action::ActionRetryQueue; -use super::bus::EventConsumerRx; use super::metrics::CoreMetrics; use super::types::WriteEvent; use super::watermark::WatermarkStore; @@ -60,6 +54,7 @@ pub fn record_event(core_id: usize, event: &WriteEvent, metrics: &CoreMetrics) { /// Flush watermark to redb if the flush interval has elapsed. pub fn maybe_flush_watermark( + shared: &SharedState, store: &WatermarkStore, core_id: usize, lsn: Lsn, @@ -67,283 +62,45 @@ pub fn maybe_flush_watermark( last_flush: &mut tokio::time::Instant, ) { if *dirty && last_flush.elapsed() >= WATERMARK_FLUSH_INTERVAL { - flush_watermark(store, core_id, lsn); + flush_watermark(shared, store, core_id, lsn); *dirty = false; *last_flush = tokio::time::Instant::now(); } } -/// Persist watermark to redb (best-effort — log on failure). -pub fn flush_watermark(store: &WatermarkStore, core_id: usize, lsn: Lsn) { +/// Persist the watermark to redb. +/// +/// A restart replays the WAL above the persisted watermark, so every sink +/// whose effect is not durable on its own persists first: the streaming +/// views flush their state with their applied keys, and the change streams +/// flush their buffers with their routed keys. A failed flush keeps the +/// watermark where it was, and the next flush tries again. Once the watermark +/// is persisted, sink keys at or below it are dropped: the consumer never +/// delivers those events again. +pub fn flush_watermark(shared: &SharedState, store: &WatermarkStore, core_id: usize, lsn: Lsn) { if lsn == Lsn::ZERO { return; } - if let Err(e) = store.save(core_id, lsn) { - warn!(core_id, lsn = lsn.as_u64(), error = %e, "failed to persist watermark"); - } else { - trace!(core_id, lsn = lsn.as_u64(), "watermark flushed"); - } -} - -/// Result of draining a Normal-mode ring-buffer batch. -/// -/// A sequence gap consumes the first event beyond the gap because an SPSC -/// receiver cannot push it back. That event is deliberately returned rather -/// than included in `events`: the caller must enter WAL catchup and recover it -/// from the durable log before any side effects are allowed for it. -#[derive(Debug)] -pub enum RingDrainOutcome { - /// Every drained event was contiguous with the prior safe sequence. - Contiguous { events: Vec }, - /// `events` contains only complete WAL records before the gap record. - /// `first_gap_event` was consumed from the ring but was neither recorded - /// nor dispatched and must be recovered by WAL replay. The initial safe - /// pair is retained so callers can distinguish a rollback to the state at - /// drain entry from a retained completed-record prefix. - Gap { - events: Vec, - first_gap_event: WriteEvent, - initial_safe_lsn: Lsn, - initial_safe_sequence: u64, - }, -} - -/// Drain available contiguous events from the ring buffer (up to -/// `DRAIN_BATCH_LIMIT`). A sequence gap means the immediately preceding -/// observed WAL record may be partial even if the consumed gap event belongs -/// to a newer LSN. The contiguous suffix sharing the last observed event's -/// LSN is therefore withheld with the gap event so normal dispatch and its -/// safe watermark stop before that potentially partial record; WAL catchup -/// then replays the whole record, including every same-LSN sibling. -pub fn drain_ring_buffer( - rx: &mut EventConsumerRx, - metrics: &CoreMetrics, - core_id: usize, - last_sequence: &mut u64, - last_lsn: &mut Lsn, -) -> RingDrainOutcome { - let initial_safe_lsn = *last_lsn; - let initial_safe_sequence = *last_sequence; - let mut events: Vec = Vec::new(); - while let Some(event) = rx.try_recv() { - if *last_sequence > 0 && event.sequence > last_sequence.saturating_add(1) { - detect_sequence_gap(core_id, &event, *last_sequence, metrics); - - // A WAL record can expand to several WriteEvents. The event - // immediately before a sequence gap can be a partial record even - // when the first post-gap event has a newer LSN, so do not dispatch - // the observed suffix of its record. Replay must start before that - // record and deliver all of its siblings together. - let last_observed_lsn = events.last().map(|prior| prior.lsn); - while events - .last() - .is_some_and(|prior| Some(prior.lsn) == last_observed_lsn) - { - events.pop(); - } - if let Some(last_complete_event) = events.last() { - *last_sequence = last_complete_event.sequence; - *last_lsn = last_complete_event.lsn; - } else { - *last_sequence = initial_safe_sequence; - *last_lsn = initial_safe_lsn; - } - for contiguous_event in &events { - record_event(core_id, contiguous_event, metrics); - } - return RingDrainOutcome::Gap { - events, - first_gap_event: event, - initial_safe_lsn, - initial_safe_sequence, - }; - } - - *last_sequence = event.sequence; - if event.lsn.is_ahead_of(*last_lsn) { - *last_lsn = event.lsn; - } - - events.push(event); - if (events.len() as u32).is_multiple_of(DRAIN_BATCH_LIMIT) { - break; - } - } - for event in &events { - record_event(core_id, event, metrics); - } - RingDrainOutcome::Contiguous { events } -} - -/// Drain ring-buffer events already covered by WAL replay (`lsn <= last_lsn`), -/// returning the first event whose LSN is beyond the replay point so the -/// caller can dispatch it. -/// -/// Reconciliation is by **LSN** — the durable, monotonic key — not by -/// `sequence`: the per-core `sequence` counter resets to 0 on process -/// restart, so at boot it collides between WAL-replayed events and freshly -/// produced live events and cannot distinguish them. `last_lsn` is always a -/// completed WAL-record boundary: a ring gap rolls it back before the gap -/// record, so replay includes every same-LSN sibling before this comparison -/// drops their stale ring copies. The ring is SPSC (a `try_recv`'d event cannot -/// be pushed back), so the first non-stale event is handed back to the caller -/// rather than dropped; every event after it in the ring is also fresh (the -/// producer emits in LSN order) and is left for the Normal-mode drain. -pub fn drain_and_skip_stale(rx: &mut EventConsumerRx, last_lsn: Lsn) -> Option { - let mut skipped = 0u32; - let mut fresh = None; - while let Some(event) = rx.try_recv() { - if event.lsn.is_ahead_of(last_lsn) { - fresh = Some(event); - break; - } - skipped += 1; - } - if skipped > 0 { - trace!( - skipped, - "drained stale events from ring buffer after WAL catchup" - ); - } - fresh -} - -/// Dispatch a single WAL-catchup write event and its awaited trigger actions. -/// -/// Normal-mode batches call [`dispatch_event_actions`] directly after their -/// audit step; replay calls it through this helper. Both paths therefore share -/// DEFINE EVENT processing without a ChangeStream cursor or epoch. -pub async fn dispatch_event( - event: &WriteEvent, - shared_state: &Arc, - retry_queue: &mut ActionRetryQueue, - cdc_router: &Arc, -) { - dispatch_event_actions(event, shared_state, retry_queue).await; - shared_state - .watermark_tracker - .advance_lsn_only(event.vshard_id.as_u32(), event.lsn.as_u64()); - cdc_router.route_event(event, &shared_state.watermark_tracker); - // A grant or hierarchy row consumed here never reaches the Normal-mode - // batch, so catchup applies it to the permission cache too. - if event.op.is_data_event() { - crate::control::security::permission_tree::event_handler::handle_permission_event( - event, - &shared_state.permission_cache, - ) - .await; - } -} - -/// Dispatch the awaited trigger actions shared by normal Event Plane delivery -/// and WAL catchup. Keeping DEFINE EVENT here ensures both paths use the same -/// WAL-recoverable ordering and complete all actions before their consumer -/// watermark may be persisted. -pub async fn dispatch_event_actions( - event: &WriteEvent, - shared_state: &Arc, - retry_queue: &mut ActionRetryQueue, -) { - if !event_actions_required(event) { + if let Err(e) = shared.mv_persistence.flush_all(&shared.mv_registry) { + warn!(core_id, lsn = lsn.as_u64(), error = %e, "watermark held: streaming views not persisted"); return; } - - super::trigger::dispatcher::dispatch_triggers(event, shared_state, retry_queue).await; - crate::control::event_trigger::process_write_event( - Arc::clone(shared_state), - event, - retry_queue, - ) - .await; -} - -fn event_actions_required(event: &WriteEvent) -> bool { - event.op.is_data_event() -} - -/// Apply the non-trigger side effects of a data write event -/// (`op.is_data_event() == true`): advances the wall-time watermark, routes -/// CDC, updates the permission cache, feeds streaming MVs and CRDT sync. -/// -/// Row/statement trigger dispatch is NOT done here — it is owned exclusively -/// by [`dispatch_triggers`] (called once per event by both the Normal-mode -/// and WAL-catchup paths) so a per-row event fires its AFTER-ROW trigger -/// exactly once regardless of which path consumed it. -pub async fn accumulate_data_event( - event: &WriteEvent, - shared_state: &Arc, - cdc_router: &Arc, -) { - let event_time_ms = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap_or_default() - .as_millis() as u64; - shared_state.watermark_tracker.advance( - event.vshard_id.as_u32(), - event.lsn.as_u64(), - event_time_ms, - ); - - cdc_router.route_event(event, &shared_state.watermark_tracker); - crate::control::security::permission_tree::event_handler::handle_permission_event( - event, - &shared_state.permission_cache, - ) - .await; - let matching_streams = shared_state.stream_registry.find_matching( - event.database_id, - event.tenant_id.as_u64(), - &event.collection, - ); - for stream_def in &matching_streams { - super::streaming_mv::processor::process_write_event_for_mvs( - event, - &shared_state.mv_registry, - &stream_def.name, - ); + if let Some(ledgers) = shared.sink_ledgers.get() + && let Err(e) = ledgers.cdc.flush(&shared.cdc_router) + { + warn!(core_id, lsn = lsn.as_u64(), error = %e, "watermark held: change streams not persisted"); + return; } - shared_state - .delta_packager - .package_and_enqueue(event, &shared_state.crdt_sync_delivery); -} - -#[cfg(test)] -mod tests { - use std::sync::Arc; - - use super::event_actions_required; - use crate::event::types::{EventSource, RowId, WriteEvent, WriteOp}; - use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; - - fn event(op: WriteOp) -> WriteEvent { - WriteEvent { - sequence: 1, - collection: Arc::from("events"), - op, - row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), - lsn: Lsn::new(1), - database_id: DatabaseId::DEFAULT, - tenant_id: TenantId::new(1), - vshard_id: VShardId::new(0), - source: EventSource::User, - new_value: None, - old_value: None, - system_time_ms: None, - valid_time_ms: None, - user_id: None, - statement_digest: None, - } + if let Err(e) = store.save(core_id, lsn) { + warn!(core_id, lsn = lsn.as_u64(), error = %e, "failed to persist watermark"); + return; } - - #[test] - fn normal_and_wal_catchup_share_data_event_action_selection() { - // Normal batches and `dispatch_event` both route through - // `dispatch_event_actions`, whose selection is independent of any - // transient ChangeStream sequence or epoch. - assert!(event_actions_required(&event(WriteOp::Insert))); - assert!(event_actions_required(&event(WriteOp::BulkDelete { - count: 2 - }))); - assert!(!event_actions_required(&event(WriteOp::Heartbeat))); + trace!(core_id, lsn = lsn.as_u64(), "watermark flushed"); + let core = u32::try_from(core_id).unwrap_or(u32::MAX); + shared.mv_registry.applied().prune(core, lsn.as_u64()); + if let Some(ledgers) = shared.sink_ledgers.get() + && let Err(e) = ledgers.prune(core, lsn.as_u64()) + { + warn!(core_id, lsn = lsn.as_u64(), error = %e, "sink ledger keys not pruned"); } } diff --git a/nodedb/src/event/crdt_sync/packager.rs b/nodedb/src/event/crdt_sync/packager.rs index 41bb39f97..535653cc6 100644 --- a/nodedb/src/event/crdt_sync/packager.rs +++ b/nodedb/src/event/crdt_sync/packager.rs @@ -23,6 +23,7 @@ use tracing::trace; use super::delivery::CrdtSyncDelivery; use super::types::{DeltaOp, OutboundDelta}; +use crate::event::sink_ledger::{CrdtLedger, SinkEventKey}; use crate::event::types::{EventSource, WriteEvent, WriteOp}; /// Per-collection sequence counter for ordering enforcement. @@ -79,8 +80,19 @@ impl DeltaPackager { /// are already captured by the triggering event's delta). Events from /// `CrdtSync` are skipped (prevent echo: Lite → Origin → Lite loop). /// + /// With a `ledger`, the event's `key` and the collection's sequence are + /// recorded durably before the delta is handed on: an event the ledger + /// already holds is not packaged again, and sequences continue across a + /// restart. + /// /// Returns `true` if the delta was enqueued, `false` if skipped. - pub fn package_and_enqueue(&self, event: &WriteEvent, delivery: &CrdtSyncDelivery) -> bool { + pub fn package_and_enqueue( + &self, + event: &WriteEvent, + key: Option<&SinkEventKey>, + ledger: Option<&CrdtLedger>, + delivery: &CrdtSyncDelivery, + ) -> bool { // Only package User-originated writes. // CrdtSync events are inbound FROM Lite — don't echo back. // Trigger/RaftFollower events are derivative — the original User @@ -123,7 +135,24 @@ impl DeltaPackager { WriteOp::Heartbeat => return false, }; - let sequence = self.sequences.next(&event.collection); + let sequence = match ledger { + None => self.sequences.next(&event.collection), + Some(ledger) => match ledger.claim(key, &event.collection) { + Ok(Some(sequence)) => sequence, + // Packaged before a restart: the delta already went out. + Ok(None) => return false, + Err(error) => { + // Packaging without the ledger could hand the same event + // on twice, so the delta is not packaged. + tracing::error!( + %error, + collection = %event.collection, + "crdt ledger unavailable; outbound delta not packaged" + ); + return false; + } + }, + }; let delta = OutboundDelta { database_id: event.database_id, @@ -171,6 +200,7 @@ mod tests { op, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("o-1")), lsn: Lsn::new(100), + record: None, database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), @@ -190,10 +220,10 @@ mod tests { let delivery = CrdtSyncDelivery::new(); let crdt_event = make_event(EventSource::CrdtSync, WriteOp::Insert); - assert!(!packager.package_and_enqueue(&crdt_event, &delivery)); + assert!(!packager.package_and_enqueue(&crdt_event, None, None, &delivery)); let trigger_event = make_event(EventSource::Trigger, WriteOp::Insert); - assert!(!packager.package_and_enqueue(&trigger_event, &delivery)); + assert!(!packager.package_and_enqueue(&trigger_event, None, None, &delivery)); } #[test] @@ -202,7 +232,7 @@ mod tests { let delivery = CrdtSyncDelivery::new(); let hb = make_event(EventSource::User, WriteOp::Heartbeat); - assert!(!packager.package_and_enqueue(&hb, &delivery)); + assert!(!packager.package_and_enqueue(&hb, None, None, &delivery)); } #[test] @@ -211,7 +241,7 @@ mod tests { let delivery = CrdtSyncDelivery::new(); // No sessions registered → no subscribers. let event = make_event(EventSource::User, WriteOp::Insert); - assert!(!packager.package_and_enqueue(&event, &delivery)); + assert!(!packager.package_and_enqueue(&event, None, None, &delivery)); assert_eq!(packager.deltas_skipped.load(Ordering::Relaxed), 1); } @@ -230,7 +260,7 @@ mod tests { .as_ref() .map(|v| v.to_vec()) .unwrap_or_default(); - let _ = packager.package_and_enqueue(&event, &delivery); + let _ = packager.package_and_enqueue(&event, None, None, &delivery); assert_eq!(expected, b"payload".to_vec()); } diff --git a/nodedb/src/event/mod.rs b/nodedb/src/event/mod.rs index ca9410e5e..363695114 100644 --- a/nodedb/src/event/mod.rs +++ b/nodedb/src/event/mod.rs @@ -17,7 +17,10 @@ pub mod graph_cdc; pub mod kafka; pub mod metrics; pub mod plane; +pub mod progress; +pub mod record_numbering; pub mod scheduler; +pub mod sink_ledger; pub mod slab_budget; pub mod streaming_mv; #[cfg(test)] diff --git a/nodedb/src/event/plane.rs b/nodedb/src/event/plane.rs index ee0676fc2..02e230967 100644 --- a/nodedb/src/event/plane.rs +++ b/nodedb/src/event/plane.rs @@ -89,6 +89,24 @@ impl EventPlane { super::action::ActionRequeueInbox::for_cores(num_cores), )); let _ = shared_state.trigger_dlq.set(Arc::clone(&trigger_dlq)); + // Coverage waits for the permission step up to these counters. + if !shared_state + .authorization_fence + .install_emit_progress(consumers_rx.iter().map(|rx| rx.progress()).collect()) + { + tracing::warn!("event emit counters already installed; the Event Plane started twice"); + } + + // Every sink's durable state loads before a consumer delivers an + // event: the streaming views with their applied keys, and the audit + // and CRDT ledgers. Without it a replayed event would reach a sink a + // second time, so the node stops instead. + if let Err(error) = + super::sink_ledger::load_sink_state(&shared_state, &wal, &watermark_store, num_cores) + { + tracing::error!(%error, "event plane sink state could not be loaded; stopping"); + drop(shutdown_bus.initiate()); + } let slab_budget = Arc::new(super::slab_budget::SlabBudget::for_cores(num_cores)); let mut slab_accounts: Vec> = Vec::new(); @@ -214,11 +232,6 @@ impl EventPlane { crate::control::shutdown::LoopHandle::Async(compaction_handle), ); - // Restore streaming MV state from redb (from last shutdown). - shared_state - .mv_persistence - .restore_all(&shared_state.mv_registry); - // Spawn MV state persistence task (flush to redb every 30s). let mv_persist_handle = super::streaming_mv::persist::spawn_persist_task( Arc::clone(&shared_state.mv_persistence), @@ -423,6 +436,7 @@ mod tests { op: WriteOp::Insert, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(seq * 10), + record: None, database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/src/event/progress.rs b/nodedb/src/event/progress.rs new file mode 100644 index 000000000..dd99661b7 --- /dev/null +++ b/nodedb/src/event/progress.rs @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! How far a Data Plane core has emitted events onto its ring. +//! +//! The core numbers every event it emits with a contiguous per-core sequence, +//! and stores the highest one here after each push, whether the push succeeded +//! or the ring was full. The Control Plane reads it to learn which events a +//! barrier must wait for: every event numbered at or below a value read here +//! was emitted before the read. +//! +//! The store happens after the push. So an event whose number a reader sees +//! here is already on the ring, or was dropped from it. + +use std::sync::atomic::{AtomicU64, Ordering}; + +/// The emitted-event counter of one core. +#[derive(Debug, Default)] +pub struct CoreEmitProgress { + emitted: AtomicU64, +} + +impl CoreEmitProgress { + pub fn new() -> Self { + Self::default() + } + + /// Record that the event numbered `sequence` left the core. Called by the + /// core's producer only, after the push. + pub fn note_emitted(&self, sequence: u64) { + self.emitted.fetch_max(sequence, Ordering::Release); + } + + /// The highest sequence the core has emitted. + pub fn emitted(&self) -> u64 { + self.emitted.load(Ordering::Acquire) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_counter_keeps_the_highest_sequence() { + let progress = CoreEmitProgress::new(); + assert_eq!(progress.emitted(), 0); + progress.note_emitted(3); + progress.note_emitted(2); + assert_eq!(progress.emitted(), 3); + } +} diff --git a/nodedb/src/event/record_numbering.rs b/nodedb/src/event/record_numbering.rs new file mode 100644 index 000000000..bb34b50de --- /dev/null +++ b/nodedb/src/event/record_numbering.rs @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Number the events of one WAL record per row. +//! +//! The Data Plane producer numbers a record's events as it emits them, and +//! WAL catch-up numbers the events it rebuilds from the same record the same +//! way. The n-th event of a record that names a given row and write kind gets +//! occurrence `n - 1` on both paths, so the Event Plane can recognise a +//! rebuilt event it already delivered from the ring, and the reverse. +//! +//! A record's events leave a core back to back: a core applies one write at +//! a time, and a transaction's install sends its held events together. So +//! the numbering restarts whenever the record changes. + +use std::collections::HashMap; +use std::sync::Arc; + +use crate::types::Lsn; + +use super::types::{RowId, WriteEvent, WriteOp}; + +/// Occurrence counters of the record being numbered. +#[derive(Debug, Default)] +pub struct RecordNumbering { + lsn: Option, + seen: HashMap<(Arc, RowId, bool), u32>, +} + +impl RecordNumbering { + pub fn new() -> Self { + Self::default() + } + + /// Stamp `event` with its occurrence within its record. An event that + /// reproduces no record is left alone. + pub fn stamp(&mut self, event: &mut WriteEvent) { + let Some(position) = &mut event.record else { + return; + }; + if self.lsn != Some(position.lsn) { + self.lsn = Some(position.lsn); + self.seen.clear(); + } + let count = self + .seen + .entry(( + Arc::clone(&event.collection), + event.row_id.clone(), + is_delete(event.op), + )) + .or_insert(0); + position.occurrence = *count; + *count += 1; + } +} + +/// Whether `op` removes rows. An insert and an update of one row share a +/// kind: the ring and catch-up can tell them apart differently. +pub fn is_delete(op: WriteOp) -> bool { + match op { + WriteOp::Delete | WriteOp::BulkDelete { .. } => true, + WriteOp::Insert | WriteOp::Update | WriteOp::BulkInsert { .. } | WriteOp::Heartbeat => { + false + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::event::types::{EventSource, RecordPosition}; + use crate::types::{DatabaseId, TenantId, VShardId}; + + fn event(lsn: u64, row: &str, op: WriteOp) -> WriteEvent { + WriteEvent { + sequence: 0, + collection: Arc::from("orders"), + op, + row_id: RowId::row(nodedb_types::RowIdentity::from_user_key(row)), + lsn: Lsn::new(lsn), + record: Some(RecordPosition::first(Lsn::new(lsn))), + database_id: DatabaseId::DEFAULT, + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + source: EventSource::User, + new_value: None, + old_value: None, + system_time_ms: None, + valid_time_ms: None, + user_id: None, + statement_digest: None, + } + } + + #[test] + fn a_repeated_row_counts_up_and_a_new_record_starts_over() { + let mut numbering = RecordNumbering::new(); + let mut events = vec![ + event(9, "a", WriteOp::Insert), + event(9, "a", WriteOp::Update), + event(9, "b", WriteOp::Insert), + event(9, "a", WriteOp::Delete), + event(10, "a", WriteOp::Insert), + ]; + for event in &mut events { + numbering.stamp(event); + } + let occurrences: Vec = events + .iter() + .map(|e| e.record.map(|r| r.occurrence).unwrap_or(u32::MAX)) + .collect(); + assert_eq!(occurrences, vec![0, 1, 0, 0, 0]); + } +} diff --git a/nodedb/src/event/sink_ledger/audit.rs b/nodedb/src/event/sink_ledger/audit.rs new file mode 100644 index 000000000..63542c240 --- /dev/null +++ b/nodedb/src/event/sink_ledger/audit.rs @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which DML audit rows the durable audit log already holds. +//! +//! A DML audit row carries the key of the event it records, in its detail +//! text. The durable audit WAL is the audit sink's ledger: each row is +//! written there with its key in one append. At startup the Event Plane +//! reads the keys above each core's watermark back out of it. A replayed +//! event whose key is present was audited before the restart and is not +//! audited again. + +use std::collections::HashSet; +use std::sync::Mutex; + +use crate::control::security::audit::{AuditEntry, AuditEvent}; + +use super::key::SinkEventKey; + +/// The text that introduces a key in a DML audit row's detail. +const KEY_MARKER: &str = " key="; + +/// The detail suffix that names `key`. +pub fn detail_suffix(key: &SinkEventKey) -> String { + format!("{KEY_MARKER}{}", key.to_token()) +} + +/// The key a DML audit row's detail names, if any. +fn key_of(detail: &str) -> Option { + let at = detail.rfind(KEY_MARKER)?; + SinkEventKey::from_token(detail.get(at + KEY_MARKER.len()..)?) +} + +/// Keys of DML audit rows written above each core's watermark. +#[derive(Debug, Default)] +pub struct AuditedKeys { + keys: Mutex>, +} + +impl AuditedKeys { + /// Collect the keys of the recovered durable audit entries that lie above + /// `watermark(core)`. + pub fn from_recovered( + entries: &[(u64, Vec)], + watermark: impl Fn(u32) -> u64, + ) -> crate::Result { + let mut keys = HashSet::new(); + for (_, bytes) in entries { + let entry: AuditEntry = + zerompk::from_msgpack(bytes).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("recovered audit entry: {e}"), + })?; + if entry.event != AuditEvent::DmlAudit { + continue; + } + if let Some(key) = key_of(&entry.detail) + && key.lsn > watermark(key.core) + { + keys.insert(key); + } + } + Ok(Self { + keys: Mutex::new(keys), + }) + } + + /// Whether the durable audit log already records `key`. + pub fn contains(&self, key: &SinkEventKey) -> bool { + self.keys + .lock() + .unwrap_or_else(|p| p.into_inner()) + .contains(key) + } + + /// Drop the keys of `core` at or below `through`. + pub fn prune(&self, core: u32, through: u64) { + self.keys + .lock() + .unwrap_or_else(|p| p.into_inner()) + .retain(|key| key.core != core || key.lsn > through); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn key(core: u32, lsn: u64) -> SinkEventKey { + SinkEventKey { + core, + lsn, + occurrence: 0, + delete: false, + collection: "orders".into(), + row_kind: 0, + row: "o-1".into(), + } + } + + #[test] + fn a_detail_carries_its_key() { + let detail = format!("INSERT orders:o-1 lsn=7{}", detail_suffix(&key(1, 7))); + assert_eq!(key_of(&detail), Some(key(1, 7))); + assert_eq!(key_of("INSERT orders:o-1 lsn=7"), None); + } + + #[test] + fn keys_at_or_below_a_watermark_are_dropped() { + let audited = AuditedKeys::default(); + audited + .keys + .lock() + .expect("lock") + .extend([key(0, 5), key(0, 9), key(1, 5)]); + audited.prune(0, 5); + assert!(!audited.contains(&key(0, 5))); + assert!(audited.contains(&key(0, 9))); + assert!(audited.contains(&key(1, 5))); + } +} diff --git a/nodedb/src/event/sink_ledger/boot.rs b/nodedb/src/event/sink_ledger/boot.rs new file mode 100644 index 000000000..0bf1a280d --- /dev/null +++ b/nodedb/src/event/sink_ledger/boot.rs @@ -0,0 +1,31 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Load every sink's durable state before a consumer delivers an event. + +use std::sync::Arc; + +use crate::control::state::SharedState; +use crate::event::watermark::WatermarkStore; +use crate::wal::WalManager; + +use super::ledgers::SinkLedgers; + +/// Restore the streaming views with their applied keys and the change-stream +/// buffers with their routed keys, and install the audit, CRDT and CDC +/// ledgers. A replayed event then reaches each sink once. +pub fn load_sink_state( + shared: &SharedState, + wal: &WalManager, + watermarks: &WatermarkStore, + num_cores: usize, +) -> crate::Result<()> { + shared.mv_persistence.restore_all(&shared.mv_registry)?; + let ledgers = SinkLedgers::open(wal, watermarks, num_cores)?; + ledgers.cdc.restore_into(&shared.cdc_router)?; + if shared.sink_ledgers.set(Arc::new(ledgers)).is_err() { + return Err(crate::Error::Internal { + detail: "event plane sink ledgers were already installed".into(), + }); + } + Ok(()) +} diff --git a/nodedb/src/event/sink_ledger/cdc.rs b/nodedb/src/event/sink_ledger/cdc.rs new file mode 100644 index 000000000..8cba8f1f3 --- /dev/null +++ b/nodedb/src/event/sink_ledger/cdc.rs @@ -0,0 +1,325 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The durable ledger of the CDC change-stream buffers. +//! +//! Routing an event pushes it into the buffer of every matching change +//! stream. Consumers read those buffers, so a push is the sink's effect. The +//! ledger keeps both halves on disk in one redb file: +//! +//! - **Events:** the retained events of every change-stream buffer. +//! - **Routed keys:** the key of every event routed since the watermark. +//! +//! Every flush writes the buffer changes and the routed keys in one +//! transaction, under the lock routing takes, so the persisted buffers and +//! keys always agree. At startup the buffers are restored first. A replayed +//! event whose key was persisted is already in a restored buffer and is not +//! routed again. An event routed after the last flush is in neither, and the +//! replay routes it once. +//! +//! Durable topics hydrate their own buffers, so the ledger holds change-stream +//! buffers only. + +use std::collections::{HashMap, HashSet}; +use std::path::Path; +use std::sync::{Arc, Mutex, MutexGuard}; + +use redb::{Database, ReadableDatabase, ReadableTable, TableDefinition}; + +use crate::event::cdc::router::BufferKey; +use crate::event::cdc::{CdcEvent, CdcOffset, CdcRouter}; +use crate::event::types::WriteEvent; +use crate::event::watermark_tracker::WatermarkTracker; +use crate::types::DatabaseId; + +use super::key::SinkEventKey; + +/// Retained events, keyed by stream and position. +const EVENTS: TableDefinition<&[u8], &[u8]> = TableDefinition::new("cdc_events"); +/// Keys of the routed events. +const ROUTED: TableDefinition<&[u8], &[u8]> = TableDefinition::new("cdc_routed"); + +fn storage(detail: impl std::fmt::Display) -> crate::Error { + crate::Error::Storage { + engine: "event_plane".into(), + detail: format!("cdc ledger: {detail}"), + } +} + +fn serialization(detail: impl std::fmt::Display) -> crate::Error { + crate::Error::Serialization { + format: "json".into(), + detail: format!("cdc ledger event: {detail}"), + } +} + +/// The row prefix of one stream: database, tenant, then the length-prefixed +/// stream name. +fn stream_prefix((database_id, tenant_id, name): &BufferKey) -> Vec { + let mut out = Vec::with_capacity(20 + name.len()); + out.extend_from_slice(&database_id.as_u64().to_be_bytes()); + out.extend_from_slice(&tenant_id.to_be_bytes()); + let name_len = u32::try_from(name.len()).unwrap_or(u32::MAX); + out.extend_from_slice(&name_len.to_be_bytes()); + out.extend_from_slice(name.as_bytes()); + out +} + +fn event_row(prefix: &[u8], position: CdcOffset) -> Vec { + let mut out = Vec::with_capacity(prefix.len() + 16); + out.extend_from_slice(prefix); + out.extend_from_slice(&position.lsn.to_be_bytes()); + out.extend_from_slice(&position.sequence.to_be_bytes()); + out +} + +/// The first and last row a stream can hold. +fn stream_bounds(key: &BufferKey) -> (Vec, Vec) { + let prefix = stream_prefix(key); + let mut last = prefix.clone(); + last.extend_from_slice(&[u8::MAX; 16]); + (event_row(&prefix, CdcOffset::ZERO), last) +} + +/// The stream a row belongs to. +fn row_stream(row: &[u8]) -> Option { + let database_id = u64::from_be_bytes(row.get(0..8)?.try_into().ok()?); + let tenant_id = u64::from_be_bytes(row.get(8..16)?.try_into().ok()?); + let name_len = u32::from_be_bytes(row.get(16..20)?.try_into().ok()?) as usize; + let name = std::str::from_utf8(row.get(20..20usize.checked_add(name_len)?)?).ok()?; + Some((DatabaseId::new(database_id), tenant_id, name.to_owned())) +} + +#[derive(Default)] +struct Routed { + keys: HashSet, + /// Routed since the last flush. + pending: Vec, +} + +/// Durable record of the change-stream buffers and the events routed into +/// them. +pub struct CdcLedger { + db: Database, + routed: Mutex, +} + +impl CdcLedger { + /// Open or create the ledger at `{dir}/cdc_ledger.redb`, loading the + /// routed keys. + pub fn open(dir: &Path) -> crate::Result { + std::fs::create_dir_all(dir) + .map_err(|e| storage(format!("create {}: {e}", dir.display())))?; + let path = dir.join("cdc_ledger.redb"); + let db = Database::create(&path) + .map_err(|e| storage(format!("open {}: {e}", path.display())))?; + let txn = db.begin_write().map_err(storage)?; + let mut keys = HashSet::new(); + { + txn.open_table(EVENTS).map_err(storage)?; + let table = txn.open_table(ROUTED).map_err(storage)?; + for entry in table.iter().map_err(storage)? { + let (row, _) = entry.map_err(storage)?; + let key = SinkEventKey::from_bytes(row.value()) + .ok_or_else(|| storage("a routed key did not decode"))?; + keys.insert(key); + } + } + txn.commit().map_err(storage)?; + Ok(Self { + db, + routed: Mutex::new(Routed { + keys, + pending: Vec::new(), + }), + }) + } + + fn lock(&self) -> MutexGuard<'_, Routed> { + self.routed.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Route `event` unless an event with `key` was routed already. An event + /// without a key reproduces no WAL record and is never replayed, so it + /// always routes. + pub fn route_once( + &self, + key: Option<&SinkEventKey>, + router: &CdcRouter, + event: &WriteEvent, + watermark_tracker: &WatermarkTracker, + ) { + let mut routed = self.lock(); + if let Some(key) = key { + if !routed.keys.insert(key.clone()) { + return; + } + routed.pending.push(key.clone()); + } + router.route_event(event, watermark_tracker); + } + + /// Push every persisted event of a registered stream back into its + /// buffer. Events of a stream that no longer exists are deleted. + pub fn restore_into(&self, router: &CdcRouter) -> crate::Result<()> { + let mut orphaned: HashSet = HashSet::new(); + { + let txn = self.db.begin_read().map_err(storage)?; + let table = txn.open_table(EVENTS).map_err(storage)?; + for entry in table.iter().map_err(storage)? { + let (row, bytes) = entry.map_err(storage)?; + let stream = row_stream(row.value()) + .ok_or_else(|| storage("an event row did not decode"))?; + let (database_id, tenant_id, name) = &stream; + let Some(def) = router.registry().get(*database_id, *tenant_id, name) else { + orphaned.insert(stream); + continue; + }; + let event: CdcEvent = sonic_rs::from_slice(bytes.value()).map_err(serialization)?; + router + .ensure_buffer(*database_id, *tenant_id, name, &def.retention) + .push(event); + } + } + if orphaned.is_empty() { + return Ok(()); + } + let txn = self.db.begin_write().map_err(storage)?; + { + let mut table = txn.open_table(EVENTS).map_err(storage)?; + for stream in &orphaned { + let (first, last) = stream_bounds(stream); + table + .retain_in(first.as_slice()..=last.as_slice(), |_, _| false) + .map_err(storage)?; + } + } + txn.commit().map_err(storage) + } + + /// Persist every buffer change and routed key since the last flush, in + /// one transaction. On an error nothing is lost: the changes stay marked + /// for the next flush. + pub fn flush(&self, router: &CdcRouter) -> crate::Result<()> { + let mut routed = self.lock(); + let removed = router.take_removed(); + let changed: Vec<(BufferKey, Arc)> = router + .stream_buffers() + .into_iter() + .filter(|(_, buffer)| buffer.take_changed()) + .collect(); + if removed.is_empty() && changed.is_empty() && routed.pending.is_empty() { + return Ok(()); + } + let snapshots: Vec<(BufferKey, Vec>)> = changed + .iter() + .map(|(key, buffer)| (key.clone(), buffer.snapshot())) + .collect(); + match self.write_changes(&removed, &snapshots, &routed.pending) { + Ok(()) => { + routed.pending.clear(); + Ok(()) + } + Err(error) => { + router.requeue_removed(removed); + for (_, buffer) in &changed { + buffer.mark_changed(); + } + Err(error) + } + } + } + + fn write_changes( + &self, + removed: &[BufferKey], + snapshots: &[(BufferKey, Vec>)], + pending: &[SinkEventKey], + ) -> crate::Result<()> { + let txn = self.db.begin_write().map_err(storage)?; + { + let mut events = txn.open_table(EVENTS).map_err(storage)?; + for stream in removed { + let (first, last) = stream_bounds(stream); + events + .retain_in(first.as_slice()..=last.as_slice(), |_, _| false) + .map_err(storage)?; + } + for (stream, snapshot) in snapshots { + let prefix = stream_prefix(stream); + let retained: HashMap, &Arc> = snapshot + .iter() + .map(|event| (event_row(&prefix, event.position()), event)) + .collect(); + let mut persisted: HashSet> = HashSet::new(); + let (first, last) = stream_bounds(stream); + events + .retain_in(first.as_slice()..=last.as_slice(), |row, _| { + let keep = retained.contains_key(row); + if keep { + persisted.insert(row.to_vec()); + } + keep + }) + .map_err(storage)?; + for (row, event) in &retained { + if persisted.contains(row) { + continue; + } + let bytes = sonic_rs::to_vec(event.as_ref()).map_err(serialization)?; + events + .insert(row.as_slice(), bytes.as_slice()) + .map_err(storage)?; + } + } + let mut keys = txn.open_table(ROUTED).map_err(storage)?; + for key in pending { + keys.insert(key.to_bytes().as_slice(), [].as_slice()) + .map_err(storage)?; + } + } + txn.commit().map_err(storage) + } + + /// Drop the routed keys of `core` at or below `through`. + pub fn prune(&self, core: u32, through: u64) -> crate::Result<()> { + { + let mut routed = self.lock(); + routed + .keys + .retain(|key| key.core != core || key.lsn > through); + routed + .pending + .retain(|key| key.core != core || key.lsn > through); + } + let from = SinkEventKey::floor_bytes(core, 0); + let to = SinkEventKey::floor_bytes(core, through.saturating_add(1)); + let txn = self.db.begin_write().map_err(storage)?; + { + let mut keys = txn.open_table(ROUTED).map_err(storage)?; + keys.retain_in(from.as_slice()..to.as_slice(), |_, _| false) + .map_err(storage)?; + } + txn.commit().map_err(storage) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_row_names_its_stream_and_rows_of_a_stream_stay_in_its_bounds() { + let stream: BufferKey = (DatabaseId::new(3), 7, "orders_stream".into()); + let prefix = stream_prefix(&stream); + let row = event_row(&prefix, CdcOffset::new(42, 5)); + assert_eq!(row_stream(&row), Some(stream.clone())); + let (first, last) = stream_bounds(&stream); + assert!(first.as_slice() <= row.as_slice() && row.as_slice() <= last.as_slice()); + + let other: BufferKey = (DatabaseId::new(3), 7, "orders_stream_2".into()); + let other_row = event_row(&stream_prefix(&other), CdcOffset::new(1, 0)); + assert!( + !(first.as_slice() <= other_row.as_slice() && other_row.as_slice() <= last.as_slice()) + ); + } +} diff --git a/nodedb/src/event/sink_ledger/crdt.rs b/nodedb/src/event/sink_ledger/crdt.rs new file mode 100644 index 000000000..248ed5ce2 --- /dev/null +++ b/nodedb/src/event/sink_ledger/crdt.rs @@ -0,0 +1,144 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The durable ledger of CRDT sync packaging. +//! +//! Packaging an event assigns it the next sequence of its collection and +//! hands the delta to the connected Lite sessions. The ledger records the +//! event's key and the collection's sequence in one redb transaction before +//! the delta is handed on. An event the ledger already holds is not packaged +//! again, and the sequence continues across a restart instead of starting +//! over at one. +//! +//! Keys at or below a core's persisted watermark are pruned: the consumer +//! never delivers them again. + +use std::path::Path; + +use redb::{Database, ReadableTable, TableDefinition}; + +use super::key::SinkEventKey; + +/// Packaged event keys. +const APPLIED: TableDefinition<&[u8], &[u8]> = TableDefinition::new("crdt_packaged"); +/// Last sequence assigned per collection. +const SEQUENCES: TableDefinition<&str, u64> = TableDefinition::new("crdt_sequences"); + +fn storage(detail: impl std::fmt::Display) -> crate::Error { + crate::Error::Storage { + engine: "event_plane".into(), + detail: format!("crdt ledger: {detail}"), + } +} + +/// Durable record of packaged events and per-collection sequences. +pub struct CrdtLedger { + db: Database, +} + +impl CrdtLedger { + /// Open or create the ledger at `{dir}/crdt_ledger.redb`. + pub fn open(dir: &Path) -> crate::Result { + std::fs::create_dir_all(dir) + .map_err(|e| storage(format!("create {}: {e}", dir.display())))?; + let path = dir.join("crdt_ledger.redb"); + let db = Database::create(&path) + .map_err(|e| storage(format!("open {}: {e}", path.display())))?; + let txn = db.begin_write().map_err(storage)?; + txn.open_table(APPLIED).map_err(storage)?; + txn.open_table(SEQUENCES).map_err(storage)?; + txn.commit().map_err(storage)?; + Ok(Self { db }) + } + + /// Claim the next sequence of `collection`. With a `key`, the claim is + /// made only when the key was never packaged, and `None` means it was. + pub fn claim( + &self, + key: Option<&SinkEventKey>, + collection: &str, + ) -> crate::Result> { + let txn = self.db.begin_write().map_err(storage)?; + let sequence = { + let mut applied = txn.open_table(APPLIED).map_err(storage)?; + if let Some(key) = key { + let bytes = key.to_bytes(); + if applied.get(bytes.as_slice()).map_err(storage)?.is_some() { + return Ok(None); + } + applied + .insert(bytes.as_slice(), [].as_slice()) + .map_err(storage)?; + } + let mut sequences = txn.open_table(SEQUENCES).map_err(storage)?; + let next = sequences + .get(collection) + .map_err(storage)? + .map_or(0, |guard| guard.value()) + .saturating_add(1); + sequences.insert(collection, next).map_err(storage)?; + next + }; + txn.commit().map_err(storage)?; + Ok(Some(sequence)) + } + + /// Drop the keys of `core` at or below `through`. + pub fn prune(&self, core: u32, through: u64) -> crate::Result<()> { + let from = SinkEventKey::floor_bytes(core, 0); + let to = SinkEventKey::floor_bytes(core, through.saturating_add(1)); + let txn = self.db.begin_write().map_err(storage)?; + { + let mut applied = txn.open_table(APPLIED).map_err(storage)?; + applied + .retain_in(from.as_slice()..to.as_slice(), |_, _| false) + .map_err(storage)?; + } + txn.commit().map_err(storage) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn key(lsn: u64) -> SinkEventKey { + SinkEventKey { + core: 0, + lsn, + occurrence: 0, + delete: false, + collection: "orders".into(), + row_kind: 0, + row: format!("o-{lsn}"), + } + } + + #[test] + fn a_key_is_packaged_once_and_sequences_survive_reopen() { + let dir = tempfile::tempdir().expect("tempdir"); + { + let ledger = CrdtLedger::open(dir.path()).expect("open"); + assert_eq!( + ledger.claim(Some(&key(1)), "orders").expect("claim"), + Some(1) + ); + assert_eq!(ledger.claim(Some(&key(1)), "orders").expect("claim"), None); + assert_eq!(ledger.claim(None, "orders").expect("claim"), Some(2)); + } + let ledger = CrdtLedger::open(dir.path()).expect("reopen"); + assert_eq!(ledger.claim(Some(&key(1)), "orders").expect("claim"), None); + assert_eq!( + ledger.claim(Some(&key(2)), "orders").expect("claim"), + Some(3) + ); + + ledger.prune(0, 1).expect("prune"); + // A pruned key is below the watermark and never delivered again; the + // ledger no longer holds it. + assert_eq!( + ledger.claim(Some(&key(1)), "orders").expect("claim"), + Some(4) + ); + assert_eq!(ledger.claim(Some(&key(2)), "orders").expect("claim"), None); + } +} diff --git a/nodedb/src/event/sink_ledger/key.rs b/nodedb/src/event/sink_ledger/key.rs new file mode 100644 index 000000000..786772c7d --- /dev/null +++ b/nodedb/src/event/sink_ledger/key.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The identity a sink remembers an applied event by. +//! +//! An event that reproduces a WAL record carries its record position. The +//! position, the row, and the write kind name the event on both delivery +//! paths and across a restart, so a sink that stored this key with its +//! effect can tell a redelivery from a new event. The core that delivered +//! the event is part of the key: each core persists its own watermark, and a +//! key at or below that core's watermark is never delivered again. + +use crate::event::record_numbering::is_delete; +use crate::event::types::{RowId, WriteEvent}; + +/// How a row identity is spelled, so identities of different kinds with the +/// same text stay distinct. +fn row_kind(row: &RowId) -> u8 { + match row { + RowId::Row(_) => 0, + RowId::Batch => 1, + RowId::Edge(_) => 2, + RowId::Heartbeat => 3, + } +} + +/// The durable identity of one delivered event. +#[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord)] +pub struct SinkEventKey { + pub core: u32, + pub lsn: u64, + pub occurrence: u32, + pub delete: bool, + pub collection: String, + pub row_kind: u8, + pub row: String, +} + +impl SinkEventKey { + /// The key of `event`, delivered by core `core`. `None` for an event that + /// reproduces no WAL record: it is delivered only once, from the ring, + /// and never replayed. + pub fn of(core: usize, event: &WriteEvent) -> Option { + let record = event.record?; + Some(Self { + core: u32::try_from(core).unwrap_or(u32::MAX), + lsn: record.lsn.as_u64(), + occurrence: record.occurrence, + delete: is_delete(event.op), + collection: event.collection.to_string(), + row_kind: row_kind(&event.row_id), + row: event.row_id.as_str().to_owned(), + }) + } + + /// A byte encoding that sorts by core, then LSN. + pub fn to_bytes(&self) -> Vec { + let mut out = Vec::with_capacity(26 + self.collection.len() + self.row.len()); + out.extend_from_slice(&self.core.to_be_bytes()); + out.extend_from_slice(&self.lsn.to_be_bytes()); + out.extend_from_slice(&self.occurrence.to_be_bytes()); + out.push(u8::from(self.delete)); + out.push(self.row_kind); + let collection_len = u32::try_from(self.collection.len()).unwrap_or(u32::MAX); + out.extend_from_slice(&collection_len.to_be_bytes()); + out.extend_from_slice(self.collection.as_bytes()); + out.extend_from_slice(self.row.as_bytes()); + out + } + + /// Decode [`Self::to_bytes`]. + pub fn from_bytes(bytes: &[u8]) -> Option { + let core = u32::from_be_bytes(bytes.get(0..4)?.try_into().ok()?); + let lsn = u64::from_be_bytes(bytes.get(4..12)?.try_into().ok()?); + let occurrence = u32::from_be_bytes(bytes.get(12..16)?.try_into().ok()?); + let delete = *bytes.get(16)? != 0; + let row_kind = *bytes.get(17)?; + let collection_len = u32::from_be_bytes(bytes.get(18..22)?.try_into().ok()?) as usize; + let collection_end = 22usize.checked_add(collection_len)?; + let collection = std::str::from_utf8(bytes.get(22..collection_end)?).ok()?; + let row = std::str::from_utf8(bytes.get(collection_end..)?).ok()?; + Some(Self { + core, + lsn, + occurrence, + delete, + collection: collection.to_owned(), + row_kind, + row: row.to_owned(), + }) + } + + /// The first key of `core` above `lsn`, for range scans. + pub fn floor_bytes(core: u32, lsn: u64) -> Vec { + let mut out = Vec::with_capacity(12); + out.extend_from_slice(&core.to_be_bytes()); + out.extend_from_slice(&lsn.to_be_bytes()); + out + } + + /// A hex token of the key, for text that carries it. + pub fn to_token(&self) -> String { + self.to_bytes().iter().map(|b| format!("{b:02x}")).collect() + } + + /// Decode [`Self::to_token`]. + pub fn from_token(token: &str) -> Option { + if !token.len().is_multiple_of(2) { + return None; + } + let bytes: Option> = (0..token.len()) + .step_by(2) + .map(|i| u8::from_str_radix(token.get(i..i + 2)?, 16).ok()) + .collect(); + Self::from_bytes(&bytes?) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn key() -> SinkEventKey { + SinkEventKey { + core: 2, + lsn: 77, + occurrence: 1, + delete: true, + collection: "orders".into(), + row_kind: 0, + row: "o:1 with spaces".into(), + } + } + + #[test] + fn a_key_survives_bytes_and_tokens() { + assert_eq!(SinkEventKey::from_bytes(&key().to_bytes()), Some(key())); + assert_eq!(SinkEventKey::from_token(&key().to_token()), Some(key())); + assert_eq!(SinkEventKey::from_token("zz"), None); + } + + #[test] + fn bytes_sort_by_core_then_lsn() { + let mut later = key(); + later.lsn = 78; + assert!(key().to_bytes() < later.to_bytes()); + assert!(SinkEventKey::floor_bytes(2, 77) <= key().to_bytes()); + assert!(SinkEventKey::floor_bytes(2, 78) > key().to_bytes()); + } +} diff --git a/nodedb/src/event/sink_ledger/ledgers.rs b/nodedb/src/event/sink_ledger/ledgers.rs new file mode 100644 index 000000000..5f8612b1c --- /dev/null +++ b/nodedb/src/event/sink_ledger/ledgers.rs @@ -0,0 +1,59 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The durable ledgers of the Event Plane's non-idempotent sinks. +//! +//! - **DML audit:** the durable audit log, read back at startup into +//! [`AuditedKeys`]. +//! - **CRDT sync packaging:** [`CrdtLedger`]. +//! - **CDC change streams:** [`CdcLedger`], which holds the stream buffers +//! with the keys of the events routed into them. +//! - **Streaming materialized views:** the applied keys persisted with the +//! view state itself (see `streaming_mv::persist`). +//! +//! Each sink stores an event's key with its effect, so an event delivered +//! again after a restart is applied once. + +use crate::event::watermark::WatermarkStore; +use crate::wal::WalManager; + +use super::audit::AuditedKeys; +use super::cdc::CdcLedger; +use super::crdt::CrdtLedger; + +/// The sink ledgers of this node. +pub struct SinkLedgers { + pub audited: AuditedKeys, + pub crdt: CrdtLedger, + pub cdc: CdcLedger, +} + +impl SinkLedgers { + /// Open the ledgers beside the watermark store, reading audited keys + /// above the persisted watermark of each of `num_cores` cores. + pub fn open( + wal: &WalManager, + watermarks: &WatermarkStore, + num_cores: usize, + ) -> crate::Result { + let mut floors = Vec::with_capacity(num_cores); + for core in 0..num_cores { + floors.push(watermarks.load(core)?.as_u64()); + } + let audited = AuditedKeys::from_recovered(&wal.recover_audit_entries()?, |core| { + floors.get(core as usize).copied().unwrap_or(0) + })?; + Ok(Self { + audited, + crdt: CrdtLedger::open(watermarks.dir())?, + cdc: CdcLedger::open(watermarks.dir())?, + }) + } + + /// Drop the keys of `core` at or below `through`: the consumer persisted + /// its watermark there and never delivers them again. + pub fn prune(&self, core: u32, through: u64) -> crate::Result<()> { + self.audited.prune(core, through); + self.crdt.prune(core, through)?; + self.cdc.prune(core, through) + } +} diff --git a/nodedb/src/event/sink_ledger/mod.rs b/nodedb/src/event/sink_ledger/mod.rs new file mode 100644 index 000000000..eed2aeda8 --- /dev/null +++ b/nodedb/src/event/sink_ledger/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod audit; +pub mod boot; +pub mod cdc; +pub mod crdt; +pub mod key; +pub mod ledgers; + +pub use audit::AuditedKeys; +pub use boot::load_sink_state; +pub use cdc::CdcLedger; +pub use crdt::CrdtLedger; +pub use key::SinkEventKey; +pub use ledgers::SinkLedgers; diff --git a/nodedb/src/event/streaming_mv/applied.rs b/nodedb/src/event/streaming_mv/applied.rs new file mode 100644 index 000000000..65cfcc6b2 --- /dev/null +++ b/nodedb/src/event/streaming_mv/applied.rs @@ -0,0 +1,105 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which events the streaming materialized views applied. +//! +//! A view's aggregates are not idempotent: an event applied twice counts +//! twice. Every event is applied to every view under this lock, and its key +//! is recorded in the same critical section. A flush persists the views and +//! the keys under the same lock, in one redb transaction, so the persisted +//! views and keys always agree. After a restart, a replayed event whose key +//! was persisted is already in the restored views and is skipped. + +use std::collections::HashSet; +use std::sync::Mutex; + +use crate::event::sink_ledger::SinkEventKey; + +#[derive(Debug, Default)] +struct Applied { + keys: HashSet, + /// Bumped on every applied event, so a flush with nothing new is skipped. + generation: u64, +} + +/// Applied event keys of the streaming materialized views. +#[derive(Debug, Default)] +pub struct MvAppliedKeys { + applied: Mutex, +} + +impl MvAppliedKeys { + fn lock(&self) -> std::sync::MutexGuard<'_, Applied> { + self.applied.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Run `apply` unless the views already applied `key`. An event without a + /// key reproduces no WAL record and is delivered once, so it always + /// applies. Returns whether `apply` ran. + pub fn apply_once(&self, key: Option<&SinkEventKey>, apply: impl FnOnce()) -> bool { + let mut applied = self.lock(); + if let Some(key) = key { + if applied.keys.contains(key) { + return false; + } + applied.keys.insert(key.clone()); + } + apply(); + applied.generation = applied.generation.wrapping_add(1); + true + } + + /// Record a change to the views made outside an event, such as a time + /// bucket finalized, so the next flush persists it. + pub fn touch(&self) { + let mut applied = self.lock(); + applied.generation = applied.generation.wrapping_add(1); + } + + /// Run `persist` with the applied keys and their generation, holding off + /// every apply until it returns. + pub fn with_consistent(&self, persist: impl FnOnce(&HashSet, u64) -> R) -> R { + let applied = self.lock(); + persist(&applied.keys, applied.generation) + } + + /// Install the keys persisted with the restored views. + pub fn restore(&self, keys: impl IntoIterator) { + self.lock().keys.extend(keys); + } + + /// Drop the keys of `core` at or below `through`. + pub fn prune(&self, core: u32, through: u64) { + self.lock() + .keys + .retain(|key| key.core != core || key.lsn > through); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn key(lsn: u64) -> SinkEventKey { + SinkEventKey { + core: 0, + lsn, + occurrence: 0, + delete: false, + collection: "orders".into(), + row_kind: 0, + row: "o-1".into(), + } + } + + #[test] + fn an_event_applies_once() { + let applied = MvAppliedKeys::default(); + let mut runs = 0; + assert!(applied.apply_once(Some(&key(1)), || runs += 1)); + assert!(!applied.apply_once(Some(&key(1)), || runs += 1)); + assert!(applied.apply_once(None, || runs += 1)); + assert_eq!(runs, 2); + applied.prune(0, 1); + assert!(applied.apply_once(Some(&key(1)), || runs += 1)); + } +} diff --git a/nodedb/src/event/streaming_mv/mod.rs b/nodedb/src/event/streaming_mv/mod.rs index ece42f53f..b22dbf16b 100644 --- a/nodedb/src/event/streaming_mv/mod.rs +++ b/nodedb/src/event/streaming_mv/mod.rs @@ -1,5 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod applied; pub mod persist; pub mod processor; pub mod query; diff --git a/nodedb/src/event/streaming_mv/persist.rs b/nodedb/src/event/streaming_mv/persist.rs index 310554367..8e9e09a21 100644 --- a/nodedb/src/event/streaming_mv/persist.rs +++ b/nodedb/src/event/streaming_mv/persist.rs @@ -20,6 +20,10 @@ use crate::types::DatabaseId; /// redb table: "v2:{database_id}:{tenant_id}:{mv_name}" → MessagePack-serialized MvSnapshot. const MV_STATE: TableDefinition<&str, &[u8]> = TableDefinition::new("mv_state"); +/// redb table: the one row `APPLIED_ROW` → MessagePack list of applied +/// event keys, written in the same transaction as the view states. +const MV_APPLIED: TableDefinition<&str, &[u8]> = TableDefinition::new("mv_applied"); +const APPLIED_ROW: &str = "applied"; /// Serialized MV state: Vec of (group_key, per-aggregate GroupState list). pub type MvSnapshot = Vec<(String, Vec)>; @@ -34,6 +38,9 @@ fn state_key(database_id: DatabaseId, tenant_id: u64, mv_name: &str) -> String { /// Manages persistence of streaming MV state. pub struct MvPersistence { db: Database, + /// Generation of the applied keys last persisted. `u64::MAX` before the + /// first flush. + flushed_generation: std::sync::atomic::AtomicU64, } impl MvPersistence { @@ -62,13 +69,21 @@ impl MvPersistence { engine: "event_plane".into(), detail: format!("open_table: {e}"), })?; + txn.open_table(MV_APPLIED) + .map_err(|e| crate::Error::Storage { + engine: "event_plane".into(), + detail: format!("open_table: {e}"), + })?; txn.commit().map_err(|e| crate::Error::Storage { engine: "event_plane".into(), detail: format!("commit: {e}"), })?; } - Ok(Self { db }) + Ok(Self { + db, + flushed_generation: std::sync::atomic::AtomicU64::new(u64::MAX), + }) } /// Persist a single MV's state snapshot. @@ -178,33 +193,95 @@ impl MvPersistence { Ok(()) } - /// Flush all MV states from the registry to redb. - pub fn flush_all(&self, registry: &MvRegistry) { - for mv_def in registry.list_all() { - if let Some(state) = - registry.get_state(mv_def.database_id, mv_def.tenant_id, &mv_def.name) + /// Persist every view's state and the applied event keys in one redb + /// transaction, holding off every apply meanwhile, so the persisted views + /// and keys agree. Nothing is written when no event applied since the + /// last flush. + pub fn flush_all(&self, registry: &MvRegistry) -> crate::Result<()> { + use std::sync::atomic::Ordering; + let storage = |e: &dyn std::fmt::Display| crate::Error::Storage { + engine: "event_plane".into(), + detail: format!("mv flush: {e}"), + }; + registry.applied().with_consistent(|keys, generation| { + if self.flushed_generation.load(Ordering::Acquire) == generation { + return Ok(()); + } + let applied: Vec> = keys.iter().map(|key| key.to_bytes()).collect(); + let applied_bytes = + zerompk::to_msgpack_vec(&applied).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("mv applied keys: {e}"), + })?; + let txn = self.db.begin_write().map_err(|e| storage(&e))?; { - let snapshot = state.snapshot(); - if !snapshot.is_empty() - && let Err(e) = self.save( - mv_def.database_id, - mv_def.tenant_id, - &mv_def.name, - &snapshot, - ) - { - warn!( - mv = %mv_def.name, - error = %e, - "failed to persist MV state" - ); + let mut states = txn.open_table(MV_STATE).map_err(|e| storage(&e))?; + for mv_def in registry.list_all() { + let Some(state) = + registry.get_state(mv_def.database_id, mv_def.tenant_id, &mv_def.name) + else { + continue; + }; + let snapshot = state.snapshot(); + if snapshot.is_empty() { + continue; + } + let bytes = zerompk::to_msgpack_vec(&snapshot).map_err(|e| { + crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("mv_state: {e}"), + } + })?; + let key = state_key(mv_def.database_id, mv_def.tenant_id, &mv_def.name); + states + .insert(key.as_str(), bytes.as_slice()) + .map_err(|e| storage(&e))?; } + let mut applied_table = txn.open_table(MV_APPLIED).map_err(|e| storage(&e))?; + applied_table + .insert(APPLIED_ROW, applied_bytes.as_slice()) + .map_err(|e| storage(&e))?; } - } + txn.commit().map_err(|e| storage(&e))?; + self.flushed_generation.store(generation, Ordering::Release); + Ok(()) + }) } - /// Restore all MV states from redb into the registry. - pub fn restore_all(&self, registry: &MvRegistry) { + /// The applied event keys persisted with the view states. + fn load_applied(&self) -> crate::Result> { + let storage = |e: &dyn std::fmt::Display| crate::Error::Storage { + engine: "event_plane".into(), + detail: format!("mv applied keys: {e}"), + }; + let txn = self.db.begin_read().map_err(|e| storage(&e))?; + let table = txn.open_table(MV_APPLIED).map_err(|e| storage(&e))?; + let Some(guard) = table.get(APPLIED_ROW).map_err(|e| storage(&e))? else { + return Ok(Vec::new()); + }; + let applied: Vec> = + zerompk::from_msgpack(guard.value()).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("mv applied keys: {e}"), + })?; + applied + .iter() + .map(|bytes| { + crate::event::sink_ledger::SinkEventKey::from_bytes(bytes).ok_or_else(|| { + crate::Error::Serialization { + format: "mv applied key".into(), + detail: "a persisted key did not decode".into(), + } + }) + }) + .collect() + } + + /// Restore all MV states and their applied event keys from redb into the + /// registry. A view whose state cannot be read, or keys that cannot be + /// read, fail the restore: applying events on top of a partial restore + /// would count some of them twice. + pub fn restore_all(&self, registry: &MvRegistry) -> crate::Result<()> { let mut restored = 0u32; for mv_def in registry.list_all() { match self.load(mv_def.database_id, mv_def.tenant_id, &mv_def.name) { @@ -217,14 +294,14 @@ impl MvPersistence { } } Ok(_) => {} - Err(e) => { - warn!(mv = %mv_def.name, error = %e, "failed to restore MV state"); - } + Err(e) => return Err(e), } } if restored > 0 { info!(restored, "restored streaming MV states from redb"); } + registry.applied().restore(self.load_applied()?); + Ok(()) } } @@ -254,6 +331,7 @@ pub fn spawn_persist_task( } } if total_finalized > 0 { + registry.applied().touch(); debug!( finalized = total_finalized, cutoff_ms = cutoff, @@ -263,12 +341,16 @@ pub fn spawn_persist_task( } // Persist state to redb. - persistence.flush_all(®istry); - trace!("MV state flushed to redb"); + match persistence.flush_all(®istry) { + Ok(()) => trace!("MV state flushed to redb"), + Err(e) => warn!(error = %e, "failed to persist MV state"), + } } _ = shutdown.changed() => { if *shutdown.borrow() { - persistence.flush_all(®istry); + if let Err(e) = persistence.flush_all(®istry) { + warn!(error = %e, "failed to persist MV state on shutdown"); + } debug!("MV persistence task: final flush on shutdown"); return; } diff --git a/nodedb/src/event/streaming_mv/registry.rs b/nodedb/src/event/streaming_mv/registry.rs index 57c2770a0..3d519e33e 100644 --- a/nodedb/src/event/streaming_mv/registry.rs +++ b/nodedb/src/event/streaming_mv/registry.rs @@ -15,6 +15,8 @@ pub struct MvRegistry { defs: RwLock>, /// (database_id, tenant_id, mv_name) → live aggregate state. states: RwLock>>, + /// Keys of the events every view applied. + applied: super::applied::MvAppliedKeys, } impl MvRegistry { @@ -22,9 +24,15 @@ impl MvRegistry { Self { defs: RwLock::new(HashMap::new()), states: RwLock::new(HashMap::new()), + applied: super::applied::MvAppliedKeys::default(), } } + /// Keys of the events every view applied. + pub fn applied(&self) -> &super::applied::MvAppliedKeys { + &self.applied + } + /// Register a streaming MV and create its state. pub fn register(&self, def: StreamingMvDef) { let key = (def.database_id, def.tenant_id, def.name.clone()); diff --git a/nodedb/src/event/trigger/dispatcher/batch.rs b/nodedb/src/event/trigger/dispatcher/batch.rs index 69ac4a8c2..58d760cb8 100644 --- a/nodedb/src/event/trigger/dispatcher/batch.rs +++ b/nodedb/src/event/trigger/dispatcher/batch.rs @@ -5,7 +5,7 @@ //! //! Not currently wired into the Normal-mode consumer loop — per-event //! dispatch (`dispatch_triggers` in `single.rs`) is the sole production path -//! for AFTER-ROW trigger firing (see `event::consumer::process_normal_batch`). +//! for AFTER-ROW trigger firing (see `event::consumer::pipeline::deliver_events`). //! This batch path (and its `TriggerBatchCollector`) remains available for a //! future WHEN-clause-batched throughput optimization; for `BatchSafe` //! triggers it could dispatch a single bulk DML, but for now it still fires diff --git a/nodedb/src/event/types.rs b/nodedb/src/event/types.rs index 5c4172d8c..e321abb4e 100644 --- a/nodedb/src/event/types.rs +++ b/nodedb/src/event/types.rs @@ -130,6 +130,12 @@ pub struct WriteEvent { /// WAL LSN for this write. Enables replay from WAL on Event Plane restart. pub lsn: Lsn, + /// The WAL record this event reproduces, when the write has one. WAL + /// catch-up rebuilds the events of such a record, so the Event Plane names + /// an event by its record position and row to deliver it once. `None` for + /// a write the WAL does not carry: only its ring copy ever arrives. + pub record: Option, + /// Database context. Producers will propagate the selected database in the /// next CDC scoping slice; existing construction sites use `DEFAULT`. pub database_id: DatabaseId, @@ -178,6 +184,27 @@ pub struct WriteEvent { pub statement_digest: Option>, } +/// Where an event sits in the WAL record it reproduces. +/// +/// A record can write one row more than once (a transaction's redo). The +/// ring and WAL catch-up both number those events per row, in record order, +/// so each names the same event the same way. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct RecordPosition { + /// LSN of the WAL record. + pub lsn: Lsn, + /// How many earlier events of the same record name the same row and the + /// same kind of write (a delete, or an insert or update). + pub occurrence: u32, +} + +impl RecordPosition { + /// The first event of record `lsn` on its row. + pub fn first(lsn: Lsn) -> Self { + Self { lsn, occurrence: 0 } + } +} + /// The type of write operation that generated this event. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum WriteOp { @@ -328,6 +355,7 @@ mod tests { op: WriteOp::Insert, row_id: RowId::row(RowIdentity::from_user_key("order-1")), lsn: Lsn::new(100), + record: None, database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/src/event/wal_replay.rs b/nodedb/src/event/wal_replay.rs index 03ab26783..c59a84757 100644 --- a/nodedb/src/event/wal_replay.rs +++ b/nodedb/src/event/wal_replay.rs @@ -82,6 +82,7 @@ fn convert_records_to_events( ) -> crate::Result> { let mut events = Vec::new(); let mut sequence = base_sequence; + let mut numbering = crate::event::record_numbering::RecordNumbering::new(); // Collection tombstones shadow any prior write in the same stream. // Extract once, then drop events whose `(tenant, collection, lsn)` @@ -102,7 +103,12 @@ fn convert_records_to_events( // A single WAL record may expand to multiple WriteEvents: a // `TransactionRedo` (Calvin cross-shard commit) decomposes into one event // per write sub-op. Raw Put/Delete records still yield at most one. - for event in record_to_events(record, &mut sequence) { + // Numbered before the tombstone filter, as the producer numbered them. + let mut record_events = record_to_events(record, &mut sequence); + for event in &mut record_events { + numbering.stamp(event); + } + for event in record_events { if tombstones.is_tombstoned( record.header.database_id, event.tenant_id.as_u64(), diff --git a/nodedb/src/event/wal_replay_parse.rs b/nodedb/src/event/wal_replay_parse.rs index aa5e6e4dd..fd45ccb84 100644 --- a/nodedb/src/event/wal_replay_parse.rs +++ b/nodedb/src/event/wal_replay_parse.rs @@ -19,7 +19,7 @@ use nodedb_types::RowIdentity; use nodedb_types::sync::wire::SyncProvenance; use tracing::warn; -use crate::event::types::{EventSource, RowId, WriteEvent, WriteOp}; +use crate::event::types::{EventSource, RecordPosition, RowId, WriteEvent, WriteOp}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; /// `(op, new_value, old_value)` for a node-label CDC event — the op tag plus @@ -109,6 +109,7 @@ pub(super) fn parse_put_record( op: WriteOp::Insert, row_id: RowId::row(RowIdentity::from_user_key(key_str.into_owned())), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -134,6 +135,7 @@ pub(super) fn parse_put_record( }, row_id: RowId::Batch, lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -164,6 +166,7 @@ pub(super) fn parse_put_record( op: WriteOp::Insert, row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -190,6 +193,7 @@ pub(super) fn parse_put_record( op: WriteOp::Insert, row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -219,6 +223,7 @@ pub(super) fn parse_put_record( op: WriteOp::Insert, row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -251,6 +256,7 @@ pub(super) fn parse_put_record( op: WriteOp::Insert, row_id: RowId::edge(src_id, label, dst_id), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -316,6 +322,7 @@ pub(super) fn parse_graph_node_label_record( op, row_id: RowId::row(RowIdentity::from_user_key(node_id)), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -352,6 +359,7 @@ pub(super) fn parse_delete_record( }, row_id: RowId::Batch, lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -378,6 +386,7 @@ pub(super) fn parse_delete_record( op: WriteOp::Delete, row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -402,6 +411,7 @@ pub(super) fn parse_delete_record( op: WriteOp::Delete, row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -424,6 +434,7 @@ pub(super) fn parse_delete_record( op: WriteOp::Delete, row_id: RowId::row(RowIdentity::from_user_key(document_id)), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, @@ -452,6 +463,7 @@ pub(super) fn parse_delete_record( op: WriteOp::Delete, row_id: RowId::edge(src_id, label, dst_id), lsn, + record: Some(RecordPosition::first(lsn)), database_id, tenant_id, vshard_id, diff --git a/nodedb/src/event/watermark.rs b/nodedb/src/event/watermark.rs index 5480a5421..09ff65e3a 100644 --- a/nodedb/src/event/watermark.rs +++ b/nodedb/src/event/watermark.rs @@ -60,6 +60,12 @@ impl WatermarkStore { Ok(Self { db, path }) } + /// The directory holding this store: `{data_dir}/event_plane`. The Event + /// Plane's other durable stores live beside it. + pub fn dir(&self) -> &Path { + self.path.parent().unwrap_or(Path::new(".")) + } + /// Load the last-processed LSN for a given core. Returns `Lsn::ZERO` if /// no watermark has been persisted yet (first startup). pub fn load(&self, core_id: usize) -> crate::Result { diff --git a/nodedb/tests/inproc/cases/bitemporal_cdc.rs b/nodedb/tests/inproc/cases/bitemporal_cdc.rs index 1cd1ec7df..dd6ae25c8 100644 --- a/nodedb/tests/inproc/cases/bitemporal_cdc.rs +++ b/nodedb/tests/inproc/cases/bitemporal_cdc.rs @@ -43,6 +43,7 @@ fn write_event(seq: u64, op: WriteOp, payload_bytes: Vec, is_delete: bool) - op, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("u-1")), lsn: Lsn::new(seq * 10), + record: None, database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/tests/inproc/cases/cdc_arc_fanout.rs b/nodedb/tests/inproc/cases/cdc_arc_fanout.rs index 59443a36b..c9de3bbed 100644 --- a/nodedb/tests/inproc/cases/cdc_arc_fanout.rs +++ b/nodedb/tests/inproc/cases/cdc_arc_fanout.rs @@ -58,6 +58,7 @@ fn write_event(seq: u64) -> WriteEvent { op: WriteOp::Insert, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key(format!("r-{seq}"))), lsn: Lsn::new(seq * 10), + record: None, database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/tests/inproc/cases/event_trigger.rs b/nodedb/tests/inproc/cases/event_trigger.rs index a5878d016..5a225f64d 100644 --- a/nodedb/tests/inproc/cases/event_trigger.rs +++ b/nodedb/tests/inproc/cases/event_trigger.rs @@ -30,6 +30,7 @@ fn make_event(source: EventSource, op: WriteOp, collection: &str) -> WriteEvent op, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(100), + record: None, database_id: DatabaseId::new(7), tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), diff --git a/nodedb/tests/inproc/cases/shutdown_event_plane.rs b/nodedb/tests/inproc/cases/shutdown_event_plane.rs index 5a9d68e2a..3cd5c00fb 100644 --- a/nodedb/tests/inproc/cases/shutdown_event_plane.rs +++ b/nodedb/tests/inproc/cases/shutdown_event_plane.rs @@ -36,6 +36,7 @@ fn make_write_event(seq: u64, lsn_val: u64) -> WriteEvent { op: WriteOp::Insert, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("row-1")), lsn: Lsn::new(lsn_val), + record: None, database_id: DatabaseId::DEFAULT, tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), @@ -81,6 +82,7 @@ async fn event_plane_watermarks_persisted_through_shutdown() { ) .expect("shared_state"); let cdc_router = Arc::clone(&shared.cdc_router); + let outcome_floor = Arc::clone(&shared.outcome_floor); let shutdown = Arc::new(ShutdownWatch::new()); let (shutdown_bus, mut shutdown_handle) = ShutdownBus::new(Arc::clone(&shutdown)); @@ -98,8 +100,13 @@ async fn event_plane_watermarks_persisted_through_shutdown() { shutdown_bus: shutdown_bus.clone(), }); - // Emit 100 events with increasing LSNs. + // Emit 100 events with increasing LSNs. Each write's outcome is final + // before its event leaves, as on the write path, so the watermark (the + // consumer's safe prefix) can reach the last LSN. for i in 1u64..=100 { + let window = outcome_floor.open_write(); + window.note_minted(Lsn::new(i * 10)); + window.settle(); producers[0].emit(make_write_event(i, i * 10)); } diff --git a/nodedb/tests/inproc/cases/write_admission_fence.rs b/nodedb/tests/inproc/cases/write_admission_fence.rs index 817155ef0..0ee7cde5d 100644 --- a/nodedb/tests/inproc/cases/write_admission_fence.rs +++ b/nodedb/tests/inproc/cases/write_admission_fence.rs @@ -3,7 +3,7 @@ //! Write-admission fence tests. //! //! The fast-path point-write gate and the deterministic Calvin scheduler share -//! ONE per-vShard lock table (`SharedState::calvin_lock_managers`). These tests +//! ONE per-vShard lock table (`CalvinLocalState::lock_managers`). These tests //! drive the gate directly against that shared table to prove the fence: //! //! - A point write whose key is held by a pending commit (a normal Calvin-band @@ -54,7 +54,8 @@ fn register_lock_manager( let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection); let lm = Arc::new(Mutex::new(LockManager::new())); shared - .calvin_lock_managers + .calvin + .lock_managers .lock() .expect("lock managers") .insert(vshard.as_u32(), Arc::clone(&lm)); @@ -70,7 +71,8 @@ fn register_promotion_channel( ) -> tokio::sync::mpsc::UnboundedReceiver> { let (tx, rx) = tokio::sync::mpsc::unbounded_channel(); shared - .calvin_promotion_senders + .calvin + .promotion_senders .lock() .expect("promotion senders") .insert(vshard.as_u32(), tx); diff --git a/nodedb/tests/wire/cases/healthz_authorization_lease.rs b/nodedb/tests/wire/cases/healthz_authorization_lease.rs new file mode 100644 index 000000000..5b06624bf --- /dev/null +++ b/nodedb/tests/wire/cases/healthz_authorization_lease.rs @@ -0,0 +1,63 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A fresh single-node server boots through its first authorization lease. +//! +//! The binary runs its default single-node cluster, which is the metadata +//! leader and its only lease holder. Boot opens the gateway only once the +//! node holds a lease, so a fresh boot must reach ready, and `/healthz` +//! must report the lease valid. + +use std::time::{Duration, Instant}; + +use tokio::io::{AsyncReadExt, AsyncWriteExt}; + +use crate::harness::TestServer; + +/// Longest a fresh boot may take to reach ready. +const BOOT_BOUND: Duration = Duration::from_secs(30); + +async fn fetch_healthz(http_port: u16) -> String { + let mut stream = tokio::net::TcpStream::connect(("127.0.0.1", http_port)) + .await + .expect("connect to /healthz"); + let req = b"GET /healthz HTTP/1.1\r\nHost: localhost\r\nConnection: close\r\n\r\n"; + stream.write_all(req).await.expect("write healthz request"); + let mut response = String::new(); + stream + .read_to_string(&mut response) + .await + .expect("read healthz response"); + response +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_fresh_single_node_boot_reaches_ready_holding_a_valid_lease() { + let started = Instant::now(); + // `start()` returns only after `/healthz` answered 200. + let server = TestServer::start().await; + assert!( + started.elapsed() < BOOT_BOUND, + "a fresh boot took {:?} to reach ready", + started.elapsed() + ); + + let response = fetch_healthz(server.http_port).await; + assert!( + response.starts_with("HTTP/1.1 200"), + "/healthz must be 200 on a booted node: {response}" + ); + assert!( + response.contains("\"authorization_lease\":\"valid\""), + "/healthz must report the authorization lease valid: {response}" + ); + + // A permission-checked statement plans at once. + server + .exec("CREATE COLLECTION lease_boot_probe") + .await + .expect("CREATE COLLECTION after boot"); + server + .exec("SELECT * FROM lease_boot_probe") + .await + .expect("the first read after boot must not be refused"); +} diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 95d642049..6b9a2a835 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -92,6 +92,7 @@ mod graph_vector_write_row_level_security; mod group_by_computed_key; mod group_by_typed_columns; mod group_by_unaliased_aggregate; +mod healthz_authorization_lease; mod healthz_calvin_readiness; mod http_result_projection; mod insert_select_cross_engine; @@ -112,6 +113,9 @@ mod native_cluster_array; mod native_index_ddl_restart; mod object_literal_dml_row_level_security; mod object_literal_trailing_clause; +mod permission_tree_fence; +mod permission_tree_restart; +mod permission_tree_support; mod pg_catalog_oid_stability; mod pg_catalog_reflection; mod pg_catalog_regclass; diff --git a/nodedb/tests/wire/cases/permission_tree_fence.rs b/nodedb/tests/wire/cases/permission_tree_fence.rs new file mode 100644 index 000000000..e71c51766 --- /dev/null +++ b/nodedb/tests/wire/cases/permission_tree_fence.rs @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Compiled only with `--features failpoints`. +//! +//! A permission-tree grant or revoke is a plain write to the permission +//! table. The Event Plane's permission step applies it to the permission +//! cache. The write is acknowledged only once the cache holds it, so every +//! statement planned after the acknowledgement plans against it. +//! +//! The fail gate `permission_tree::before_apply` parks the permission step +//! while its file is absent. The test sets the tree up with the gate open, +//! removes the file, and issues a revoke: the acknowledgement must wait. +//! Writing the file releases the step, and the revoke then binds the other +//! session's very next statement. + +#[cfg(feature = "failpoints")] +use std::time::Duration; + +#[cfg(feature = "failpoints")] +use super::permission_tree_support::{ + SELECT_DOCS, connect_probe, create_tree, grant_d1, select_ids, +}; +#[cfg(feature = "failpoints")] +use crate::harness::TestServer; + +/// How long the revoke must stay unacknowledged while the step is parked. +#[cfg(feature = "failpoints")] +const PARKED_FOR: Duration = Duration::from_millis(1500); + +#[cfg(feature = "failpoints")] +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_revoke_is_acknowledged_only_after_the_permission_step_applies_it() { + let gate_dir = tempfile::tempdir().expect("gate tempdir"); + let release = gate_dir.path().join("release-permission-step"); + // The gate starts open: the file exists. + std::fs::write(&release, b"").expect("open the gate"); + let server = TestServer::start_with_failpoints(&format!( + "permission_tree::before_apply=wait_file({})", + release.display() + )) + .await; + + create_tree(&server).await; + grant_d1(&server).await; + let (first, first_handle) = connect_probe(&server).await; + let (second, second_handle) = connect_probe(&server).await; + // The grant was acknowledged, so both sessions see it at once. + assert_eq!(select_ids(&first, SELECT_DOCS).await, vec!["d1"]); + assert_eq!(select_ids(&second, SELECT_DOCS).await, vec!["d1"]); + + // Park the permission step, then revoke on a superuser connection of its + // own, so the test can watch the acknowledgement wait. + let (revoker, revoker_handle) = server + .connect_as("nodedb", "nodedb") + .await + .unwrap_or_else(|e| panic!("connect the revoker: {e}")); + std::fs::remove_file(&release).expect("park the permission step"); + let revoke = { + let client = revoker; + tokio::spawn(async move { + client + .simple_query( + "DELETE FROM pt_grants WHERE resource_id = 'd1' AND grantee = 'pt_role'", + ) + .await + .map(|_| ()) + .map_err(|e| e.to_string()) + }) + }; + tokio::time::sleep(PARKED_FOR).await; + assert!( + !revoke.is_finished(), + "the revoke was acknowledged while the permission step had not applied it" + ); + + // Release the step: the revoke is acknowledged, and binds the other + // session's next statement, cached or not. + std::fs::write(&release, b"").expect("release the permission step"); + revoke + .await + .expect("revoke task") + .unwrap_or_else(|e| panic!("revoke: {e}")); + assert_eq!( + select_ids(&second, SELECT_DOCS).await, + Vec::::new(), + "a session planned against the permission state before the acknowledged revoke" + ); + assert_eq!( + select_ids(&first, SELECT_DOCS).await, + Vec::::new(), + "a cached plan kept the revoked grant" + ); + + drop(first); + drop(second); + first_handle.abort(); + second_handle.abort(); + revoker_handle.abort(); +} diff --git a/nodedb/tests/wire/cases/permission_tree_restart.rs b/nodedb/tests/wire/cases/permission_tree_restart.rs new file mode 100644 index 000000000..17a951b07 --- /dev/null +++ b/nodedb/tests/wire/cases/permission_tree_restart.rs @@ -0,0 +1,67 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Permission-tree grants survive a restart. +//! +//! The permission cache is in-memory. Its grants and hierarchy edges live in +//! collections, and the boot sequence loads them before the gateway serves. +//! A non-superuser's first statement after the restart plans against the +//! grants in force before it: a granted row stays visible, and a revoked row +//! stays hidden. + +use super::permission_tree_support::{ + SELECT_DOCS, connect_probe, create_tree, grant_d1, revoke_d1, select_ids, +}; +use crate::harness::TestServer; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_grant_stays_visible_after_restart() { + let server = TestServer::start().await; + create_tree(&server).await; + grant_d1(&server).await; + { + let (probe, handle) = connect_probe(&server).await; + assert_eq!(select_ids(&probe, SELECT_DOCS).await, vec!["d1"]); + drop(probe); + handle.abort(); + } + + let (server, dir) = server.take_dir(); + server.graceful_shutdown().await; + let (server, _dir) = TestServer::open_on_path(dir).await; + + let (probe, handle) = connect_probe(&server).await; + assert_eq!( + select_ids(&probe, SELECT_DOCS).await, + vec!["d1"], + "the grant on d1 was lost across the restart" + ); + drop(probe); + handle.abort(); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_revoke_stays_in_force_after_restart() { + let server = TestServer::start().await; + create_tree(&server).await; + grant_d1(&server).await; + revoke_d1(&server).await; + { + let (probe, handle) = connect_probe(&server).await; + assert_eq!(select_ids(&probe, SELECT_DOCS).await, Vec::::new()); + drop(probe); + handle.abort(); + } + + let (server, dir) = server.take_dir(); + server.graceful_shutdown().await; + let (server, _dir) = TestServer::open_on_path(dir).await; + + let (probe, handle) = connect_probe(&server).await; + assert_eq!( + select_ids(&probe, SELECT_DOCS).await, + Vec::::new(), + "the revoked grant on d1 came back after the restart" + ); + drop(probe); + handle.abort(); +} diff --git a/nodedb/tests/wire/cases/permission_tree_support.rs b/nodedb/tests/wire/cases/permission_tree_support.rs new file mode 100644 index 000000000..3f37a7eea --- /dev/null +++ b/nodedb/tests/wire/cases/permission_tree_support.rs @@ -0,0 +1,91 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared setup for the permission-tree wire tests. +//! +//! The governed collection `pt_docs` holds `d1` and `d2`. The permission +//! table `pt_grants` feeds the tree. The probe user holds general read +//! access through `readwrite` and receives tree grants through the role +//! `pt_role`. A role name is the grantee because the numeric user id is +//! assigned by the catalog and the test cannot predict it. +//! +//! The probe is a non-superuser: a superuser skips the permission tree, so +//! only a non-superuser read exercises it. + +use crate::harness::TestServer; + +pub const PROBE_USER: &str = "pt_probe"; +pub const PROBE_PASSWORD: &str = "pt-probe-password-7"; + +/// The read every test issues. Its text is fixed, so a repeat on one +/// connection is served from the session plan cache. +pub const SELECT_DOCS: &str = "SELECT id FROM pt_docs ORDER BY id"; + +/// Create the governed collection, the permission table, the tree, and the +/// probe user. No grant exists yet. +pub async fn create_tree(server: &TestServer) { + for sql in [ + "CREATE COLLECTION pt_docs (id TEXT PRIMARY KEY, title TEXT) \ + WITH (engine='document_strict')", + "INSERT INTO pt_docs (id, title) VALUES ('d1', 'Doc One')", + "INSERT INTO pt_docs (id, title) VALUES ('d2', 'Doc Two')", + "CREATE COLLECTION pt_grants", + "ALTER COLLECTION pt_docs SET PERMISSION_TREE = '{\ + \"resource_column\":\"id\",\ + \"graph_index\":\"pt_docs_tree\",\ + \"permission_table\":\"pt_grants\"\ + }'", + "CREATE ROLE pt_role", + "CREATE USER pt_probe PASSWORD 'pt-probe-password-7'", + "GRANT ROLE readwrite TO pt_probe", + "GRANT ROLE pt_role TO pt_probe", + ] { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } +} + +/// Grant `pt_role` view access to `d1`. +pub async fn grant_d1(server: &TestServer) { + server + .exec( + "INSERT INTO pt_grants (resource_id, grantee, level, inherited) \ + VALUES ('d1', 'pt_role', 'viewer', false)", + ) + .await + .unwrap_or_else(|e| panic!("grant d1: {e}")); +} + +/// Revoke the grant on `d1`. +pub async fn revoke_d1(server: &TestServer) { + server + .exec("DELETE FROM pt_grants WHERE resource_id = 'd1' AND grantee = 'pt_role'") + .await + .unwrap_or_else(|e| panic!("revoke d1: {e}")); +} + +/// Open a connection as the probe user. +pub async fn connect_probe( + server: &TestServer, +) -> (tokio_postgres::Client, tokio::task::JoinHandle<()>) { + server + .connect_as(PROBE_USER, PROBE_PASSWORD) + .await + .unwrap_or_else(|e| panic!("connect as {PROBE_USER}: {e}")) +} + +/// Run `sql` on `client` and return the first column of each row. +pub async fn select_ids(client: &tokio_postgres::Client, sql: &str) -> Vec { + let messages = client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + let mut rows = Vec::new(); + for message in messages { + if let tokio_postgres::SimpleQueryMessage::Row(row) = message { + rows.push(row.get(0).unwrap_or("").to_string()); + } + } + rows +} diff --git a/nodedb/tests/wire/cases/session_plan_cache_permission_tree_revoke.rs b/nodedb/tests/wire/cases/session_plan_cache_permission_tree_revoke.rs index be9d0c489..3d855be5f 100644 --- a/nodedb/tests/wire/cases/session_plan_cache_permission_tree_revoke.rs +++ b/nodedb/tests/wire/cases/session_plan_cache_permission_tree_revoke.rs @@ -3,43 +3,29 @@ //! Regression coverage: a revoked permission-tree grant must not keep being //! served from a stale session plan cache entry on the same connection. //! -//! Unlike an RLS policy write (synchronous — the policy store bumps its -//! tenant version inside the DDL handler), a permission-tree grant has no -//! synchronous SQL surface. It lands as a plain `INSERT`/`DELETE` on the -//! collection named `permission_table` in the tree definition, and the -//! in-memory `PermissionCache` is updated asynchronously off `WriteEvent`s -//! consumed by the Event Plane (`control/security/permission_tree/event_handler.rs`). -//! That asynchrony is why this test polls: the grant/revoke DML returns as -//! soon as it is WAL/Raft-durable, before CDC has necessarily applied it to -//! the cache. +//! A permission-tree grant has no dedicated SQL surface. It lands as a plain +//! `INSERT`/`DELETE` on the collection named `permission_table` in the tree +//! definition. The Event Plane applies it to the in-memory `PermissionCache`, +//! and the write is acknowledged only once the cache holds it. So the +//! statement right after an acknowledged grant or revoke plans against it, +//! with no polling. //! -//! The poll uses a FRESH, differently-worded `SELECT` every iteration (a -//! trivially-true extra predicate makes the SQL text unique) so it always -//! misses the session plan cache and reads the live `PermissionCache` on -//! every attempt — this establishes ground truth for "has CDC applied the -//! write yet" without touching the cache path under test. The actual -//! assertion then reuses one FIXED statement text, issued repeatedly on one -//! connection, so a cache hit is the only way it can be served: that is what -//! exercises `DescriptorVersionSet::permission_tree_version` re-validation -//! in `PlanCache::get`. +//! The ground-truth read uses SQL text unique to it (a trivially-true extra +//! predicate), so it always misses the session plan cache. The actual +//! assertion reuses one FIXED statement text on one connection, so a cache +//! hit is the only way it can be served: that is what exercises +//! `DescriptorVersionSet::permission_tree_version` re-validation in +//! `PlanCache::get`. //! //! The probing identity is a non-superuser: a superuser produces no //! `PermCtx` at all (`inject_permission_tree` returns early for one), so a //! superuser-issued read cannot exercise this path no matter what the cache //! does (`control/planner/rls_injection/permission_tree/plan.rs`). -use std::time::Duration; - use crate::harness::TestServer; const PASSWORD: &str = "perm-tree-cache-probe-19"; -/// How long to wait for asynchronous CDC application to the permission -/// cache, and separately for the session-cache eviction it drives. Generous -/// — every poll returns as soon as its condition holds, so a high ceiling -/// costs nothing on the passing path. -const CDC_TIMEOUT: Duration = Duration::from_secs(30); - /// Run `sql` on `client` and return the first column of each row. async fn select_ids(client: &tokio_postgres::Client, sql: &str) -> Vec { let messages = client @@ -55,37 +41,16 @@ async fn select_ids(client: &tokio_postgres::Client, sql: &str) -> Vec { rows } -/// Poll with SQL text unique to each attempt (an always-true extra predicate) -/// until the live permission-tree state yields exactly `expected`, or panic -/// once `timeout` elapses. Never reuses one statement text across attempts, -/// so this always replans from the current `PermissionCache` and never -/// depends on — or pollutes — the session plan cache the real assertion -/// below exercises. -async fn wait_for_live_visibility( - probe: &tokio_postgres::Client, - tag: &str, - expected: &[&str], - timeout: Duration, -) { - let deadline = tokio::time::Instant::now() + timeout; - let mut attempt: u64 = 0; - loop { - attempt += 1; - let sql = format!( - "SELECT id FROM perm_tree_docs WHERE '{tag}-{attempt}' = '{tag}-{attempt}' \ - ORDER BY id" - ); - let got = select_ids(probe, &sql).await; - if got == expected { - return; - } - assert!( - tokio::time::Instant::now() < deadline, - "timed out waiting for live permission-tree state to reach {expected:?}, \ - last observed {got:?}" - ); - tokio::time::sleep(Duration::from_millis(100)).await; - } +/// Read with SQL text unique to `tag` (an always-true extra predicate), so +/// the read plans from the current `PermissionCache` and never touches the +/// session plan cache the real assertion below exercises. +async fn assert_live_visibility(probe: &tokio_postgres::Client, tag: &str, expected: &[&str]) { + let sql = format!("SELECT id FROM perm_tree_docs WHERE '{tag}' = '{tag}' ORDER BY id"); + assert_eq!( + select_ids(probe, &sql).await, + expected, + "live permission-tree state after the acknowledged write" + ); } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] @@ -171,9 +136,8 @@ async fn revoked_permission_tree_grant_is_not_served_from_stale_session_plan_cac .await .unwrap_or_else(|e| panic!("connect as perm_tree_probe: {e}")); - // Ground truth: wait for CDC to apply the grant, via SQL text that can - // never hit the session plan cache. - wait_for_live_visibility(&probe, "grant-landed", &["d1"], CDC_TIMEOUT).await; + // Ground truth, via SQL text that can never hit the session plan cache. + assert_live_visibility(&probe, "grant-landed", &["d1"]).await; let select = "SELECT id FROM perm_tree_docs ORDER BY id"; @@ -197,9 +161,8 @@ async fn revoked_permission_tree_grant_is_not_served_from_stale_session_plan_cac .await .unwrap(); - // Ground truth again: wait for CDC to apply the revoke, independent of - // the cached statement below. - wait_for_live_visibility(&probe, "revoke-landed", &[], CDC_TIMEOUT).await; + // Ground truth again, independent of the cached statement below. + assert_live_visibility(&probe, "revoke-landed", &[]).await; // SAME probe connection, SAME statement text, issued exactly once: must // reflect the revoke rather than replaying the plan cached before it. @@ -214,8 +177,7 @@ async fn revoked_permission_tree_grant_is_not_served_from_stale_session_plan_cac ); // And once more, on the same connection, to confirm the revoke keeps - // applying rather than only taking effect for the one query issued - // immediately after CDC caught up. + // applying. let after_again = select_ids(&probe, select).await; assert_eq!(after_again, after); From 6572818f22ae02a9ce5720c2de77fde281a963c8 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 10:51:06 +0800 Subject: [PATCH 34/64] feat(security): refuse a role assignment or drop that breaks a role rule A statement could assign a user a custom role that does not exist in its tenant, or drop a custom role while a user still held it or another role inherited from it, silently leaving the catalog in an inconsistent state. Add role_assignment::{check_assignable, check_droppable} and run them at every DDL entry point (CREATE/ALTER USER, CREATE ROLE, DROP ROLE, GRANT ROLE, service accounts, API keys, OIDC claim mappings, and external JWT/OIDC logins) before proposing, each seeing the committed catalog overlaid with its own transaction's buffered role and user entries. The metadata applier runs the same check at each entry's log position and reports ApplyOutcome::Refused for an entry that breaks a rule there, so a racing change that committed first is refused identically on every node instead of diverging the catalog. A transaction's COMMIT replays its whole batch of buffered role and user entries against the committed catalog before writing anything, so DROP ROLE after CREATE USER ... ROLE in the same batch is refused before any of it applies. Add role_checks::{visible_user, visible_roles} so a DDL handler sees a user or role created earlier in its own transaction and not one already dropped in it, and move user-record construction into a dedicated credential/store/user_builders module shared by the new and existing builders. --- .../control/catalog_entry/apply/dispatch.rs | 41 ++- nodedb/src/control/catalog_entry/apply/mod.rs | 2 + .../control/catalog_entry/apply/outcome.rs | 26 ++ .../src/control/catalog_entry/apply/role.rs | 45 ++- .../src/control/catalog_entry/apply/user.rs | 23 +- nodedb/src/control/catalog_entry/mod.rs | 1 + .../src/control/catalog_entry/role_rules.rs | 74 +++++ .../cluster/metadata_applier/catalog_ddl.rs | 7 +- .../control/security/credential/store/crud.rs | 25 -- .../control/security/credential/store/mod.rs | 1 + .../security/credential/store/replication.rs | 173 +++--------- .../credential/store/user_builders.rs | 206 ++++++++++++++ .../src/control/security/jwt_policy/gate.rs | 19 +- nodedb/src/control/security/jwt_policy/mod.rs | 2 + .../src/control/security/jwt_policy/roles.rs | 149 ++++++++++ nodedb/src/control/security/mod.rs | 1 + nodedb/src/control/security/oidc/verify.rs | 2 +- nodedb/src/control/security/role.rs | 79 ++++-- .../src/control/security/role_assignment.rs | 257 ++++++++++++++++++ .../shared/ddl/neutral/apikey/create.rs | 9 +- .../server/shared/ddl/neutral/apikey/parse.rs | 9 +- .../ddl/neutral/grant/database_permission.rs | 11 +- .../shared/ddl/neutral/grant/permission.rs | 6 +- .../server/shared/ddl/neutral/grant/role.rs | 42 ++- .../control/server/shared/ddl/neutral/mod.rs | 1 + .../control/server/shared/ddl/neutral/oidc.rs | 26 +- .../control/server/shared/ddl/neutral/role.rs | 90 ++++-- .../server/shared/ddl/neutral/role_checks.rs | 248 +++++++++++++++++ .../shared/ddl/neutral/service_account.rs | 105 ++++--- .../server/shared/ddl/neutral/user/alter.rs | 42 ++- .../server/shared/ddl/neutral/user/create.rs | 53 ++-- .../server/shared/ddl/neutral/user/drop.rs | 14 +- .../server/shared/session/ddl_flush.rs | 25 +- nodedb/src/error/types.rs | 12 + nodedb/src/error_classify.rs | 8 + nodedb/tests/inproc/cases/mod.rs | 1 + .../inproc/cases/role_assignment_rules.rs | 177 ++++++++++++ nodedb/tests/wire/cases/mod.rs | 2 + .../wire/cases/role_assignment_transaction.rs | 109 ++++++++ nodedb/tests/wire/cases/user_transaction.rs | 147 ++++++++++ 40 files changed, 1909 insertions(+), 361 deletions(-) create mode 100644 nodedb/src/control/catalog_entry/apply/outcome.rs create mode 100644 nodedb/src/control/catalog_entry/role_rules.rs create mode 100644 nodedb/src/control/security/credential/store/user_builders.rs create mode 100644 nodedb/src/control/security/jwt_policy/roles.rs create mode 100644 nodedb/src/control/security/role_assignment.rs create mode 100644 nodedb/src/control/server/shared/ddl/neutral/role_checks.rs create mode 100644 nodedb/tests/inproc/cases/role_assignment_rules.rs create mode 100644 nodedb/tests/wire/cases/role_assignment_transaction.rs create mode 100644 nodedb/tests/wire/cases/user_transaction.rs diff --git a/nodedb/src/control/catalog_entry/apply/dispatch.rs b/nodedb/src/control/catalog_entry/apply/dispatch.rs index 5a75212b6..b89cbb37e 100644 --- a/nodedb/src/control/catalog_entry/apply/dispatch.rs +++ b/nodedb/src/control/catalog_entry/apply/dispatch.rs @@ -8,6 +8,8 @@ use crate::control::catalog_entry::entry::CatalogEntry; use crate::control::security::catalog::SystemCatalog; use crate::control::security::catalog::types::CheckpointDoc; +use super::outcome::ApplyOutcome; + use super::{ alert_rule, api_key, auth_user, change_stream, checkpoint, collection, column_stats, consumer_group, continuous_aggregate, custom_type, database, function, index_registry, @@ -20,22 +22,34 @@ use super::{ /// Apply `entry` to `catalog`. /// /// A failed catalog write raises: skipping a committed metadata entry -/// diverges this node from the quorum. `Ok(false)` reports that the entry -/// wrote nothing, which still concludes its DDL. Debug builds verify +/// diverges this node from the quorum. [`ApplyOutcome::Unchanged`] reports +/// that the entry wrote nothing, which still concludes its DDL. +/// [`ApplyOutcome::Refused`] reports an entry that breaks a role rule at its +/// log position; every node refuses it alike. Debug builds verify /// referential integrity after every apply — release-gated because a full /// rescan would wedge `raft_tick_loop` on a node with a pre-existing orphan. -pub fn apply_to(entry: &CatalogEntry, catalog: &SystemCatalog) -> Result { - let applied = match entry { +pub fn apply_to( + entry: &CatalogEntry, + catalog: &SystemCatalog, +) -> Result { + let outcome = match entry { CatalogEntry::PutTenantWithAdmin { tenant, admin } => { - tenant::put_with_admin(tenant, admin, catalog)? + if tenant::put_with_admin(tenant, admin, catalog)? { + ApplyOutcome::Applied + } else { + ApplyOutcome::Unchanged + } } + CatalogEntry::PutUser(stored) => user::put(stored, catalog)?, + CatalogEntry::PutRole(stored) => role::put(stored, catalog)?, + CatalogEntry::DeleteRole { name } => role::delete(name, catalog)?, _ => { apply_to_inner(entry, catalog)?; - true + ApplyOutcome::Applied } }; - if !applied { - return Ok(false); + if !outcome.wrote() { + return Ok(outcome); } #[cfg(debug_assertions)] { @@ -62,7 +76,7 @@ pub fn apply_to(entry: &CatalogEntry, catalog: &SystemCatalog) -> Result crate::Result<()> { @@ -144,10 +158,13 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul tenant_id, name, } => change_stream::delete(*database_id, *tenant_id, name, catalog), - CatalogEntry::PutUser(stored) => user::put(stored, catalog), + // Applied by `apply_to`, which reports a refused entry. + CatalogEntry::PutUser(_) => Ok(()), CatalogEntry::DropUser { username } => user::delete(username, catalog), - CatalogEntry::PutRole(stored) => role::put(stored, catalog), - CatalogEntry::DeleteRole { name } => role::delete(name, catalog), + // Applied by `apply_to`, which reports a refused entry. + CatalogEntry::PutRole(_) => Ok(()), + // Applied by `apply_to`, which reports a refused entry. + CatalogEntry::DeleteRole { .. } => Ok(()), CatalogEntry::PutApiKey(stored) => api_key::put(stored, catalog), CatalogEntry::RevokeApiKey { key_id } => api_key::revoke(key_id, catalog), CatalogEntry::PutAuthUser(stored) => auth_user::put(stored, catalog), diff --git a/nodedb/src/control/catalog_entry/apply/mod.rs b/nodedb/src/control/catalog_entry/apply/mod.rs index 24d244877..2a2380265 100644 --- a/nodedb/src/control/catalog_entry/apply/mod.rs +++ b/nodedb/src/control/catalog_entry/apply/mod.rs @@ -23,6 +23,7 @@ pub mod index_registry; pub mod local; pub mod materialized_view; pub mod oidc_provider; +pub mod outcome; pub mod owner; pub mod permission; pub mod procedure; @@ -45,3 +46,4 @@ pub mod vector; pub mod wal_tombstone; pub use dispatch::apply_to; +pub use outcome::ApplyOutcome; diff --git a/nodedb/src/control/catalog_entry/apply/outcome.rs b/nodedb/src/control/catalog_entry/apply/outcome.rs new file mode 100644 index 000000000..ce88e89af --- /dev/null +++ b/nodedb/src/control/catalog_entry/apply/outcome.rs @@ -0,0 +1,26 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What applying one catalog entry did. + +use crate::control::security::role_assignment::RoleRefusal; + +/// The result of applying a committed catalog entry. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum ApplyOutcome { + /// The entry was written. Its post-apply side effects run. + Applied, + /// The entry wrote nothing and still concludes its DDL, such as an + /// if-absent create for a descriptor that exists. No side effects run. + Unchanged, + /// The entry breaks a role rule at its log position, and is skipped. + /// Every node applies the same log in the same order, so every node + /// skips it alike. No side effects run. + Refused(RoleRefusal), +} + +impl ApplyOutcome { + /// Whether the entry was written, so its side effects must run. + pub fn wrote(&self) -> bool { + matches!(self, Self::Applied) + } +} diff --git a/nodedb/src/control/catalog_entry/apply/role.rs b/nodedb/src/control/catalog_entry/apply/role.rs index 482625fd9..8d2710d42 100644 --- a/nodedb/src/control/catalog_entry/apply/role.rs +++ b/nodedb/src/control/catalog_entry/apply/role.rs @@ -1,17 +1,54 @@ // SPDX-License-Identifier: BUSL-1.1 //! Apply Role catalog entries to `SystemCatalog` redb. +//! +//! Both entries check the role rules against the catalog at their log +//! position. The statement checked before proposing; a change that +//! committed in between is caught here, on every node alike. use crate::control::security::catalog::{StoredRole, SystemCatalog, catalog_err}; +use crate::control::security::role_assignment::{self, RoleRefusal}; -pub fn put(stored: &StoredRole, catalog: &SystemCatalog) -> crate::Result<()> { +use super::outcome::ApplyOutcome; + +/// Write the role, unless its inheritance parent is neither built in nor +/// defined. +pub fn put(stored: &StoredRole, catalog: &SystemCatalog) -> crate::Result { + let parent = stored.parent.as_str(); + if !parent.is_empty() && !role_assignment::is_builtin_role_name(parent) { + let roles = catalog.load_all_roles()?; + if !roles.iter().any(|role| role.name == parent) { + let refusal = RoleRefusal::Undefined { + name: parent.to_string(), + }; + tracing::warn!( + role = %stored.name, + %refusal, + "catalog_entry: role entry refused; its parent is undefined" + ); + return Ok(ApplyOutcome::Refused(refusal)); + } + } catalog .put_role(stored) - .map_err(|e| catalog_err(&format!("put_role '{}'", stored.name), e)) + .map_err(|e| catalog_err(&format!("put_role '{}'", stored.name), e))?; + Ok(ApplyOutcome::Applied) } -pub fn delete(name: &str, catalog: &SystemCatalog) -> crate::Result<()> { +/// Delete the role, unless a user holds it or a role inherits from it. +pub fn delete(name: &str, catalog: &SystemCatalog) -> crate::Result { + let users = catalog.load_all_users()?; + let roles = catalog.load_all_roles()?; + if let Err(refusal) = role_assignment::check_stored_drop(name, &users, &roles) { + tracing::warn!( + role = %name, + %refusal, + "catalog_entry: role drop refused; users or roles still depend on it" + ); + return Ok(ApplyOutcome::Refused(refusal)); + } catalog .delete_role(name) - .map_err(|e| catalog_err(&format!("delete_role '{name}'"), e)) + .map_err(|e| catalog_err(&format!("delete_role '{name}'"), e))?; + Ok(ApplyOutcome::Applied) } diff --git a/nodedb/src/control/catalog_entry/apply/user.rs b/nodedb/src/control/catalog_entry/apply/user.rs index 7b47f8323..91adcb9cf 100644 --- a/nodedb/src/control/catalog_entry/apply/user.rs +++ b/nodedb/src/control/catalog_entry/apply/user.rs @@ -3,11 +3,30 @@ //! Apply User catalog entries to `SystemCatalog` redb. use crate::control::security::catalog::{StoredUser, SystemCatalog, catalog_err}; +use crate::control::security::role_assignment; -pub fn put(stored: &StoredUser, catalog: &SystemCatalog) -> crate::Result<()> { +use super::outcome::ApplyOutcome; + +/// Write the user, unless an active user names a role that is neither built +/// in nor defined in its tenant at this log position. The statement checked +/// before proposing; a role dropped after that check and before this entry +/// committed is caught here, on every node alike. +pub fn put(stored: &StoredUser, catalog: &SystemCatalog) -> crate::Result { + if stored.is_active { + let roles = catalog.load_all_roles()?; + if let Err(refusal) = role_assignment::check_stored_user(stored, &roles) { + tracing::warn!( + user = %stored.username, + %refusal, + "catalog_entry: user entry refused; it names an undefined role" + ); + return Ok(ApplyOutcome::Refused(refusal)); + } + } catalog .put_user(stored) - .map_err(|e| catalog_err(&format!("put_user '{}'", stored.username), e)) + .map_err(|e| catalog_err(&format!("put_user '{}'", stored.username), e))?; + Ok(ApplyOutcome::Applied) } /// Fully remove the user record from redb. `delete_user` is idempotent — a diff --git a/nodedb/src/control/catalog_entry/mod.rs b/nodedb/src/control/catalog_entry/mod.rs index 5e6f8420f..e175ee82a 100644 --- a/nodedb/src/control/catalog_entry/mod.rs +++ b/nodedb/src/control/catalog_entry/mod.rs @@ -39,6 +39,7 @@ pub mod entry; pub mod kind; pub mod persist_collection; pub mod post_apply; +pub mod role_rules; pub use codec::{decode, encode}; pub use entry::CatalogEntry; diff --git a/nodedb/src/control/catalog_entry/role_rules.rs b/nodedb/src/control/catalog_entry/role_rules.rs new file mode 100644 index 000000000..e6a1b2d76 --- /dev/null +++ b/nodedb/src/control/catalog_entry/role_rules.rs @@ -0,0 +1,74 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Role rules over a batch of catalog entries, in order. +//! +//! A transaction buffers its DDL and applies it at COMMIT. One statement can +//! break a role rule for another in the same batch: `DROP ROLE r` after +//! `CREATE USER u ROLE r`, for example. Each statement's own check saw only +//! the state before the batch. This check replays the batch's role and user +//! entries over the committed catalog, in order, and refuses the whole batch +//! before anything is written, so a COMMIT never applies part of a batch. + +use std::collections::HashMap; + +use crate::control::security::catalog::{StoredRole, StoredUser, SystemCatalog}; +use crate::control::security::role_assignment::{self, RoleRefusal}; + +use super::entry::CatalogEntry; + +/// Check every role and user entry of `entries` against the committed +/// catalog and the entries before it. +pub fn check_batch<'a>( + entries: impl IntoIterator, + catalog: &SystemCatalog, +) -> crate::Result<()> { + let mut users: HashMap = catalog + .load_all_users()? + .into_iter() + .map(|user| (user.username.clone(), user)) + .collect(); + let mut roles: HashMap = catalog + .load_all_roles()? + .into_iter() + .map(|role| (role.name.clone(), role)) + .collect(); + for entry in entries { + match entry { + CatalogEntry::PutUser(user) => { + if user.is_active { + let defined: Vec = roles.values().cloned().collect(); + role_assignment::check_stored_user(user, &defined)?; + } + users.insert(user.username.clone(), (**user).clone()); + } + CatalogEntry::PutTenantWithAdmin { admin, .. } => { + users.insert(admin.username.clone(), (**admin).clone()); + } + CatalogEntry::DropUser { username } => { + users.remove(username); + } + CatalogEntry::PutRole(role) => { + let parent = role.parent.as_str(); + if !parent.is_empty() + && !role_assignment::is_builtin_role_name(parent) + && !roles.contains_key(parent) + { + return Err(RoleRefusal::Undefined { + name: parent.to_string(), + } + .into()); + } + roles.insert(role.name.clone(), (**role).clone()); + } + CatalogEntry::DeleteRole { name } => { + let held: Vec = users.values().cloned().collect(); + let defined: Vec = roles.values().cloned().collect(); + role_assignment::check_stored_drop(name, &held, &defined)?; + roles.remove(name); + } + // Every other entry leaves users and roles as they are. + _ => {} + } + } + Ok(()) +} diff --git a/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs b/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs index a1e95c75b..b7e919308 100644 --- a/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs +++ b/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs @@ -107,9 +107,12 @@ impl MetadataCommitApplier { } debug!(kind = stamped.kind(), "catalog_entry: applying to redb"); - if !catalog_entry::apply::apply_to(&stamped, catalog)? { + let outcome = catalog_entry::apply::apply_to(&stamped, catalog)?; + if !outcome.wrote() { // A `Put*` that wrote nothing (e.g. an if-absent create for a - // descriptor that already exists) still concludes its DDL. + // descriptor that already exists) still concludes its DDL. So + // does a refused entry: every node refuses it at this position, + // and the proposer reports the refusal to its client. self.clear_implicit_drain(&stamped); return Ok(()); } diff --git a/nodedb/src/control/security/credential/store/crud.rs b/nodedb/src/control/security/credential/store/crud.rs index 69ce91aa9..34b795cf1 100644 --- a/nodedb/src/control/security/credential/store/crud.rs +++ b/nodedb/src/control/security/credential/store/crud.rs @@ -293,31 +293,6 @@ impl CredentialStore { Ok(()) } - /// Replace the `accessible_databases` list on a service account. - /// - /// Requires the caller to have already verified superuser authority. - /// For non-service-account users, returns an error. - pub fn set_service_account_databases( - &self, - name: &str, - databases: Vec, - ) -> crate::Result<()> { - let mut users = write_lock(&self.users); - let record = users - .get_mut(name) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("service account '{name}' not found"), - })?; - if !record.is_service_account { - return Err(crate::Error::BadRequest { - detail: format!("'{name}' is a user, not a service account"), - }); - } - record.accessible_databases = databases; - self.commit_user_mutation(record, Some(SessionInvalidationReason::RoleAltered))?; - Ok(()) - } - /// Remove a role from a user. Triggers `RoleRevoked` soft-revoke on /// open sessions. pub fn remove_role(&self, username: &str, role: &Role) -> crate::Result<()> { diff --git a/nodedb/src/control/security/credential/store/mod.rs b/nodedb/src/control/security/credential/store/mod.rs index c94a29b60..a330c7116 100644 --- a/nodedb/src/control/security/credential/store/mod.rs +++ b/nodedb/src/control/security/credential/store/mod.rs @@ -6,6 +6,7 @@ pub mod core; pub mod crud; pub mod list; pub mod replication; +pub mod user_builders; pub use auth::{AuthRejection, PasswordVerification, ScramCredentials, ScramLookup}; pub use core::CredentialStore; diff --git a/nodedb/src/control/security/credential/store/replication.rs b/nodedb/src/control/security/credential/store/replication.rs index a72525c00..069529063 100644 --- a/nodedb/src/control/security/credential/store/replication.rs +++ b/nodedb/src/control/security/credential/store/replication.rs @@ -27,14 +27,8 @@ use crate::types::TenantId; use super::super::super::catalog::StoredUser; use super::super::super::identity::Role; -use super::super::super::time::now_secs; -use super::super::hash::{ - compute_scram_salted_password, generate_scram_salt, hash_password_argon2, -}; use super::super::record::UserRecord; -use super::core::{ - CredentialStore, PasswordPrincipal, read_lock, validate_password_assignment, write_lock, -}; +use super::core::{CredentialStore, read_lock, write_lock}; impl CredentialStore { /// Build a `StoredUser` ready for replication via @@ -49,116 +43,62 @@ impl CredentialStore { tenant_id: TenantId, roles: Vec, ) -> crate::Result { - { - let users = read_lock(&self.users); - if users.contains_key(username) { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' already exists"), - }); - } + if read_lock(&self.users).contains_key(username) { + return Err(crate::Error::BadRequest { + detail: format!("user '{username}' already exists"), + }); } - validate_password_assignment(password, PasswordPrincipal::New)?; + self.prepare_new_user(username, password, tenant_id, roles) + } - let salt = generate_scram_salt(); - let scram_salted_password = compute_scram_salted_password(password, &salt); - let password_hash = hash_password_argon2(password, &self.argon2_config)?; - let user_id = self.alloc_user_id()?; - let is_superuser = roles.contains(&Role::Superuser); - let now = now_secs(); + /// Build a service-account `StoredUser` ready for replication via + /// `CatalogEntry::PutUser`. + pub fn prepare_service_account( + &self, + name: &str, + tenant_id: TenantId, + roles: Vec, + accessible_databases: Vec, + ) -> crate::Result { + if read_lock(&self.users).contains_key(name) { + return Err(crate::Error::BadRequest { + detail: format!("user or service account '{name}' already exists"), + }); + } + self.prepare_new_service_account(name, tenant_id, roles, accessible_databases) + } - Ok(StoredUser { - user_id, - username: username.to_string(), - tenant_id: tenant_id.as_u64(), - password_hash, - scram_salt: salt, - scram_salted_password, - roles: roles.iter().map(|r| r.to_string()).collect(), - is_superuser, - is_active: true, - is_service_account: false, - created_at: now, - updated_at: now, - password_expires_at: self.compute_expiry(), - must_change_password: false, - password_changed_at: now, - default_database_id: 0, - accessible_databases: vec![], - }) + /// Build the updated `StoredUser` of committed service account `name` + /// restricted to `databases`. + pub fn prepare_service_account_databases( + &self, + name: &str, + databases: Vec, + ) -> crate::Result { + let base = self.existing_active(name)?; + self.prepare_service_account_databases_from(base, databases) } - /// Build an updated `StoredUser` from an existing user with - /// specific fields replaced. Used by `ALTER USER SET PASSWORD` - /// and `ALTER USER SET ROLE`. Returns the updated record - /// ready for propose. + /// Build an updated `StoredUser` from committed user `username` with a + /// new password and/or role set. pub fn prepare_user_update( &self, username: &str, new_password: Option<&str>, new_roles: Option>, ) -> crate::Result { - let users = read_lock(&self.users); - let existing = users - .get(username) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("user '{username}' not found"), - })?; - if !existing.is_active { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' is inactive"), - }); - } - if let Some(password) = new_password { - validate_password_assignment( - password, - PasswordPrincipal::Existing { - is_service_account: existing.is_service_account, - }, - )?; - } - let mut stored = existing.to_stored(); - drop(users); - - if let Some(pw) = new_password { - let salt = generate_scram_salt(); - stored.scram_salted_password = compute_scram_salted_password(pw, &salt); - stored.scram_salt = salt; - stored.password_hash = hash_password_argon2(pw, &self.argon2_config)?; - stored.password_expires_at = self.compute_expiry(); - stored.must_change_password = false; - stored.password_changed_at = now_secs(); - } - if let Some(roles) = new_roles { - stored.is_superuser = roles.contains(&Role::Superuser); - stored.roles = roles.iter().map(|r| r.to_string()).collect(); - } - stored.updated_at = now_secs(); - Ok(stored) + let base = self.existing_active(username)?; + self.prepare_user_update_from(base, new_password, new_roles) } /// Build an updated `StoredUser` that sets `must_change_password`. - /// Used by `ALTER USER MUST CHANGE PASSWORD`. pub fn prepare_set_must_change_password( &self, username: &str, required: bool, ) -> crate::Result { - let users = read_lock(&self.users); - let existing = users - .get(username) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("user '{username}' not found"), - })?; - if !existing.is_active { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' is inactive"), - }); - } - let mut stored = existing.to_stored(); - drop(users); - stored.must_change_password = required; - stored.updated_at = now_secs(); - Ok(stored) + let base = self.existing_active(username)?; + Ok(self.prepare_set_must_change_password_from(base, required)) } /// Build an updated `StoredUser` that sets `password_expires_at`. @@ -168,47 +108,18 @@ impl CredentialStore { username: &str, expires_at: u64, ) -> crate::Result { - let users = read_lock(&self.users); - let existing = users - .get(username) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("user '{username}' not found"), - })?; - if !existing.is_active { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' is inactive"), - }); - } - let mut stored = existing.to_stored(); - drop(users); - stored.password_expires_at = expires_at; - stored.updated_at = now_secs(); - Ok(stored) + let base = self.existing_active(username)?; + Ok(self.prepare_set_password_expires_at_from(base, expires_at)) } /// Build an updated `StoredUser` that sets `default_database_id`. - /// Used by `ALTER USER SET DEFAULT DATABASE `. pub fn prepare_set_default_database( &self, username: &str, database_id: u64, ) -> crate::Result { - let users = read_lock(&self.users); - let existing = users - .get(username) - .ok_or_else(|| crate::Error::BadRequest { - detail: format!("user '{username}' not found"), - })?; - if !existing.is_active { - return Err(crate::Error::BadRequest { - detail: format!("user '{username}' is inactive"), - }); - } - let mut stored = existing.to_stored(); - drop(users); - stored.default_database_id = database_id; - stored.updated_at = now_secs(); - Ok(stored) + let base = self.existing_active(username)?; + Ok(self.prepare_set_default_database_from(base, database_id)) } /// Install a replicated `StoredUser` into the in-memory cache and diff --git a/nodedb/src/control/security/credential/store/user_builders.rs b/nodedb/src/control/security/credential/store/user_builders.rs new file mode 100644 index 000000000..74a17867a --- /dev/null +++ b/nodedb/src/control/security/credential/store/user_builders.rs @@ -0,0 +1,206 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Build the `StoredUser` a user-mutating statement proposes. +//! +//! Each builder takes the record it starts from rather than looking it up. +//! A statement inside a transaction starts from the user as the transaction +//! sees it: a user it created earlier is visible, and a user it dropped is +//! not. Outside a transaction that is the committed record, which +//! [`CredentialStore::stored_user`] returns. The builders touch neither the +//! in-memory map nor redb: the applier installs the record after commit. + +use crate::types::TenantId; + +use super::super::super::catalog::StoredUser; +use super::super::super::identity::Role; +use super::super::super::time::now_secs; +use super::super::hash::{ + compute_scram_salted_password, generate_scram_salt, hash_password_argon2, +}; +use super::core::{CredentialStore, PasswordPrincipal, read_lock, validate_password_assignment}; + +impl CredentialStore { + /// The committed record of active user `username`. + pub fn stored_user(&self, username: &str) -> Option { + let users = read_lock(&self.users); + users + .get(username) + .filter(|user| user.is_active) + .map(|user| user.to_stored()) + } + + /// Build a new password user. Allocates a user_id and hashes the password + /// (Argon2 + SCRAM salt). The caller has checked the name is free. + pub fn prepare_new_user( + &self, + username: &str, + password: &str, + tenant_id: TenantId, + roles: Vec, + ) -> crate::Result { + validate_password_assignment(password, PasswordPrincipal::New)?; + + let salt = generate_scram_salt(); + let scram_salted_password = compute_scram_salted_password(password, &salt); + let password_hash = hash_password_argon2(password, &self.argon2_config)?; + let user_id = self.alloc_user_id()?; + let is_superuser = roles.contains(&Role::Superuser); + let now = now_secs(); + + Ok(StoredUser { + user_id, + username: username.to_string(), + tenant_id: tenant_id.as_u64(), + password_hash, + scram_salt: salt, + scram_salted_password, + roles: roles.iter().map(|r| r.to_string()).collect(), + is_superuser, + is_active: true, + is_service_account: false, + created_at: now, + updated_at: now, + password_expires_at: self.compute_expiry(), + must_change_password: false, + password_changed_at: now, + default_database_id: 0, + accessible_databases: vec![], + }) + } + + /// Build a new service account. It authenticates by API key only, so it + /// carries no password hash. The caller has checked the name is free. + pub fn prepare_new_service_account( + &self, + name: &str, + tenant_id: TenantId, + roles: Vec, + accessible_databases: Vec, + ) -> crate::Result { + let user_id = self.alloc_user_id()?; + let now = now_secs(); + Ok(StoredUser { + user_id, + username: name.to_string(), + tenant_id: tenant_id.as_u64(), + password_hash: String::new(), + scram_salt: Vec::new(), + scram_salted_password: Vec::new(), + is_superuser: roles.contains(&Role::Superuser), + roles: roles.iter().map(|r| r.to_string()).collect(), + is_active: true, + is_service_account: true, + created_at: now, + updated_at: now, + password_expires_at: 0, + must_change_password: false, + password_changed_at: now, + default_database_id: 0, + accessible_databases: accessible_databases + .iter() + .map(|database_id| database_id.as_u64()) + .collect(), + }) + } + + /// `base` with a new password and/or role set. Used by `ALTER USER SET + /// PASSWORD`, `ALTER USER SET ROLE`, and `GRANT` / `REVOKE` of roles. + pub fn prepare_user_update_from( + &self, + mut base: StoredUser, + new_password: Option<&str>, + new_roles: Option>, + ) -> crate::Result { + if let Some(password) = new_password { + validate_password_assignment( + password, + PasswordPrincipal::Existing { + is_service_account: base.is_service_account, + }, + )?; + let salt = generate_scram_salt(); + base.scram_salted_password = compute_scram_salted_password(password, &salt); + base.scram_salt = salt; + base.password_hash = hash_password_argon2(password, &self.argon2_config)?; + base.password_expires_at = self.compute_expiry(); + base.must_change_password = false; + base.password_changed_at = now_secs(); + } + if let Some(roles) = new_roles { + base.is_superuser = roles.contains(&Role::Superuser); + base.roles = roles.iter().map(|r| r.to_string()).collect(); + } + base.updated_at = now_secs(); + Ok(base) + } + + /// `base` with `must_change_password` set to `required`. + pub fn prepare_set_must_change_password_from( + &self, + mut base: StoredUser, + required: bool, + ) -> StoredUser { + base.must_change_password = required; + base.updated_at = now_secs(); + base + } + + /// `base` with `password_expires_at`. `0` means "NEVER EXPIRES". + pub fn prepare_set_password_expires_at_from( + &self, + mut base: StoredUser, + expires_at: u64, + ) -> StoredUser { + base.password_expires_at = expires_at; + base.updated_at = now_secs(); + base + } + + /// `base` with `default_database_id`. + pub fn prepare_set_default_database_from( + &self, + mut base: StoredUser, + database_id: u64, + ) -> StoredUser { + base.default_database_id = database_id; + base.updated_at = now_secs(); + base + } + + /// Service account `base` restricted to `databases`. Refuses a password + /// user. + pub fn prepare_service_account_databases_from( + &self, + mut base: StoredUser, + databases: Vec, + ) -> crate::Result { + if !base.is_service_account { + return Err(crate::Error::BadRequest { + detail: format!("'{}' is a user, not a service account", base.username), + }); + } + base.accessible_databases = databases + .iter() + .map(|database_id| database_id.as_u64()) + .collect(); + base.updated_at = now_secs(); + Ok(base) + } + + /// The committed active record of `username`, or the error a + /// username-addressed builder reports. + pub(super) fn existing_active(&self, username: &str) -> crate::Result { + let users = read_lock(&self.users); + let existing = users + .get(username) + .ok_or_else(|| crate::Error::BadRequest { + detail: format!("user '{username}' not found"), + })?; + if !existing.is_active { + return Err(crate::Error::BadRequest { + detail: format!("user '{username}' is inactive"), + }); + } + Ok(existing.to_stored()) + } +} diff --git a/nodedb/src/control/security/jwt_policy/gate.rs b/nodedb/src/control/security/jwt_policy/gate.rs index 192a29a79..259937194 100644 --- a/nodedb/src/control/security/jwt_policy/gate.rs +++ b/nodedb/src/control/security/jwt_policy/gate.rs @@ -10,28 +10,33 @@ //! [`SharedState`] carries, so it hangs off the two call sites that own one: //! the HTTP bearer path and the native/OIDC bearer path. +use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::jwt::JwtClaims; use crate::control::state::SharedState; -use crate::types::TenantId; -use super::{provisioning, scopes}; +use super::{provisioning, roles, scopes}; /// Apply the state-dependent half of the JWT policy to a verified token. /// -/// `tenant_id` is the tenant the *identity* was bound to by its provider — -/// never a tenant asserted by the token's claims. +/// `identity` is the identity the token bound to: its tenant is the one its +/// provider is bound to — never a tenant asserted by the token's claims — +/// and its roles are the resolved ones the session will hold. /// -/// A deployment with no JWKS registry has no JWT authentication at all, so -/// there is no policy to apply and the gate is a no-op. +/// An identity holding a custom role not defined in its tenant is refused +/// first, before any record is provisioned. A deployment with no JWKS +/// registry has no JWT authentication at all, so there is no other policy to +/// apply. pub fn enforce_stateful_jwt_policy( state: &SharedState, claims: &JwtClaims, - tenant_id: TenantId, + identity: &AuthenticatedIdentity, ) -> crate::Result<()> { + roles::refuse_undefined_roles(state, identity)?; let Some(registry) = state.jwks_registry.as_ref() else { return Ok(()); }; let config = registry.jwt_config(); + let tenant_id = identity.tenant_id; scopes::enforce_declared_scopes(config.enforce_scopes, &state.scope_defs, claims, tenant_id)?; provisioning::provision_and_check_status( diff --git a/nodedb/src/control/security/jwt_policy/mod.rs b/nodedb/src/control/security/jwt_policy/mod.rs index 4c0c56b5b..4ab535fe8 100644 --- a/nodedb/src/control/security/jwt_policy/mod.rs +++ b/nodedb/src/control/security/jwt_policy/mod.rs @@ -4,6 +4,7 @@ pub mod claim_path; pub mod gate; pub mod provisioning; pub mod remap; +pub mod roles; pub mod scopes; pub mod status; @@ -11,5 +12,6 @@ pub use claim_path::{resolve_claim, string_list}; pub use gate::enforce_stateful_jwt_policy; pub use provisioning::provision_and_check_status; pub use remap::{REMAPPABLE_FIELDS, remap_claims, validate_claim_remap}; +pub use roles::refuse_undefined_roles; pub use scopes::enforce_declared_scopes; pub use status::check_blocked_status; diff --git a/nodedb/src/control/security/jwt_policy/roles.rs b/nodedb/src/control/security/jwt_policy/roles.rs new file mode 100644 index 000000000..abf93e64c --- /dev/null +++ b/nodedb/src/control/security/jwt_policy/roles.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Refuse an external identity that resolved to an undefined custom role. +//! +//! A JWT or OIDC login takes its roles from the identity provider: the +//! token's claims, remapped by `[auth.jwt]`, or the provider's claim-mapping +//! rules. A resolved name that is not built in parses to a custom role, and a +//! custom role grants something only while it is defined in the tenant the +//! provider is bound to. An identity holding an undefined role would hold +//! nothing, and a just-in-time provisioned record would store it. So the +//! login is refused with a typed error naming the role, and the refusal is +//! recorded in the audit log, before anything is provisioned. + +use crate::control::security::audit::AuditEvent; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::security::role_assignment::{self, RoleRefusal}; +use crate::control::state::SharedState; + +/// Refuse `identity` when any of its roles is neither built in nor defined +/// in its tenant. +pub fn refuse_undefined_roles( + state: &SharedState, + identity: &AuthenticatedIdentity, +) -> crate::Result<()> { + let tenant_id = identity.tenant_id; + let checked = role_assignment::check_assignable(&identity.roles, tenant_id.as_u64(), |name| { + state + .roles + .get_role(name) + .map(|role| role.tenant_id.as_u64()) + }); + let Err(refusal) = checked else { + return Ok(()); + }; + let role = match refusal { + RoleRefusal::Undefined { name } => name, + other => other.to_string(), + }; + state.audit_record( + AuditEvent::AuthFailure, + Some(tenant_id), + &identity.username, + &format!("external login refused: role \"{role}\" is not defined in tenant {tenant_id}"), + ); + Err(crate::Error::ExternalRoleUndefined { + subject: identity.username.clone(), + role, + tenant_id: tenant_id.as_u64(), + }) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::control::security::identity::{AuthMethod, Role}; + use crate::types::TenantId; + use crate::wal::WalManager; + + fn state(dir: &tempfile::TempDir) -> Arc { + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("roles.wal")).expect("open WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + SharedState::new(dispatcher, wal).expect("shared state") + } + + fn external(tenant: u64, roles: Vec) -> AuthenticatedIdentity { + AuthenticatedIdentity::new_regular( + 7, + "idp-alice", + TenantId::new(tenant), + AuthMethod::OidcBearer, + roles, + None, + AuthenticatedIdentity::default_database_set(false), + ) + } + + fn auth_failures(state: &SharedState) -> Vec { + state + .audit + .lock() + .unwrap_or_else(|p| p.into_inner()) + .query_by_event(&AuditEvent::AuthFailure) + .into_iter() + .map(|entry| entry.detail.clone()) + .collect() + } + + #[test] + fn an_undefined_claimed_role_refuses_the_login_and_is_audited() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = state(&dir); + let identity = external(5, vec![Role::ReadOnly, Role::Custom("ghost".into())]); + + match refuse_undefined_roles(&state, &identity) { + Err(crate::Error::ExternalRoleUndefined { + subject, + role, + tenant_id, + }) => { + assert_eq!(subject, "idp-alice"); + assert_eq!(role, "ghost"); + assert_eq!(tenant_id, 5); + } + other => panic!("expected ExternalRoleUndefined, got {other:?}"), + } + let failures = auth_failures(&state); + assert!( + failures.iter().any(|detail| detail.contains("\"ghost\"")), + "the refusal must be audited: {failures:?}" + ); + } + + #[test] + fn a_defined_role_of_the_bound_tenant_is_accepted() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = state(&dir); + state + .roles + .create_role("analyst", TenantId::new(5), None, None) + .expect("create role"); + + refuse_undefined_roles( + &state, + &external(5, vec![Role::ReadWrite, Role::Custom("analyst".into())]), + ) + .expect("a defined role is accepted"); + assert!(auth_failures(&state).is_empty()); + } + + #[test] + fn a_role_defined_only_in_another_tenant_is_refused() { + let dir = tempfile::tempdir().expect("tempdir"); + let state = state(&dir); + state + .roles + .create_role("analyst", TenantId::new(5), None, None) + .expect("create role"); + + assert!(matches!( + refuse_undefined_roles(&state, &external(6, vec![Role::Custom("analyst".into())])), + Err(crate::Error::ExternalRoleUndefined { .. }) + )); + } +} diff --git a/nodedb/src/control/security/mod.rs b/nodedb/src/control/security/mod.rs index 7ae4ea187..87c455765 100644 --- a/nodedb/src/control/security/mod.rs +++ b/nodedb/src/control/security/mod.rs @@ -43,6 +43,7 @@ pub mod request_scope; pub mod risk; pub mod rls; pub mod role; +pub mod role_assignment; pub mod scope; pub mod session_handle; pub mod session_registry; diff --git a/nodedb/src/control/security/oidc/verify.rs b/nodedb/src/control/security/oidc/verify.rs index f953b2aa0..5188cd385 100644 --- a/nodedb/src/control/security/oidc/verify.rs +++ b/nodedb/src/control/security/oidc/verify.rs @@ -157,7 +157,7 @@ pub async fn verify_bearer_token( crate::control::security::jwt_policy::enforce_stateful_jwt_policy( state, verified_claims, - identity.tenant_id, + &identity, )?; Ok((identity, verified)) diff --git a/nodedb/src/control/security/role.rs b/nodedb/src/control/security/role.rs index 11d59feec..c6bf16d08 100644 --- a/nodedb/src/control/security/role.rs +++ b/nodedb/src/control/security/role.rs @@ -126,30 +126,7 @@ impl RoleStore { tenant_id: TenantId, parent: Option<&str>, ) -> crate::Result { - if is_builtin(name) { - return Err(crate::Error::BadRequest { - detail: format!("'{name}' is a built-in role and cannot be created"), - }); - } - let roles = self.roles.read(); - if roles.contains_key(name) { - return Err(crate::Error::BadRequest { - detail: format!("role '{name}' already exists"), - }); - } - if let Some(parent_name) = parent { - validate_parent(name, parent_name, &roles)?; - } - let now = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap_or_default() - .as_secs(); - Ok(StoredRole { - name: name.to_string(), - tenant_id: tenant_id.as_u64(), - parent: parent.unwrap_or("").to_string(), - created_at: now, - }) + prepare_role_against(name, tenant_id, parent, &self.roles.read()) } /// Create a custom role. Returns error if it already exists or would create a cycle. @@ -309,6 +286,52 @@ impl RoleStore { } } +/// Build a `StoredRole` ready for replication via `CatalogEntry::PutRole`, +/// validated against `roles`: the custom roles the creating statement sees. +/// Inside a transaction those include the roles it created earlier. Rejects a +/// built-in name, a duplicate, an undefined parent, and a parent that would +/// close a cycle or exceed [`MAX_ROLE_INHERITANCE_DEPTH`]. +pub fn prepare_role_against( + name: &str, + tenant_id: TenantId, + parent: Option<&str>, + roles: &HashMap, +) -> crate::Result { + if is_builtin(name) { + return Err(crate::Error::BadRequest { + detail: format!("'{name}' is a built-in role and cannot be created"), + }); + } + if roles.contains_key(name) { + return Err(crate::Error::BadRequest { + detail: format!("role '{name}' already exists"), + }); + } + if let Some(parent_name) = parent { + validate_parent(name, parent_name, roles)?; + } + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or_default() + .as_secs(); + Ok(StoredRole { + name: name.to_string(), + tenant_id: tenant_id.as_u64(), + parent: parent.unwrap_or("").to_string(), + created_at: now, + }) +} + +/// [`RoleStore::check_inheritance_cycle`] against `roles`, the custom roles +/// the altering statement sees. +pub fn check_inheritance_cycle_against( + role_name: &str, + parent: &str, + roles: &HashMap, +) -> crate::Result<()> { + check_inheritance_chain(role_name, parent, roles) +} + /// Walk the inheritance chain starting from `start_name` upward through the /// given `roles` map. Returns the chain length (number of hops including /// `start_name` itself). If the chain is a cycle or exceeds @@ -375,11 +398,11 @@ fn validate_parent( check_inheritance_chain(child_name, parent_name, roles) } +/// Every name the role parser maps to a built-in role, so a custom role can +/// never shadow one: `cluster_admin` and the `database_*:{id}` forms as well +/// as the five tenant roles. fn is_builtin(name: &str) -> bool { - matches!( - name, - "superuser" | "tenant_admin" | "readwrite" | "readonly" | "monitor" - ) + super::role_assignment::is_builtin_role_name(name) } #[cfg(test)] diff --git a/nodedb/src/control/security/role_assignment.rs b/nodedb/src/control/security/role_assignment.rs new file mode 100644 index 000000000..c09bead8f --- /dev/null +++ b/nodedb/src/control/security/role_assignment.rs @@ -0,0 +1,257 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which role names a user can hold, and when a custom role can be dropped. +//! +//! A role name parses to a built-in [`Role`] or to [`Role::Custom`]. A custom +//! name is only a name: it grants something only once a custom role of that +//! name is defined in the user's tenant. So a user may hold a custom role only +//! while it is defined there, and a custom role may be dropped only while no +//! user holds it and no role inherits from it, as PostgreSQL refuses to drop +//! a role that objects still depend on. +//! +//! The same rules run in two places, each against its own view of the +//! committed catalog: +//! +//! - **Proposal:** the DDL handler checks the in-memory role and user stores, +//! and refuses the statement before anything is proposed. +//! - **Apply:** the metadata applier checks the redb catalog at the entry's +//! log position. Every node applies the same log in the same order, so a +//! `PutUser` or `DeleteRole` that raced another change is skipped on every +//! node alike, and a replayed log never produces a user holding an +//! undefined role. + +use crate::control::security::catalog::{StoredRole, StoredUser}; + +use super::identity::Role; + +/// Why a role assignment or a role drop is refused. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum RoleRefusal { + /// The role is neither built in nor defined in the user's tenant. + #[error("role \"{name}\" does not exist")] + Undefined { name: String }, + /// Users still hold the role. + #[error( + "role \"{name}\" cannot be dropped because users still hold it: {}", + .users.join(", ") + )] + HeldByUsers { name: String, users: Vec }, + /// Other roles inherit from the role. + #[error( + "role \"{name}\" cannot be dropped because other roles inherit from it: {}", + .children.join(", ") + )] + InheritedBy { name: String, children: Vec }, +} + +impl RoleRefusal { + /// The SQLSTATE PostgreSQL gives for the same refusal. + pub fn sqlstate(&self) -> &'static str { + match self { + Self::Undefined { .. } => "42704", + Self::HeldByUsers { .. } | Self::InheritedBy { .. } => "2BP01", + } + } +} + +impl From for crate::Error { + fn from(refusal: RoleRefusal) -> Self { + match refusal { + RoleRefusal::Undefined { name } => crate::Error::UndefinedObject { kind: "role", name }, + other => crate::Error::BadRequest { + detail: other.to_string(), + }, + } + } +} + +/// Parse a role name. Unknown names become [`Role::Custom`]. +pub fn parse_role_name(name: &str) -> Role { + match name.parse() { + Ok(role) => role, + Err(e) => match e {}, + } +} + +/// Whether `name` names a built-in role, so it can never be a custom role. +pub fn is_builtin_role_name(name: &str) -> bool { + !matches!(parse_role_name(name), Role::Custom(_)) +} + +/// Check that a user of `tenant_id` can hold every role in `roles`. +/// +/// `custom_tenant` returns the tenant a custom role of the given name is +/// defined in, or `None` when no such role is defined. +pub fn check_assignable<'a>( + roles: impl IntoIterator, + tenant_id: u64, + custom_tenant: impl Fn(&str) -> Option, +) -> Result<(), RoleRefusal> { + for role in roles { + if let Role::Custom(name) = role + && custom_tenant(name) != Some(tenant_id) + { + return Err(RoleRefusal::Undefined { name: name.clone() }); + } + } + Ok(()) +} + +/// Check that the custom role `name` can be dropped: no user in `users` +/// holds it, and no role in `roles` inherits from it. +pub fn check_droppable<'a>( + name: &str, + users: impl IntoIterator, + roles: impl IntoIterator, +) -> Result<(), RoleRefusal> { + let mut holders: Vec = users + .into_iter() + .filter(|(_, held)| held.iter().any(|role| role == name)) + .map(|(user, _)| user.to_string()) + .collect(); + if !holders.is_empty() { + holders.sort(); + return Err(RoleRefusal::HeldByUsers { + name: name.to_string(), + users: holders, + }); + } + let mut children: Vec = roles + .into_iter() + .filter(|(_, parent)| *parent == name) + .map(|(child, _)| child.to_string()) + .collect(); + if !children.is_empty() { + children.sort(); + return Err(RoleRefusal::InheritedBy { + name: name.to_string(), + children, + }); + } + Ok(()) +} + +/// [`check_assignable`] for a stored user against the stored roles. +pub fn check_stored_user(user: &StoredUser, roles: &[StoredRole]) -> Result<(), RoleRefusal> { + let parsed: Vec = user + .roles + .iter() + .map(String::as_str) + .map(parse_role_name) + .collect(); + check_assignable(&parsed, user.tenant_id, |name| { + roles + .iter() + .find(|role| role.name == name) + .map(|role| role.tenant_id) + }) +} + +/// [`check_droppable`] for the stored role `name` against the stored users +/// and roles. Inactive users are dropped users and hold nothing. +pub fn check_stored_drop( + name: &str, + users: &[StoredUser], + roles: &[StoredRole], +) -> Result<(), RoleRefusal> { + check_droppable( + name, + users + .iter() + .filter(|user| user.is_active) + .map(|user| (user.username.as_str(), user.roles.as_slice())), + roles + .iter() + .map(|role| (role.name.as_str(), role.parent.as_str())), + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn custom(name: &str) -> Role { + Role::Custom(name.to_string()) + } + + #[test] + fn built_in_roles_are_always_assignable() { + let roles = [ + Role::Superuser, + Role::TenantAdmin, + Role::ReadWrite, + Role::ReadOnly, + Role::Monitor, + ]; + assert_eq!(check_assignable(&roles, 1, |_| None), Ok(())); + } + + #[test] + fn an_undefined_custom_role_is_refused_by_name() { + let roles = [Role::ReadWrite, custom("read_write")]; + assert_eq!( + check_assignable(&roles, 1, |_| None), + Err(RoleRefusal::Undefined { + name: "read_write".into() + }) + ); + assert_eq!( + RoleRefusal::Undefined { name: "x".into() }.sqlstate(), + "42704" + ); + } + + #[test] + fn a_custom_role_is_assignable_only_in_its_own_tenant() { + let lookup = |name: &str| (name == "analyst").then_some(7); + assert_eq!(check_assignable(&[custom("analyst")], 7, lookup), Ok(())); + assert!(check_assignable(&[custom("analyst")], 8, lookup).is_err()); + } + + #[test] + fn a_held_or_inherited_role_cannot_be_dropped() { + let held = vec!["analyst".to_string()]; + let none: Vec = Vec::new(); + assert_eq!( + check_droppable( + "analyst", + [("bob", held.as_slice()), ("amy", none.as_slice())], + [] + ), + Err(RoleRefusal::HeldByUsers { + name: "analyst".into(), + users: vec!["bob".into()] + }) + ); + assert_eq!( + check_droppable( + "analyst", + [("amy", none.as_slice())], + [("junior", "analyst")] + ), + Err(RoleRefusal::InheritedBy { + name: "analyst".into(), + children: vec!["junior".into()] + }) + ); + assert_eq!( + check_droppable("analyst", [("amy", none.as_slice())], [("junior", "")]), + Ok(()) + ); + } + + #[test] + fn every_parsed_built_in_name_is_built_in() { + for name in [ + "superuser", + "cluster_admin", + "tenant_admin", + "readwrite", + "readonly", + "monitor", + ] { + assert!(is_builtin_role_name(name), "{name}"); + } + assert!(!is_builtin_role_name("read_write")); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs index ba471fc00..8eb0dd41a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs @@ -54,10 +54,9 @@ pub fn create_api_key( require_tenant_admin(identity, "create API keys for other users")?; } - // Look up the target user. - let target_user = state - .credentials - .get_user(target_username) + // Look up the target user as this statement sees it: a user created + // earlier in the transaction counts, one dropped in it does not. + let target_user = super::super::role_checks::visible_user(state, target_username) .ok_or_else(|| err("42704", format!("user '{target_username}' not found")))?; // Parse optional EXPIRES. @@ -119,7 +118,7 @@ pub fn create_api_key( .prepare_key(crate::control::security::apikey::CreateKeyParams { username: target_username, user_id: target_user.user_id, - tenant_id: target_user.tenant_id, + tenant_id: crate::types::TenantId::new(target_user.tenant_id), expires_secs, scope: key_scopes, accessible_databases, diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs index cc1669afd..a807a353d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs @@ -120,17 +120,20 @@ pub(super) fn parse_with_databases( Ok(Some(ids)) } -/// Build the owner's `DatabaseSet` from a `UserRecord` for CREATE-time subset validation. +/// Build the owner's `DatabaseSet` for CREATE-time subset validation, from +/// the user as the statement sees it. pub(super) fn build_owner_database_set_for_user( state: &SharedState, - user: &crate::control::security::credential::record::UserRecord, + user: &crate::control::security::catalog::auth_types::user::StoredUser, ) -> Result { if user.is_superuser { return Ok(DatabaseSet::All); } if user.is_service_account && !user.accessible_databases.is_empty() { return Ok(DatabaseSet::Some(SmallVec::from_iter( - user.accessible_databases.iter().copied(), + user.accessible_databases + .iter() + .map(|&id| crate::types::DatabaseId::new(id)), ))); } // Regular user or legacy service account: read from database_grants. diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs index 757a44869..b83c4a99c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs @@ -48,10 +48,9 @@ pub fn grant_database( .map_err(|e| DdlError::new("XX000", format!("catalog lookup: {e}")))? .ok_or_else(|| DdlError::new("42704", format!("database '{db_name}' does not exist")))?; - // Resolve the target user_id from the grantee name. - let user_record = state - .credentials - .get_user(grantee) + // Resolve the target user_id from the grantee name, as this statement + // sees it: a user created earlier in the transaction counts. + let user_record = super::super::role_checks::visible_user(state, grantee) .ok_or_else(|| DdlError::new("42704", format!("user '{grantee}' does not exist")))?; let privileges: Vec<&str> = if privilege.eq_ignore_ascii_case("ALL") { @@ -105,9 +104,7 @@ pub fn revoke_database( .map_err(|e| DdlError::new("XX000", format!("catalog lookup: {e}")))? .ok_or_else(|| DdlError::new("42704", format!("database '{db_name}' does not exist")))?; - let user_record = state - .credentials - .get_user(grantee) + let user_record = super::super::role_checks::visible_user(state, grantee) .ok_or_else(|| DdlError::new("42704", format!("user '{grantee}' does not exist")))?; let privileges: Vec<&str> = if privilege.eq_ignore_ascii_case("ALL") { diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs index 472a922bb..3ddf80bf9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs @@ -33,7 +33,9 @@ use super::support::{require_tenant_admin, status}; /// names that resolve to neither, so unresolved typos don't sink into the /// store as silently unenforceable rows. fn canonicalize_grantee(state: &SharedState, raw: &str) -> Result { - if state.credentials.get_user(raw).is_some() { + // Users and roles the statement sees: created earlier in the + // transaction counts, dropped earlier in it does not. + if super::super::role_checks::visible_user(state, raw).is_some() { return Ok(format!("user:{raw}")); } let parsed: Role = match raw.parse() { @@ -41,7 +43,7 @@ fn canonicalize_grantee(state: &SharedState, raw: &str) -> Result match e {}, }; let is_known_role = match &parsed { - Role::Custom(name) => state.roles.get_role(name).is_some(), + Role::Custom(name) => super::super::role_checks::visible_roles(state).contains_key(name), _ => true, }; if is_known_role { diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/role.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/role.rs index d83666659..9540cb9b3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/role.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/role.rs @@ -26,23 +26,28 @@ use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; use super::support::{parse_role, require_tenant_admin, status}; -fn current_roles(state: &SharedState, username: &str) -> Result, DdlError> { - state - .credentials - .get_user(username) - .map(|r| r.roles) - .ok_or_else(|| DdlError::new("42704", format!("user '{username}' not found"))) +/// The roles and tenant of user `username` as this statement sees it: a user +/// created earlier in the transaction is visible, one dropped in it is not. +fn current_roles( + state: &SharedState, + username: &str, +) -> Result<(Vec, crate::types::TenantId), DdlError> { + let user = super::super::role_checks::visible_user_or_missing(state, username)?; + let roles = user.roles.iter().map(|name| parse_role(name)).collect(); + Ok((roles, crate::types::TenantId::new(user.tenant_id))) } fn propose_user_with_roles( state: &SharedState, username: &str, + tenant_id: crate::types::TenantId, new_roles: Vec, invalidation: crate::control::security::buses::SessionInvalidationReason, ) -> Result<(), DdlError> { + let base = super::super::role_checks::visible_user_or_missing(state, username)?; let stored = state .credentials - .prepare_user_update(username, None, Some(new_roles)) + .prepare_user_update_from(base, None, Some(new_roles.clone())) .map_err(|e| DdlError::new("42704", e.to_string()))?; let entry = CatalogEntry::PutUser(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) @@ -57,6 +62,8 @@ fn propose_user_with_roles( state .credentials .install_replicated_user(&stored, Some(invalidation)); + } else if outcome.is_replicated() { + super::super::role_checks::confirm_user_roles(state, username, &new_roles, tenant_id)?; } Ok(()) } @@ -79,9 +86,9 @@ pub fn grant_role( return Err(DdlError::new("42601", "GRANT: missing role name")); } - if state.credentials.get_user(grantee).is_some() { + if super::super::role_checks::visible_user(state, grantee).is_some() { grant_roles_to_user(state, identity, roles, grantee) - } else if state.roles.get_role(grantee).is_some() { + } else if super::super::role_checks::visible_roles(state).contains_key(grantee) { grant_role_to_role(state, identity, roles, grantee) } else { Err(DdlError::new( @@ -97,9 +104,12 @@ fn grant_roles_to_user( role_names: &[String], username: &str, ) -> Result, DdlError> { - let mut roles = current_roles(state, username)?; - for name in role_names { - let role = parse_role(name); + let (mut roles, tenant_id) = current_roles(state, username)?; + let granted: Vec = role_names.iter().map(|name| parse_role(name)).collect(); + // A role that is neither built in nor defined in the user's tenant would + // grant nothing: refuse it by name. + super::super::role_checks::check_user_roles(state, &granted, tenant_id)?; + for role in granted { if matches!(role, Role::Superuser) && !identity.is_superuser { return Err(DdlError::new( "42501", @@ -113,6 +123,7 @@ fn grant_roles_to_user( propose_user_with_roles( state, username, + tenant_id, roles, crate::control::security::buses::SessionInvalidationReason::RoleGranted, )?; @@ -181,9 +192,9 @@ pub fn revoke_role( )); } - if state.credentials.get_user(grantee).is_some() { + if super::super::role_checks::visible_user(state, grantee).is_some() { revoke_roles_from_user(state, identity, roles, grantee) - } else if state.roles.get_role(grantee).is_some() { + } else if super::super::role_checks::visible_roles(state).contains_key(grantee) { revoke_role_from_role(state, identity, roles, grantee) } else { Err(DdlError::new( @@ -199,7 +210,7 @@ fn revoke_roles_from_user( role_names: &[String], username: &str, ) -> Result, DdlError> { - let mut roles = current_roles(state, username)?; + let (mut roles, tenant_id) = current_roles(state, username)?; let revoked: Vec = role_names.iter().map(|n| parse_role(n)).collect(); for role in &revoked { if !roles.contains(role) { @@ -213,6 +224,7 @@ fn revoke_roles_from_user( propose_user_with_roles( state, username, + tenant_id, roles, crate::control::security::buses::SessionInvalidationReason::RoleRevoked, )?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/mod.rs index 0022b6393..eae9717e6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/mod.rs @@ -64,6 +64,7 @@ pub mod replicate; pub mod retention_policy; pub mod rls; pub mod role; +mod role_checks; pub mod router; pub mod schedule; pub mod scope_ddl; diff --git a/nodedb/src/control/server/shared/ddl/neutral/oidc.rs b/nodedb/src/control/server/shared/ddl/neutral/oidc.rs index 44820fae6..fab4e8aed 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/oidc.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/oidc.rs @@ -70,7 +70,14 @@ fn has_ambiguous_issuer_route(existing_audience: Option<&str>, audience: Option< } } -fn validate_claim_mapping_roles(claim_mappings: &[OidcClaimMappingClause]) -> Result<(), DdlError> { +/// Refuse a claim mapping that grants superuser, or a role that is neither +/// built in nor defined in the provider's tenant: a login mapped to it would +/// hold nothing. A role dropped after this check refuses the login instead. +fn validate_claim_mapping_roles( + state: &SharedState, + claim_mappings: &[OidcClaimMappingClause], + tenant_id: Option, +) -> Result<(), DdlError> { if claim_mappings .iter() .flat_map(|mapping| mapping.add_roles.iter()) @@ -81,6 +88,19 @@ fn validate_claim_mapping_roles(claim_mappings: &[OidcClaimMappingClause]) -> Re "OIDC claim mappings cannot grant the database-owned superuser role", )); } + if let Some(tenant_id) = tenant_id { + let roles: Vec = claim_mappings + .iter() + .flat_map(|mapping| mapping.add_roles.iter()) + .map(String::as_str) + .map(crate::control::security::role_assignment::parse_role_name) + .collect(); + super::role_checks::check_user_roles( + state, + &roles, + crate::types::TenantId::new(tenant_id), + )?; + } Ok(()) } @@ -116,7 +136,6 @@ pub fn create_oidc_provider( if jwks_uri.is_empty() { return Err(DdlError::new("22023", "JWKS_URI must not be empty")); } - validate_claim_mapping_roles(claim_mappings)?; let catalog = state.credentials.catalog(); @@ -131,6 +150,7 @@ pub fn create_oidc_provider( format!("tenant '{tenant_id}' does not exist"), )); } + validate_claim_mapping_roles(state, claim_mappings, Some(tenant_id))?; // Check for duplicate by provider name. match catalog.get_oidc_provider(name) { @@ -223,7 +243,7 @@ pub fn alter_oidc_provider_claim_mapping( .get_oidc_provider(name) .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? .ok_or_else(|| DdlError::new("42704", format!("OIDC provider '{name}' does not exist")))?; - validate_claim_mapping_roles(claim_mappings)?; + validate_claim_mapping_roles(state, claim_mappings, provider.tenant_id)?; let stored_mappings: Vec = claim_mappings .iter() diff --git a/nodedb/src/control/server/shared/ddl/neutral/role.rs b/nodedb/src/control/server/shared/ddl/neutral/role.rs index 911a13261..c8bde6f4a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/role.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/role.rs @@ -40,8 +40,13 @@ pub fn create_role( let name = parts[2]; + // The roles this statement sees: committed ones, and inside a + // transaction those it created earlier, so a parent created in the same + // transaction resolves. COMMIT checks the whole batch again. + let visible = super::role_checks::visible_roles(state); + // `IF NOT EXISTS`: re-creating an existing role is a no-op success. - if if_not_exists && state.roles.get_role(name).is_some() { + if if_not_exists && visible.contains_key(name) { return Ok(status("CREATE ROLE")); } @@ -51,12 +56,15 @@ pub fn create_role( None }; - // Build the `StoredRole` on the proposer (runs the same - // validation as `create_role` but without touching state). - let stored = state - .roles - .prepare_role(name, identity.tenant_id, parent) - .map_err(|e| DdlError::new("42710", e.to_string()))?; + // Build the `StoredRole` on the proposer: the same validation as + // `create_role`, against the visible roles, without touching state. + let stored = crate::control::security::role::prepare_role_against( + name, + identity.tenant_id, + parent, + &visible, + ) + .map_err(|e| DdlError::new("42710", e.to_string()))?; let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) @@ -67,6 +75,8 @@ pub fn create_role( .put_role(&stored) .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; state.roles.install_replicated_role(&stored); + } else if outcome.is_replicated() { + super::role_checks::confirm_role(state, name, parent)?; } state.audit_record( @@ -100,7 +110,7 @@ pub fn drop_role( } let name = parts[2]; - let exists_before = state.roles.get_role(name).is_some(); + let exists_before = super::role_checks::visible_roles(state).contains_key(name); if !exists_before { // `IF EXISTS`: dropping a missing role is a no-op success. if if_exists { @@ -112,6 +122,11 @@ pub fn drop_role( )); } + // As PostgreSQL does, a role that users hold or roles inherit from is + // not dropped: dropping it would leave them naming a role that grants + // nothing. + super::role_checks::check_role_droppable(state, name)?; + let entry = crate::control::catalog_entry::CatalogEntry::DeleteRole { name: name.to_string(), }; @@ -122,11 +137,22 @@ pub fn drop_role( state .roles .drop_role(name, Some(catalog)) - .map_err(|e| DdlError::new("42704", e.to_string()))? + .map_err(|e| DdlError::new("2BP01", e.to_string()))? + } else if outcome.is_replicated() { + // The synchronous post-apply removed the role from this node's + // cache before the applied index advanced. A role still present + // means the applier skipped the entry: a user or role came to + // depend on it before the drop committed. + if state.roles.get_role(name).is_some() { + super::role_checks::check_role_droppable(state, name)?; + return Err(DdlError::new( + "40001", + format!("transient: the drop of role '{name}' was superseded, retry"), + )); + } + true } else { - // Cluster mode: the raft entry committed, trust the - // log index. The in-memory cache update runs in a - // spawned tokio task and may not be visible yet. + // Buffered in an open transaction: COMMIT applies it. true }; @@ -159,11 +185,14 @@ pub fn alter_role_typed( ) -> Result, DdlError> { require_tenant_admin(identity, "alter roles")?; - // The role must exist before we mutate it. - state - .roles - .get_role(role_name) - .ok_or_else(|| DdlError::new("42704", format!("role '{role_name}' not found")))?; + // The role must exist before we mutate it: committed, or created earlier + // in this transaction. + if !super::role_checks::visible_roles(state).contains_key(role_name) { + return Err(DdlError::new( + "42704", + format!("role '{role_name}' not found"), + )); + } match sub_op { AlterRoleOp::Grant { @@ -220,17 +249,18 @@ pub fn set_role_parent( role_name: &str, parent: Option<&str>, ) -> Result<(), DdlError> { - let old_role = state - .roles - .get_role(role_name) + // Roles created earlier in this transaction count, as they do for + // `CREATE ROLE`. + let visible = super::role_checks::visible_roles(state); + let old_role = visible + .get(role_name) + .cloned() .ok_or_else(|| DdlError::new("42704", format!("role '{role_name}' not found")))?; if let Some(parent) = parent { - let parent_is_builtin = matches!( - parent, - "superuser" | "tenant_admin" | "readwrite" | "readonly" | "monitor" - ); - if !parent_is_builtin && state.roles.get_role(parent).is_none() { + let parent_is_builtin = + crate::control::security::role_assignment::is_builtin_role_name(parent); + if !parent_is_builtin && !visible.contains_key(parent) { return Err(DdlError::new( "42704", format!("parent role '{parent}' does not exist"), @@ -238,10 +268,10 @@ pub fn set_role_parent( } // Reject self-inheritance and multi-hop cycles, and enforce the // inheritance-depth cap — the same invariant `CREATE ROLE` checks. - state - .roles - .check_inheritance_cycle(role_name, parent) - .map_err(|e| DdlError::new("42P16", e.to_string()))?; + crate::control::security::role::check_inheritance_cycle_against( + role_name, parent, &visible, + ) + .map_err(|e| DdlError::new("42P16", e.to_string()))?; } let now = std::time::SystemTime::now() @@ -264,6 +294,8 @@ pub fn set_role_parent( .put_role(&stored) .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; state.roles.install_replicated_role(&stored); + } else if outcome.is_replicated() { + super::role_checks::confirm_role(state, role_name, parent)?; } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/role_checks.rs b/nodedb/src/control/server/shared/ddl/neutral/role_checks.rs new file mode 100644 index 000000000..85d64cb77 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/role_checks.rs @@ -0,0 +1,248 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Role rules at the DDL entry points: a user may hold only a built-in role +//! or a custom role defined in its tenant, and a custom role may be dropped +//! only while no user holds it and no role inherits from it. +//! +//! Each statement checks before it proposes. The metadata applier runs the +//! same rules at the entry's log position and skips an entry that breaks +//! them, on every node alike. A statement whose entry was skipped, because +//! a racing change committed first, learns it here after the apply and +//! reports the refusal instead of a success. +//! +//! Inside a transaction a statement sees the committed users and roles +//! overlaid with the role and user entries its transaction buffered, in +//! order: a role created earlier in the transaction can be assigned, and a +//! role a user created earlier in it holds cannot be dropped. COMMIT checks +//! the whole batch again before it applies anything. + +use std::collections::HashMap; + +use crate::control::catalog_entry::CatalogEntry; +use crate::control::security::identity::Role; +use crate::control::security::role_assignment::{self, RoleRefusal}; +use crate::control::state::SharedState; +use crate::types::TenantId; + +use super::super::result::DdlError; + +/// The DDL error for a role refusal, with PostgreSQL's SQLSTATE. +pub(super) fn refusal(refusal: RoleRefusal) -> DdlError { + DdlError::new(refusal.sqlstate(), refusal.to_string()) +} + +/// Users and roles as the running statement sees them. +struct RoleView { + /// Active users: name to held role names. + users: HashMap>, + /// Custom roles: name to (tenant, parent, empty when none). + roles: HashMap, +} + +impl RoleView { + /// The committed users and roles, overlaid with the entries this + /// connection's open transaction buffered, in statement order. + fn current(state: &SharedState) -> Self { + let mut view = Self { + users: state + .credentials + .list_user_details() + .into_iter() + .map(|user| { + let held = user.roles.iter().map(ToString::to_string).collect(); + (user.username, held) + }) + .collect(), + roles: state + .roles + .list_roles() + .into_iter() + .map(|role| { + let parent = role.parent.unwrap_or_default(); + (role.name, (role.tenant_id.as_u64(), parent)) + }) + .collect(), + }; + crate::control::server::shared::session::ddl_buffer::with_buffered(|buffered| { + for item in buffered { + view.overlay(&item.entry); + } + }); + view + } + + fn overlay(&mut self, entry: &CatalogEntry) { + match entry { + CatalogEntry::PutUser(user) if user.is_active => { + self.users.insert(user.username.clone(), user.roles.clone()); + } + CatalogEntry::PutUser(user) => { + self.users.remove(&user.username); + } + CatalogEntry::PutTenantWithAdmin { admin, .. } => { + self.users + .insert(admin.username.clone(), admin.roles.clone()); + } + CatalogEntry::DropUser { username } => { + self.users.remove(username); + } + CatalogEntry::PutRole(role) => { + self.roles + .insert(role.name.clone(), (role.tenant_id, role.parent.clone())); + } + CatalogEntry::DeleteRole { name } => { + self.roles.remove(name); + } + // Every other entry leaves users and roles as they are. + _ => {} + } + } +} + +/// Active user `username` as the running statement sees it: the committed +/// record, overlaid with the user entries its transaction buffered, in +/// order. A user created earlier in the transaction is visible; one dropped +/// earlier in it is not. +pub(super) fn visible_user( + state: &SharedState, + username: &str, +) -> Option { + let committed = state.credentials.stored_user(username); + crate::control::server::shared::session::ddl_buffer::with_buffered(|buffered| { + let mut user = committed.clone(); + for item in buffered { + match &item.entry { + CatalogEntry::PutUser(stored) if stored.username == username => { + user = stored.is_active.then(|| (**stored).clone()); + } + CatalogEntry::PutTenantWithAdmin { admin, .. } if admin.username == username => { + user = Some((**admin).clone()); + } + CatalogEntry::DropUser { username: dropped } if dropped == username => { + user = None; + } + // Every other entry leaves this user as it is. + _ => {} + } + } + user + }) + .unwrap_or(committed) +} + +/// [`visible_user`], or the 42704 refusal a statement naming a missing user +/// reports. +pub(super) fn visible_user_or_missing( + state: &SharedState, + username: &str, +) -> Result { + visible_user(state, username) + .ok_or_else(|| DdlError::new("42704", format!("user '{username}' not found"))) +} + +/// Every custom role the running statement sees, keyed by name: the +/// committed ones overlaid with those its transaction buffered. +pub(super) fn visible_roles( + state: &SharedState, +) -> HashMap { + RoleView::current(state) + .roles + .into_iter() + .map(|(name, (tenant_id, parent))| { + let role = crate::control::security::role::CustomRole { + name: name.clone(), + tenant_id: TenantId::new(tenant_id), + parent: (!parent.is_empty()).then_some(parent), + created_at: 0, + }; + (name, role) + }) + .collect() +} + +/// Refuse any role in `roles` a user of `tenant_id` cannot hold. +pub(super) fn check_user_roles( + state: &SharedState, + roles: &[Role], + tenant_id: TenantId, +) -> Result<(), DdlError> { + let view = RoleView::current(state); + role_assignment::check_assignable(roles, tenant_id.as_u64(), |name| { + view.roles.get(name).map(|(tenant, _)| *tenant) + }) + .map_err(refusal) +} + +/// After a replicated user entry applied, confirm the user of `tenant_id` +/// holds exactly `roles`. The applier skips a user entry naming a role that +/// was dropped before the entry committed; that surfaces here as the role's +/// refusal. Any other mismatch is a concurrent change or a truncated entry, +/// which the client retries. +pub(super) fn confirm_user_roles( + state: &SharedState, + username: &str, + roles: &[Role], + tenant_id: TenantId, +) -> Result<(), DdlError> { + let held = state.credentials.get_user(username).map(|user| user.roles); + let held_exactly = held.as_ref().is_some_and(|held| { + roles.iter().all(|role| held.contains(role)) && held.iter().all(|role| roles.contains(role)) + }); + if held_exactly { + return Ok(()); + } + check_user_roles(state, roles, tenant_id)?; + Err(DdlError::new( + "40001", + format!( + "transient: the entry for user '{username}' was superseded or truncated by a \ + leader change, retry" + ), + )) +} + +/// Refuse to drop the custom role `name` while a user holds it or a role +/// inherits from it. +pub(super) fn check_role_droppable(state: &SharedState, name: &str) -> Result<(), DdlError> { + let view = RoleView::current(state); + role_assignment::check_droppable( + name, + view.users + .iter() + .map(|(user, held)| (user.as_str(), held.as_slice())), + view.roles + .iter() + .map(|(role, (_, parent))| (role.as_str(), parent.as_str())), + ) + .map_err(refusal) +} + +/// After a replicated role entry applied, confirm role `name` exists with +/// inheritance parent `parent`. The applier skips a role entry whose parent +/// was dropped before the entry committed; that surfaces here as the +/// parent's refusal. +pub(super) fn confirm_role( + state: &SharedState, + name: &str, + parent: Option<&str>, +) -> Result<(), DdlError> { + let installed = state + .roles + .get_role(name) + .is_some_and(|role| role.parent.as_deref() == parent); + if installed { + return Ok(()); + } + if let Some(parent) = parent + && !role_assignment::is_builtin_role_name(parent) + && state.roles.get_role(parent).is_none() + { + return Err(refusal(RoleRefusal::Undefined { + name: parent.to_string(), + })); + } + Err(DdlError::new( + "40001", + format!("transient: the entry for role '{name}' was superseded or truncated, retry"), + )) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/service_account.rs b/nodedb/src/control/server/shared/ddl/neutral/service_account.rs index 99699d16c..1e34068ff 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/service_account.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/service_account.rs @@ -2,14 +2,10 @@ //! Protocol-neutral service-account DDL — CREATE / DROP / ALTER SET DATABASES. //! -//! Ported from the pgwire `ddl::service_account` and -//! `ddl::service_account_alter` handlers. All non-return logic (permission -//! checks, `IF [NOT] EXISTS` token stripping, ROLE / TENANT / FOR DATABASE / -//! IN DATABASE parsing, credential-store `create_service_account` / -//! `drop_user` / `set_service_account_databases`, database-name resolution via -//! the system catalog, and the `audit_record` calls) is preserved verbatim; -//! only the result construction changed from pgwire `Response` / `PgWireError` -//! to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! Every change replicates through the metadata group as a user entry: +//! CREATE and ALTER SET DATABASES propose `PutUser`, and DROP runs the +//! replicated `DROP USER` path. A single node applies the entry locally. +//! Database names resolve through the system catalog. use crate::control::security::audit::AuditEvent; use crate::control::security::identity::{AuthenticatedIdentity, Role}; @@ -70,9 +66,17 @@ pub fn create_service_account( let name = parts[3]; + // The name is taken when the statement sees the account: created + // earlier in its transaction counts, dropped earlier in it frees it. // `IF NOT EXISTS`: re-creating an existing service account is a no-op. - if if_not_exists && state.credentials.get_user(name).is_some() { - return Ok(status("CREATE SERVICE ACCOUNT")); + if super::role_checks::visible_user(state, name).is_some() { + if if_not_exists { + return Ok(status("CREATE SERVICE ACCOUNT")); + } + return Err(DdlError::new( + "42710", + format!("user or service account '{name}' already exists"), + )); } // Parse optional ROLE, TENANT, FOR DATABASE / IN DATABASE. @@ -170,10 +174,35 @@ pub fn create_service_account( } let _ = seen_for_tenant; // suppress unused warning - state + // A role that is neither built in nor defined in the tenant would leave + // the account with no permissions: refuse it by name. + super::role_checks::check_user_roles(state, std::slice::from_ref(&role), tenant_id)?; + + // The account replicates through the metadata group like any user, so + // every node authenticates it and every node sees its role. + let stored = state .credentials - .create_service_account(name, tenant_id, vec![role], accessible_databases) + .prepare_new_service_account(name, tenant_id, vec![role.clone()], accessible_databases) .map_err(|e| DdlError::new("42710", e.to_string()))?; + let entry = crate::control::catalog_entry::CatalogEntry::PutUser(Box::new(stored.clone())); + let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + if outcome.needs_local_apply() { + state + .credentials + .catalog() + .put_user(&stored) + .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + // A new account has no open sessions to invalidate. + state.credentials.install_replicated_user(&stored, None); + } else if outcome.is_replicated() { + super::role_checks::confirm_user_roles( + state, + name, + std::slice::from_ref(&role), + tenant_id, + )?; + } state.audit_record( AuditEvent::PrivilegeChange, @@ -204,8 +233,8 @@ pub fn drop_service_account( let name = parts[3]; - // Verify it's actually a service account. - let user = match state.credentials.get_user(name) { + // Verify it's actually a service account, as this statement sees it. + let user = match super::role_checks::visible_user(state, name) { Some(u) => u, None => { // `IF EXISTS`: dropping a missing account is a no-op success. @@ -225,25 +254,11 @@ pub fn drop_service_account( )); } - let dropped = state - .credentials - .drop_user(name) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; - - if dropped { - state.audit_record( - AuditEvent::PrivilegeChange, - Some(identity.tenant_id), - &identity.username, - &format!("dropped service account '{name}'"), - ); - Ok(status("DROP SERVICE ACCOUNT")) - } else { - Err(DdlError::new( - "42704", - format!("service account '{name}' not found"), - )) - } + // Dropped through the replicated `DROP USER` path, so the account goes + // on every node, and its owned objects and grants are handled as a + // user's are. That path records the audit entry. + super::user::drop_user(state, identity, &["DROP", "USER", name])?; + Ok(status("DROP SERVICE ACCOUNT")) } /// ALTER SERVICE ACCOUNT SET DATABASES (db1, db2, ...) @@ -277,10 +292,8 @@ pub fn alter_service_account_set_databases( let name = parts[3]; - // Verify it's actually a service account. - let user = state - .credentials - .get_user(name) + // Verify it's actually a service account, as this statement sees it. + let user = super::role_checks::visible_user(state, name) .ok_or_else(|| DdlError::new("42704", format!("service account '{name}' not found")))?; if !user.is_service_account { return Err(DdlError::new( @@ -324,10 +337,24 @@ pub fn alter_service_account_set_databases( } } - state + let stored = state .credentials - .set_service_account_databases(name, db_ids) + .prepare_service_account_databases_from(user, db_ids) .map_err(|e| DdlError::new("XX000", e.to_string()))?; + let entry = crate::control::catalog_entry::CatalogEntry::PutUser(Box::new(stored.clone())); + let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + if outcome.needs_local_apply() { + state + .credentials + .catalog() + .put_user(&stored) + .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + state.credentials.install_replicated_user( + &stored, + Some(crate::control::security::buses::SessionInvalidationReason::RoleAltered), + ); + } state.audit_record( AuditEvent::PrivilegeChange, diff --git a/nodedb/src/control/server/shared/ddl/neutral/user/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/user/alter.rs index f40ced5d5..572fd7525 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/user/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/user/alter.rs @@ -19,6 +19,7 @@ use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::{parse_role, require_tenant_admin, status}; +use super::super::role_checks::visible_user_or_missing; use super::iso8601::parse_iso8601_to_unix; /// ALTER USER ... — typed dispatch for all AlterUserOp forms. @@ -53,9 +54,10 @@ pub fn alter_user( "password must be a non-empty single-quoted string", )); } + let base = visible_user_or_missing(state, username)?; let stored = state .credentials - .prepare_user_update(username, Some(password.as_str()), None) + .prepare_user_update_from(base, Some(password.as_str()), None) .map_err(|e| DdlError::new("XX000", e.to_string()))?; // Password change — no role/access change; no invalidation. propose_and_install(state, stored, None)?; @@ -78,15 +80,24 @@ pub fn alter_user( return Err(DdlError::new("42601", "expected role name after SET ROLE")); } let parsed_role: Role = parse_role(role); + let base = visible_user_or_missing(state, username)?; + let tenant_id = crate::types::TenantId::new(base.tenant_id); + let new_roles = vec![parsed_role.clone()]; + super::super::role_checks::check_user_roles(state, &new_roles, tenant_id)?; let stored = state .credentials - .prepare_user_update(username, None, Some(vec![parsed_role.clone()])) + .prepare_user_update_from(base, None, Some(new_roles.clone())) .map_err(|e| DdlError::new("XX000", e.to_string()))?; - propose_and_install( + let replicated = propose_and_install( state, stored, Some(crate::control::security::buses::SessionInvalidationReason::RoleAltered), )?; + if replicated { + super::super::role_checks::confirm_user_roles( + state, username, &new_roles, tenant_id, + )?; + } state.audit_record( AuditEvent::PrivilegeChange, @@ -99,10 +110,10 @@ pub fn alter_user( AlterUserOp::MustChangePassword => { require_tenant_admin(identity, "set must_change_password")?; + let base = visible_user_or_missing(state, username)?; let stored = state .credentials - .prepare_set_must_change_password(username, true) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .prepare_set_must_change_password_from(base, true); propose_and_install(state, stored, None)?; state.audit_record( @@ -116,10 +127,10 @@ pub fn alter_user( AlterUserOp::PasswordNeverExpires => { require_tenant_admin(identity, "set password expiry")?; + let base = visible_user_or_missing(state, username)?; let stored = state .credentials - .prepare_set_password_expires_at(username, 0) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .prepare_set_password_expires_at_from(base, 0); propose_and_install(state, stored, None)?; state.audit_record( @@ -139,10 +150,10 @@ pub fn alter_user( format!("invalid ISO-8601 datetime '{iso8601}': {e}"), ) })?; + let base = visible_user_or_missing(state, username)?; let stored = state .credentials - .prepare_set_password_expires_at(username, expires_at) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .prepare_set_password_expires_at_from(base, expires_at); propose_and_install(state, stored, None)?; state.audit_record( @@ -163,10 +174,10 @@ pub fn alter_user( )); } let expires_at = crate::control::security::time::now_secs() + (*days as u64) * 86400; + let base = visible_user_or_missing(state, username)?; let stored = state .credentials - .prepare_set_password_expires_at(username, expires_at) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .prepare_set_password_expires_at_from(base, expires_at); propose_and_install(state, stored, None)?; state.audit_record( @@ -200,10 +211,10 @@ pub fn alter_user( .ok_or_else(|| { DdlError::new("42704", format!("database '{db_name}' does not exist")) })?; + let base = visible_user_or_missing(state, username)?; let stored = state .credentials - .prepare_set_default_database(username, db_id.as_u64()) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .prepare_set_default_database_from(base, db_id.as_u64()); propose_and_install(state, stored, None)?; state.audit_record( @@ -218,6 +229,7 @@ pub fn alter_user( } /// Propose a `StoredUser` via Raft and install it locally on single-node. +/// Returns whether the entry was replicated through the metadata group. /// /// `invalidation` is passed to `install_replicated_user` for in-process /// session notification in single-node mode. Cluster-mode notifications @@ -226,7 +238,7 @@ fn propose_and_install( state: &SharedState, stored: crate::control::security::catalog::StoredUser, invalidation: Option, -) -> Result<(), DdlError> { +) -> Result { let entry = crate::control::catalog_entry::CatalogEntry::PutUser(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; @@ -241,5 +253,5 @@ fn propose_and_install( .credentials .install_replicated_user(&stored, invalidation); } - Ok(()) + Ok(outcome.is_replicated()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/user/create.rs b/nodedb/src/control/server/shared/ddl/neutral/user/create.rs index 13b1ce20b..0bc8aad7b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/user/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/user/create.rs @@ -58,9 +58,17 @@ pub fn create_user( )); } + // The name is taken when the statement sees the user: created earlier + // in its transaction counts, dropped earlier in it frees the name. // `IF NOT EXISTS`: re-creating an existing user is a no-op success. - if if_not_exists && state.credentials.get_user(username).is_some() { - return Ok(status("CREATE USER")); + if super::super::role_checks::visible_user(state, username).is_some() { + if if_not_exists { + return Ok(status("CREATE USER")); + } + return Err(DdlError::new( + "42710", + format!("user '{username}' already exists"), + )); } if password.is_empty() { @@ -80,13 +88,17 @@ pub fn create_user( identity.tenant_id }; + // A role that is neither built in nor defined in the tenant would leave + // the user with no permissions: refuse it by name. + super::super::role_checks::check_user_roles(state, std::slice::from_ref(&role), tenant_id)?; + // Build the full `StoredUser` locally (hash + salt + user_id). // Followers cannot reproduce the random salt, so this step // MUST happen on the proposer node. The computed record is // then replicated verbatim. let stored = state .credentials - .prepare_user(username, password, tenant_id, vec![role]) + .prepare_new_user(username, password, tenant_id, vec![role.clone()]) .map_err(|e| DdlError::new("42710", e.to_string()))?; let entry = crate::control::catalog_entry::CatalogEntry::PutUser(Box::new(stored.clone())); @@ -109,26 +121,21 @@ pub fn create_user( // CREATE USER: no open sessions exist for a brand-new user. state.credentials.install_replicated_user(&stored, None); } else if outcome.is_replicated() { - // Cluster mode: `propose_catalog_entry` waits for the - // entry to be applied on THIS node, which runs the - // synchronous post_apply (`install_replicated_user`) - // inline BEFORE the applied-index watermark bumps. So if - // our entry really committed, `get_user` must see it now. - // - // If `get_user` returns None, the Raft log entry at the - // index our leader assigned has been truncated and - // overwritten with a noop from a new leader term (a known - // Raft subtlety: `propose` returns the assigned log index - // without waiting for commit; if leadership changes - // before the quorum ack, the entry is dropped). Return a - // retryable error so `exec_ddl_on_any_leader` re-proposes - // on the next attempt against whoever is now leader. - if state.credentials.get_user(username).is_none() { - return Err(DdlError::new( - "40001", - "transient: metadata entry truncated by leader change, retry", - )); - } + // Cluster mode: `propose_catalog_entry` waits for the entry to apply + // on THIS node, and the synchronous post_apply + // (`install_replicated_user`) runs before the applied-index watermark + // bumps. So a committed entry is visible now. A missing user means + // the applier skipped the entry because its role was dropped first, + // or the entry was truncated by a leader change (`propose` returns + // the assigned index before the quorum ack). The first is refused by + // name; the second is retryable, and `exec_ddl_on_any_leader` + // re-proposes against whoever leads now. + super::super::role_checks::confirm_user_roles( + state, + username, + std::slice::from_ref(&role), + tenant_id, + )?; } // A `Buffered` outcome falls through: the open transaction owns the entry // until COMMIT, so neither the cache install nor the truncation check diff --git a/nodedb/src/control/server/shared/ddl/neutral/user/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/user/drop.rs index 21b89aa0b..92766fef5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/user/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/user/drop.rs @@ -70,17 +70,19 @@ fn drop_user_inner( return Err(DdlError::new("42501", "cannot drop your own user")); } + // The user as this statement sees it: created earlier in the + // transaction counts, dropped earlier in it does not. + let visible = super::super::role_checks::visible_user(state, username); + // Look up user's tenant before dropping (for ownership reassignment). - let user_tenant = state - .credentials - .get_user(username) - .map(|u| u.tenant_id) + let user_tenant = visible + .as_ref() + .map(|u| crate::types::TenantId::new(u.tenant_id)) .unwrap_or(identity.tenant_id); // Pre-check existence so a DROP USER on a missing user is a // clean error that doesn't touch raft. - let exists_before = state.credentials.get_user(username).is_some(); - if !exists_before { + if visible.is_none() { // `IF EXISTS`: dropping a missing user is a no-op success. if if_exists { return Ok(status("DROP USER")); diff --git a/nodedb/src/control/server/shared/session/ddl_flush.rs b/nodedb/src/control/server/shared/session/ddl_flush.rs index 3a022ad3c..4abc02dc5 100644 --- a/nodedb/src/control/server/shared/session/ddl_flush.rs +++ b/nodedb/src/control/server/shared/session/ddl_flush.rs @@ -90,13 +90,27 @@ pub(super) fn flush_local(state: &SharedState, buffered: DdlBuffer) -> Option return Some(AbortReason::DdlPropose(error)), }; let catalog = shared.credentials.catalog(); + // A statement can break a role rule for a later one in the same + // transaction. Refuse the batch before anything is written. + if let Err(error) = crate::control::catalog_entry::role_rules::check_batch( + buffered.iter().map(|item| &item.entry), + catalog, + ) { + return Some(AbortReason::DdlPropose(error)); + } let total = buffered.len(); for (position, item) in buffered.into_iter().enumerate() { match crate::control::catalog_entry::apply::apply_to(&item.entry, catalog) { // Wrote nothing (if-absent create for an existing descriptor): // the applier suppresses post-apply here, so this must too. - Ok(false) => continue, - Ok(true) => {} + Ok(crate::control::catalog_entry::apply::ApplyOutcome::Unchanged) => continue, + // An earlier statement of this transaction broke a role rule for + // this one, such as dropping a role a user created earlier in + // the same transaction holds. The COMMIT fails with the refusal. + Ok(crate::control::catalog_entry::apply::ApplyOutcome::Refused(refusal)) => { + return Some(AbortReason::DdlPropose(crate::Error::from(refusal))); + } + Ok(crate::control::catalog_entry::apply::ApplyOutcome::Applied) => {} Err(error) => { return Some(AbortReason::DdlPropose(crate::Error::Internal { detail: format!( @@ -248,6 +262,13 @@ fn propose_pending_buffered<'a>( })?; let distributed_guard = crate::control::metadata_proposer::acquire_ddl_prepare_lease(state, handle.as_ref())?; + // Checked under the preparation lease, which every DDL takes: no other + // role or user change commits between this check and the finalize, so + // the finalize never applies part of the batch. + crate::control::catalog_entry::role_rules::check_batch( + buffered.iter().map(|item| &item.entry), + state.credentials.catalog(), + )?; for item in &buffered { if let Some((descriptor_id, prior_version)) = diff --git a/nodedb/src/error/types.rs b/nodedb/src/error/types.rs index e133cfc41..96ea7fd2f 100644 --- a/nodedb/src/error/types.rs +++ b/nodedb/src/error/types.rs @@ -557,6 +557,18 @@ pub enum Error { #[error("OIDC token rejected: authenticated provider tenant is unavailable")] OidcProviderTenantUnavailable { tenant_id: u64 }, + /// An external identity (JWT or OIDC bearer) resolved to a custom role + /// that is not defined in its provider-bound tenant. The login is + /// refused: an identity holding an undefined role would hold nothing. + #[error( + "external identity '{subject}' claims role \"{role}\", which is not defined in tenant {tenant_id}" + )] + ExternalRoleUndefined { + subject: String, + role: String, + tenant_id: u64, + }, + /// OIDC bearer token rejected: claim mapping produced no default database. #[error("OIDC token rejected: claim mapping produced no default database for subject '{sub}'")] OidcNoDefaultDatabase { sub: String }, diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index 0a27a56ef..1add05e5d 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -308,6 +308,14 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { Error::OidcNoDefaultDatabase { sub } => NodeDbError::bad_request(format!( "OIDC: no default database resolved for sub '{sub}'" )), + Error::ExternalRoleUndefined { + subject, + role, + tenant_id, + } => NodeDbError::bad_request(format!( + "external identity '{subject}' claims role \"{role}\", which is not defined in \ + tenant {tenant_id}" + )), // Exhaustive by construction: a code falling through to `internal` // reaches the client as an indistinguishable NDB-9000. Error::DataPlane(code) => { diff --git a/nodedb/tests/inproc/cases/mod.rs b/nodedb/tests/inproc/cases/mod.rs index 23dd214d4..c100a849c 100644 --- a/nodedb/tests/inproc/cases/mod.rs +++ b/nodedb/tests/inproc/cases/mod.rs @@ -151,6 +151,7 @@ mod request_tracker_backpressure; mod resp_row_level_security; mod retention_policy_replication_apply; mod rls_fuzz; +mod role_assignment_rules; mod scope_quota_hard_refusal; mod scope_quota_replication_apply; mod security; diff --git a/nodedb/tests/inproc/cases/role_assignment_rules.rs b/nodedb/tests/inproc/cases/role_assignment_rules.rs new file mode 100644 index 000000000..dfd56aa4e --- /dev/null +++ b/nodedb/tests/inproc/cases/role_assignment_rules.rs @@ -0,0 +1,177 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A user may hold only a built-in role or a custom role defined in its +//! tenant, on every statement that assigns one. A custom role that a user +//! holds is not dropped. Both refusals name the role and carry PostgreSQL's +//! SQLSTATE: 42704 for an undefined role, 2BP01 for a role still in use. + +use nodedb::control::security::identity::Role; +use nodedb_test_support::pgwire_auth_helpers::{ + ddl_err, ddl_ok, make_state, make_state_with_catalog, superuser, +}; + +fn assert_undefined(error: &str, role: &str) { + assert!( + error.contains("42704") && error.contains(role), + "expected 42704 naming role '{role}', got: {error}" + ); +} + +#[tokio::test] +async fn every_role_assignment_refuses_an_undefined_role() { + let state = make_state(); + let su = superuser(); + ddl_ok( + &state, + &su, + "CREATE USER rita WITH PASSWORD 'pass' ROLE readonly", + ) + .await; + + for sql in [ + "CREATE USER ghost WITH PASSWORD 'pass' ROLE read_write", + "ALTER USER rita SET ROLE read_write", + "GRANT ROLE read_write TO rita", + "GRANT read_write TO rita", + "CREATE SERVICE ACCOUNT ghost_svc ROLE read_write", + ] { + assert_undefined(&ddl_err(&state, &su, sql).await, "read_write"); + } + + assert!(state.credentials.get_user("ghost").is_none()); + assert!(state.credentials.get_user("ghost_svc").is_none()); + let rita = state.credentials.get_user("rita").expect("rita"); + assert_eq!( + rita.roles, + vec![Role::ReadOnly], + "a refused statement changed rita" + ); +} + +#[tokio::test] +async fn every_role_assignment_accepts_a_defined_custom_role() { + let state = make_state(); + let su = superuser(); + ddl_ok(&state, &su, "CREATE ROLE analyst").await; + let analyst = Role::Custom("analyst".into()); + + ddl_ok( + &state, + &su, + "CREATE USER ana WITH PASSWORD 'pass' ROLE analyst", + ) + .await; + ddl_ok( + &state, + &su, + "CREATE USER ben WITH PASSWORD 'pass' ROLE readonly", + ) + .await; + ddl_ok(&state, &su, "ALTER USER ben SET ROLE analyst").await; + ddl_ok( + &state, + &su, + "CREATE USER cat WITH PASSWORD 'pass' ROLE readonly", + ) + .await; + ddl_ok(&state, &su, "GRANT ROLE analyst TO cat").await; + ddl_ok(&state, &su, "CREATE SERVICE ACCOUNT ana_svc ROLE analyst").await; + + for user in ["ana", "ben", "cat", "ana_svc"] { + let record = state + .credentials + .get_user(user) + .unwrap_or_else(|| panic!("{user} exists")); + assert!( + record.roles.contains(&analyst), + "{user}: {:?}", + record.roles + ); + } +} + +#[tokio::test] +async fn a_custom_role_of_another_tenant_is_undefined_here() { + let state = make_state(); + let su = superuser(); + ddl_ok(&state, &su, "CREATE ROLE home_only").await; + + let error = ddl_err( + &state, + &su, + "CREATE USER stray WITH PASSWORD 'pass' ROLE home_only TENANT 4242", + ) + .await; + assert_undefined(&error, "home_only"); + assert!(state.credentials.get_user("stray").is_none()); +} + +#[tokio::test] +async fn a_held_role_is_not_dropped_until_no_user_holds_it() { + let state = make_state(); + let su = superuser(); + ddl_ok(&state, &su, "CREATE ROLE reviewer").await; + ddl_ok( + &state, + &su, + "CREATE USER rex WITH PASSWORD 'pass' ROLE reviewer", + ) + .await; + + let error = ddl_err(&state, &su, "DROP ROLE reviewer").await; + assert!( + error.contains("2BP01") && error.contains("rex"), + "expected 2BP01 naming the holder, got: {error}" + ); + assert!(state.roles.get_role("reviewer").is_some()); + + ddl_ok(&state, &su, "REVOKE reviewer FROM rex").await; + ddl_ok(&state, &su, "DROP ROLE reviewer").await; + assert!(state.roles.get_role("reviewer").is_none()); +} + +#[tokio::test] +async fn a_role_another_role_inherits_from_is_not_dropped() { + let state = make_state(); + let su = superuser(); + ddl_ok(&state, &su, "CREATE ROLE senior").await; + ddl_ok(&state, &su, "CREATE ROLE junior INHERIT senior").await; + + let error = ddl_err(&state, &su, "DROP ROLE senior").await; + assert!( + error.contains("2BP01") && error.contains("junior"), + "expected 2BP01 naming the child role, got: {error}" + ); + assert!(state.roles.get_role("senior").is_some()); +} + +/// An OIDC claim mapping may add only roles its provider's tenant defines: +/// a login mapped to an undefined role would hold nothing. +#[tokio::test] +async fn an_oidc_claim_mapping_refuses_an_undefined_role() { + let state = make_state_with_catalog(); + let su = superuser(); + ddl_ok(&state, &su, "CREATE TENANT mapped_roles ID 43").await; + + let error = ddl_err( + &state, + &su, + "CREATE OIDC PROVIDER ghost_mapping \ + ISSUER 'https://ghost-idp.example/' \ + JWKS_URI 'https://ghost-idp.example/jwks' \ + AUDIENCE 'nodedb-api' \ + TENANT 43 \ + CLAIM MAPPING WHEN sub = '*' SET DEFAULT_DATABASE = 1 ADD ROLES ['ghost_role']", + ) + .await; + assert_undefined(&error, "ghost_role"); + assert!( + state + .credentials + .catalog() + .get_oidc_provider("ghost_mapping") + .expect("catalog read") + .is_none(), + "a refused provider must not be stored" + ); +} diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 6b9a2a835..2303a3dbc 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -144,6 +144,7 @@ mod redaction_policy_database_scope; mod reindex_concurrent; mod restart_refused_write_not_resurrected; mod rls_policy_database_scope; +mod role_assignment_transaction; mod router_misroute_literals; mod scalar_aggregate_empty_input; mod scalar_aggregate_multicore_merge; @@ -309,6 +310,7 @@ mod trigger_e2e; mod truncate_engine_conformance; mod truncate_engine_conformance_columnar_family; mod txn_ddl_commit_registry_sync; +mod user_transaction; mod vector_index_bulk_delete_reindex; mod vector_index_bulk_update_reindex; mod vector_index_merge_reindex; diff --git a/nodedb/tests/wire/cases/role_assignment_transaction.rs b/nodedb/tests/wire/cases/role_assignment_transaction.rs new file mode 100644 index 000000000..c3e11d101 --- /dev/null +++ b/nodedb/tests/wire/cases/role_assignment_transaction.rs @@ -0,0 +1,109 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Role rules inside a transaction. +//! +//! A transaction buffers its DDL until COMMIT. A statement sees the roles and +//! users committed before the transaction and those its own transaction +//! created earlier: a role created in the transaction can be assigned in it, +//! and a role a user created in it holds cannot be dropped in it. + +use crate::harness::TestServer; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_role_created_in_a_transaction_can_be_assigned_in_it() { + let server = TestServer::start().await; + + for sql in [ + "BEGIN", + "CREATE ROLE txn_role", + "CREATE USER txn_user WITH PASSWORD 'txn-user-pass-1' ROLE txn_role", + "COMMIT", + ] { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + // The committed user holds the committed role, so the role is in use. + server.expect_error("DROP ROLE txn_role", "2BP01").await; + server.expect_error("DROP ROLE txn_role", "txn_user").await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_role_a_user_created_in_the_transaction_holds_is_not_dropped_in_it() { + let server = TestServer::start().await; + server + .exec("CREATE ROLE txn_held_role") + .await + .expect("create role"); + + server.exec("BEGIN").await.expect("begin"); + server + .exec("CREATE USER txn_holder WITH PASSWORD 'txn-holder-pass-1' ROLE txn_held_role") + .await + .expect("create user in transaction"); + server + .expect_error("DROP ROLE txn_held_role", "2BP01") + .await; + server.exec("ROLLBACK").await.expect("rollback"); + + // Nothing of the transaction survives, and the role is free to drop. + server + .exec("DROP ROLE txn_held_role") + .await + .expect("drop the role once no user holds it"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn an_undefined_role_is_refused_inside_a_transaction() { + let server = TestServer::start().await; + + server.exec("BEGIN").await.expect("begin"); + server + .expect_error( + "CREATE USER txn_ghost WITH PASSWORD 'txn-ghost-pass-1' ROLE read_write", + "42704", + ) + .await; + server.exec("ROLLBACK").await.expect("rollback"); +} + +/// A parent role and a child that inherits it, created in one transaction, +/// both exist after COMMIT, and the child inherits the parent. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_parent_and_child_role_created_in_one_transaction_both_commit() { + let server = TestServer::start().await; + + for sql in [ + "BEGIN", + "CREATE ROLE txn_parent", + "CREATE ROLE txn_child INHERIT txn_parent", + "COMMIT", + ] { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + let rows = server + .query_named_rows("SHOW ROLES") + .await + .expect("SHOW ROLES"); + let parent_of = |name: &str| { + rows.iter() + .find(|row| row.get("name").map(String::as_str) == Some(name)) + .map(|row| row.get("parent").cloned().unwrap_or_default()) + }; + assert_eq!( + parent_of("txn_parent"), + Some(String::new()), + "txn_parent exists" + ); + assert_eq!( + parent_of("txn_child"), + Some("txn_parent".to_string()), + "txn_child exists and inherits txn_parent" + ); +} diff --git a/nodedb/tests/wire/cases/user_transaction.rs b/nodedb/tests/wire/cases/user_transaction.rs new file mode 100644 index 000000000..85752d7bf --- /dev/null +++ b/nodedb/tests/wire/cases/user_transaction.rs @@ -0,0 +1,147 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! User DDL inside a transaction. +//! +//! A statement sees the users committed before its transaction and those its +//! own transaction created earlier, and not those it dropped. So a user +//! created in a transaction can be granted roles and altered in it, and a +//! user created and dropped in one transaction leaves nothing behind. + +use crate::harness::TestServer; + +/// The roles SHOW USERS reports for `username`, or `None` when no such user +/// exists. +async fn roles_of(server: &TestServer, username: &str) -> Option> { + let rows = server + .query_named_rows("SHOW USERS") + .await + .expect("SHOW USERS"); + rows.iter() + .find(|row| row.get("username").map(String::as_str) == Some(username)) + .map(|row| { + let mut roles: Vec = row + .get("roles") + .map(|roles| roles.split(", ").map(str::to_string).collect()) + .unwrap_or_default(); + roles.sort(); + roles + }) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_user_created_in_a_transaction_can_be_granted_and_altered_in_it() { + // Password mode: a trust-mode server accepts any password, so only this + // mode shows which password the committed user holds. + let server = TestServer::start_password().await; + + for sql in [ + "BEGIN", + "CREATE USER txn_user_u WITH PASSWORD 'txn-first-pass-1' ROLE readonly", + "GRANT ROLE readwrite TO txn_user_u", + "ALTER USER txn_user_u SET PASSWORD 'txn-second-pass-2'", + "COMMIT", + ] { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + assert_eq!( + roles_of(&server, "txn_user_u").await, + Some(vec!["readonly".to_string(), "readwrite".to_string()]), + "the committed user holds the created and the granted role" + ); + let (client, handle) = server + .connect_as("txn_user_u", "txn-second-pass-2") + .await + .unwrap_or_else(|e| panic!("log in with the password set in the transaction: {e}")); + drop(client); + handle.abort(); + assert!( + server + .connect_as("txn_user_u", "txn-first-pass-1") + .await + .is_err(), + "the password replaced in the transaction must not log in" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_user_created_and_dropped_in_one_transaction_leaves_nothing() { + let server = TestServer::start().await; + + for sql in [ + "BEGIN", + "CREATE USER txn_user_gone WITH PASSWORD 'txn-gone-pass-1' ROLE readonly", + "GRANT ROLE readwrite TO txn_user_gone", + "DROP USER txn_user_gone", + "COMMIT", + ] { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + assert_eq!(roles_of(&server, "txn_user_gone").await, None); + // The name is free again. + server + .exec("CREATE USER txn_user_gone WITH PASSWORD 'txn-gone-pass-2' ROLE readonly") + .await + .expect("the name of a user dropped in its own transaction is free"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_rolled_back_user_leaves_nothing() { + // Password mode, so a refused login proves the user is absent rather + // than trust-mode acceptance of any password. + let server = TestServer::start_password().await; + + for sql in [ + "BEGIN", + "CREATE USER txn_user_rolled WITH PASSWORD 'txn-rolled-pass-1' ROLE readonly", + "GRANT ROLE readwrite TO txn_user_rolled", + "ALTER USER txn_user_rolled SET PASSWORD 'txn-rolled-pass-2'", + "ROLLBACK", + ] { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + assert_eq!(roles_of(&server, "txn_user_rolled").await, None); + assert!( + server + .connect_as("txn_user_rolled", "txn-rolled-pass-2") + .await + .is_err(), + "a rolled-back user must not log in" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_user_dropped_in_a_transaction_is_not_visible_in_it() { + let server = TestServer::start().await; + server + .exec("CREATE USER txn_user_dropped WITH PASSWORD 'txn-dropped-pass-1' ROLE readonly") + .await + .expect("create user"); + + server.exec("BEGIN").await.expect("begin"); + server + .exec("DROP USER txn_user_dropped") + .await + .expect("drop user in transaction"); + server + .expect_error("GRANT ROLE readwrite TO txn_user_dropped", "42704") + .await; + server.exec("ROLLBACK").await.expect("rollback"); + + assert_eq!( + roles_of(&server, "txn_user_dropped").await, + Some(vec!["readonly".to_string()]), + "the rolled-back drop leaves the user as it was" + ); +} From ad7c0aed871262b33da34e86ed6b7b49b3cf4faa Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 10:52:55 +0800 Subject: [PATCH 35/64] feat(backup): back up and restore on a replicated consistent cut MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A backup used to snapshot each engine against its local watermark, with no guarantee that every replica's copy, or every group's copy of a multi-group write, landed on the same side of the cut. Add a consistent-cut protocol: the cut proposes a watermark-carrying barrier into every Raft data group and Calvin scheduler this node hosts and waits for this node's apply of it, so every write below the watermark is captured and every write at or above it is excluded on every replica alike. `state::tenant_marks` and the new `_system.tenant_group_marks` / `_system.calvin_applied` catalog tables make each group's applied floor and a scheduler's Calvin progress durable across a restart, so replicated state, not a single node's memory, answers what a group already covered. RESTORE gains a staleness guard (`backup::restore::guard`) that reads every data group's durable tenant write marks — from a local replica once it is caught up, or from a current replica of a group elsewhere — before applying anything, and a quorum check (`backup::restore::quorum`) that fails at once, naming the unreachable group and nodes, rather than waiting out a deadline on a group with no reachable majority. Durable and local write marks (`backup::restore::durable`, local_write_stamps) and dedicated per-engine reissue paths (`kv_reissue`, `redo_reissue`) replace the prior ad hoc restore-time write paths so a restored row's write order is preserved. Split the distributed applier's apply loop into a directory (`context`, `group_watch`, `lane`, `metadata_floor`, `pipeline`, `start`) built around a per-group `lane` that tracks entries in log order from enqueue through settle, and an `apply_window` that bounds how far ahead of a group's durable floor its entries may run. Add a per-node outcome-floor snapshot helper (`capture_memory`) and give diagnostics a dedicated `raft_apply` context/recording module for tracing an entry through propose, apply, and settle. Extend the authorization lease's coverage rebuild to settle against the restart-recovered Calvin applied state, rate-limit its withheld-renewal warning per node, and report which groups fall short of their floor. Cover the new paths with cluster tests for a remote consistent cut and a restart mid-restore, and wire-protocol and in-process tests for calvin cut markers, durable and local restore marks, restore staleness, and apply-pipeline group independence. --- .config/nextest.toml | 2 +- nodedb-cluster-tests/Cargo.toml | 4 + .../cases/cluster_backup_remote_cut.rs | 271 +++++++++++ .../cases/cluster_backup_restore.rs | 113 ++--- .../cluster_restore_documents_restart.rs | 235 +++++++++ .../cluster_restore_refuses_on_non_replica.rs | 125 +++++ .../tests/common_suite/cases/mod.rs | 3 + .../auth_objects.rs | 117 ++++- nodedb-cluster/src/calvin/sequencer/entry.rs | 6 + nodedb-cluster/src/calvin/sequencer/replay.rs | 11 + .../src/calvin/sequencer/service/core.rs | 1 + .../calvin/sequencer/state_machine/apply.rs | 34 ++ .../src/calvin/types/scheduler_input.rs | 4 + .../src/raft_loop/handle_rpc/plan_dispatch.rs | 5 +- nodedb-cluster/src/raft_loop/proposals.rs | 10 +- nodedb-cluster/src/rpc_codec/data_propose.rs | 94 +++- nodedb-cluster/src/rpc_codec/mod.rs | 4 +- nodedb-cluster/src/transport/client/close.rs | 40 ++ nodedb-cluster/src/transport/client/mod.rs | 1 + .../src/physical_plan/cluster_event.rs | 5 + .../src/physical_plan/collection.rs | 1 + nodedb-physical/src/physical_plan/meta.rs | 13 +- nodedb-physical/src/physical_plan/mod.rs | 2 + .../src/physical_plan/redo_origin.rs | 27 ++ .../src/cluster_harness/cluster/bringup.rs | 154 +----- .../src/cluster_harness/cluster/mod.rs | 7 +- .../src/cluster_harness/cluster/ready.rs | 162 +++++++ .../src/cluster_harness/cluster/restart.rs | 39 ++ .../cluster_harness/cluster/spawn_variants.rs | 20 + .../cluster_harness/node/inspect/snapshot.rs | 1 + .../cluster_harness/node/inspect/topology.rs | 21 + .../src/cluster_harness/node/lifecycle/mod.rs | 4 +- .../cluster_harness/node/lifecycle/restart.rs | 102 ++++ .../node/lifecycle/spawn_full.rs | 13 +- .../node/lifecycle/spawn_variants.rs | 2 +- .../node/lifecycle/teardown.rs | 14 + nodedb-test-support/src/tx_commit.rs | 1 + nodedb/src/bridge/dispatch/outcome_floor.rs | 26 + .../src/control/array_sync/inbound_propose.rs | 6 +- .../src/control/array_sync/raft_apply/cell.rs | 3 + .../control/array_sync/raft_apply/common.rs | 23 +- .../src/control/array_sync/raft_apply/op.rs | 3 + .../control/array_sync/raft_apply/schema.rs | 1 + nodedb/src/control/backup/cut.rs | 251 ++++++++++ nodedb/src/control/backup/mod.rs | 1 + nodedb/src/control/backup/orchestrator.rs | 232 +++++---- .../backup/restore/columnar_reissue.rs | 71 --- .../control/backup/restore/crdt_reissue.rs | 2 +- nodedb/src/control/backup/restore/durable.rs | 114 +++++ nodedb/src/control/backup/restore/guard.rs | 416 ++++++++++++++++ .../src/control/backup/restore/kv_reissue.rs | 118 +++++ nodedb/src/control/backup/restore/mod.rs | 12 +- .../control/backup/restore/orchestrate/mod.rs | 5 +- .../backup/restore/orchestrate/rebind.rs | 123 +++-- .../backup/restore/orchestrate/reissue.rs | 12 +- .../backup/restore/orchestrate/restore.rs | 201 +++----- .../backup/restore/orchestrate/stats.rs | 14 +- nodedb/src/control/backup/restore/quorum.rs | 110 +++++ .../backup/restore/redo_reissue/commit.rs | 206 ++++++++ .../backup/restore/redo_reissue/documents.rs | 369 ++++++++++++++ .../backup/restore/redo_reissue/edges.rs | 136 ++++++ .../backup/restore/redo_reissue/mod.rs | 13 + .../backup/restore/redo_reissue/reissue.rs | 61 +++ .../backup/restore/redo_reissue/sub_record.rs | 148 ++++++ .../backup/restore/redo_reissue/units.rs | 34 ++ nodedb/src/control/backup/restore/remote.rs | 96 ---- nodedb/src/control/backup/restore/sections.rs | 46 +- .../backup/restore/timeseries_reissue.rs | 72 +-- nodedb/src/control/backup/restore/topology.rs | 188 ------- .../control/backup/restore/vector_reissue.rs | 73 --- nodedb/src/control/backup/snapshot_keys.rs | 14 +- nodedb/src/control/checkpoint_manager.rs | 46 ++ nodedb/src/control/checkpoint_task.rs | 2 + .../control/cluster/array_executor/write.rs | 1 + .../cluster/calvin/scheduler/applied_gate.rs | 10 + .../calvin/scheduler/applied_mirror.rs | 30 ++ .../cluster/calvin/scheduler/cut_floor.rs | 173 +++++++ .../driver/core/commit_resolve/apply_tail.rs | 39 +- .../scheduler/driver/core/completion_route.rs | 35 +- .../scheduler/driver/core/cut_marker.rs | 106 ++++ .../calvin/scheduler/driver/core/intake.rs | 4 +- .../calvin/scheduler/driver/core/mod.rs | 1 + .../scheduler/driver/core/owed/retry.rs | 4 +- .../calvin/scheduler/driver/core/process.rs | 5 +- .../calvin/scheduler/driver/core/scheduler.rs | 17 +- .../driver/core/sequencer_proposer/raft.rs | 49 +- .../driver/core/sequencer_proposer/seam.rs | 3 + .../control/cluster/calvin/scheduler/mod.rs | 5 +- .../cluster/calvin/scheduler/recovery.rs | 88 ++++ .../src/control/cluster/snapshot_applier.rs | 12 + .../src/control/cluster/snapshot_builder.rs | 19 +- .../control/cluster/start_raft/group_setup.rs | 9 +- .../cluster/start_raft/proposer_wiring.rs | 223 +++++++-- .../src/control/cluster/start_raft_helpers.rs | 18 +- nodedb/src/control/crdt_admission.rs | 26 +- .../control/distributed_applier/applier.rs | 144 +++--- .../apply_loop/array_dispatch.rs | 53 -- .../apply_loop/calvin_read_result.rs | 1 + .../distributed_applier/apply_loop/context.rs | 83 ++++ .../distributed_applier/apply_loop/driver.rs | 277 ++--------- .../apply_loop/group_watch.rs | 82 ++++ .../distributed_applier/apply_loop/helpers.rs | 21 + .../distributed_applier/apply_loop/lane.rs | 458 ++++++++++++++++++ .../apply_loop/metadata_floor.rs | 130 +++++ .../distributed_applier/apply_loop/mod.rs | 37 +- .../apply_loop/pipeline.rs | 285 +++++++++++ .../apply_loop/proposal_gate.rs | 182 +++++-- .../distributed_applier/apply_loop/start.rs | 207 ++++++++ .../apply_loop/transaction_redo.rs | 68 ++- .../apply_loop/write_dispatch.rs | 154 ++++-- .../distributed_applier/apply_window.rs | 101 ++++ nodedb/src/control/distributed_applier/mod.rs | 4 +- .../distributed_applier/propose_tracker.rs | 107 +++- .../src/control/exec_receiver/backup_cut.rs | 38 ++ nodedb/src/control/exec_receiver/executor.rs | 25 +- nodedb/src/control/exec_receiver/mod.rs | 2 + .../src/control/exec_receiver/tenant_marks.rs | 63 +++ nodedb/src/control/fail_gate.rs | 10 +- nodedb/src/control/gateway/core.rs | 11 - nodedb/src/control/gateway/router.rs | 1 + nodedb/src/control/otel/receiver.rs | 2 +- .../control/planner/rls_injection/array.rs | 4 +- .../src/control/planner/rls_injection/meta.rs | 5 +- .../rls_injection/permission_tree/array.rs | 4 +- .../rls_injection/permission_tree/meta.rs | 5 +- .../security/auth_lease/calvin_acks.rs | 48 ++ nodedb/src/control/security/auth_lease/mod.rs | 1 + .../control/security/auth_lease/renew_loop.rs | 26 +- .../control/security/auth_lease/service.rs | 50 +- .../src/control/security/auth_lease/table.rs | 32 ++ .../security/auth_lease/withheld_warn.rs | 50 ++ .../security/catalog/bootstrap_tables.rs | 2 + .../security/catalog/calvin_applied.rs | 181 +++++++ nodedb/src/control/security/catalog/mod.rs | 2 + .../security/catalog/tenant_group_marks.rs | 95 ++++ .../control/server/dispatch_utils/dispatch.rs | 4 + .../dispatch_utils/durability_barrier.rs | 1 + .../src/control/server/dispatch_utils/mod.rs | 3 +- .../submit_write/funnel/dispatch.rs | 50 +- .../submit_write/funnel/driver.rs | 116 +++-- .../dispatch_utils/submit_write/funnel/mod.rs | 5 +- .../submit_write/funnel/pending.rs | 85 ++++ .../submit_write/funnel/response.rs | 53 +- .../submit_write/funnel/wal_append.rs | 1 + .../server/dispatch_utils/submit_write/mod.rs | 2 +- .../dispatch_utils/submit_write/params.rs | 6 + .../server/exchange/all_cores/dispatch.rs | 34 +- .../server/exchange/all_cores/snapshot.rs | 6 + .../control/server/exchange/owning_core.rs | 47 +- .../server/pgwire/handler/dispatch/entry.rs | 33 +- .../server/pgwire/handler/dispatch/local.rs | 1 + .../handler/routing/gateway_dispatch.rs | 10 +- .../pgwire/handler/routing/gateway_fold.rs | 64 ++- .../control/server/pgwire/types/error_map.rs | 10 + .../response_shape/compose/materialized.rs | 8 +- .../control/server/session_auth/bearer_jwt.rs | 2 +- .../neutral/tenant/move_tenant/snapshot.rs | 1 + .../shared/ddl/sync_dispatch/dispatch.rs | 14 +- .../control/server/shared/returning/inject.rs | 4 +- .../shared/session/commit/single_shard.rs | 2 +- .../server/shared/write_admission/mod.rs | 4 +- .../shared/write_admission/predicate/mod.rs | 2 + .../predicate/txn_buffering/classify.rs | 5 +- .../predicate/user_data_write.rs | 75 +++ .../server/sync/raft_dispatch/propose.rs | 13 +- .../server/sync/raft_dispatch/write.rs | 25 +- .../src/control/server/wal_dispatch/core.rs | 1 + nodedb/src/control/shutdown/registry.rs | 61 ++- nodedb/src/control/state/calvin_cuts.rs | 98 ++++ nodedb/src/control/state/calvin_local.rs | 11 +- nodedb/src/control/state/fields.rs | 5 +- nodedb/src/control/state/init.rs | 1 + nodedb/src/control/state/init_prod/open.rs | 3 + .../src/control/state/local_write_stamps.rs | 134 +++++ nodedb/src/control/state/methods.rs | 37 -- nodedb/src/control/state/mod.rs | 6 + nodedb/src/control/state/tenant_marks.rs | 373 ++++++++++++++ nodedb/src/control/state/tenant_write.rs | 104 ++++ nodedb/src/control/vshard_admission.rs | 14 +- .../control/wal_replication/decode/entry.rs | 3 + .../decode/transaction_redo.rs | 9 +- .../encode/transaction_redo.rs | 1 + .../control/wal_replication/legacy_entry.rs | 2 + nodedb/src/control/wal_replication/propose.rs | 92 ++-- .../wal_replication/transaction_redo/apply.rs | 29 +- .../wal_replication/transaction_redo/mod.rs | 2 +- .../transaction_redo/payload.rs | 6 +- .../control/wal_replication/types/aliases.rs | 8 +- .../wal_replication/types/replicated_entry.rs | 17 +- .../wal_replication/types/replicated_write.rs | 9 + .../data/executor/core_loop/calvin_fence.rs | 19 + nodedb/src/data/executor/dispatch/meta.rs | 6 +- .../handlers/snapshot/capture_memory.rs | 177 +++++++ .../data/executor/handlers/snapshot/create.rs | 203 ++------ .../data/executor/handlers/snapshot/mod.rs | 1 + .../handlers/snapshot/restore/engines.rs | 41 +- .../snapshot/restore/tenant_snapshot.rs | 132 +++-- .../handlers/transaction/redo_apply/entry.rs | 17 +- .../transaction/redo_apply/validate.rs | 24 +- nodedb/src/diag/context/mod.rs | 2 + nodedb/src/diag/context/raft_apply.rs | 79 +++ nodedb/src/diag/mod.rs | 9 +- nodedb/src/diag/recording/mod.rs | 4 + nodedb/src/diag/recording/raft_apply.rs | 65 +++ nodedb/src/engine/bitemporal/enforcement.rs | 10 +- nodedb/src/engine/sparse/btree_scan.rs | 2 +- .../src/engine/sparse/btree_versioned/mod.rs | 1 + .../engine/sparse/btree_versioned/snapshot.rs | 144 ++++++ .../retention_policy/enforcement.rs | 11 +- nodedb/src/error/types.rs | 25 + nodedb/src/error_classify.rs | 2 + nodedb/src/event/alert/executor.rs | 8 +- nodedb/src/main_boot/shutdown_wiring.rs | 58 +++ nodedb/src/types/snapshot.rs | 24 + .../apply_pipeline_group_independence.rs | 108 +++++ nodedb/tests/calvin_hold_liveness.rs | 113 +++++ nodedb/tests/crash_harness/mod.rs | 1 + nodedb/tests/crash_harness/vshards.rs | 53 ++ nodedb/tests/crash_replay_stamp_calvin.rs | 67 +-- .../cases/auth_service_account_scope.rs | 59 ++- nodedb/tests/wire/cases/backup_support.rs | 60 +++ .../cases/graph_analytics_authorization.rs | 6 + .../wire/cases/graph_match_authorization.rs | 6 + .../cases/graph_traversal_authorization.rs | 6 + nodedb/tests/wire/cases/mod.rs | 7 + .../cases/query_function_authorization.rs | 12 + .../wire/cases/sorted_index_authorization.rs | 6 + .../tests/wire/cases/sql_backup_calvin_cut.rs | 162 +++++++ .../wire/cases/sql_backup_consistent_cut.rs | 126 +++++ .../cases/sql_backup_restore_documents.rs | 157 ++++++ .../cases/sql_backup_restore_durable_marks.rs | 109 +++++ .../cases/sql_backup_restore_local_marks.rs | 124 +++++ .../cases/sql_backup_restore_staleness.rs | 87 ++++ .../cases/version_history_authorization.rs | 6 + nodedb/tests/wire/harness/lifecycle.rs | 14 + 235 files changed, 10677 insertions(+), 2232 deletions(-) create mode 100644 nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs create mode 100644 nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_documents_restart.rs create mode 100644 nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_refuses_on_non_replica.rs create mode 100644 nodedb-cluster/src/transport/client/close.rs create mode 100644 nodedb-physical/src/physical_plan/redo_origin.rs create mode 100644 nodedb-test-support/src/cluster_harness/cluster/ready.rs create mode 100644 nodedb-test-support/src/cluster_harness/cluster/restart.rs create mode 100644 nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs create mode 100644 nodedb/src/control/backup/cut.rs create mode 100644 nodedb/src/control/backup/restore/durable.rs create mode 100644 nodedb/src/control/backup/restore/guard.rs create mode 100644 nodedb/src/control/backup/restore/kv_reissue.rs create mode 100644 nodedb/src/control/backup/restore/quorum.rs create mode 100644 nodedb/src/control/backup/restore/redo_reissue/commit.rs create mode 100644 nodedb/src/control/backup/restore/redo_reissue/documents.rs create mode 100644 nodedb/src/control/backup/restore/redo_reissue/edges.rs create mode 100644 nodedb/src/control/backup/restore/redo_reissue/mod.rs create mode 100644 nodedb/src/control/backup/restore/redo_reissue/reissue.rs create mode 100644 nodedb/src/control/backup/restore/redo_reissue/sub_record.rs create mode 100644 nodedb/src/control/backup/restore/redo_reissue/units.rs delete mode 100644 nodedb/src/control/backup/restore/remote.rs delete mode 100644 nodedb/src/control/backup/restore/topology.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/cut_floor.rs create mode 100644 nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs delete mode 100644 nodedb/src/control/distributed_applier/apply_loop/array_dispatch.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/context.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/group_watch.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/lane.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/pipeline.rs create mode 100644 nodedb/src/control/distributed_applier/apply_loop/start.rs create mode 100644 nodedb/src/control/distributed_applier/apply_window.rs create mode 100644 nodedb/src/control/exec_receiver/backup_cut.rs create mode 100644 nodedb/src/control/exec_receiver/tenant_marks.rs create mode 100644 nodedb/src/control/security/auth_lease/withheld_warn.rs create mode 100644 nodedb/src/control/security/catalog/calvin_applied.rs create mode 100644 nodedb/src/control/security/catalog/tenant_group_marks.rs create mode 100644 nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs create mode 100644 nodedb/src/control/server/shared/write_admission/predicate/user_data_write.rs create mode 100644 nodedb/src/control/state/calvin_cuts.rs create mode 100644 nodedb/src/control/state/local_write_stamps.rs create mode 100644 nodedb/src/control/state/tenant_marks.rs create mode 100644 nodedb/src/control/state/tenant_write.rs create mode 100644 nodedb/src/data/executor/handlers/snapshot/capture_memory.rs create mode 100644 nodedb/src/diag/context/raft_apply.rs create mode 100644 nodedb/src/diag/recording/raft_apply.rs create mode 100644 nodedb/src/engine/sparse/btree_versioned/snapshot.rs create mode 100644 nodedb/tests/apply_pipeline_group_independence.rs create mode 100644 nodedb/tests/calvin_hold_liveness.rs create mode 100644 nodedb/tests/crash_harness/vshards.rs create mode 100644 nodedb/tests/wire/cases/backup_support.rs create mode 100644 nodedb/tests/wire/cases/sql_backup_calvin_cut.rs create mode 100644 nodedb/tests/wire/cases/sql_backup_consistent_cut.rs create mode 100644 nodedb/tests/wire/cases/sql_backup_restore_documents.rs create mode 100644 nodedb/tests/wire/cases/sql_backup_restore_durable_marks.rs create mode 100644 nodedb/tests/wire/cases/sql_backup_restore_local_marks.rs create mode 100644 nodedb/tests/wire/cases/sql_backup_restore_staleness.rs diff --git a/.config/nextest.toml b/.config/nextest.toml index 591f66946..dd8f29c48 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -147,7 +147,7 @@ slow-timeout = { period = "30s", terminate-after = 8 } # hide the cause rather than fix it. The tail this costs is bounded — roughly a # dozen crash/shutdown tests at about ten seconds each. [[profile.default.overrides]] -filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | binary(crash_replay_stamp) | binary(crash_replay_stamp_calvin) | binary(crash_kv_atomic_autocommit) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' +filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | binary(crash_replay_stamp) | binary(crash_replay_stamp_calvin) | binary(calvin_hold_liveness) | binary(apply_pipeline_group_independence) | binary(crash_kv_atomic_autocommit) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' test-group = 'server-process' threads-required = 'num-test-threads' diff --git a/nodedb-cluster-tests/Cargo.toml b/nodedb-cluster-tests/Cargo.toml index 40e10decf..fd6403f2c 100644 --- a/nodedb-cluster-tests/Cargo.toml +++ b/nodedb-cluster-tests/Cargo.toml @@ -10,6 +10,10 @@ description = "3-node integration test suite for NodeDB. Heavy; runs in its own [lib] path = "src/lib.rs" +[features] +# Arms the in-process fail points the cluster tests park writes at. +failpoints = ["nodedb/failpoints", "nodedb-types/failpoints"] + [dependencies] [dev-dependencies] diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs new file mode 100644 index 000000000..bb0d1d613 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs @@ -0,0 +1,271 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's consistent cut binds a remote source node too. +//! +//! The backup's coordinator snapshots every vShard from the leader of its +//! group. The test parks a write on the collection's group leader only, at +//! the fail gate `funnel::before_dispatch::node::`, between +//! its record and its core. The coordinator is another node, so its own +//! replica applies the write and its own cut passes. The leader must take the +//! same cut before it snapshots: the backup waits until the parked write +//! applies there, and the envelope holds the row. After the purge settles on +//! every node, a restore brings the row back on every node. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +use std::time::Duration; + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; +use nodedb_types::backup_envelope::{DEFAULT_MAX_TOTAL_BYTES, parse_encrypted}; +use nodedb_types::fail_point::{FailAction, FailGuard}; + +use crate::common; +use common::cluster_harness::wait::wait_for; +use common::cluster_harness::{TestCluster, TestClusterNode}; + +/// Fixed test KEK the cluster harness injects into every node. +const TEST_KEK: [u8; 32] = [0x42u8; 32]; + +const TENANT: u64 = 1; +const COLLECTION: &str = "rcut_docs"; + +/// How long the parked write and the backup must stay unfinished. +const PARKED_FOR: Duration = Duration::from_millis(1500); + +fn db_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn drain_backup(client: &tokio_postgres::Client) -> Result, String> { + let stream = client + .copy_out(&format!("COPY (BACKUP TENANT {TENANT}) TO STDOUT")) + .await + .map_err(|e| db_detail(&e))?; + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice(&chunk.map_err(|e| db_detail(&e))?); + } + Ok(bytes) +} + +async fn push_restore(client: &tokio_postgres::Client, envelope: Vec) -> Result<(), String> { + let sink = client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({TENANT}) FROM STDIN")) + .await + .map_err(|e| db_detail(&e))?; + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .map_err(|e| db_detail(&e))?; + sink.as_mut() + .finish() + .await + .map(|_| ()) + .map_err(|e| db_detail(&e)) +} + +/// The leader of `group_id` in `node`'s routing table: the node a backup +/// coordinated there snapshots the group's vShards from. +fn routing_leader(node: &TestClusterNode, group_id: u64) -> u64 { + node.shared + .cluster_routing + .as_ref() + .and_then(|routing| { + routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_info(group_id) + .map(|info| info.leader) + }) + .unwrap_or(0) +} + +/// Whether `node` purged `collection`: its catalog row is gone and its WAL +/// tombstone is recorded, so the async purge ran on its Data Plane. +fn purged_on(node: &TestClusterNode, collection: &str) -> bool { + let catalog = node.shared.credentials.catalog(); + let active = matches!( + catalog.get_collection(nodedb_types::DatabaseId::DEFAULT, TENANT, collection), + Ok(Some(c)) if c.is_active + ); + let tombstoned = catalog + .load_wal_tombstones() + .map(|set| { + set.iter() + .any(|(_, tenant, name, lsn)| tenant == TENANT && name == collection && lsn > 0) + }) + .unwrap_or(false); + !active && tombstoned +} + +/// Whether the backup `envelope` holds a KV row of `collection` whose key +/// carries `key`. +fn envelope_holds_kv_row(envelope: &[u8], collection: &str, key: &[u8]) -> bool { + let parsed = + parse_encrypted(envelope, DEFAULT_MAX_TOTAL_BYTES, &TEST_KEK).expect("parse the envelope"); + parsed + .sections + .iter() + .filter_map(|section| { + zerompk::from_msgpack::(§ion.body).ok() + }) + .flat_map(|snapshot| snapshot.kv_tables) + .filter(|(name, _)| name == collection) + .filter_map(|(_, rows)| zerompk::from_msgpack::, Vec, u64)>>(&rows).ok()) + .flatten() + .any(|(row_key, _, _)| row_key.windows(key.len()).any(|window| window == key)) +} + +async fn connect(pg_addr: std::net::SocketAddr) -> tokio_postgres::Client { + let conn_str = format!( + "host={} port={} user=nodedb dbname=default", + pg_addr.ip(), + pg_addr.port() + ); + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .expect("connect a second client"); + tokio::spawn(connection); + client +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_backup_waits_for_a_write_held_on_a_remote_source_node() { + let cluster = TestCluster::spawn_three().await.expect("cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')" + )) + .await + .expect("CREATE COLLECTION"); + + // The coordinator snapshots the collection's vShard from the group leader + // its routing table names. Wait until every node's table names the + // elected leader, so the coordinator picked below names it too. + let group_id = cluster.nodes[0] + .group_id_for_collection(COLLECTION) + .expect("the collection's data group"); + let elected = || { + cluster.nodes[0] + .all_group_leaders() + .into_iter() + .find(|(group, _)| *group == group_id) + .map_or(0, |(_, leader)| leader) + }; + wait_for( + "every routing table names the elected leader of the collection's group", + Duration::from_secs(10), + Duration::from_millis(50), + || { + let leader = elected(); + leader != 0 + && cluster + .nodes + .iter() + .all(|node| routing_leader(node, group_id) == leader) + }, + ) + .await; + let source = elected(); + let coordinator = cluster + .nodes + .iter() + .position(|node| node.node_id != source) + .expect("a node that is not the source"); + + // Park the write on the source node only. + let gate_dir = tempfile::tempdir().expect("gate tempdir"); + let release = gate_dir.path().join("release-source-apply"); + let _gate = FailGuard::install( + &format!("funnel::before_dispatch::node{source}::{COLLECTION}"), + FailAction::WaitForFile(release.clone()), + ); + + let writer = connect(cluster.nodes[coordinator].pg_addr).await; + let insert = tokio::spawn(async move { + writer + .simple_query(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('held', 'x')" + )) + .await + .map(|_| ()) + .map_err(|e| db_detail(&e)) + }); + tokio::time::sleep(PARKED_FOR).await; + + let backup_client = connect(cluster.nodes[coordinator].pg_addr).await; + let backup = tokio::spawn(async move { drain_backup(&backup_client).await }); + tokio::time::sleep(PARKED_FOR).await; + assert!( + !backup.is_finished(), + "the backup snapshotted node {source} before a write held there had its outcome" + ); + + std::fs::write(&release, b"release").expect("release the held apply"); + insert + .await + .expect("insert task") + .unwrap_or_else(|e| panic!("insert: {e}")); + let envelope = backup + .await + .expect("backup task") + .unwrap_or_else(|e| panic!("backup: {e}")); + + assert!( + envelope_holds_kv_row(&envelope, COLLECTION, b"held"), + "the backup missed a write committed before its cut" + ); + + cluster + .exec_ddl_on_any_leader(&format!("DROP COLLECTION {COLLECTION} PURGE")) + .await + .expect("purge the collection"); + // The purge reaches each node's Data Plane asynchronously. A purge that + // lands after the restore would remove the restored row. + wait_for( + "every node purged the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || cluster.nodes.iter().all(|node| purged_on(node, COLLECTION)), + ) + .await; + push_restore(&cluster.nodes[coordinator].client, envelope) + .await + .unwrap_or_else(|e| panic!("restore: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + for node in &cluster.nodes { + let rows = node + .client + .simple_query(&format!( + "SELECT value FROM {COLLECTION} WHERE key = 'held'" + )) + .await + .unwrap_or_else(|e| panic!("read the restored row: {}", db_detail(&e))); + let values: Vec = rows + .iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .collect(); + assert_eq!( + values, + vec!["x".to_owned()], + "node {} does not hold the restored row", + node.node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore.rs index df7195af2..4df7e4912 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore.rs @@ -313,21 +313,23 @@ async fn backup_watermark_advances_after_writes() { // 3-node spawn cost is acceptable. // ──────────────────────────────────────────────────────────────────── -// Mid-flight node failure during restore fan-out. +// Restore into a group that lost its quorum. // -// The restore orchestrator iterates per-node sub-snapshots via -// `sync_dispatch` (local) or `RaftRpc::ExecuteRequest` (remote) and -// surfaces the first per-node error. Two contracts must hold: +// A restore commits every write through Raft: catalog rows through the +// metadata group, rows through each collection's data group. One dead node +// of three leaves every group a majority, and the restore succeeds. Two dead +// nodes leave no group a majority, and nothing can commit. Two contracts +// must hold: // -// 1. Loud failure — the client receives a structured error that -// names the failing node, never silent partial success. -// 2. Idempotent retry — the engine-level PointPut writes are -// idempotent, so a subsequent retry after the failed node is -// restored converges to the expected state. +// 1. Loud, prompt failure — the client receives a typed error that names +// the group and the nodes it cannot reach, well within the statement +// deadline, never a hang and never a silent success. +// 2. Nothing applied — no group on the surviving node applied an entry of +// the refused restore. // ──────────────────────────────────────────────────────────────────── #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn restore_surfaces_failing_node_id_on_midflight_failure() { +async fn restore_refuses_a_group_without_quorum_and_applies_nothing() { let mut cluster = TestCluster::spawn_three().await.expect("cluster"); cluster @@ -347,64 +349,55 @@ async fn restore_surfaces_failing_node_id_on_midflight_failure() { .await .unwrap_or_else(|e| panic!("insert f{i}: {}", db_detail(&e))); } + let bytes = drain_backup(0, &cluster, TENANT).await; - // Fault-inject: take down node 2, then pin node 0's routing table - // so it still believes node 2 is the leader for every raft group. - // Without the stale-route pin, quorum-replicated failover hides the - // fault entirely — restore transparently re-routes to a surviving - // replica and succeeds. Pinning forces the restore fan-out to - // attempt dispatch against the dead peer so the structured - // error-naming contract is actually exercised. - let downed_node_id = cluster.nodes[2].node_id; - let downed = cluster.nodes.remove(2); - downed.shutdown().await; - // Wait for the surviving nodes' SWIM/topology subsystem to observe - // the peer death before pinning the stale routes. If we pin before - // SWIM converges, an in-flight SWIM update can land *after* the pin - // and overwrite it with a fresh leader hint, defeating the - // fault-injection. The pin must be the last write to the routing - // table for `downed_node_id`'s groups. + // Take down two of the three nodes: no group keeps a majority. + let mut downed_ids = Vec::new(); + for _ in 0..2 { + let downed = cluster.nodes.remove(1); + downed_ids.push(downed.node_id); + downed.shutdown().await; + } + downed_ids.sort_unstable(); wait_for( - "node 0 marks downed peer as inactive in its topology view", - Duration::from_secs(10), + "the surviving node marks both downed peers inactive", + Duration::from_secs(20), Duration::from_millis(20), - || cluster.nodes[0].active_topology_size() < 3, + || cluster.nodes[0].active_topology_size() == 1, ) .await; - // Back up only once the topology has settled, and before the routes - // are pinned. - // - // The restore path refuses an envelope whose watermark predates the - // destination's last observed write-HLC, and that high-water advances on - // every successful dispatch for the tenant — not only on writes the test - // issues itself. Taking the backup first leaves the node teardown and the - // SWIM convergence window sitting between the envelope and the restore, - // and anything dispatched in there carries the high-water past the - // envelope. The restore then fails the staleness check instead of - // reaching the fan-out, and the assertion below sees the wrong loud - // failure. Capturing after the teardown keeps the envelope dominant; - // pinning afterwards adds nothing, since it only rewrites a local routing - // table. - // - // The backup itself still succeeds with the peer down: routing has failed - // over to the surviving replicas at this point, which is exactly the - // transparent recovery the pin below goes on to defeat. - let bytes = drain_backup(0, &cluster, TENANT).await; - - for group_id in 0..8u64 { - cluster.nodes[0].force_stale_route_for_test(group_id, downed_node_id); - } - - let err = push_restore(0, &cluster, TENANT, bytes) - .await - .expect_err("restore must fail loudly when fan-out targets a dead node"); + let applied_before = cluster.nodes[0].shared.group_watchers().snapshot(); + let started = std::time::Instant::now(); + let err = tokio::time::timeout( + Duration::from_secs(10), + push_restore(0, &cluster, TENANT, bytes), + ) + .await + .expect("a restore into a group without quorum must fail promptly, not hang") + .expect_err("a restore into a group without quorum must fail loudly"); assert!( - err.contains(&format!("node {downed_node_id}")) - || err.contains(&downed_node_id.to_string()), - "restore error must name the failing node id {downed_node_id} so the \ - operator can act; got: {err}" + err.contains("has no reachable quorum"), + "expected the typed quorum refusal, got: {err}" + ); + assert!( + err.contains(&format!("unreachable {downed_ids:?}")), + "the refusal must name the unreachable nodes {downed_ids:?}, got: {err}" + ); + assert!( + err.contains("raft group "), + "the refusal must name the group, got: {err}" + ); + assert!( + started.elapsed() < Duration::from_secs(5), + "the refusal took {:?}; it must not wait out a commit deadline", + started.elapsed() + ); + assert_eq!( + cluster.nodes[0].shared.group_watchers().snapshot(), + applied_before, + "the refused restore applied an entry on the surviving node" ); cluster.shutdown().await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_documents_restart.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_documents_restart.rs new file mode 100644 index 000000000..57bab4ecc --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_documents_restart.rs @@ -0,0 +1,235 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A cluster RESTORE re-issues document rows, their index entries, and graph +//! edges as replicated writes, so every replica holds them durably. +//! +//! One cluster backs up a schemaless collection with a secondary index and a +//! unique index, a strict collection, a strict `bitemporal=true` collection +//! with two versions of one row, and two edges. A second cluster restores the +//! backup through one node, then every node restarts. Before and after the +//! restart, each node reads its own replica: every row, both index lookups, +//! both versions, and both edges. + +use std::time::Duration; + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; + +use crate::common; +use common::cluster_harness::{TestCluster, TestClusterNode, read_once_a_leader_exists}; + +const TENANT: u64 = 1; + +const COLLECTIONS: &[&str] = &[ + "CREATE COLLECTION crd_people (id STRING PRIMARY KEY, city STRING, email STRING) \ + WITH (engine='document_schemaless')", + "CREATE COLLECTION crd_accounts (id STRING PRIMARY KEY, owner STRING, balance INT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION crd_ledger (id STRING PRIMARY KEY, value STRING) \ + WITH (engine='document_strict', bitemporal=true)", + "CREATE INDEX ON crd_people (city)", + "CREATE UNIQUE INDEX crd_people_email ON crd_people (email)", +]; + +const WRITES: &[&str] = &[ + "INSERT INTO crd_people (id, city, email) VALUES ('alice', 'paris', 'a@x')", + "INSERT INTO crd_people (id, city, email) VALUES ('bob', 'rome', 'b@x')", + "INSERT INTO crd_people (id, city, email) VALUES ('carol', 'paris', 'c@x')", + "GRAPH INSERT EDGE IN 'crd_people' FROM 'alice' TO 'bob' TYPE 'knows'", + "GRAPH INSERT EDGE IN 'crd_people' FROM 'bob' TO 'carol' TYPE 'knows'", + "INSERT INTO crd_accounts (id, owner, balance) VALUES ('acc1', 'alice', 10)", + "INSERT INTO crd_accounts (id, owner, balance) VALUES ('acc2', 'bob', 20)", + "INSERT INTO crd_ledger (id, value) VALUES ('e1', 'draft')", + "UPDATE crd_ledger SET value = 'final' WHERE id = 'e1'", +]; + +fn db_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn drain_backup(client: &tokio_postgres::Client) -> Vec { + let stream = client + .copy_out(&format!("COPY (BACKUP TENANT {TENANT}) TO STDOUT")) + .await + .unwrap_or_else(|e| panic!("copy_out: {}", db_detail(&e))); + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice(&chunk.unwrap_or_else(|e| panic!("chunk: {}", db_detail(&e)))); + } + bytes +} + +async fn push_restore(client: &tokio_postgres::Client, envelope: Vec) { + let sink = client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({TENANT}) FROM STDIN")) + .await + .unwrap_or_else(|e| panic!("copy_in: {}", db_detail(&e))); + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .unwrap_or_else(|e| panic!("send: {}", db_detail(&e))); + sink.as_mut() + .finish() + .await + .unwrap_or_else(|e| panic!("restore: {}", db_detail(&e))); +} + +/// The first column of every row `sql` returns on `node`, sorted. +async fn column(node: &TestClusterNode, sql: &str) -> Vec { + let messages = read_once_a_leader_exists( + sql, + Duration::from_secs(30), + Duration::from_millis(100), + || node.client.simple_query(sql), + ) + .await; + let mut values: Vec = messages + .iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .collect(); + values.sort(); + values +} + +/// Every text cell `sql` returns on `node`, joined. +async fn text(node: &TestClusterNode, sql: &str) -> String { + let messages = read_once_a_leader_exists( + sql, + Duration::from_secs(30), + Duration::from_millis(100), + || node.client.simple_query(sql), + ) + .await; + messages + .iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => Some( + (0..row.len()) + .filter_map(|i| row.get(i)) + .collect::(), + ), + _ => None, + }) + .collect() +} + +/// Every restored row, index entry, version and edge reads back from +/// `node`'s own replica. +async fn assert_restored_on(node: &TestClusterNode, stage: &str) { + let id = node.node_id; + node.client + .simple_query("SET default_read_consistency = 'eventual'") + .await + .unwrap_or_else(|e| panic!("node {id}: set eventual reads: {}", db_detail(&e))); + assert_eq!( + column(node, "SELECT id FROM crd_people WHERE city = 'paris'").await, + vec!["alice", "carol"], + "{stage}, node {id}: the secondary index lookup" + ); + assert_eq!( + column(node, "SELECT id FROM crd_people WHERE email = 'b@x'").await, + vec!["bob"], + "{stage}, node {id}: the unique index lookup" + ); + assert_eq!( + column(node, "SELECT balance FROM crd_accounts WHERE id = 'acc2'").await, + vec!["20"], + "{stage}, node {id}: the strict row by primary key" + ); + assert_eq!( + column(node, "SELECT owner FROM crd_accounts").await, + vec!["alice", "bob"], + "{stage}, node {id}: every strict row" + ); + assert_eq!( + column(node, "SELECT value FROM crd_ledger WHERE id = 'e1'").await, + vec!["final"], + "{stage}, node {id}: the bitemporal row's current version" + ); + assert_eq!( + column(node, "SELECT value FROM crd_ledger AS OF SYSTEM TIME NULL").await, + vec!["draft", "final"], + "{stage}, node {id}: the bitemporal row keeps both versions" + ); + let from_alice = text( + node, + "GRAPH NEIGHBORS IN 'crd_people' OF 'alice' LABEL 'knows' DIRECTION out", + ) + .await; + assert!( + from_alice.contains("bob"), + "{stage}, node {id}: the edge alice -> bob, got {from_alice}" + ); + let from_bob = text( + node, + "GRAPH NEIGHBORS IN 'crd_people' OF 'bob' LABEL 'knows' DIRECTION out", + ) + .await; + assert!( + from_bob.contains("carol"), + "{stage}, node {id}: the edge bob -> carol, got {from_bob}" + ); +} + +async fn assert_restored(cluster: &TestCluster, stage: &str) { + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + for node in &cluster.nodes { + assert_restored_on(node, stage).await; + } + let duplicate = cluster.nodes[0] + .exec("INSERT INTO crd_people (id, city, email) VALUES ('dave', 'oslo', 'a@x')") + .await; + assert!( + duplicate.is_err(), + "{stage}: the unique index must refuse a restored row's email" + ); +} + +async fn source_backup() -> Vec { + let source = TestCluster::spawn_three().await.expect("source cluster"); + for sql in COLLECTIONS { + source + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + for sql in WRITES { + source.nodes[0] + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + source + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let backup = drain_backup(&source.nodes[0].client).await; + source.shutdown().await; + backup +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn restored_documents_indexes_and_edges_survive_a_full_cluster_restart() { + let backup = source_backup().await; + + let target = TestCluster::spawn_three().await.expect("target cluster"); + push_restore(&target.nodes[1].client, backup).await; + assert_restored(&target, "after the restore").await; + + let target = target + .restart_all() + .await + .unwrap_or_else(|e| panic!("restart every node: {e}")); + assert_restored(&target, "after every node restarted").await; + + target.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_refuses_on_non_replica.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_refuses_on_non_replica.rs new file mode 100644 index 000000000..10042fb46 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_refuses_on_non_replica.rs @@ -0,0 +1,125 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's staleness guard reads the write marks of every data group, not +//! the restoring node's memory. +//! +//! With a replication factor of 1, each data group lives on one node. A write +//! after the backup applies only on that node. A restore issued on another +//! node, which never applied the write, still refuses: the guard asks the +//! group's replica for its newest write. + +use std::time::Duration; + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; + +use crate::common; +use common::cluster_harness::TestCluster; +use common::cluster_harness::wait::wait_for; + +const TENANT: u64 = 1; +const COLLECTION: &str = "nonreplica_docs"; + +fn db_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn drain_backup(client: &tokio_postgres::Client) -> Vec { + let stream = client + .copy_out(&format!("COPY (BACKUP TENANT {TENANT}) TO STDOUT")) + .await + .unwrap_or_else(|e| panic!("copy_out: {}", db_detail(&e))); + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice(&chunk.unwrap_or_else(|e| panic!("chunk: {}", db_detail(&e)))); + } + bytes +} + +async fn push_restore(client: &tokio_postgres::Client, envelope: Vec) -> Result<(), String> { + let sink = client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({TENANT}) FROM STDIN")) + .await + .map_err(|e| db_detail(&e))?; + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .map_err(|e| db_detail(&e))?; + sink.as_mut() + .finish() + .await + .map(|_| ()) + .map_err(|e| db_detail(&e)) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_restore_on_a_node_that_never_applied_the_write_refuses() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')" + )) + .await + .expect("CREATE COLLECTION"); + let group_id = cluster.nodes[0] + .group_id_for_collection(COLLECTION) + .expect("the collection's data group"); + // Every joiner enters every group as a learner. Placement convergence + // then removes the nodes outside the group's placement, one per tick, + // after a leadership transfer when the leader itself must leave. A + // removed node keeps its mounted replica, so membership decides. + wait_for( + "exactly one node replicates the collection's group", + Duration::from_secs(30), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(group_id)) + .count() + == 1 + }, + ) + .await; + let restorer = cluster + .nodes + .iter() + .find(|node| !node.replicates_data_group(group_id)) + .expect("a node that does not replicate the group"); + + let backup = drain_backup(&restorer.client).await; + restorer + .client + .simple_query(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('after', 'x')" + )) + .await + .unwrap_or_else(|e| panic!("insert: {}", db_detail(&e))); + assert!( + restorer.shared.tenant_write_mark(TENANT).is_none(), + "the restoring node applied no write of the tenant, so its memory holds no mark" + ); + + let error = push_restore(&restorer.client, backup) + .await + .expect_err("a write after the backup must refuse the restore on every node"); + assert!( + error.contains("restore refused"), + "expected the staleness refusal, got: {error}" + ); + assert!( + error.contains(COLLECTION), + "the refusal must name the collection of the newer write, got: {error}" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs index 8f71c9950..534f235cb 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs @@ -25,6 +25,7 @@ mod catalog_put_if_absent; mod checkpoint_cross_node; mod cluster_array; mod cluster_array_cell_raft_replication; +mod cluster_backup_remote_cut; mod cluster_backup_restore; mod cluster_backup_restore_engines; mod cluster_cdc_publish_once; @@ -34,6 +35,8 @@ mod cluster_epoch_self_fence; mod cluster_execute_request; mod cluster_partition_strategy_replication; mod cluster_post_apply_follower_dispatch; +mod cluster_restore_documents_restart; +mod cluster_restore_refuses_on_non_replica; mod cluster_surrogate_replication; mod column_stats_cross_node; mod constraint_delivery; diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/auth_objects.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/auth_objects.rs index aa19f6621..2e83ce568 100644 --- a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/auth_objects.rs +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/auth_objects.rs @@ -11,7 +11,7 @@ async fn user_create_visible_on_every_node() { let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); cluster - .exec_ddl_on_any_leader("CREATE USER alice WITH PASSWORD 'sekret123' ROLE read_write") + .exec_ddl_on_any_leader("CREATE USER alice WITH PASSWORD 'sekret123' ROLE readwrite") .await .expect("create user"); @@ -76,38 +76,43 @@ async fn role_create_visible_on_every_node() { async fn alter_user_role_replicates() { let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); - cluster - .exec_ddl_on_any_leader("CREATE USER bob WITH PASSWORD 'initial-pass' ROLE read_only") - .await - .expect("create user"); + for sql in [ + "CREATE ROLE auditor", + "CREATE USER bob WITH PASSWORD 'initial-pass' ROLE readonly", + ] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } wait_for( - "all 3 nodes see bob with read_only role", + "all 3 nodes see bob with the readonly role", Duration::from_secs(10), Duration::from_millis(50), || { cluster .nodes .iter() - .all(|n| n.user_has_role("bob", "read_only")) + .all(|n| n.user_has_role("bob", "readonly")) }, ) .await; cluster - .exec_ddl_on_any_leader("ALTER USER bob SET ROLE read_write") + .exec_ddl_on_any_leader("ALTER USER bob SET ROLE auditor") .await .expect("alter user set role"); wait_for( - "all 3 nodes see bob with read_write role", + "all 3 nodes see bob with the custom auditor role", Duration::from_secs(10), Duration::from_millis(50), || { cluster .nodes .iter() - .all(|n| n.user_has_role("bob", "read_write")) + .all(|n| n.user_has_role("bob", "auditor")) }, ) .await; @@ -115,6 +120,98 @@ async fn alter_user_role_replicates() { cluster.shutdown().await; } +/// A role name that is neither built in nor defined is refused on every +/// entry point, on every node, with SQLSTATE 42704. No node ends up with a +/// user holding it. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn an_undefined_role_is_refused_on_every_node() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + cluster + .exec_ddl_on_any_leader("CREATE USER carol WITH PASSWORD 'carol-pass-1' ROLE readonly") + .await + .expect("create carol"); + + for node in &cluster.nodes { + for sql in [ + "CREATE USER dave WITH PASSWORD 'dave-pass-1' ROLE read_write", + "ALTER USER carol SET ROLE read_write", + "GRANT ROLE read_write TO carol", + ] { + let error = node + .exec(sql) + .await + .expect_err("an undefined role must be refused"); + assert!( + error.contains("42704") && error.contains("read_write"), + "node {}: {sql}: expected 42704 naming the role, got {error}", + node.node_id + ); + } + } + for node in &cluster.nodes { + assert!( + !node.has_active_user("dave"), + "node {} created dave", + node.node_id + ); + assert!( + node.user_has_role("carol", "readonly") && !node.user_has_role("carol", "read_write"), + "node {} changed carol's roles", + node.node_id + ); + } + + cluster.shutdown().await; +} + +/// A role a user holds is not dropped, as PostgreSQL refuses: the drop +/// fails with SQLSTATE 2BP01 naming the user. Once no user holds it, the +/// drop succeeds on every node. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_held_role_is_not_dropped() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + for sql in [ + "CREATE ROLE reviewer", + "CREATE USER erin WITH PASSWORD 'erin-pass-1' ROLE reviewer", + ] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + let error = cluster.nodes[0] + .exec("DROP ROLE reviewer") + .await + .expect_err("a held role must not be dropped"); + assert!( + error.contains("2BP01") && error.contains("erin"), + "expected 2BP01 naming the holder, got {error}" + ); + assert!( + cluster.nodes.iter().all(|n| n.has_role("reviewer")), + "the refused drop removed the role on some node" + ); + + for sql in ["ALTER USER erin SET ROLE readonly", "DROP ROLE reviewer"] { + cluster + .exec_ddl_on_any_leader(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + wait_for( + "all 3 nodes no longer see the dropped role", + Duration::from_secs(10), + Duration::from_millis(50), + || cluster.nodes.iter().all(|n| !n.has_role("reviewer")), + ) + .await; + + cluster.shutdown().await; +} + #[tokio::test(flavor = "multi_thread", worker_threads = 6)] async fn api_key_create_and_revoke_replicates() { let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); diff --git a/nodedb-cluster/src/calvin/sequencer/entry.rs b/nodedb-cluster/src/calvin/sequencer/entry.rs index 318539a40..c0b3656b3 100644 --- a/nodedb-cluster/src/calvin/sequencer/entry.rs +++ b/nodedb-cluster/src/calvin/sequencer/entry.rs @@ -133,6 +133,12 @@ pub enum SequencerEntry { position: u32, reason: AbortReason, }, + /// A backup's consistent-cut marker, carrying the backup's watermark + /// `hlc`. Every replica fans it out to each of its vShard schedulers in + /// log order. A scheduler reports the marker once every transaction + /// delivered to it before the marker finished, and gives every + /// transaction delivered after it a commit HLC above `hlc`. + CutMarker { hlc: u64 }, } #[cfg(test)] diff --git a/nodedb-cluster/src/calvin/sequencer/replay.rs b/nodedb-cluster/src/calvin/sequencer/replay.rs index c20e3bfb9..7a30d65cb 100644 --- a/nodedb-cluster/src/calvin/sequencer/replay.rs +++ b/nodedb-cluster/src/calvin/sequencer/replay.rs @@ -56,6 +56,7 @@ impl SequencerStateMachine { /// computed identically to the live [`SequencerStateMachine::apply`] path. /// * `ReserveRead` targeting `vshard_id` → [`SchedulerInput::Reserve`]. /// * `ReleaseReservation` targeting `vshard_id` → [`SchedulerInput::Release`]. + /// * `CutMarker` → [`SchedulerInput::CutMarker`], for every vShard. /// * All other variants carry no per-vShard scheduler input. /// /// Entries are emitted in Raft-log order (and, within an epoch batch, in @@ -81,6 +82,11 @@ impl SequencerStateMachine { // No-op entry (newly elected leader heartbeat). continue; } + if crate::conf_change::ConfChange::is_conf_change(&entry.data) { + // A membership change of the sequencer group, applied by the + // Raft layer; it carries no sequencer input. + continue; + } let seq_entry: SequencerEntry = match zerompk::from_msgpack(&entry.data) { Ok(e) => e, Err(err) => { @@ -140,6 +146,11 @@ impl SequencerStateMachine { } if vshard == vshard_id => { result.push(SchedulerInput::Release { owner, reason }); } + // A cut marker reaches every vShard, exactly as the live + // `CutMarker` arm fans it out. + SequencerEntry::CutMarker { hlc } => { + result.push(SchedulerInput::CutMarker { hlc }); + } // Reservation entries for a different vShard carry nothing for us. SequencerEntry::ReserveRead { .. } => {} SequencerEntry::ReleaseReservation { .. } => {} diff --git a/nodedb-cluster/src/calvin/sequencer/service/core.rs b/nodedb-cluster/src/calvin/sequencer/service/core.rs index 297289909..601a6d0c6 100644 --- a/nodedb-cluster/src/calvin/sequencer/service/core.rs +++ b/nodedb-cluster/src/calvin/sequencer/service/core.rs @@ -509,6 +509,7 @@ fn entry_txn_count(entry: &SequencerEntry) -> usize { SequencerEntry::AbortVerdict { .. } => 0, SequencerEntry::ReserveRead { .. } => 0, SequencerEntry::ReleaseReservation { .. } => 0, + SequencerEntry::CutMarker { .. } => 0, } } diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs index df81c52b1..4b507b81b 100644 --- a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs @@ -70,6 +70,12 @@ impl SequencerStateMachine { // committed at `index` regardless, so it is a safe replay upper bound. self.last_committed_index = index; + // A membership change of the sequencer group commits in its log too. + // It is no sequencer entry, and the Raft layer applied it already. + if crate::conf_change::ConfChange::is_conf_change(data) { + return; + } + let entry: SequencerEntry = match zerompk::from_msgpack(data) { Ok(e) => e, Err(err) => { @@ -394,6 +400,34 @@ impl SequencerStateMachine { } } } + // Fan a backup's cut marker out to every vShard scheduler this + // node hosts. Same `try_send` discipline as `ReserveRead`: a + // dropped marker is recovered by the scheduler's catch-up drain, + // which replays it in log order. + SequencerEntry::CutMarker { hlc } => { + for (&vshard, sender) in &self.vshard_senders { + match sender.try_send(SchedulerInput::CutMarker { hlc }) { + Ok(()) => {} + Err(mpsc::error::TrySendError::Full(_)) => { + warn!( + vshard, + hlc, + "sequencer apply: vshard channel full (backpressure); \ + dropping cut marker" + ); + self.record_catch_up(vshard, index); + } + Err(mpsc::error::TrySendError::Closed(_)) => { + warn!( + vshard, + "sequencer apply: vshard sender gone; \ + scheduler may have exited (cut marker)" + ); + self.record_catch_up(vshard, index); + } + } + } + } // Fan a reservation release out to its owning vShard's scheduler. // Same `try_send` discipline as `ReserveRead`. SequencerEntry::ReleaseReservation { diff --git a/nodedb-cluster/src/calvin/types/scheduler_input.rs b/nodedb-cluster/src/calvin/types/scheduler_input.rs index 1ceb6a37c..4e0e7be50 100644 --- a/nodedb-cluster/src/calvin/types/scheduler_input.rs +++ b/nodedb-cluster/src/calvin/types/scheduler_input.rs @@ -29,4 +29,8 @@ pub enum SchedulerInput { owner: TxnIdWire, reason: ReleaseReason, }, + /// A backup's consistent-cut marker carrying its watermark `hlc`. Every + /// transaction delivered before it must finish before the scheduler + /// reports it; every transaction delivered after it commits above `hlc`. + CutMarker { hlc: u64 }, } diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs index 765e3354f..e0115c725 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs @@ -88,10 +88,7 @@ fn propose_forwarded(mr: &mut MultiRaft, req: DataProposeRequest) -> DataPropose }; match proposed { Ok((group_id, log_index)) => DataProposeResponse::ok(group_id, log_index), - Err(ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint })) => { - DataProposeResponse::err("not leader", leader_hint) - } - Err(e) => DataProposeResponse::err(e.to_string(), None), + Err(error) => DataProposeResponse::refused(&error), } } diff --git a/nodedb-cluster/src/raft_loop/proposals.rs b/nodedb-cluster/src/raft_loop/proposals.rs index f3acd8f22..50df25f26 100644 --- a/nodedb-cluster/src/raft_loop/proposals.rs +++ b/nodedb-cluster/src/raft_loop/proposals.rs @@ -252,16 +252,8 @@ impl RaftLoop { crate::rpc_codec::RaftRpc::DataProposeResponse(r) => { if r.success { Ok((r.group_id, r.log_index)) - } else if let Some(hint) = r.leader_hint { - Err(crate::error::ClusterError::Raft( - nodedb_raft::RaftError::NotLeader { - leader_hint: Some(hint), - }, - )) } else { - Err(crate::error::ClusterError::Transport { - detail: format!("data propose forward failed: {}", r.error_message), - }) + Err(r.refusal_error()) } } other => Err(crate::error::ClusterError::Transport { diff --git a/nodedb-cluster/src/rpc_codec/data_propose.rs b/nodedb-cluster/src/rpc_codec/data_propose.rs index c1f459c0f..e22527ce0 100644 --- a/nodedb-cluster/src/rpc_codec/data_propose.rs +++ b/nodedb-cluster/src/rpc_codec/data_propose.rs @@ -30,6 +30,19 @@ pub struct DataProposeRequest { pub bytes: Vec, } +/// Why a leader refused a forwarded proposal, typed so the forwarding node +/// can tell a transient refusal from a final one. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum ForwardedProposeRefusal { + /// The node does not lead the target group. `leader_hint` names the + /// leader it knows, if any. + NotLeader, + /// A leadership transfer of the target group is in flight. + LeadershipTransferInProgress, + /// Any other failure; `error_message` describes it. + Failed, +} + /// Response to a forwarded data-group proposal. #[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub struct DataProposeResponse { @@ -37,6 +50,8 @@ pub struct DataProposeResponse { pub group_id: u64, pub log_index: u64, pub leader_hint: Option, + /// The typed reason of a refusal. `None` on success. + pub refusal: Option, pub error_message: String, } @@ -47,17 +62,48 @@ impl DataProposeResponse { group_id, log_index, leader_hint: None, + refusal: None, error_message: String::new(), } } - pub fn err(message: impl Into, leader_hint: Option) -> Self { + /// The response for a proposal the leader refused with `error`. + pub fn refused(error: &ClusterError) -> Self { + let (refusal, leader_hint) = match error { + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint }) => { + (ForwardedProposeRefusal::NotLeader, *leader_hint) + } + ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) => { + (ForwardedProposeRefusal::LeadershipTransferInProgress, None) + } + _ => (ForwardedProposeRefusal::Failed, None), + }; Self { success: false, group_id: 0, log_index: 0, leader_hint, - error_message: message.into(), + refusal: Some(refusal), + error_message: error.to_string(), + } + } + + /// The typed error a refused response stands for on the forwarding node. + /// A refusal the leader typed keeps its Raft error; any other stays a + /// transport error carrying the leader's message. + pub fn refusal_error(&self) -> ClusterError { + match self.refusal { + Some(ForwardedProposeRefusal::NotLeader) => { + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: self.leader_hint, + }) + } + Some(ForwardedProposeRefusal::LeadershipTransferInProgress) => { + ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) + } + Some(ForwardedProposeRefusal::Failed) | None => ClusterError::Transport { + detail: format!("data propose forward failed: {}", self.error_message), + }, } } } @@ -135,4 +181,48 @@ mod tests { let req = roundtrip(ProposeTarget::VShard(42)); assert_eq!(req.target, ProposeTarget::VShard(42)); } + + fn refusal_across_the_wire(error: ClusterError) -> ClusterError { + let rpc = RaftRpc::DataProposeResponse(DataProposeResponse::refused(&error)); + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + match decode(&encoded, &epoch).expect("decode") { + RaftRpc::DataProposeResponse(resp) => { + assert!(!resp.success); + resp.refusal_error() + } + other => panic!("decoded the wrong variant: {other:?}"), + } + } + + #[test] + fn a_transfer_in_progress_keeps_its_raft_error_across_the_wire() { + let error = refusal_across_the_wire(ClusterError::Raft( + nodedb_raft::RaftError::LeadershipTransferInProgress, + )); + assert!(matches!( + error, + ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) + )); + } + + #[test] + fn a_not_leader_keeps_its_hint_across_the_wire() { + let error = + refusal_across_the_wire(ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: Some(3), + })); + assert!(matches!( + error, + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: Some(3) + }) + )); + } + + #[test] + fn any_other_refusal_stays_a_transport_error() { + let error = refusal_across_the_wire(ClusterError::VShardNotMapped { vshard_id: 7 }); + assert!(matches!(error, ClusterError::Transport { .. })); + } } diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index 1f8826468..4b10a22cd 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -44,7 +44,9 @@ pub use cluster_mgmt::{ PongResponse, TopologyAck, TopologyUpdate, }; pub use data_plane_error::{DataPlaneCounterFault, DataPlaneErrorCode}; -pub use data_propose::{DataProposeRequest, DataProposeResponse, ProposeTarget}; +pub use data_propose::{ + DataProposeRequest, DataProposeResponse, ForwardedProposeRefusal, ProposeTarget, +}; pub use execute::{ DescriptorVersionEntry, ExecuteRequest, ExecuteResponse, ExecuteStreamChunk, ExecuteStreamEnd, PLAN_DECODE_FAILED, TypedClusterError, diff --git a/nodedb-cluster/src/transport/client/close.rs b/nodedb-cluster/src/transport/client/close.rs new file mode 100644 index 000000000..3b40b1db9 --- /dev/null +++ b/nodedb-cluster/src/transport/client/close.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Close this node's QUIC endpoint. +//! +//! Dropping every handle of an endpoint does not release its UDP socket. The +//! endpoint's driver task owns the socket, and it runs until every connection +//! is gone. A peer that keeps its connection alive keeps the socket bound +//! until the idle timeout, so a node that stops and starts again on the same +//! address fails to bind. Closing the endpoint ends every connection at once: +//! peers see the close immediately, and the driver releases the socket once +//! the last handle drops. + +use std::time::Duration; + +use super::transport::NexarTransport; + +/// The QUIC application close code a node sends when it shuts down. +const SHUTDOWN_CLOSE_CODE: u32 = 0; + +impl NexarTransport { + /// Close every connection, refuse new ones, and wait until the peers + /// acknowledged the close or `timeout` passed. Returns `false` when + /// `timeout` passed first; the connections are closed either way. + pub async fn close(&self, timeout: Duration) -> bool { + let endpoint = self.listener.endpoint(); + endpoint.close( + quinn::VarInt::from_u32(SHUTDOWN_CLOSE_CODE), + b"node shutdown", + ); + // The cached connections are closed: a send after this fails at once + // instead of reusing one. + self.peers + .write() + .unwrap_or_else(|p| p.into_inner()) + .clear(); + tokio::time::timeout(timeout, endpoint.wait_idle()) + .await + .is_ok() + } +} diff --git a/nodedb-cluster/src/transport/client/mod.rs b/nodedb-cluster/src/transport/client/mod.rs index 3ddc6d012..2b2a55b5c 100644 --- a/nodedb-cluster/src/transport/client/mod.rs +++ b/nodedb-cluster/src/transport/client/mod.rs @@ -9,6 +9,7 @@ //! //! [`RaftTransport`]: nodedb_raft::transport::RaftTransport +pub mod close; pub mod pool; pub mod raft_impl; pub mod send; diff --git a/nodedb-physical/src/physical_plan/cluster_event.rs b/nodedb-physical/src/physical_plan/cluster_event.rs index 7052c29cd..2407d4849 100644 --- a/nodedb-physical/src/physical_plan/cluster_event.rs +++ b/nodedb-physical/src/physical_plan/cluster_event.rs @@ -44,6 +44,11 @@ pub enum ClusterEventOp { topic_name: String, payload: String, }, + /// Read the receiving node's tenant write marks of `group_ids`, once it + /// applied every entry the groups committed before the request. RESTORE's + /// staleness guard asks a replica of each group this way when the + /// restoring node does not replicate the group. + TenantWriteMarks { tenant_id: u64, group_ids: Vec }, } #[cfg(test)] diff --git a/nodedb-physical/src/physical_plan/collection.rs b/nodedb-physical/src/physical_plan/collection.rs index a5891fbe8..b031081d0 100644 --- a/nodedb-physical/src/physical_plan/collection.rs +++ b/nodedb-physical/src/physical_plan/collection.rs @@ -188,6 +188,7 @@ mod tests { redo: Vec::new(), collections: collections.clone(), sum_targets: Vec::new(), + origin: crate::physical_plan::RedoOrigin::Commit, }); let flush = PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch: 1, diff --git a/nodedb-physical/src/physical_plan/meta.rs b/nodedb-physical/src/physical_plan/meta.rs index 5b2519435..da6a48196 100644 --- a/nodedb-physical/src/physical_plan/meta.rs +++ b/nodedb-physical/src/physical_plan/meta.rs @@ -93,7 +93,16 @@ pub enum MetaOp { /// Snapshot a tenant's data from the sparse engine. /// Returns serialized `(documents, indexes)` as JSON payload. - CreateTenantSnapshot { tenant_id: u64 }, + CreateTenantSnapshot { + tenant_id: u64, + /// `Some(W)` asks the node that receives the plan to take a backup's + /// consistent cut at watermark `W` before it snapshots: every write + /// committed below `W` has its final outcome there first. The + /// receiving Control Plane takes the cut and clears the field. The + /// Data Plane never reads it. + #[serde(default)] + cut_watermark: Option, + }, /// Restore a tenant's data across all engines from a snapshot. /// `snapshot` is a MessagePack-serialized `TenantDataSnapshot`. @@ -586,9 +595,11 @@ pub enum MetaOp { /// collection the transaction wrote; each gets a collection-floor write /// version at the record's LSN. `sum_targets` is the materialized-sum /// resolution the transaction's document writes fold into their targets. + /// `origin` decides which commit-boundary checks the apply runs. ApplyTransactionRedo { redo: Vec, collections: Vec, sum_targets: Vec, + origin: super::RedoOrigin, }, } diff --git a/nodedb-physical/src/physical_plan/mod.rs b/nodedb-physical/src/physical_plan/mod.rs index 644c10781..2f00b5eb8 100644 --- a/nodedb-physical/src/physical_plan/mod.rs +++ b/nodedb-physical/src/physical_plan/mod.rs @@ -20,6 +20,7 @@ pub mod meta; pub mod meta_calvin; pub mod plan; pub mod query; +pub mod redo_origin; pub mod rls_write_check_accessor; pub mod routing; pub mod set_op; @@ -54,6 +55,7 @@ pub use kv::{ pub use meta::{MetaOp, SAVEPOINT_MARKER_BYTES}; pub use plan::PhysicalPlan; pub use query::{AggregateSpec, GroupKeySpec, JoinProjection, QueryOp}; +pub use redo_origin::RedoOrigin; pub use routing::plan_contains_cluster_partitioned_leaf; pub use set_op::SetOpKind; pub use sort_key::SortKeySpec; diff --git a/nodedb-physical/src/physical_plan/redo_origin.rs b/nodedb-physical/src/physical_plan/redo_origin.rs new file mode 100644 index 000000000..0a57970ce --- /dev/null +++ b/nodedb-physical/src/physical_plan/redo_origin.rs @@ -0,0 +1,27 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Where a committed redo record comes from, and so which checks its apply +//! runs. + +/// The source of a redo record a replica installs. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum RedoOrigin { + /// A transaction commit. The apply runs every commit-boundary check: + /// BALANCED, UNIQUE, and the stateless PUT and DELETE rules. + Commit, + /// A RESTORE re-installing rows a backup captured. Each row passed its + /// collection's rules when it was first written, and a bitemporal row's + /// earlier versions are history, not new writes. The apply checks only + /// that every unique value has one owner in the post-state. + Restore, +} diff --git a/nodedb-test-support/src/cluster_harness/cluster/bringup.rs b/nodedb-test-support/src/cluster_harness/cluster/bringup.rs index a29a8212a..cda7141b3 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/bringup.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/bringup.rs @@ -1,9 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! The shared 3-node bringup body (`spawn_three_inner`) and its -//! post-join convergence barriers: topology size, rolling-upgrade -//! compat-mode exit, metadata-group leader stability, and per-group -//! Raft leader stability. +//! The shared 3-node bringup body (`spawn_three_inner`). The post-join +//! convergence barriers live in `ready`. use std::time::Duration; @@ -12,7 +10,6 @@ use nodedb_types::config::tuning::ClusterTransportTuning; use super::TestCluster; use super::types::ClusterSpawnConfig; use crate::cluster_harness::node::TestClusterNode; -use crate::cluster_harness::wait::wait_for; impl TestCluster { /// Shared 3-node spawn body. Threads an optional Raft @@ -74,152 +71,7 @@ impl TestCluster { spawn_config: config, }; - wait_for( - "all 3 nodes report topology_size == 3", - Duration::from_secs(30), - Duration::from_millis(50), - || cluster.nodes.iter().all(|n| n.topology_size() == 3), - ) - .await; - - // CRITICAL: wait for every node to exit rolling-upgrade - // compat mode before letting the test issue any DDL. - // - // `metadata_proposer::propose_catalog_entry` consults - // `cluster_version_view().can_activate_feature(DISTRIBUTED_CATALOG_VERSION)` - // and, while even one node still reports a lower wire - // version, returns `Ok(0)` without going through the raft - // group. The pgwire DDL handlers (CREATE USER, etc.) then - // fall through to a LEGACY path that writes the record - // directly on the proposing node — **with zero - // replication** to followers. Any subsequent - // `has_active_user` check on a follower returns false and - // the test flakes. - // - // Topology has three members the moment the join request - // completes, but the `wire_version` field on each node's - // topology entry is updated asynchronously by the gossip - // path. That's why `topology_size == 3` converges fast yet - // `can_activate_feature(...)` can still be false for - // several hundred milliseconds afterwards. Waiting here - // closes the window deterministically — no retries, no - // flakes, no compat-mode fallback silently breaking - // replication. - wait_for( - "all 3 nodes exit rolling-upgrade compat mode", - Duration::from_secs(30), - Duration::from_millis(20), - || { - cluster.nodes.iter().all(|n| { - n.shared.cluster_version_view().can_activate_feature( - nodedb::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, - ) - }) - }, - ) - .await; - - // CRITICAL: wait for the metadata Raft group to elect a leader - // and for every node's local view to agree on the same leader id. - // - // Topology convergence + rolling-upgrade exit only guarantees - // membership and wire version are agreed; they say nothing about - // election state. Under heavy host load (e.g. running this test - // immediately after another full-suite cluster test exits and - // the unit-test pool ramps back up), the initial Raft heartbeat - // window can be missed and the first `acquire`/`propose` issued - // by the test races a re-election — surfacing as - // `raft error: not leader (leader hint: None)` from a - // descriptor-lease or DDL call. - // - // Waiting until every node reports the same non-zero leader id - // closes the window deterministically. Symmetric to the - // rolling-upgrade wait above: no retries, no flakes, no - // wasted CI minutes on cleanup of a doomed cluster bringup. - wait_for( - "metadata group has stable leader visible on every node", - Duration::from_secs(30), - Duration::from_millis(20), - || { - let leaders: Vec = cluster - .nodes - .iter() - .map(|n| n.metadata_group_leader()) - .collect(); - let first = leaders[0]; - first != 0 && leaders.iter().all(|&l| l == first) - }, - ) - .await; - - // CRITICAL: wait for EVERY data Raft group to elect a stable - // leader visible on every node. Without this barrier, the - // first data-group write after `spawn_three()` returns can - // race a still-electing group: - // - // 1. Proposer's local `propose()` runs on a node that thinks - // it's leader (stale routing-table hint), gets an Ok back - // with a `log_index` that was never actually committed. - // 2. `ProposeTracker::register((group_id, log_index))`. - // 3. Some unrelated entry that *does* commit at that index - // (e.g., a leadership-change no-op) fires `tracker.complete`, - // waking the waiter with `Ok([])` even though the user's - // `INSERT` row was never replicated. - // 4. `simple_query` returns success; the row is permanently - // lost. - // - // The metadata-group-only wait above is insufficient because - // data groups elect independently and lag the metadata group - // by hundreds of milliseconds under load. Waiting until every - // group on every node reports a non-zero leader closes the - // window deterministically. - wait_for( - "every Raft group has a stable leader visible on every node", - Duration::from_secs(30), - Duration::from_millis(20), - || { - // Snapshot every node's per-group leader view. A group - // is "ready" iff every node reports the same non-zero - // leader for it. - let per_node: Vec> = cluster - .nodes - .iter() - .map(|n| n.all_group_leaders()) - .collect(); - if per_node.iter().any(|v| v.is_empty()) { - return false; - } - // The Calvin sequencer group is an internal Raft group that is - // not part of the data/metadata routing topology. Cluster - // readiness for data operations does not depend on it, and its - // leader is surfaced to the observer on a slower/independent path - // than the routing groups — so gating general cluster startup on - // it makes every test (Calvin or not) flake when the sequencer - // group's observed leader lags. Calvin tests gate on the - // sequencer separately (`wait_for_sequencer_leader`). Exclude it - // from the general readiness gate. - let group_ids: std::collections::BTreeSet = per_node - .iter() - .flat_map(|v| v.iter().map(|(gid, _)| *gid)) - .filter(|gid| *gid != nodedb_cluster::calvin::SEQUENCER_GROUP_ID) - .collect(); - if group_ids.is_empty() { - return false; - } - group_ids.iter().all(|gid| { - let leaders: Vec = per_node - .iter() - .filter_map(|v| v.iter().find(|(g, _)| g == gid).map(|(_, l)| *l)) - .collect(); - if leaders.len() != per_node.len() { - return false; - } - let first = leaders[0]; - first != 0 && leaders.iter().all(|&l| l == first) - }) - }, - ) - .await; + cluster.await_ready().await; Ok(cluster) } diff --git a/nodedb-test-support/src/cluster_harness/cluster/mod.rs b/nodedb-test-support/src/cluster_harness/cluster/mod.rs index 899e9ade0..dcafb7432 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/mod.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/mod.rs @@ -4,11 +4,14 @@ //! //! [`TestCluster`] + [`ClusterSpawnConfig`] type definitions live in //! [`types`]; spawn-variant convenience wrappers in [`spawn_variants`]; -//! the heavy bringup/convergence body in [`bringup`]; post-spawn -//! membership + DDL helpers in [`membership`]. +//! the bringup body in [`bringup`]; the convergence barriers in [`ready`]; +//! the in-place restart in [`restart`]; post-spawn membership + DDL helpers +//! in [`membership`]. mod bringup; mod membership; +mod ready; +mod restart; mod spawn_variants; mod types; diff --git a/nodedb-test-support/src/cluster_harness/cluster/ready.rs b/nodedb-test-support/src/cluster_harness/cluster/ready.rs new file mode 100644 index 000000000..0dca12245 --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/cluster/ready.rs @@ -0,0 +1,162 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The convergence barriers a cluster passes before a test issues anything: +//! topology size, rolling-upgrade compat-mode exit, metadata-group leader +//! stability, and per-group Raft leader stability. A fresh bringup and an +//! in-place restart both wait here. + +use std::time::Duration; + +use super::TestCluster; +use crate::cluster_harness::wait::wait_for; + +impl TestCluster { + /// Wait until every node agrees on the topology, every node left compat + /// mode, and every Raft group has one leader every node sees. + pub(super) async fn await_ready(&self) { + let node_count = self.nodes.len(); + wait_for( + "every node reports the full topology", + Duration::from_secs(30), + Duration::from_millis(50), + || self.nodes.iter().all(|n| n.topology_size() == node_count), + ) + .await; + + // CRITICAL: wait for every node to exit rolling-upgrade + // compat mode before letting the test issue any DDL. + // + // `metadata_proposer::propose_catalog_entry` consults + // `cluster_version_view().can_activate_feature(DISTRIBUTED_CATALOG_VERSION)` + // and, while even one node still reports a lower wire + // version, returns `Ok(0)` without going through the raft + // group. The pgwire DDL handlers (CREATE USER, etc.) then + // fall through to a LEGACY path that writes the record + // directly on the proposing node — **with zero + // replication** to followers. Any subsequent + // `has_active_user` check on a follower returns false and + // the test flakes. + // + // Topology has three members the moment the join request + // completes, but the `wire_version` field on each node's + // topology entry is updated asynchronously by the gossip + // path. That's why `topology_size == 3` converges fast yet + // `can_activate_feature(...)` can still be false for + // several hundred milliseconds afterwards. Waiting here + // closes the window deterministically — no retries, no + // flakes, no compat-mode fallback silently breaking + // replication. + wait_for( + "every node exits rolling-upgrade compat mode", + Duration::from_secs(30), + Duration::from_millis(20), + || { + self.nodes.iter().all(|n| { + n.shared.cluster_version_view().can_activate_feature( + nodedb::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, + ) + }) + }, + ) + .await; + + // CRITICAL: wait for the metadata Raft group to elect a leader + // and for every node's local view to agree on the same leader id. + // + // Topology convergence + rolling-upgrade exit only guarantees + // membership and wire version are agreed; they say nothing about + // election state. Under heavy host load (e.g. running this test + // immediately after another full-suite cluster test exits and + // the unit-test pool ramps back up), the initial Raft heartbeat + // window can be missed and the first `acquire`/`propose` issued + // by the test races a re-election — surfacing as + // `raft error: not leader (leader hint: None)` from a + // descriptor-lease or DDL call. + // + // Waiting until every node reports the same non-zero leader id + // closes the window deterministically. Symmetric to the + // rolling-upgrade wait above: no retries, no flakes, no + // wasted CI minutes on cleanup of a doomed cluster bringup. + wait_for( + "metadata group has stable leader visible on every node", + Duration::from_secs(30), + Duration::from_millis(20), + || { + let leaders: Vec = self + .nodes + .iter() + .map(|n| n.metadata_group_leader()) + .collect(); + let first = leaders[0]; + first != 0 && leaders.iter().all(|&l| l == first) + }, + ) + .await; + + // CRITICAL: wait for EVERY data Raft group to elect a stable + // leader visible on every node. Without this barrier, the + // first data-group write after `spawn_three()` returns can + // race a still-electing group: + // + // 1. Proposer's local `propose()` runs on a node that thinks + // it's leader (stale routing-table hint), gets an Ok back + // with a `log_index` that was never actually committed. + // 2. `ProposeTracker::register((group_id, log_index))`. + // 3. Some unrelated entry that *does* commit at that index + // (e.g., a leadership-change no-op) fires `tracker.complete`, + // waking the waiter with `Ok([])` even though the user's + // `INSERT` row was never replicated. + // 4. `simple_query` returns success; the row is permanently + // lost. + // + // The metadata-group-only wait above is insufficient because + // data groups elect independently and lag the metadata group + // by hundreds of milliseconds under load. Waiting until every + // group on every node reports a non-zero leader closes the + // window deterministically. + wait_for( + "every Raft group has a stable leader visible on every node", + Duration::from_secs(30), + Duration::from_millis(20), + || { + // Snapshot every node's per-group leader view. A group + // is "ready" iff every node reports the same non-zero + // leader for it. + let per_node: Vec> = + self.nodes.iter().map(|n| n.all_group_leaders()).collect(); + if per_node.iter().any(|v| v.is_empty()) { + return false; + } + // The Calvin sequencer group is an internal Raft group that is + // not part of the data/metadata routing topology. Cluster + // readiness for data operations does not depend on it, and its + // leader is surfaced to the observer on a slower/independent path + // than the routing groups — so gating general cluster startup on + // it makes every test (Calvin or not) flake when the sequencer + // group's observed leader lags. Calvin tests gate on the + // sequencer separately (`wait_for_sequencer_leader`). Exclude it + // from the general readiness gate. + let group_ids: std::collections::BTreeSet = per_node + .iter() + .flat_map(|v| v.iter().map(|(gid, _)| *gid)) + .filter(|gid| *gid != nodedb_cluster::calvin::SEQUENCER_GROUP_ID) + .collect(); + if group_ids.is_empty() { + return false; + } + group_ids.iter().all(|gid| { + let leaders: Vec = per_node + .iter() + .filter_map(|v| v.iter().find(|(g, _)| g == gid).map(|(_, l)| *l)) + .collect(); + if leaders.len() != per_node.len() { + return false; + } + let first = leaders[0]; + first != 0 && leaders.iter().all(|&l| l == first) + }) + }, + ) + .await; + } +} diff --git a/nodedb-test-support/src/cluster_harness/cluster/restart.rs b/nodedb-test-support/src/cluster_harness/cluster/restart.rs new file mode 100644 index 000000000..a46d6ea2e --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/cluster/restart.rs @@ -0,0 +1,39 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Restart every node of a [`TestCluster`] in place. + +use super::TestCluster; +use crate::cluster_harness::node::TestClusterNode; + +impl TestCluster { + /// Stop every node, then bring every one back on its node id, listen + /// address and data directory, and wait until the cluster is ready. + /// + /// Every node stops before any restarts, so nothing survives in memory: + /// what each node serves afterwards comes from its own disk. The nodes + /// restart together, since each Raft group needs a quorum to elect. + pub async fn restart_all(self) -> Result> { + let TestCluster { + nodes, + spawn_config, + } = self; + let mut stopped = Vec::with_capacity(nodes.len()); + for node in nodes { + stopped.push(node.stop_for_restart().await?); + } + let seeds: Vec = + stopped.iter().map(|node| node.listen_addr()).collect(); + let nodes = futures::future::try_join_all( + stopped + .into_iter() + .map(|node| TestClusterNode::restart(node, seeds.clone(), &spawn_config)), + ) + .await?; + let cluster = TestCluster { + nodes, + spawn_config, + }; + cluster.await_ready().await; + Ok(cluster) + } +} diff --git a/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs b/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs index 6e9dd08e4..f67106008 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs @@ -165,6 +165,26 @@ impl TestCluster { .await } + /// Spawn a 3-node cluster whose data groups each place + /// `replication_factor` of the three nodes. With a factor below 3 some + /// node replicates no copy of a group, so a test can act on a node that + /// never applied the group's writes. + /// + /// Uses the standard fast-election tuning and 1 Data-Plane core per node. + pub async fn spawn_three_with_replication_factor( + replication_factor: usize, + ) -> Result> { + Self::spawn_three_inner( + fast_cluster_tuning(), + nodedb_types::config::tuning::GraphTuning::default(), + nodedb_types::config::tuning::QueryTuning::default(), + 1, + None, + replication_factor, + ) + .await + } + /// Spawn a 3-node cluster with custom cluster-transport, graph engine tuning, /// query execution tuning, and a specific core count per node. /// diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs b/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs index 2d8878e84..680d3500b 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs @@ -44,6 +44,7 @@ impl TestClusterNode { vshard_id, plan: PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: tenant.as_u64(), + cut_watermark: None, }), deadline: std::time::Instant::now() + std::time::Duration::from_secs(5), priority: Priority::Normal, diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs b/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs index 1a155b0e1..de1ab5df2 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs @@ -145,6 +145,27 @@ impl TestClusterNode { .any(|g| g.group_id == group_id) } + /// True iff this node is a current replica of data group `group_id`: it + /// has the group mounted, and its own routing table lists it as a voter or + /// learner of the group. + /// + /// [`Self::hosts_data_group`] alone does not answer this. A node removed + /// from a group keeps its mounted replica: a join adds the joiner as a + /// learner of every group, and placement convergence later removes the + /// nodes outside the group's placement from its membership only. + pub fn replicates_data_group(&self, group_id: u64) -> bool { + if !self.hosts_data_group(group_id) { + return false; + } + let Some(routing) = self.shared.cluster_routing.as_ref() else { + return false; + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing.group_info(group_id).is_some_and(|info| { + info.members.contains(&self.node_id) || info.learners.contains(&self.node_id) + }) + } + /// The local `snapshot_index` for `group_id` from this node's own Raft /// state, or `0` if the group isn't hosted here. /// diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs index 50cce7a13..dc0df838d 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs @@ -27,9 +27,11 @@ //! //! Struct definition in [`types`]; thin `spawn*` convenience wrappers in //! [`spawn_variants`]; the full spawn body in [`spawn_full`]; query -//! execution + shutdown + `Drop` teardown in [`teardown`]. +//! execution + shutdown + `Drop` teardown in [`teardown`]; the in-place +//! restart in [`restart`]. mod client_slot; +mod restart; mod spawn_full; mod spawn_variants; mod teardown; diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs new file mode 100644 index 000000000..8893a4fba --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs @@ -0,0 +1,102 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! In-place restart of a [`TestClusterNode`]: the same node id, the same QUIC +//! listen address, and the same data directory. + +use std::net::SocketAddr; +use std::time::Duration; + +use crate::cluster_harness::cluster::ClusterSpawnConfig; + +use super::types::{DataDir, TestClusterNode}; + +/// A node stopped for an in-place restart. It owns the node's data +/// directory, which outlives the stopped node. +pub(crate) struct StoppedNode { + node_id: u64, + listen_addr: SocketAddr, + data_dir: tempfile::TempDir, +} + +impl StoppedNode { + /// The QUIC address the node listened on, which its peers still hold. + pub(crate) fn listen_addr(&self) -> SocketAddr { + self.listen_addr + } +} + +impl TestClusterNode { + /// Stop the node, flush its WAL and release every file handle, keeping + /// its data directory for [`Self::restart`]. + pub(crate) async fn stop_for_restart( + mut self, + ) -> Result> { + let DataDir::Owned(data_dir) = std::mem::replace(&mut self._data_dir, DataDir::Borrowed) + else { + return Err(format!( + "node {} runs on a caller-supplied data directory, which an in-place \ + restart does not own", + self.node_id + ) + .into()); + }; + let (node_id, listen_addr) = (self.node_id, self.listen_addr); + self.graceful_shutdown_wal_only().await; + await_port_released(node_id, listen_addr).await?; + Ok(StoppedNode { + node_id, + listen_addr, + data_dir, + }) + } + + /// Bring `stopped` back on its node id, listen address and data. Its + /// persisted cluster state makes it rejoin as the same member. + pub(crate) async fn restart( + stopped: StoppedNode, + seed_nodes: Vec, + config: &ClusterSpawnConfig, + ) -> Result> { + let StoppedNode { + node_id, + listen_addr, + data_dir, + } = stopped; + let mut node = Self::spawn_with_full_config_at( + node_id, + seed_nodes, + config, + Some(data_dir.path().to_path_buf()), + Some(listen_addr), + ) + .await?; + node._data_dir = DataDir::Owned(data_dir); + Ok(node) + } +} + +/// Wait until `listen_addr`'s UDP port is free: the stopped node's QUIC +/// endpoint releases its socket once its last handle dropped, after the +/// close drained every connection. +async fn await_port_released( + node_id: u64, + listen_addr: SocketAddr, +) -> Result<(), Box> { + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + loop { + match tokio::net::UdpSocket::bind(listen_addr).await { + Ok(probe) => { + drop(probe); + return Ok(()); + } + Err(error) if tokio::time::Instant::now() >= deadline => { + return Err(format!( + "node {node_id} stopped, but its QUIC port {listen_addr} is still bound \ + after 10s ({error}): a task still holds the node's transport" + ) + .into()); + } + Err(_) => tokio::time::sleep(Duration::from_millis(20)).await, + } + } +} diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs index 273f7b7ea..33b163290 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs @@ -33,7 +33,7 @@ impl TestClusterNode { seed_nodes: Vec, config: &ClusterSpawnConfig, ) -> Result> { - Self::spawn_with_full_config_at(node_id, seed_nodes, config, None).await + Self::spawn_with_full_config_at(node_id, seed_nodes, config, None, None).await } /// Lowest-level cluster-node spawn. In addition to the tuning knobs of @@ -67,11 +67,16 @@ impl TestClusterNode { /// before this parameter existed); on a reopened directory it rebuilds /// in-memory-only structures (e.g. the vector HNSW index) from the /// persisted `TransactionRedo` / `Put` / etc. records. + /// + /// `listen_override`: `None` binds the QUIC transport on an ephemeral + /// port. `Some(addr)` binds it on `addr`, the address a restarted node's + /// peers already hold for it. pub(crate) async fn spawn_with_full_config_at( node_id: u64, seed_nodes: Vec, config: &ClusterSpawnConfig, data_dir_path_override: Option, + listen_override: Option, ) -> Result> { // Every cluster node funnels through here, so installing tracing at // this one point means no test has to opt in to see server-side logs. @@ -143,9 +148,13 @@ impl TestClusterNode { } else { // Pre-bind the QUIC transport on a random port so we know the // listen address before wiring seeds / cluster settings. + let bind_addr = match listen_override { + Some(addr) => addr, + None => "127.0.0.1:0".parse()?, + }; let transport = Arc::new(nodedb_cluster::NexarTransport::new( node_id, - "127.0.0.1:0".parse()?, + bind_addr, nodedb_cluster::TransportCredentials::Insecure, )?); let listen_addr = transport.local_addr(); diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs index d223e1684..dffccc7ca 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs @@ -159,6 +159,6 @@ impl TestClusterNode { replication_factor: 1, single_node_calvin: true, }; - Self::spawn_with_full_config_at(1, vec![], &config, data_dir_path).await + Self::spawn_with_full_config_at(1, vec![], &config, data_dir_path, None).await } } diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs index 0e8c5c555..0c77dc508 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs @@ -136,6 +136,20 @@ impl TestClusterNode { } } + // Close the QUIC endpoint. Its driver owns the UDP socket and runs + // until every connection is gone, and a live peer keeps its + // connection open until the idle timeout. Without the close, a + // restart on the same address finds the port still bound. + if let Some(transport) = self.shared.cluster_transport.clone() + && !transport.close(Duration::from_secs(2)).await + { + eprintln!( + "graceful_shutdown_wal_only: node {} peers did not acknowledge the transport \ + close within 2s", + self.node_id + ); + } + // `start_raft` fans out to background tasks (raft apply loop, tick // loop, sequencer service, RPC server, health monitor, per-vShard // Calvin schedulers, reconcile loop) that each hold an diff --git a/nodedb-test-support/src/tx_commit.rs b/nodedb-test-support/src/tx_commit.rs index bab25f2fa..62bc8fade 100644 --- a/nodedb-test-support/src/tx_commit.rs +++ b/nodedb-test-support/src/tx_commit.rs @@ -83,6 +83,7 @@ pub fn commit_plans( redo, collections: written_collections(&plans), sum_targets: redo_sum_targets(&plans), + origin: nodedb_physical::physical_plan::RedoOrigin::Commit, })); install.wal_lsn = Some(Lsn::new(lsn)); send_request(core, tx, rx, install) diff --git a/nodedb/src/bridge/dispatch/outcome_floor.rs b/nodedb/src/bridge/dispatch/outcome_floor.rs index a47c3eca2..b712bf673 100644 --- a/nodedb/src/bridge/dispatch/outcome_floor.rs +++ b/nodedb/src/bridge/dispatch/outcome_floor.rs @@ -73,6 +73,8 @@ pub struct OutcomeFloor { windows: Mutex, /// Windows dropped without a settle or a hold. leaked: AtomicU64, + /// Woken each time a window closes, so a waiter re-reads the floor. + closed: tokio::sync::Notify, } #[derive(Debug)] @@ -294,6 +296,12 @@ impl OutcomeFloor { Lsn::new(self.lock().floor()) } + /// The highest LSN any window noted: every record a write window minted + /// so far is at or below it. + pub fn max_noted(&self) -> Lsn { + Lsn::new(self.lock().max_noted) + } + /// Windows dropped without a settle or a hold since the process started. pub fn leaked_windows(&self) -> u64 { self.leaked.load(Ordering::Relaxed) @@ -331,6 +339,24 @@ impl OutcomeFloor { fn close(&self, ticket: u64) { self.lock().close(ticket); + self.closed.notify_waiters(); + } + + /// Wait until the floor reaches `target`: every record minted at or below + /// it has a final outcome. Returns `false` when `deadline` passes first, + /// for example behind a window held until restart. + pub async fn await_floor(&self, target: Lsn, deadline: tokio::time::Instant) -> bool { + loop { + let notified = self.closed.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + if self.floor() >= target { + return true; + } + if tokio::time::timeout_at(deadline, notified).await.is_err() { + return self.floor() >= target; + } + } } /// Count a leaked window and report it. The window stays open. diff --git a/nodedb/src/control/array_sync/inbound_propose.rs b/nodedb/src/control/array_sync/inbound_propose.rs index 6c0bddcd4..1b741faac 100644 --- a/nodedb/src/control/array_sync/inbound_propose.rs +++ b/nodedb/src/control/array_sync/inbound_propose.rs @@ -74,7 +74,11 @@ impl OriginArrayInbound { // Use the async proposer (with transparent leader forwarding + apply // wait) when available. It returns the apply payload directly. if let Some(async_proposer) = self.shared().async_raft_proposer().map(|a| a.as_ref()) { - return match async_proposer(vshard_id, idempotency_key, data).await { + let deadline = tokio::time::Instant::now() + + std::time::Duration::from_secs( + self.shared().tuning.network.default_deadline_secs, + ); + return match async_proposer(vshard_id, idempotency_key, data, deadline).await { Ok((_payload, _committed_version)) => Ok(()), Err(e) => { warn!(array = %array, error = %e, "array_inbound: raft propose+apply failed"); diff --git a/nodedb/src/control/array_sync/raft_apply/cell.rs b/nodedb/src/control/array_sync/raft_apply/cell.rs index 905b9f9c9..cb9e0f1a4 100644 --- a/nodedb/src/control/array_sync/raft_apply/cell.rs +++ b/nodedb/src/control/array_sync/raft_apply/cell.rs @@ -63,10 +63,12 @@ pub(crate) async fn apply_array_cell_write( target: ArrayCellTarget, plan: PhysicalPlan, ) -> bool { + let commit_hlc = pos.carried_commit_hlc(); let AppliedPosition { group_id, log_index, applied_key, + .. } = pos; let ArrayCellTarget { tenant_id, @@ -118,6 +120,7 @@ pub(crate) async fn apply_array_cell_write( event_source: crate::event::EventSource::User, resolved_now_ms, apply_key: applied_key, + commit_hlc, op_label: "array cell write", }, ) diff --git a/nodedb/src/control/array_sync/raft_apply/common.rs b/nodedb/src/control/array_sync/raft_apply/common.rs index 205002273..7ef31acf0 100644 --- a/nodedb/src/control/array_sync/raft_apply/common.rs +++ b/nodedb/src/control/array_sync/raft_apply/common.rs @@ -21,15 +21,26 @@ use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; /// Identifies a committed Raft entry within the apply loop. /// -/// Groups the three fields that always travel together: the Raft group, the -/// log index within that group, and the idempotency key extracted from the -/// `ReplicatedEntry` header. All three are forwarded together to -/// `ProposeTracker::complete` after each apply. +/// Groups the fields that always travel together: the Raft group, the log +/// index within that group, and the idempotency key extracted from the +/// `ReplicatedEntry` header, all forwarded to `ProposeTracker::complete` after +/// each apply, plus the entry's commit HLC. #[derive(Debug, Clone, Copy)] pub(crate) struct AppliedPosition { pub group_id: u64, pub log_index: u64, pub applied_key: u64, + /// HLC wall time, in nanoseconds, the proposer stamped on the entry. `0` + /// when the entry carries none. It is the write's commit instant on the + /// tenant's observed write high-water, however late this replica applies. + pub commit_hlc: u64, +} + +impl AppliedPosition { + /// The entry's commit HLC for the write funnel, `None` when it carries none. + pub(crate) fn carried_commit_hlc(&self) -> Option { + (self.commit_hlc != 0).then_some(self.commit_hlc) + } } /// One committed array write, ready for the Control-Plane write funnel. @@ -46,6 +57,8 @@ pub(super) struct ArrayWriteSubmit { /// The idempotency key of the committed entry, carried by the redo /// record's header. pub apply_key: u64, + /// The committed entry's commit HLC, `None` when it carries none. + pub commit_hlc: Option, /// Contextual label for the error surfaced to the propose waiter. pub op_label: &'static str, } @@ -74,6 +87,7 @@ pub(super) async fn submit_array_write( event_source, resolved_now_ms, apply_key, + commit_hlc, op_label, } = params; @@ -97,6 +111,7 @@ pub(super) async fn submit_array_write( durability: WalDurability::AppendHere { now_override: resolved_now_ms, apply_key, + commit_hlc, }, // Raft committed this entry at a fixed log index and every replica // applies it in that order; re-entering the write-admission gate diff --git a/nodedb/src/control/array_sync/raft_apply/op.rs b/nodedb/src/control/array_sync/raft_apply/op.rs index f61675c19..500f0380c 100644 --- a/nodedb/src/control/array_sync/raft_apply/op.rs +++ b/nodedb/src/control/array_sync/raft_apply/op.rs @@ -35,10 +35,12 @@ pub(crate) async fn apply_array_op( database_id, array, } = target; + let commit_hlc = pos.carried_commit_hlc(); let AppliedPosition { group_id, log_index, applied_key, + .. } = pos; use nodedb_array::sync::op_codec; @@ -202,6 +204,7 @@ pub(crate) async fn apply_array_op( // KV writes resolve one, and no array op is such a write. resolved_now_ms: None, apply_key: applied_key, + commit_hlc, op_label: "array op", }, ) diff --git a/nodedb/src/control/array_sync/raft_apply/schema.rs b/nodedb/src/control/array_sync/raft_apply/schema.rs index 615a1bc1a..f5db9f3f8 100644 --- a/nodedb/src/control/array_sync/raft_apply/schema.rs +++ b/nodedb/src/control/array_sync/raft_apply/schema.rs @@ -45,6 +45,7 @@ pub(crate) fn apply_array_schema( group_id, log_index, applied_key, + .. } = pos; use nodedb_array::sync::hlc::Hlc; diff --git a/nodedb/src/control/backup/cut.rs b/nodedb/src/control/backup/cut.rs new file mode 100644 index 000000000..ba8f7f4a5 --- /dev/null +++ b/nodedb/src/control/backup/cut.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's consistent cut. +//! +//! The cut picks the envelope watermark `W`, then waits until every user +//! write committed below `W` has a final outcome on this node, and only then +//! lets the backup snapshot. Afterwards every write is one of two kinds: +//! +//! - committed below `W`: applied before the snapshot, so the backup holds it; +//! - committed at or above `W`: its mark is above `W`, so a restore of this +//! backup refuses it. +//! +//! Three waits make the cut: +//! +//! - **Raft data groups.** A replicated write carries its proposer's commit +//! stamp and applies in log order. The cut proposes a +//! [`ReplicatedWrite::CutBarrier`] carrying `W` into every data group this +//! node hosts and waits for this node's apply of it. Every entry before the +//! barrier applied first. Every entry after it records a commit HLC above +//! `W`, however early its proposer stamped it. +//! - **Calvin transactions.** A Calvin transaction commits at its place in +//! the sequencer log. The cut proposes a `CutMarker` carrying `W` into the +//! sequencer log and waits until every Calvin scheduler on this node passed +//! it: every transaction delivered before the marker installed or dropped. +//! Every transaction delivered after it records a commit HLC above `W`. +//! - **Local write windows.** A write the funnel appends here stamps itself +//! after its record is minted inside an outcome-floor window. The cut reads +//! the highest LSN any window minted, after it picks `W`, and waits for the +//! outcome floor to reach it. A record minted later stamps itself above `W`. +//! On a server with no Raft groups a write stamps itself before it mints, so +//! its mark is durable first. The cut first waits for every such stamp at or +//! below `W` to mint, then reads the highest LSN. +//! +//! Two stamps can share a wall time. Once it picks `W`, the cut moves this +//! node's clock past `W`, so every later stamp here reads above it. +//! +//! The waits bind this node's replicas. A remote source node takes the same +//! cut at `W` itself before it snapshots: the snapshot plan sent to it carries +//! `W`, and its Control Plane runs [`cut_at`] first. + +use std::collections::BTreeMap; +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::calvin::SequencerEntry; +use nodedb_types::Hlc; + +use crate::Error; +use crate::control::security::auth_fence::cluster::{group_of_vshard, hosts_group, routed_groups}; +use crate::control::state::SharedState; +use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite, propose_replicated_entry}; +use crate::types::{DatabaseId, VShardId}; + +/// Pick the envelope watermark and wait until every write committed below it +/// has a final outcome on this node. Returns the watermark. +pub(super) async fn consistent_cut(state: &Arc, tenant_id: u64) -> Result { + let watermark = state.hlc_clock.now().wall_ns; + cut_at(state, tenant_id, watermark).await?; + Ok(watermark) +} + +/// Take the consistent cut at `watermark` on this node: wait until every +/// write committed below it has a final outcome here. A remote source node +/// runs it for the watermark the backup's coordinator picked. +pub(crate) async fn cut_at( + state: &Arc, + tenant_id: u64, + watermark: u64, +) -> Result<(), Error> { + state + .hlc_clock + .update(Hlc::new(watermark.saturating_add(1), 0)); + let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); + let deadline = tokio::time::Instant::now() + timeout; + // A local write stamped at or below the watermark has not always minted + // its record yet. Every stamp taken from here on reads above it. + if !state + .tenant_marks + .await_local_stamps_minted(watermark, deadline) + .await + { + return Err(Error::Internal { + detail: format!( + "backup: local writes stamped at or below watermark {watermark} did not mint \ + their WAL records within {}s, so the backup cannot take a consistent cut. \ + Retry the backup", + timeout.as_secs(), + ), + }); + } + // Read after the watermark: a record minted after this read stamps its + // write above the watermark. + let target = state.outcome_floor.max_noted(); + + let (groups, calvin) = tokio::join!( + cut_data_groups(state, tenant_id, watermark), + cut_calvin(state, watermark, deadline), + ); + groups?; + calvin?; + + if !state.outcome_floor.await_floor(target, deadline).await { + return Err(Error::Internal { + detail: format!( + "backup: writes minted at or below WAL LSN {} had no final outcome within \ + {}s, so the backup cannot take a consistent cut (outcome floor at {}). \ + Retry the backup. A write window held until restart keeps the floor \ + below it: restart the node if the floor does not move", + target.as_u64(), + timeout.as_secs(), + state.outcome_floor.floor().as_u64(), + ), + }); + } + Ok(()) +} + +/// Propose a cut barrier carrying `watermark` into every data group this node +/// hosts, and wait for this node's apply of each. +async fn cut_data_groups( + state: &Arc, + tenant_id: u64, + watermark: u64, +) -> Result<(), Error> { + let Some(proposer) = state.async_raft_proposer() else { + return Ok(()); + }; + let barriers = futures::future::join_all(barrier_vshards(state).into_iter().map( + |(group_id, vshard_id)| { + let entry = ReplicatedEntry::new( + tenant_id, + DatabaseId::DEFAULT.as_u64(), + vshard_id, + ReplicatedWrite::CutBarrier { hlc: watermark }, + ); + async move { + propose_replicated_entry(state, proposer, entry) + .await + .map_err(|error| (group_id, error)) + } + }, + )) + .await; + for barrier in barriers { + if let Err((group_id, error)) = barrier { + // A group this node left while the barrier waited holds no + // replica here to cut: the source node that snapshots it takes + // its own cut. + if !hosts_group(state, group_id) { + tracing::info!( + group_id, + %error, + "backup: this node left the group before its cut barrier applied here; \ + the group needs no cut on this node" + ); + continue; + } + return Err(Error::Internal { + detail: format!( + "backup: the consistent-cut barrier of raft group {group_id} did not \ + apply on this node: {error}. Retry the backup" + ), + }); + } + } + Ok(()) +} + +/// How long the cut waits for its Calvin marker before it proposes the +/// marker again. A leader change can drop a proposed marker; a second copy is +/// harmless, since a scheduler passes each marker once its earlier +/// transactions finished. +const CUT_MARKER_RETRY: Duration = Duration::from_secs(1); + +/// Propose a Calvin cut marker carrying `watermark`, and wait until every +/// Calvin scheduler on this node passed it, or `deadline`. +pub(crate) async fn cut_calvin( + state: &Arc, + watermark: u64, + deadline: tokio::time::Instant, +) -> Result<(), Error> { + let cuts = &state.calvin.cuts; + if cuts.is_empty() { + // No Calvin scheduler runs here: this node applies no Calvin write. + return Ok(()); + } + let proposer = state + .calvin + .sequencer_proposer + .get() + .ok_or_else(|| Error::Internal { + detail: "backup: Calvin schedulers run on this node, but no sequencer proposer \ + is set, so the consistent cut cannot place its marker. Retry the backup \ + once the cluster finished starting" + .into(), + })?; + let marker = zerompk::to_msgpack_vec(&SequencerEntry::CutMarker { hlc: watermark }).map_err( + |error| Error::Internal { + detail: format!("backup: encode the Calvin cut marker: {error}"), + }, + )?; + let mut last_refusal = None; + loop { + if let Err(error) = proposer.propose(marker.clone()) { + last_refusal = Some(error.to_string()); + } + let attempt_deadline = deadline.min(tokio::time::Instant::now() + CUT_MARKER_RETRY); + let lagging = cuts.await_passed(watermark, attempt_deadline).await; + if lagging.is_empty() { + return Ok(()); + } + if tokio::time::Instant::now() >= deadline { + return Err(Error::Internal { + detail: format!( + "backup: the Calvin schedulers of vShards {lagging:?} did not pass the \ + consistent-cut marker in time, so Calvin transactions sequenced before \ + the backup may not be installed (last marker refusal: {}). Retry the \ + backup", + last_refusal.as_deref().unwrap_or("none") + ), + }); + } + } +} + +/// One vShard per data group this node hosts: the barrier of a group is +/// proposed through any vShard the group homes. +fn barrier_vshards(state: &SharedState) -> BTreeMap { + let hosted: Vec = routed_groups(state) + .into_iter() + .filter(|group_id| { + *group_id != METADATA_GROUP_ID + && *group_id != SEQUENCER_GROUP_ID + && hosts_group(state, *group_id) + }) + .collect(); + let mut vshards = BTreeMap::new(); + for vshard_id in 0..VShardId::COUNT { + if vshards.len() == hosted.len() { + break; + } + if let Ok(group_id) = group_of_vshard(state, vshard_id) + && hosted.contains(&group_id) + { + vshards.entry(group_id).or_insert(vshard_id); + } + } + vshards +} diff --git a/nodedb/src/control/backup/mod.rs b/nodedb/src/control/backup/mod.rs index f12b96bca..710a692b4 100644 --- a/nodedb/src/control/backup/mod.rs +++ b/nodedb/src/control/backup/mod.rs @@ -1,5 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod cut; pub mod detect; pub mod orchestrator; pub mod restore; diff --git a/nodedb/src/control/backup/orchestrator.rs b/nodedb/src/control/backup/orchestrator.rs index 74535cefa..b003d346f 100644 --- a/nodedb/src/control/backup/orchestrator.rs +++ b/nodedb/src/control/backup/orchestrator.rs @@ -43,20 +43,29 @@ pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result1: keep only the @@ -66,13 +75,6 @@ pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result, tenant_id: u64) -> Result = Vec::new(); - for coll in all.iter().filter(|c| c.tenant_id == tenant_id) { - if let Ok(bytes) = zerompk::to_msgpack_vec(coll) { - blobs.push(nodedb_types::backup_envelope::StoredCollectionBlob { - name: coll.name.clone(), - bytes, - }); - } - } - if !blobs.is_empty() - && let Ok(body) = zerompk::to_msgpack_vec(&blobs) - { - writer - .push_section( - nodedb_types::backup_envelope::SECTION_ORIGIN_CATALOG_ROWS, - body, - ) - .map_err(|e| Error::Internal { - detail: format!("backup envelope (catalog rows): {e}"), - })?; - } - } - - // PK→surrogate identity map for the tenant's collections. This is - // DATA-derived per-node state that the per-node engine sections do NOT - // carry (the Data-Plane snapshot handler has no catalog access). Without - // it a restored node has documents but cannot resolve PK point-lookups - // (`WHERE id=`) — full scans work, point-lookups silently miss. The - // restore path rebinds these into the destination catalog. - if let Ok(all) = catalog.load_all_collections(DatabaseId::DEFAULT) { - let mut binds: Vec = Vec::new(); - for coll in all.iter().filter(|c| c.tenant_id == tenant_id) { - if let Ok(rows) = catalog.scan_surrogates_for_collection( - DatabaseId::DEFAULT, - TenantId::new(tenant_id), - &coll.name, - ) { - for (pk, surrogate) in rows { - binds.push(nodedb_types::backup_envelope::SurrogateBindBlob { - tenant_id, - collection: coll.name.clone(), - pk, - surrogate: surrogate.as_u32(), - }); - } - } - } - if !binds.is_empty() - && let Ok(body) = zerompk::to_msgpack_vec(&binds) - { - writer - .push_section( - nodedb_types::backup_envelope::SECTION_ORIGIN_SURROGATE_PK, - body, - ) - .map_err(|e| Error::Internal { - detail: format!("backup envelope (surrogate pk): {e}"), - })?; - } - } - - if let Ok(tset) = catalog.load_wal_tombstones() { - let mut tombs: Vec = Vec::new(); - for (database_id, tid, name, purge_lsn) in tset.iter() { - if database_id == DatabaseId::DEFAULT.as_u64() && tid == tenant_id { - tombs.push(nodedb_types::backup_envelope::SourceTombstoneEntry { - collection: name.to_string(), - purge_lsn, - }); - } - } - if !tombs.is_empty() - && let Ok(body) = zerompk::to_msgpack_vec(&tombs) - { - writer - .push_section( - nodedb_types::backup_envelope::SECTION_ORIGIN_SOURCE_TOMBSTONES, - body, - ) - .map_err(|e| Error::Internal { - detail: format!("backup envelope (source tombstones): {e}"), - })?; - } - } - } + push_metadata_sections(state, tenant_id, &mut writer)?; // A backup KEK must be configured; plaintext backup envelopes are no // longer supported. @@ -254,8 +169,7 @@ fn source_assignment(state: &SharedState) -> Vec<(u64, HashSet)> { /// /// The per-section vshard classification is shared with the Raft snapshot SEND /// builder via `snapshot_keys::retain_tenant_data_for_vshards`. The vshard-of -/// closure is the canonical routing function, matching both the snapshot -/// builder and the restore topology splitter. +/// closure is the canonical routing function, matching the snapshot builder. fn filter_node_snapshot( body: Vec, tenant_id: u64, @@ -278,6 +192,112 @@ fn filter_node_snapshot( }) } +/// Push the metadata sections: the tenant's catalog rows, its PK-to-surrogate +/// binds, and its source-side tombstones. A catalog read or encode error fails +/// the backup: an envelope without these sections restores rows a point +/// lookup cannot find, or resurrects a purged collection. +fn push_metadata_sections( + state: &SharedState, + tenant_id: u64, + writer: &mut EnvelopeWriter, +) -> Result<(), Error> { + let catalog = state.credentials.catalog(); + let collections: Vec<_> = catalog + .load_all_collections(DatabaseId::DEFAULT)? + .into_iter() + .filter(|coll| coll.tenant_id == tenant_id) + .collect(); + + let mut blobs = Vec::with_capacity(collections.len()); + for coll in &collections { + blobs.push(nodedb_types::backup_envelope::StoredCollectionBlob { + name: coll.name.clone(), + bytes: encode_section_part("catalog row", coll)?, + }); + } + if !blobs.is_empty() { + push_encoded( + writer, + nodedb_types::backup_envelope::SECTION_ORIGIN_CATALOG_ROWS, + "catalog rows", + &blobs, + )?; + } + + // PK→surrogate identity map for the tenant's collections. This is + // DATA-derived per-node state that the per-node engine sections do NOT + // carry (the Data-Plane snapshot handler has no catalog access). Without + // it a restored node has documents but cannot resolve PK point-lookups + // (`WHERE id=`): full scans work, point-lookups silently miss. The + // restore path rebinds these into the destination catalog. + let mut binds: Vec = Vec::new(); + for coll in &collections { + let rows = catalog.scan_surrogates_for_collection( + DatabaseId::DEFAULT, + TenantId::new(tenant_id), + &coll.name, + )?; + for (pk, surrogate) in rows { + binds.push(nodedb_types::backup_envelope::SurrogateBindBlob { + tenant_id, + collection: coll.name.clone(), + pk, + surrogate: surrogate.as_u32(), + }); + } + } + if !binds.is_empty() { + push_encoded( + writer, + nodedb_types::backup_envelope::SECTION_ORIGIN_SURROGATE_PK, + "surrogate pk", + &binds, + )?; + } + + let mut tombs: Vec = Vec::new(); + for (database_id, tid, name, purge_lsn) in catalog.load_wal_tombstones()?.iter() { + if database_id == DatabaseId::DEFAULT.as_u64() && tid == tenant_id { + tombs.push(nodedb_types::backup_envelope::SourceTombstoneEntry { + collection: name.to_string(), + purge_lsn, + }); + } + } + if !tombs.is_empty() { + push_encoded( + writer, + nodedb_types::backup_envelope::SECTION_ORIGIN_SOURCE_TOMBSTONES, + "source tombstones", + &tombs, + )?; + } + Ok(()) +} + +/// Encode one part of a metadata section. +fn encode_section_part(what: &str, value: &T) -> Result, Error> { + zerompk::to_msgpack_vec(value).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("backup envelope ({what}): encode: {e}"), + }) +} + +/// Encode `value` and push it as the section `origin`. +fn push_encoded( + writer: &mut EnvelopeWriter, + origin: u64, + what: &str, + value: &T, +) -> Result<(), Error> { + let body = encode_section_part(what, value)?; + writer + .push_section(origin, body) + .map_err(|e| Error::Internal { + detail: format!("backup envelope ({what}): {e}"), + }) +} + fn is_self(state: &SharedState, node_id: u64) -> bool { node_id == state.node_id || node_id == 0 || state.cluster_transport.is_none() } diff --git a/nodedb/src/control/backup/restore/columnar_reissue.rs b/nodedb/src/control/backup/restore/columnar_reissue.rs index d10ed4225..323fb5105 100644 --- a/nodedb/src/control/backup/restore/columnar_reissue.rs +++ b/nodedb/src/control/backup/restore/columnar_reissue.rs @@ -9,7 +9,6 @@ //! on single-node — the same branch a normal write takes. use std::collections::HashMap; -use std::time::Duration; use nodedb_columnar::{ColumnarEngineSnapshot, MutationEngine, materialize_segment_live_rows}; use nodedb_types::RlsWriteCheck; @@ -18,10 +17,6 @@ use nodedb_types::value::Value; use crate::Error; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; -use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, VShardId}; use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp}; /// Live rows of a decoded snapshot, ready to re-issue. @@ -161,72 +156,6 @@ pub fn build_columnar_insert_plan( })) } -/// Re-issue a restored columnar collection's rows durably. -/// -/// Branches identically to a normal write: -/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: append the redo under an outcome-floor window, then -/// `sync_dispatch::dispatch_system`, which closes the window. -pub async fn reissue_columnar_durably( - state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - collection: &str, - plan: PhysicalPlan, -) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); - - if let Some(proposer) = state.async_raft_proposer() { - let entry = crate::control::wal_replication::to_replicated_entry( - tenant_id, - database_id, - vshard, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or_else(|| Error::Internal { - detail: format!( - "restore reissue: columnar plan for '{collection}' did not map to a \ - replicated write" - ), - })?; - crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; - return Ok(()); - } - - // Single-node: WAL first (durable for restart replay), then install live. - // The record's outcome-floor window opens before the append and closes - // from the install's outcome. - let owner = RecordOwner { - tenant_id, - database_id, - vshard_id: vshard, - }; - let minted = MintedRecords::open(&state.outcome_floor); - if let Err(error) = minted.append_plan(&state.wal, owner, &plan) { - // Any record appended before the error never reaches a core. - minted.cancel(&state.wal, owner, 0).await?; - return Err(error); - } - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - tenant_id, - database_id, - collection, - plan, - ) - .with_minted(minted), - REISSUE_TIMEOUT, - ) - .await?; - Ok(()) -} - -/// Per-collection re-issue dispatch timeout. Generous: a restored collection may -/// carry many flushed segments' worth of rows in one insert. -const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); - /// Convert a row's positional `Value`s (schema column order) into a field-keyed /// `Value::Object`. Errors if the arity does not match the schema. fn row_values_to_object( diff --git a/nodedb/src/control/backup/restore/crdt_reissue.rs b/nodedb/src/control/backup/restore/crdt_reissue.rs index 4603205c0..3dbbf83e6 100644 --- a/nodedb/src/control/backup/restore/crdt_reissue.rs +++ b/nodedb/src/control/backup/restore/crdt_reissue.rs @@ -27,7 +27,7 @@ const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); /// Re-issue one collection's snapshot import to the data group owning its /// vshard. /// -/// Branches identically to a normal write (and to `reissue_timeseries_durably`): +/// Branches identically to a normal write (and to `durable::reissue_plan_durably`): /// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. /// - Single-node: the autocommit funnel appends the redo and installs it. async fn reissue_crdt_collection( diff --git a/nodedb/src/control/backup/restore/durable.rs b/nodedb/src/control/backup/restore/durable.rs new file mode 100644 index 000000000..54a5963f9 --- /dev/null +++ b/nodedb/src/control/backup/restore/durable.rs @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The durable write every RESTORE re-issue goes through. +//! +//! A restored row installed straight into a Data-Plane map has no WAL record +//! and no Raft entry: it is lost on restart, and only the one node that +//! installed it holds it. A re-issued row is a normal write instead: every +//! replica of its group applies it, and each one's WAL makes it durable. + +use std::time::Duration; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; +use crate::control::server::shared::ddl::sync_dispatch; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, VShardId}; + +/// Dispatch timeout of one re-issued write. Generous: one restored +/// collection's rows may travel in a single write. +const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); + +/// Write `plan`, restored into `collection`, durably. +/// +/// Branches identically to a normal write: +/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. +/// - Single-node: append the redo under an outcome-floor window, then +/// `sync_dispatch::dispatch_system`, which closes the window. +pub async fn reissue_plan_durably( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + plan: PhysicalPlan, +) -> crate::Result<()> { + let vshard = VShardId::from_collection_in_database(database_id, collection); + + if let Some(proposer) = state.async_raft_proposer() { + let entry = crate::control::wal_replication::to_replicated_entry( + tenant_id, + database_id, + vshard, + &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, + )? + .ok_or_else(|| Error::Internal { + detail: format!( + "restore reissue: the plan restored into '{collection}' did not map to a \ + replicated write" + ), + })?; + let (_, write_version) = + crate::control::wal_replication::propose_replicated_entry(state, proposer, entry) + .await?; + tracing::debug!( + collection, + vshard_id = vshard.as_u32(), + write_version = write_version.as_u64(), + "restore: re-issued write applied on this node" + ); + return Ok(()); + } + + // Single-node: WAL first (durable for restart replay), then install live. + // The record's outcome-floor window opens before the append and closes + // from the install's outcome. + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id: vshard, + }; + let minted = MintedRecords::open(&state.outcome_floor); + if let Err(error) = minted.append_plan(&state.wal, owner, &plan) { + // Any record appended before the error never reaches a core. + minted.cancel(&state.wal, owner, 0).await?; + return Err(error); + } + sync_dispatch::dispatch_system( + state, + sync_dispatch::SystemTask::new( + sync_dispatch::SystemReason::BackupRestore, + tenant_id, + database_id, + collection, + plan, + ) + .with_minted(minted), + REISSUE_TIMEOUT, + ) + .await?; + Ok(()) +} + +/// Log one restore re-issue step: what it writes, where, and who proposes +/// it, so every step of a restore shows in the log. +pub(super) fn log_reissue_step( + state: &SharedState, + step: &'static str, + collection: &str, + vshard: VShardId, + rows: usize, +) { + let group_id = + crate::control::security::auth_fence::cluster::group_of_vshard(state, vshard.as_u32()).ok(); + tracing::info!( + step, + collection, + vshard_id = vshard.as_u32(), + group_id = ?group_id, + rows, + proposer_node = state.node_id, + replicated = state.async_raft_proposer().is_some(), + "restore: re-issuing rows" + ); +} diff --git a/nodedb/src/control/backup/restore/guard.rs b/nodedb/src/control/backup/restore/guard.rs new file mode 100644 index 000000000..6365a05b1 --- /dev/null +++ b/nodedb/src/control/backup/restore/guard.rs @@ -0,0 +1,416 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's staleness guard: the newest committed write of a tenant. +//! +//! The answer comes from replicated state. Every data group's replicas derive +//! the same durable per-tenant marks from the entries they apply, and a +//! committed Calvin transaction records its mark in the data group that homes +//! its vShard before its COMMIT is acknowledged. The guard reads the marks of +//! every data group: +//! +//! - a group this node replicates, here, once this node applied every entry +//! the group committed before the read and every Calvin transaction +//! sequenced before it installed; +//! - any other group, from a current replica, which does the same before it +//! answers. +//! +//! A replica that left the group refuses with a typed `NotLeader`. The guard +//! then asks another replica, until one statement deadline shared by every +//! group. A group no replica answered for by then refuses the restore. +//! +//! A node that restarted, or that never applied the write, answers alike. + +use std::collections::{BTreeMap, BTreeSet}; +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::rpc_codec::{ExecuteRequest, ExecuteResponse, RaftRpc, TypedClusterError}; +use nodedb_physical::physical_plan::{ClusterEventOp, PhysicalPlan, wire as plan_wire}; + +use crate::Error; +use crate::control::security::auth_fence::cluster::{ + confirmed_read_index, hosts_group, routed_groups, wait_applied, +}; +use crate::control::state::SharedState; +use crate::control::state::tenant_marks::{GroupMark, LOCAL_MARK_GROUP}; +use crate::types::{DatabaseId, TraceId}; + +/// First wait before the guard asks again for the marks of groups whose +/// replica refused. Each round doubles it up to [`MAX_ASK_BACKOFF`]. +const FIRST_ASK_BACKOFF: Duration = Duration::from_millis(10); + +/// Longest wait between two rounds of asking. +const MAX_ASK_BACKOFF: Duration = Duration::from_millis(200); + +/// The newest committed write of a tenant, and where it was recorded. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(super) struct NewestWrite { + /// HLC wall time, in nanoseconds, of its commit. + pub hlc: u64, + /// The apply path that recorded it. + pub site: String, + /// The collection it named, when it named one. + pub collection: Option, +} + +/// One group's mark on the wire: `(group_id, commit_hlc, site_code, +/// collection)`. +type WireMark = (u64, u64, u8, String); + +/// The newest committed write of `tenant_id` across every data group, and +/// this node's own mark of writes no data group carries. +pub(super) async fn newest_committed_write( + state: &Arc, + tenant_id: u64, +) -> Result, Error> { + let mut newest = state.tenant_write_mark(tenant_id).map(|mark| NewestWrite { + hlc: mark.hlc, + site: mark.origin.site.to_owned(), + collection: mark.origin.collection, + }); + let mut consider = |mark: GroupMark| { + if newest.as_ref().is_none_or(|current| mark.hlc > current.hlc) { + newest = Some(NewestWrite { + hlc: mark.hlc, + site: mark.site.as_str().to_owned(), + collection: mark.collection, + }); + } + }; + // The durable mark of this node's writes while it ran with no Raft groups. + if let Some(mark) = state.tenant_marks.get(LOCAL_MARK_GROUP, tenant_id) { + consider(mark); + } + + // The statement deadline, shared by every group and every attempt. + let deadline = tokio::time::Instant::now() + + Duration::from_secs(state.tuning.network.default_deadline_secs); + let groups: Vec = routed_groups(state) + .into_iter() + .filter(|group| *group != METADATA_GROUP_ID && *group != SEQUENCER_GROUP_ID) + .collect(); + for (_, mark) in group_marks(state, tenant_id, groups, deadline).await? { + consider(mark); + } + Ok(newest) +} + +/// The marks of `tenant_id` in every group of `groups`, each from a current +/// replica of the group. +/// +/// Each round reads the groups this node replicates here, and asks one +/// replica of every other group. A replica that refuses a group, because it +/// does not replicate it, is not asked for that group again until every known +/// replica refused. Every round ends by `deadline`. A group still unanswered +/// then fails the call with [`Error::GroupMarksUnavailable`]. Any other error +/// fails it at once. +async fn group_marks( + state: &Arc, + tenant_id: u64, + groups: Vec, + deadline: tokio::time::Instant, +) -> Result, Error> { + let mut pending: BTreeMap = groups + .into_iter() + .map(|group| (group, GroupAsk::default())) + .collect(); + let mut marks = Vec::new(); + let mut backoff = FIRST_ASK_BACKOFF; + loop { + let (local, remote): (Vec, Vec) = pending + .keys() + .copied() + .partition(|group| hosts_group(state, *group)); + if !local.is_empty() { + marks.extend(local_tenant_marks(state, tenant_id, &local, deadline).await?); + for group in &local { + pending.remove(group); + } + } + let mut by_node: BTreeMap> = BTreeMap::new(); + for group_id in remote { + if let Some(ask) = pending.get_mut(&group_id) + && let Some(node_id) = ask.next_target(state, group_id) + { + by_node.entry(node_id).or_default().push(group_id); + } + } + let answers = futures::future::join_all(by_node.into_iter().map(|(node_id, groups)| { + let asked = groups.clone(); + async move { + let answer = remote_tenant_marks(state, node_id, tenant_id, groups, deadline).await; + (node_id, asked, answer) + } + })) + .await; + for (node_id, asked, answer) in answers { + match answer { + Ok(answered) => { + marks.extend(answered); + for group in &asked { + pending.remove(group); + } + } + Err(RemoteMarksError::NotReplica { group_id, hint }) => { + if let Some(ask) = pending.get_mut(&group_id) { + ask.refused(node_id, hint); + } + } + Err(RemoteMarksError::Failed(error)) => return Err(error), + } + } + let Some((&group_id, ask)) = pending.iter().next() else { + return Ok(marks); + }; + if tokio::time::Instant::now() + backoff >= deadline { + return Err(Error::GroupMarksUnavailable { + group_id, + refused_by: ask.refused_by.iter().copied().collect(), + }); + } + tokio::time::sleep(backoff).await; + backoff = (backoff * 2).min(MAX_ASK_BACKOFF); + } +} + +/// Which replicas of one group the guard asked, and which refused. +#[derive(Debug, Default)] +struct GroupAsk { + /// Nodes that refused the group since the last time every known replica + /// refused it. + refused: BTreeSet, + /// Every node that refused the group, for the deadline error. + refused_by: BTreeSet, + /// The replica a refusing node's routing table named. + hint: Option, +} + +impl GroupAsk { + /// The node to ask next: the last refusal's hint, then the group's leader, + /// voters and learners in this node's routing table, skipping this node + /// and every node that refused. When every candidate refused, the refused + /// set is cleared for the next round, since routing tables converge, and + /// this round asks no node. + fn next_target(&mut self, state: &SharedState, group_id: u64) -> Option { + let candidates = self.hint.into_iter().chain(group_replicas(state, group_id)); + let mut any = false; + for node in candidates { + if node == 0 || node == state.node_id { + continue; + } + any = true; + if !self.refused.contains(&node) { + return Some(node); + } + } + if any { + self.refused.clear(); + self.hint = None; + } + None + } + + fn refused(&mut self, node_id: u64, hint: Option) { + self.refused.insert(node_id); + self.refused_by.insert(node_id); + self.hint = hint.filter(|hint| !self.refused.contains(hint)); + } +} + +/// The replicas of `group_id` in this node's routing table: its leader when +/// known, then its voters, then its learners. +fn group_replicas(state: &SharedState, group_id: u64) -> Vec { + let Some(routing) = state.cluster_routing.as_ref() else { + return Vec::new(); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let Some(info) = routing.group_info(group_id) else { + return Vec::new(); + }; + std::iter::once(info.leader) + .chain(info.members.iter().copied()) + .chain(info.learners.iter().copied()) + .collect() +} + +/// This node's marks of `tenant_id` in `group_ids`, once this node applied +/// every entry the groups committed before the call and every Calvin +/// transaction sequenced before it installed here. Every wait ends by +/// `deadline`. +pub(crate) async fn local_tenant_marks( + state: &Arc, + tenant_id: u64, + group_ids: &[u64], + deadline: tokio::time::Instant, +) -> Result, Error> { + if !group_ids.is_empty() { + let marker = state.hlc_clock.now().wall_ns; + crate::control::backup::cut::cut_calvin(state, marker, deadline).await?; + } + let mut marks = Vec::new(); + for &group_id in group_ids { + let index = confirmed_read_index(state, group_id, remaining(deadline)?).await?; + wait_applied(state, group_id, index, remaining(deadline)?).await?; + if let Some(mark) = state.tenant_marks.get(group_id, tenant_id) { + marks.push((group_id, mark)); + } + } + Ok(marks) +} + +/// Encode marks for the wire. +pub(crate) fn encode_marks(marks: &[(u64, GroupMark)]) -> Result, Error> { + let wire: Vec = marks + .iter() + .map(|(group_id, mark)| { + ( + *group_id, + mark.hlc, + mark.site.code(), + mark.collection.clone().unwrap_or_default(), + ) + }) + .collect(); + zerompk::to_msgpack_vec(&wire).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("tenant write marks: encode: {e}"), + }) +} + +fn decode_marks(bytes: &[u8]) -> Result, Error> { + let wire: Vec = zerompk::from_msgpack(bytes).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("tenant write marks: decode: {e}"), + })?; + Ok(wire + .into_iter() + .map(|(group_id, hlc, site, collection)| { + ( + group_id, + GroupMark { + hlc, + site: crate::control::state::tenant_marks::MarkSite::from_code(site), + collection: (!collection.is_empty()).then_some(collection), + }, + ) + }) + .collect()) +} + +/// The time left before `deadline`, or the deadline error once it passed. +fn remaining(deadline: tokio::time::Instant) -> Result { + let left = deadline.saturating_duration_since(tokio::time::Instant::now()); + if left.is_zero() { + return Err(Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + }); + } + Ok(left) +} + +/// Why a replica gave no marks. +enum RemoteMarksError { + /// The node does not replicate `group_id`. `hint` is the replica its + /// routing table names. + NotReplica { group_id: u64, hint: Option }, + /// Any other error. It ends the guard. + Failed(Error), +} + +impl From for RemoteMarksError { + fn from(error: Error) -> Self { + Self::Failed(error) + } +} + +/// Ask `node_id` for its marks of `tenant_id` in `group_ids`, within what +/// remains of `deadline`. +async fn remote_tenant_marks( + state: &SharedState, + node_id: u64, + tenant_id: u64, + group_ids: Vec, + deadline: tokio::time::Instant, +) -> Result, RemoteMarksError> { + let transport = state + .cluster_transport + .as_ref() + .ok_or_else(|| Error::Internal { + detail: format!( + "restore: node {node_id} replicates a data group of the tenant, but this node \ + has no cluster transport to ask it for the group's newest write" + ), + })?; + let plan = PhysicalPlan::ClusterEvent(ClusterEventOp::TenantWriteMarks { + tenant_id, + group_ids, + }); + let plan_bytes = plan_wire::encode(&plan).map_err(|e| Error::Internal { + detail: format!("restore: encode the tenant write-mark request: {e}"), + })?; + let budget = remaining(deadline)?; + let request = RaftRpc::ExecuteRequest(ExecuteRequest { + plan_bytes, + tenant_id, + database_id: DatabaseId::DEFAULT.as_u64(), + deadline_remaining_ms: u64::try_from(budget.as_millis()).unwrap_or(u64::MAX), + trace_id: TraceId::generate().0, + descriptor_versions: Vec::new(), + txn_id: None, + }); + let response = tokio::time::timeout_at(deadline, transport.send_rpc(node_id, request)) + .await + .map_err(|_| Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + })? + .map_err(|e| Error::Internal { + detail: format!("restore: tenant write-mark request to node {node_id} failed: {e}"), + })?; + match response { + RaftRpc::ExecuteResponse(ExecuteResponse { + success: true, + payloads, + .. + }) => match payloads.as_slice() { + [payload] => Ok(decode_marks(payload)?), + _ => Err(Error::Internal { + detail: format!( + "restore: node {node_id} answered the tenant write-mark request with {} \ + payloads, expected 1", + payloads.len() + ), + } + .into()), + }, + RaftRpc::ExecuteResponse(ExecuteResponse { + error: + Some(TypedClusterError::NotLeader { + group_id, + leader_node_id, + .. + }), + .. + }) => Err(RemoteMarksError::NotReplica { + group_id, + hint: leader_node_id, + }), + RaftRpc::ExecuteResponse(ExecuteResponse { + error: Some(error), .. + }) => Err(Error::from(error).into()), + RaftRpc::ExecuteResponse(ExecuteResponse { error: None, .. }) => Err(Error::Internal { + detail: format!( + "restore: node {node_id} failed the tenant write-mark request without an error" + ), + } + .into()), + other => Err(Error::Internal { + detail: format!( + "restore: unexpected reply to the tenant write-mark request from node \ + {node_id}: {other:?}" + ), + } + .into()), + } +} diff --git a/nodedb/src/control/backup/restore/kv_reissue.rs b/nodedb/src/control/backup/restore/kv_reissue.rs new file mode 100644 index 000000000..7b98a67c1 --- /dev/null +++ b/nodedb/src/control/backup/restore/kv_reissue.rs @@ -0,0 +1,118 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable re-issue of restored KV rows. +//! +//! The per-node snapshot install puts a KV row straight into the target +//! node's memtable, with no WAL record and no Raft entry: only that node holds +//! it, and it is gone after a restart. RESTORE re-issues each row as a +//! `KvOp::Put` instead, so every replica of the collection's group applies it +//! and its WAL makes it durable. + +use nodedb_physical::physical_plan::KvOp; +use nodedb_types::QualifiedCollection; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; + +/// One restored KV table's rows: `(key, value, expire_at_ms)`, the shape the +/// KV snapshot captures. `expire_at_ms` is `0` for a row with no TTL. +type KvRows = Vec<(Vec, Vec, u64)>; + +/// Split a KV snapshot's stored collection name into its database and bare +/// collection. A collection outside the default database is stored as +/// `{database_id}/{collection}`. +fn split_stored_collection(stored: &str) -> (DatabaseId, &str) { + match stored.split_once('/') { + Some((prefix, name)) => match prefix.parse::() { + Ok(database_id) => (DatabaseId::new(database_id), name), + Err(_) => (DatabaseId::DEFAULT, stored), + }, + None => (DatabaseId::DEFAULT, stored), + } +} + +/// The TTL a restored row keeps at `now_ms`: `Some(0)` for no TTL, the time +/// left for a row that has not expired, `None` for a row already expired. +fn remaining_ttl_ms(expire_at_ms: u64, now_ms: u64) -> Option { + match expire_at_ms { + 0 => Some(0), + at if at > now_ms => Some(at - now_ms), + _ => None, + } +} + +/// Decode and durably re-issue every restored KV table. Returns the number of +/// rows re-issued. +pub(in crate::control::backup::restore) async fn reissue_kv_tables( + state: &SharedState, + tenant_id: u64, + tables: Vec<(String, Vec)>, +) -> Result { + let tenant = TenantId::new(tenant_id); + let mut reissued = 0usize; + for (stored, bytes) in tables { + let rows: KvRows = zerompk::from_msgpack(&bytes).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("restore reissue: deserialize KV table '{stored}': {e}"), + })?; + let (database_id, collection) = split_stored_collection(&stored); + super::durable::log_reissue_step( + state, + "kv", + collection, + crate::types::VShardId::from_collection_in_database(database_id, collection), + rows.len(), + ); + let now_ms = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_millis() as u64) + .unwrap_or(0); + for (key, value, expire_at_ms) in rows { + let Some(ttl_ms) = remaining_ttl_ms(expire_at_ms, now_ms) else { + continue; + }; + let surrogate = state + .surrogate_assigner + .assign(database_id, tenant, &stored, &key)?; + let plan = PhysicalPlan::Kv(KvOp::Put { + collection: QualifiedCollection::from_stored(stored.clone()), + key, + value, + ttl_ms, + surrogate, + returning: None, + rls_filters: Vec::new(), + }); + super::durable::reissue_plan_durably(state, tenant, database_id, collection, plan) + .await?; + reissued += 1; + } + } + Ok(reissued) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_stored_name_splits_into_its_database_and_collection() { + assert_eq!( + split_stored_collection("orders"), + (DatabaseId::DEFAULT, "orders") + ); + assert_eq!( + split_stored_collection("7/orders"), + (DatabaseId::new(7), "orders") + ); + } + + #[test] + fn an_expired_row_is_not_restored() { + assert_eq!(remaining_ttl_ms(0, 500), Some(0)); + assert_eq!(remaining_ttl_ms(900, 500), Some(400)); + assert_eq!(remaining_ttl_ms(400, 500), None); + } +} diff --git a/nodedb/src/control/backup/restore/mod.rs b/nodedb/src/control/backup/restore/mod.rs index 103f79bbf..24468244c 100644 --- a/nodedb/src/control/backup/restore/mod.rs +++ b/nodedb/src/control/backup/restore/mod.rs @@ -3,17 +3,19 @@ //! RESTORE TENANT — module root. //! //! Submodule wiring only. All restore orchestrator logic lives in -//! [`orchestrate`]; column/timeseries/vector re-issue helpers in their -//! respective submodules; supporting primitives in `remote`, `sections`, and -//! `topology`. +//! [`orchestrate`]; each engine's re-issue lives in its own submodule; the +//! section decoding lives in `sections`. pub mod columnar_reissue; pub(crate) mod crdt_reissue; +mod durable; +pub(crate) mod guard; +mod kv_reissue; mod orchestrate; -mod remote; +mod quorum; +mod redo_reissue; mod sections; pub mod timeseries_reissue; -mod topology; pub mod vector_reissue; pub use orchestrate::{RestoreStats, restore_tenant}; diff --git a/nodedb/src/control/backup/restore/orchestrate/mod.rs b/nodedb/src/control/backup/restore/orchestrate/mod.rs index c0c717022..362eee893 100644 --- a/nodedb/src/control/backup/restore/orchestrate/mod.rs +++ b/nodedb/src/control/backup/restore/orchestrate/mod.rs @@ -3,9 +3,8 @@ //! RESTORE TENANT orchestrator logic. //! //! Validates a backup envelope, merges all sections into a single -//! `TenantDataSnapshot`, then splits the merged snapshot into per-node -//! sub-snapshots according to the *current* cluster topology and -//! dispatches `MetaOp::RestoreTenantSnapshot` to each owning node. +//! `TenantDataSnapshot`, then re-issues every section as durable, replicated +//! writes. //! //! Durable re-issue of columnar/timeseries/vector rows lives in [`reissue`]; //! post-install surrogate rebinding and tombstone warnings live in diff --git a/nodedb/src/control/backup/restore/orchestrate/rebind.rs b/nodedb/src/control/backup/restore/orchestrate/rebind.rs index d8027fa9e..9444749e4 100644 --- a/nodedb/src/control/backup/restore/orchestrate/rebind.rs +++ b/nodedb/src/control/backup/restore/orchestrate/rebind.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Post-install surrogate rebinding and tombstoned-collection warnings for +//! Surrogate rebinding and tombstoned-collection warnings for //! [`super::restore_tenant`]. use std::sync::Arc; @@ -8,26 +8,26 @@ use std::sync::Arc; use nodedb_types::Surrogate; use crate::Error; +use crate::control::backup::snapshot_keys::extract_db_tenant_scoped_collection; use crate::control::state::SharedState; +use crate::engine::graph::edge_store::parse_versioned_edge_key; use crate::types::{DatabaseId, SurrogateBindEntry, TenantDataSnapshot, TenantId}; -/// Rebind every PK→surrogate identity carried in the backup into the -/// destination catalog so restored rows resolve by PK point-lookup. +/// Bind every PK→surrogate identity the backup carries on this node, before +/// any re-issue, so a re-issued row keeps the surrogate it was stored under. /// -/// No-op when the snapshot carried no bindings (e.g. an older backup created -/// before the surrogate-pk section existed) or when the node has no catalog. -/// Any catalog write failure is FATAL. +/// Binding is first-wins: a key this node already binds keeps its surrogate, +/// and the re-issue writes that row over it. Each bind also raises this +/// node's surrogate high-water mark past the backup's surrogate, so no later +/// allocation here reuses it. Every replica binds the identities a re-issued +/// write carries as it applies the write. Any bind error is fatal. pub(super) fn rebind_surrogates( state: &Arc, - binds: Vec, + binds: &[SurrogateBindEntry], ) -> Result<(), Error> { - if binds.is_empty() { - return Ok(()); - } - let catalog = state.credentials.catalog(); let database_id = crate::types::DatabaseId::DEFAULT; - for e in &binds { - catalog.put_surrogate( + for e in binds { + state.surrogate_assigner.bind( database_id, TenantId::new(e.tenant_id), &e.collection, @@ -52,24 +52,7 @@ pub(super) fn warn_on_tombstoned_restores( return; } - let mut names = std::collections::BTreeSet::new(); - let sections: [&[(String, Vec)]; 6] = [ - &merged.documents, - &merged.indexes, - &merged.vectors, - &merged.kv_tables, - &merged.timeseries, - &merged.edges, - ]; - for section in sections { - for (key, _) in section { - if let Some(name) = collection_from_key(key) { - names.insert(name.to_string()); - } - } - } - - for name in &names { + for name in &restored_collection_names(tenant_id, merged) { let Some(purge_lsn) = tombstones.purge_lsn(DatabaseId::DEFAULT.as_u64(), tenant_id, name) else { continue; @@ -96,32 +79,72 @@ pub(super) fn warn_on_tombstoned_restores( } } -fn collection_from_key(key: &str) -> Option<&str> { - let tail = key.split_once(':')?.1; - tail.split([':', '\0']).next() +/// Every collection the backup restores rows into, read from each section's +/// key in that section's own format. +fn restored_collection_names( + tenant_id: u64, + merged: &TenantDataSnapshot, +) -> std::collections::BTreeSet { + let mut names = std::collections::BTreeSet::new(); + let db_tenant_scoped: [&[(String, Vec)]; 5] = [ + &merged.documents, + &merged.documents_versioned, + &merged.indexes, + &merged.vectors, + &merged.timeseries, + ]; + for section in db_tenant_scoped { + for (key, _) in section { + if let Some(name) = extract_db_tenant_scoped_collection(key, tenant_id) { + names.insert(name.to_string()); + } + } + } + // A KV table's key is its collection name. + for (name, _) in &merged.kv_tables { + names.insert(name.clone()); + } + for (key, _) in &merged.edges { + if let Some((name, ..)) = parse_versioned_edge_key(key) { + names.insert(name.to_string()); + } + } + names } #[cfg(test)] -mod collection_key_tests { - use super::collection_from_key; +mod collection_name_tests { + use super::*; #[test] - fn extracts_collection_with_colon_separator() { - assert_eq!(collection_from_key("1:users:doc-1"), Some("users")); - } - - #[test] - fn extracts_collection_with_null_separator() { - assert_eq!(collection_from_key("1:src\0label\0"), Some("src")); - } - - #[test] - fn vector_and_kv_key_shapes() { - assert_eq!(collection_from_key("1:events"), Some("events")); + fn every_section_names_its_collection() { + let snap = TenantDataSnapshot { + documents: vec![("0:7:users:0000002a".into(), vec![])], + documents_versioned: vec![( + "0:7:ledger:0000002a\x0000000000000000000001".into(), + vec![], + )], + vectors: vec![("0:7:embeddings".into(), vec![])], + kv_tables: vec![("sessions".into(), vec![])], + edges: vec![( + "follows\x00a\x00L\x00b\x0000000000000000000001".into(), + vec![], + )], + ..Default::default() + }; + let names: Vec = restored_collection_names(7, &snap).into_iter().collect(); + assert_eq!( + names, + vec!["embeddings", "follows", "ledger", "sessions", "users"] + ); } #[test] - fn no_tenant_prefix_returns_none() { - assert_eq!(collection_from_key("no_colon"), None); + fn another_tenants_key_names_nothing() { + let snap = TenantDataSnapshot { + documents: vec![("0:8:users:0000002a".into(), vec![])], + ..Default::default() + }; + assert!(restored_collection_names(7, &snap).is_empty()); } } diff --git a/nodedb/src/control/backup/restore/orchestrate/reissue.rs b/nodedb/src/control/backup/restore/orchestrate/reissue.rs index 013647bbb..2ef107821 100644 --- a/nodedb/src/control/backup/restore/orchestrate/reissue.rs +++ b/nodedb/src/control/backup/restore/orchestrate/reissue.rs @@ -1,9 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 //! Durable re-issue of columnar, timeseries, and vector rows drained from the -//! snapshot before the topology split (see [`super::restore_tenant`]'s -//! doc comments on why these engines bypass the per-node snapshot -//! install path). +//! merged backup snapshot (see [`super::restore_tenant`]). use std::sync::Arc; @@ -80,7 +78,7 @@ pub(super) async fn reissue_timeseries_snapshots( let plan = super::super::timeseries_reissue::build_timeseries_ingest_plan(&collection, rows)?; - super::super::timeseries_reissue::reissue_timeseries_durably( + super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), database_id, @@ -138,7 +136,7 @@ pub(super) async fn reissue_columnar_snapshots( let plan = super::super::columnar_reissue::build_columnar_insert_plan(&collection, decoded)?; - super::super::columnar_reissue::reissue_columnar_durably( + super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), database_id, @@ -203,7 +201,7 @@ pub(super) async fn reissue_vector_snapshots( vector, surrogate, ); - super::super::vector_reissue::reissue_vector_durably( + super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), database_id, @@ -308,7 +306,7 @@ pub(super) async fn reissue_vector_params( &field_name, &config, ); - super::super::vector_reissue::reissue_vector_durably( + super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), database_id, diff --git a/nodedb/src/control/backup/restore/orchestrate/restore.rs b/nodedb/src/control/backup/restore/orchestrate/restore.rs index 0391e3f87..cb10686e7 100644 --- a/nodedb/src/control/backup/restore/orchestrate/restore.rs +++ b/nodedb/src/control/backup/restore/orchestrate/restore.rs @@ -1,8 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 //! `restore_tenant`: validates a backup envelope, merges all sections into -//! a single `TenantDataSnapshot`, splits it by current cluster topology, and -//! dispatches `MetaOp::RestoreTenantSnapshot` to each owning node. +//! a single `TenantDataSnapshot`, and re-issues every section as durable, +//! replicated writes. use std::sync::Arc; @@ -11,16 +11,10 @@ use nodedb_types::backup_envelope::{ }; use crate::Error; -use crate::bridge::envelope::PhysicalPlan; use crate::control::server::shared::ddl::neutral::collection::dispatch_register_from_stored; -use crate::control::server::shared::ddl::sync_dispatch; use crate::control::state::SharedState; -use crate::types::TenantId; -use nodedb_physical::physical_plan::MetaOp; -use super::super::remote::{NODE_RESTORE_TIMEOUT, dispatch_remote}; use super::super::sections::{apply_metadata_sections, merge_sections}; -use super::super::topology::{SplitOutput, is_self, split_by_current_topology}; use super::rebind; use super::reissue; use super::stats::RestoreStats; @@ -51,14 +45,27 @@ pub async fn restore_tenant( .into()); } - if !dry_run && env.meta.snapshot_watermark != 0 { - let current_high_water = state.tenant_write_hlc(tenant_id); + // Every group the restore reads or writes has a reachable majority, or + // the restore fails here, before it proposes anything. + if !dry_run { + super::super::quorum::require_quorum(state)?; + } + + let newest = if !dry_run && env.meta.snapshot_watermark != 0 { + super::super::guard::newest_committed_write(state, tenant_id).await? + } else { + None + }; + if let Some(mark) = newest { + let current_high_water = mark.hlc; if env.meta.snapshot_watermark < current_high_water { if force { tracing::warn!( tenant_id, envelope_watermark = env.meta.snapshot_watermark, current_high_water, + newest_write_site = mark.site.as_str(), + newest_write_collection = mark.collection.as_deref().unwrap_or(""), "restore staleness protection explicitly overridden via FORCE: \ envelope watermark is older than the destination cluster's last \ observed write-HLC for this tenant — newer writes will be overwritten" @@ -68,8 +75,13 @@ pub async fn restore_tenant( detail: format!( "restore refused: envelope watermark {} is older than the \ destination cluster's last observed write-HLC {} for tenant \ - {} — newer writes would be silently overwritten", - env.meta.snapshot_watermark, current_high_water, tenant_id + {} (newest write: {} on collection '{}') — newer writes would \ + be silently overwritten", + env.meta.snapshot_watermark, + current_high_water, + tenant_id, + mark.site, + mark.collection.as_deref().unwrap_or(""), ), }); } @@ -109,8 +121,8 @@ pub async fn restore_tenant( } let mut merged = merge_sections(&env.sections)?; - stats.documents = merged.documents.len(); - stats.indexes = merged.indexes.len(); + stats.documents = merged.documents.len() + merged.documents_versioned.len(); + stats.indexes = merged.indexes.len() + merged.indexes_versioned.len(); stats.edges = merged.edges.len(); stats.vectors = merged.vectors.len(); stats.kv_tables = merged.kv_tables.len(); @@ -127,136 +139,50 @@ pub async fn restore_tenant( return Ok(stats); } - // Plain-columnar engine state is NOT installed via the snapshot path (that - // lands in in-memory-only Data Plane maps — lost on restart, never - // replicated). Drain it here and re-issue durably below as - // `ColumnarOp::Insert`s. The topology split must therefore never see - // columnar engines. + // Every section re-issues as durable writes: Raft-replicated to every + // replica of its group in cluster mode, WAL-appended then installed on a + // single node. None is installed straight into a Data-Plane map, which + // would hold it on one node only and lose it on restart. let columnar_snapshots = std::mem::take(&mut merged.columnar_engines); - - // Timeseries engine state (memtable section + flushed on-disk segments) is - // likewise NOT installed via the snapshot path — `restore_timeseries` and - // `restore_flushed_ts_segments` do a per-node DIRECT install that is never - // Raft-replicated, so on a multi-replica cluster the data lands on only one - // node. Drain both sections here and re-issue durably below as - // `TimeseriesOp::Ingest`s (Raft-replicated in cluster mode; WAL-appended - // then installed in single-node mode). The topology split must therefore - // never see timeseries data — otherwise it would be double-installed. let timeseries_memtables = std::mem::take(&mut merged.timeseries); let flushed_ts_segments = std::mem::take(&mut merged.flushed_ts_segments); - - // CRDT state is NOT installed via the per-node snapshot fan-out: that - // dispatch is race-prone (skips data groups with no leader yet) and not - // durable across restart. Drain the per-collection CRDT section here and - // re-issue durably below as `CrdtOp::ImportSnapshot` (Raft-replicated in - // cluster mode; WAL-appended then installed in single-node mode). The - // topology split must therefore never see CRDT state — otherwise the - // coordinator would double-import. let crdt_state = std::mem::take(&mut merged.crdt_state); - - // Vector engine state is likewise NOT installed via the snapshot path — - // `restore_vector_collection` installs straight into the in-memory-only - // `vector_collections` Data Plane map with no WAL record and no Raft - // entry, so it is lost on restart (single-node) and never replicated - // (cluster). Drain it here and re-issue durably below, one - // `VectorOp::Insert` per restored vector (Raft-replicated in cluster - // mode; WAL-appended then installed in single-node mode). The topology - // split must therefore never see vector data — otherwise it would be - // double-installed. + let kv_tables = std::mem::take(&mut merged.kv_tables); let vector_snapshots = std::mem::take(&mut merged.vectors); - - // Vector-index HNSW/PQ/IVF configuration (metric, M, ef_construction, - // quantization/index_type) is captured at backup alongside the raw - // vectors above (see `TenantDataSnapshot::vector_params` / - // `::index_configs` doc comments) but is likewise NOT installed via the - // snapshot path. Drain both here and re-issue durably below as - // `VectorOp::SetParams` — BEFORE the vector `Insert` re-issue, since - // `get_or_create_vector_index` lazily creates the Data Plane HNSW index - // from `self.vector_params` on the first `Insert` it sees for a - // (collection, field), defaulting silently if no `SetParams` landed - // first. The topology split must therefore never see these sections. + // Vector-index config re-issues as `VectorOp::SetParams` before the first + // vector `Insert`: the Data Plane creates a (collection, field) HNSW index + // on its first `Insert`, from whatever params it holds by then. let vector_params_snapshots = std::mem::take(&mut merged.vector_params); let index_config_snapshots = std::mem::take(&mut merged.index_configs); - // Drain the PK→surrogate identity map before the topology split (the split - // only routes per-key engine data). It is rebound into the destination - // catalog after the data install dispatches succeed — without it restored - // documents are unreachable by PK point-lookup (`WHERE id=`). + // The PK→surrogate identity map. It is bound on this node before any + // re-issue, so a re-issued row keeps the surrogate the backup stored it + // under unless this node already binds its key. let surrogate_binds = std::mem::take(&mut merged.surrogate_pk); - - let SplitOutput { - buckets, - malformed_keys, - route_fallbacks, - } = split_by_current_topology(state, tenant_id, merged); - stats.nodes_dispatched = buckets.len(); - stats.malformed_keys = malformed_keys; - stats.route_fallbacks = route_fallbacks; - if malformed_keys > 0 { - tracing::warn!( - tenant_id, - count = malformed_keys, - "restore: snapshot contained keys that did not parse — possible corruption" - ); - } - if route_fallbacks > 0 { - tracing::warn!( - tenant_id, - count = route_fallbacks, - "restore: routed some entries to local node because no current leader was visible" - ); - } - - let mut local_plan: Option = None; - let mut remote_futs = Vec::with_capacity(buckets.len()); - for (node_id, sub) in buckets { - let payload = zerompk::to_msgpack_vec(&sub).map_err(|e| Error::Internal { - detail: format!("restore: snapshot encode failed: {e}"), - })?; - let plan = PhysicalPlan::Meta(MetaOp::RestoreTenantSnapshot { - tenant_id, - snapshot: payload, - // User RESTORE keeps the fail-closed collision behavior. - replace_mode: false, - clear_vshards: Vec::new(), - collections_to_clear: Vec::new(), - }); - if is_self(state, node_id) { - local_plan = Some(plan); - } else { - let state = state.clone(); - remote_futs - .push(async move { dispatch_remote(&state, node_id, tenant_id, plan).await }); - } - } - if let Some(plan) = local_plan { - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - TenantId::new(tenant_id), - // TODO(A8-followup): backup/restore not yet multi-database. - crate::types::DatabaseId::DEFAULT, - "__system", - plan, - ), - NODE_RESTORE_TIMEOUT, - ) - .await?; - } - let results = futures::future::join_all(remote_futs).await; - if let Some(first_err) = results.into_iter().find_map(Result::err) { - return Err(first_err); - } - - // Rebind the PK→surrogate identity map into the destination catalog now - // that the data is installed. The catalog is the SOURCE OF TRUTH the - // planner consults for PK point-lookups (`surrogate_assigner.lookup(pk)`); - // a missing binding makes a restored row unreachable by PK even though it - // is present in the doc store. A rebind failure is FATAL — silently - // shipping unqueryable rows is the partial-success anti-pattern this - // codebase forbids. - rebind::rebind_surrogates(state, surrogate_binds)?; + rebind::rebind_surrogates(state, &surrogate_binds)?; + + // Document rows, their versions and graph edges re-issue as committed + // redo records through each collection's apply log: every replica binds + // the rows' identities, appends the record to its WAL, installs the rows + // and derives their secondary index entries. The backup's own index + // entries are therefore not installed. + let documents = std::mem::take(&mut merged.documents); + let documents_versioned = std::mem::take(&mut merged.documents_versioned); + let edges = std::mem::take(&mut merged.edges); + let redo = super::super::redo_reissue::reissue_rows_and_edges( + state, + tenant_id, + super::super::redo_reissue::RestoredRows { + documents, + documents_versioned, + edges, + binds: &surrogate_binds, + }, + ) + .await?; + stats.documents_reissued = redo.documents; + stats.edges_reissued = redo.edges; + stats.redo_records = redo.records; // Durable re-issue of plain-columnar rows. Each restored collection's live // rows are decoded from the snapshot and replayed as a durable @@ -289,6 +215,11 @@ pub async fn restore_tenant( stats.crdt_reissued = super::super::crdt_reissue::reissue_crdt_snapshots(state, crdt_state).await?; + // Durable re-issue of KV rows, one `KvOp::Put` per live row. Any failure + // is fatal — no warn-and-continue. + stats.kv_reissued = + super::super::kv_reissue::reissue_kv_tables(state, tenant_id, kv_tables).await?; + // Durable re-issue of vector-index configuration. Each restored // (collection, field) HNSW/PQ/IVF config is replayed as a // `VectorOp::SetParams` (Raft-replicated in cluster mode; WAL-appended diff --git a/nodedb/src/control/backup/restore/orchestrate/stats.rs b/nodedb/src/control/backup/restore/orchestrate/stats.rs index 5a72d210e..39050aa01 100644 --- a/nodedb/src/control/backup/restore/orchestrate/stats.rs +++ b/nodedb/src/control/backup/restore/orchestrate/stats.rs @@ -27,14 +27,18 @@ pub struct RestoreStats { pub crdt_reissued: usize, /// Number of individual vectors re-issued durably (Raft/WAL) on restore. pub vectors_reissued: usize, + /// Number of individual KV rows re-issued durably (Raft/WAL) on restore. + pub kv_reissued: usize, /// Number of (collection, field) vector-index HNSW/PQ/IVF configs /// re-issued durably (Raft/WAL) on restore. pub vector_params_reissued: usize, /// Number of PK→surrogate identity bindings rebound into the catalog. pub surrogate_pk: usize, - pub nodes_dispatched: usize, - /// Non-zero = snapshot contained unparseable keys (possible corruption). - pub malformed_keys: usize, - /// Non-zero = some entries were routed to local node due to missing shard leader. - pub route_fallbacks: usize, + /// Document sub-records re-issued: one per current row, one per version + /// of a `bitemporal=true` row. + pub documents_reissued: usize, + /// Edge versions re-issued. + pub edges_reissued: usize, + /// Redo records the document and edge re-issue committed. + pub redo_records: usize, } diff --git a/nodedb/src/control/backup/restore/quorum.rs b/nodedb/src/control/backup/restore/quorum.rs new file mode 100644 index 000000000..506759f21 --- /dev/null +++ b/nodedb/src/control/backup/restore/quorum.rs @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's quorum check: every Raft group the restore reads or writes has a +//! reachable majority before the restore proposes anything. +//! +//! A restore reads every data group's write marks, writes catalog rows +//! through the metadata group, places a cut marker in the sequencer group, +//! and re-issues rows through the data groups. A group with no reachable +//! majority commits nothing, so every step against it waits out its +//! deadline. Checked first, the restore fails at once with the group and the +//! nodes it cannot reach, and nothing of it is applied anywhere. +//! +//! The check reads this node's membership view: the routing table's voters +//! and the topology's active nodes. A majority lost after the check fails the +//! step that needs it at its deadline instead. + +use std::collections::BTreeSet; + +use crate::Error; +use crate::control::state::SharedState; + +/// Fail with [`Error::GroupQuorumUnavailable`] for the first Raft group whose +/// voters have no reachable majority. A node with no cluster routing has no +/// group to check. +pub(super) fn require_quorum(state: &SharedState) -> Result<(), Error> { + let (Some(routing), Some(topology)) = ( + state.cluster_routing.as_ref(), + state.cluster_topology.as_ref(), + ) else { + return Ok(()); + }; + let active: BTreeSet = topology + .read() + .unwrap_or_else(|p| p.into_inner()) + .active_nodes() + .iter() + .map(|node| node.node_id) + .collect(); + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let mut group_ids = routing.group_ids(); + group_ids.sort_unstable(); + for group_id in group_ids { + let Some(info) = routing.group_info(group_id) else { + continue; + }; + if let Some(error) = quorum_error(group_id, &info.members, &active) { + return Err(error); + } + } + Ok(()) +} + +/// The error for `group_id` when fewer than a majority of `voters` are in +/// `active`, else `None`. +fn quorum_error(group_id: u64, voters: &[u64], active: &BTreeSet) -> Option { + if voters.is_empty() { + return None; + } + let mut unreachable: Vec = voters + .iter() + .copied() + .filter(|voter| !active.contains(voter)) + .collect(); + unreachable.sort_unstable(); + let reachable = voters.len() - unreachable.len(); + if reachable * 2 > voters.len() { + return None; + } + let mut voters = voters.to_vec(); + voters.sort_unstable(); + Some(Error::GroupQuorumUnavailable { + group_id, + voters, + unreachable, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_group_with_a_reachable_majority_passes() { + let active = BTreeSet::from([1, 2]); + assert!(quorum_error(4, &[1, 2, 3], &active).is_none()); + } + + #[test] + fn a_group_without_a_reachable_majority_names_its_unreachable_voters() { + let active = BTreeSet::from([1]); + match quorum_error(4, &[3, 1, 2], &active) { + Some(Error::GroupQuorumUnavailable { + group_id, + voters, + unreachable, + }) => { + assert_eq!(group_id, 4); + assert_eq!(voters, vec![1, 2, 3]); + assert_eq!(unreachable, vec![2, 3]); + } + other => panic!("expected GroupQuorumUnavailable, got {other:?}"), + } + } + + #[test] + fn half_of_an_even_voter_set_is_not_a_majority() { + let active = BTreeSet::from([1, 2]); + assert!(quorum_error(4, &[1, 2, 3, 4], &active).is_some()); + } +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/commit.rs b/nodedb/src/control/backup/restore/redo_reissue/commit.rs new file mode 100644 index 000000000..0e7f9aab3 --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/commit.rs @@ -0,0 +1,206 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Commit restored units as redo records through their vShard's apply log. +//! +//! A restored row installs exactly as a committed transaction's row does. With +//! Raft the record is proposed to the collection's data group, and every +//! replica binds its identities, appends it to its own WAL and installs it. +//! With no Raft this node runs the same apply alone. Either way the call +//! returns once the record is durable and installed here. + +use std::collections::HashSet; + +use nodedb_physical::physical_plan::RedoOrigin; + +use crate::bridge::envelope::{ErrorCode, Status}; +use crate::control::state::SharedState; +use crate::control::surrogate::CarriedIdentity; +use crate::control::wal_replication::encode::transaction_redo_entry; +use crate::control::wal_replication::propose_replicated_entry; +use crate::control::wal_replication::transaction_redo::{ + RedoTarget, TransactionRedoPayload, apply_transaction_redo, +}; +use crate::event::EventSource; +use crate::types::{TenantId, VShardId}; +use crate::wal::{RedoRecord, RedoSubRecord}; + +use super::units::{CollectionUnits, RowUnit}; + +/// Most sub-records one restore record carries. +const MAX_OPS_PER_RECORD: usize = 512; + +/// Most encoded bytes one restore record carries. A single unit larger than +/// this still commits, alone in its own record. +const MAX_BYTES_PER_RECORD: usize = 4 * 1024 * 1024; + +/// Split `units` into record-sized batches, in order. A unit never splits. +fn batch_units(units: Vec) -> Vec> { + let mut batches = Vec::new(); + let mut current: Vec = Vec::new(); + let (mut ops, mut bytes) = (0usize, 0usize); + for unit in units { + let (unit_ops, unit_bytes) = (unit.ops.len(), unit.byte_len()); + if !current.is_empty() + && (ops + unit_ops > MAX_OPS_PER_RECORD || bytes + unit_bytes > MAX_BYTES_PER_RECORD) + { + batches.push(std::mem::take(&mut current)); + (ops, bytes) = (0, 0); + } + ops += unit_ops; + bytes += unit_bytes; + current.push(unit); + } + if !current.is_empty() { + batches.push(current); + } + batches +} + +/// One batch as the payload every replica applies. +fn batch_payload(collection: &str, batch: Vec) -> TransactionRedoPayload { + let mut ops: Vec = Vec::new(); + let mut identities: Vec = Vec::new(); + let mut seen: HashSet<(String, Vec)> = HashSet::new(); + for unit in batch { + ops.extend(unit.ops); + for identity in unit.identities { + if seen.insert((identity.collection.clone(), identity.pk_bytes.clone())) { + identities.push(identity); + } + } + } + TransactionRedoPayload { + redo: RedoRecord { + version: 1, + ops, + calvin_stamp: None, + }, + collections: vec![collection.to_string()], + // The backup holds every target row with its total already folded in. + sum_targets: Vec::new(), + identities, + event_source: EventSource::User, + origin: RedoOrigin::Restore, + } +} + +/// Commit one record and wait until it is durable and installed here. +async fn commit_record( + state: &SharedState, + target: RedoTarget, + payload: &TransactionRedoPayload, +) -> crate::Result<()> { + super::super::durable::log_reissue_step( + state, + "redo", + payload.collections.first().map_or("", String::as_str), + target.vshard_id, + payload.redo.ops.len(), + ); + if let Some(proposer) = state.async_raft_proposer() { + let entry = transaction_redo_entry( + target.tenant_id, + target.database_id, + target.vshard_id, + payload, + ); + propose_replicated_entry(state, proposer, entry).await?; + return Ok(()); + } + let outcome = apply_transaction_redo(state, target, payload, 0, None).await?; + if outcome.response.status == Status::Ok { + return Ok(()); + } + Err(crate::Error::DataPlane( + outcome + .response + .error_code + .as_deref() + .cloned() + .unwrap_or_else(|| ErrorCode::Internal { + detail: "restore redo apply returned an error status with no error code".into(), + }), + )) +} + +/// Commit every unit of `units` in order. Returns the records committed. +pub(super) async fn commit_collection( + state: &SharedState, + tenant_id: TenantId, + units: CollectionUnits, +) -> crate::Result { + let CollectionUnits { + database_id, + collection, + units, + } = units; + let target = RedoTarget { + tenant_id, + database_id, + vshard_id: VShardId::from_collection_in_database(database_id, &collection), + }; + let mut records = 0usize; + for batch in batch_units(units) { + let payload = batch_payload(&collection, batch); + commit_record(state, target, &payload) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("restore: re-issuing rows of '{collection}' failed: {e}"), + })?; + records += 1; + } + Ok(records) +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + + fn unit(ops: usize, payload_len: usize, pk: &str) -> RowUnit { + RowUnit { + ops: (0..ops) + .map(|_| RedoSubRecord { + record_type: 0, + payload: vec![0; payload_len], + }) + .collect(), + identities: vec![CarriedIdentity { + collection: "c".into(), + pk_bytes: pk.as_bytes().to_vec(), + surrogate: Surrogate::new(1), + }], + } + } + + #[test] + fn batches_cut_between_units_at_the_op_limit() { + let units = (0..3) + .map(|i| unit(MAX_OPS_PER_RECORD / 2, 1, &i.to_string())) + .collect(); + let batches = batch_units(units); + let sizes: Vec = batches.iter().map(Vec::len).collect(); + assert_eq!(sizes, vec![2, 1]); + } + + #[test] + fn an_oversized_unit_commits_alone() { + let units = vec![ + unit(1, 1, "a"), + unit(1, MAX_BYTES_PER_RECORD + 1, "b"), + unit(1, 1, "c"), + ]; + let sizes: Vec = batch_units(units).iter().map(Vec::len).collect(); + assert_eq!(sizes, vec![1, 1, 1]); + } + + #[test] + fn a_payload_carries_each_identity_once_and_restores_without_folds() { + let payload = batch_payload("c", vec![unit(1, 1, "a"), unit(1, 1, "a")]); + assert_eq!(payload.redo.ops.len(), 2); + assert_eq!(payload.identities.len(), 1); + assert!(payload.sum_targets.is_empty()); + assert_eq!(payload.origin, RedoOrigin::Restore); + } +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/documents.rs b/nodedb/src/control/backup/restore/redo_reissue/documents.rs new file mode 100644 index 000000000..17e05f7fe --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/documents.rs @@ -0,0 +1,369 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Restored document rows as redo units. +//! +//! A backup carries each row in its stored form, keyed by its storage key: +//! +//! * `documents` — `"{db}:{tid}:{collection}:{storage_key}"`, the current row +//! of a collection that keeps no history; +//! * `documents_versioned` — `"{db}:{tid}:{collection}:{storage_key}\x00{sys:020}"`, +//! every version of a `bitemporal=true` row. +//! +//! Each row becomes one unit: its sub-records in version order plus the +//! identity every replica binds before it installs them. A strict row's Binary +//! Tuple decodes back to MessagePack with the collection's schema, the same +//! conversion a transaction commit applies to a staged strict row. Every +//! replica re-derives the row's secondary index entries as it installs it. + +use std::collections::{BTreeMap, HashMap}; + +use nodedb_types::columnar::StrictSchema; +use nodedb_types::{CollectionType, DocumentMode, RowIdentity, StorageKey}; + +use crate::control::state::SharedState; +use crate::control::surrogate::CarriedIdentity; +use crate::data::executor::strict_format::{binary_tuple_to_msgpack, undecodable_strict_row}; +use crate::engine::sparse::btree_versioned::{TAG_LIVE, TAG_TOMBSTONE, decode_value}; +use crate::types::{DatabaseId, SurrogateBindEntry, TenantId}; + +use super::sub_record::{VersionStamp, document_put, document_tombstone}; +use super::units::{CollectionUnits, RowUnit}; + +/// A row as the backup stored it. +enum StoredRow { + /// The one current body of a row with no history. + Current(Vec), + /// Every `(sys_from_ms, versioned value)` of a bitemporal row. + Versions(Vec<(i64, Vec)>), +} + +/// What decoding a collection's rows reads from its catalog entry. +struct CollectionShape { + strict: Option, + declared_primary_key: Option, +} + +fn malformed(key: &str) -> crate::Error { + let prefix: String = key.chars().take(64).collect(); + crate::Error::Serialization { + format: "backup".into(), + detail: format!("restore: document key '{prefix}' is malformed"), + } +} + +/// Split `"{db}:{tid}:{collection}:{rest}"`, checking the tenant. +fn split_key(key: &str, tenant_id: u64) -> crate::Result<(u64, &str, &str)> { + let mut parts = key.splitn(4, ':'); + let (Some(db), Some(tid), Some(collection), Some(rest)) = + (parts.next(), parts.next(), parts.next(), parts.next()) + else { + return Err(malformed(key)); + }; + let db = db.parse::().map_err(|_| malformed(key))?; + if tid.parse::().ok() != Some(tenant_id) || collection.is_empty() { + return Err(malformed(key)); + } + Ok((db, collection, rest)) +} + +/// Group every restored row by `(database, collection)`, then by storage key. +fn group_rows( + tenant_id: u64, + documents: Vec<(String, Vec)>, + documents_versioned: Vec<(String, Vec)>, +) -> crate::Result>> { + let mut grouped: BTreeMap<(u64, String), BTreeMap> = BTreeMap::new(); + for (key, body) in documents { + let (db, collection, rest) = split_key(&key, tenant_id)?; + let storage_key = StorageKey::parse(rest).ok_or_else(|| malformed(&key))?; + grouped + .entry((db, collection.to_string())) + .or_default() + .insert(storage_key, StoredRow::Current(body)); + } + for (key, value) in documents_versioned { + let (db, collection, rest) = split_key(&key, tenant_id)?; + let (hex, sys) = rest.split_once('\x00').ok_or_else(|| malformed(&key))?; + let storage_key = StorageKey::parse(hex).ok_or_else(|| malformed(&key))?; + let sys_from_ms = sys.parse::().map_err(|_| malformed(&key))?; + let rows = grouped.entry((db, collection.to_string())).or_default(); + match rows + .entry(storage_key) + .or_insert_with(|| StoredRow::Versions(Vec::new())) + { + StoredRow::Versions(versions) => versions.push((sys_from_ms, value)), + StoredRow::Current(_) => { + return Err(crate::Error::Serialization { + format: "backup".into(), + detail: format!( + "restore: row {storage_key} of '{collection}' is both current-only \ + and versioned" + ), + }); + } + } + } + for rows in grouped.values_mut() { + for row in rows.values_mut() { + if let StoredRow::Versions(versions) = row { + versions.sort_by_key(|(sys, _)| *sys); + } + } + } + Ok(grouped) +} + +fn collection_shape( + state: &SharedState, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, +) -> crate::Result { + let stored = state + .credentials + .catalog() + .get_collection(database_id, tenant_id, collection)? + .ok_or_else(|| crate::Error::Internal { + detail: format!( + "restore: the backup holds rows of '{collection}' but restored no catalog \ + entry for it" + ), + })?; + // The storage mode the Data Plane registers for the collection, so a row + // decodes with the schema it was encoded with. + let strict = match stored.collection_type { + CollectionType::Document(DocumentMode::Strict(schema)) => Some(schema), + CollectionType::KeyValue(config) => Some(config.schema), + CollectionType::Document(DocumentMode::Schemaless) | CollectionType::Columnar(_) => None, + }; + Ok(CollectionShape { + strict, + declared_primary_key: stored.declared_primary_key, + }) +} + +/// A stored body as the MessagePack a put carries. +fn body_msgpack( + shape: &CollectionShape, + collection: &str, + key: StorageKey, + body: &[u8], +) -> crate::Result> { + match &shape.strict { + Some(schema) => binary_tuple_to_msgpack(body, schema) + .ok_or_else(|| undecodable_strict_row(collection, key.to_identity().as_str())), + None => Ok(body.to_vec()), + } +} + +/// Builds each row's unit for one collection. +struct RowBuilder<'a> { + state: &'a SharedState, + database_id: DatabaseId, + tenant: TenantId, + collection: &'a str, + shape: CollectionShape, + /// `storage surrogate → primary key` the backup bound for this collection. + binds: HashMap, +} + +impl RowBuilder<'_> { + /// The row's client identity: the backup's binding, else the identity + /// INSERT derives from the row body. + fn identity(&self, key: StorageKey, body: Option<&[u8]>) -> crate::Result { + if let Some(pk) = self.binds.get(&key.surrogate().as_u32()) { + let pk = std::str::from_utf8(pk).map_err(|_| crate::Error::Serialization { + format: "backup".into(), + detail: format!( + "restore: the backup binds row {key} of '{}' to a key that is not UTF-8", + self.collection + ), + })?; + return Ok(RowIdentity::from_user_key(pk)); + } + Ok(match body { + Some(body) => { + RowIdentity::of_stored_row(body, self.shape.declared_primary_key.as_deref(), key) + } + None => key.to_identity(), + }) + } + + /// Bind the row's identity on this node. The backup's surrogate wins + /// unless this node already binds the identity: the row then installs + /// under that surrogate, over the row it names. + fn bind(&self, identity: &RowIdentity, key: StorageKey) -> crate::Result { + let surrogate = self.state.surrogate_assigner.bind( + self.database_id, + self.tenant, + self.collection, + identity.as_str().as_bytes(), + key.surrogate(), + )?; + Ok(CarriedIdentity { + collection: self.collection.to_string(), + pk_bytes: identity.as_str().as_bytes().to_vec(), + surrogate, + }) + } + + fn current(&self, key: StorageKey, body: &[u8]) -> crate::Result { + let value = body_msgpack(&self.shape, self.collection, key, body)?; + let identity = self.identity(key, Some(&value))?; + let carried = self.bind(&identity, key)?; + let op = document_put( + self.collection, + identity.as_str(), + value, + carried.surrogate.as_u32(), + None, + )?; + Ok(RowUnit { + ops: vec![op], + identities: vec![carried], + }) + } + + fn versions(&self, key: StorageKey, versions: &[(i64, Vec)]) -> crate::Result { + // Decode every version first: the identity comes from a live body. + let mut decoded = Vec::with_capacity(versions.len()); + for (sys_from_ms, raw) in versions { + let version = decode_value(raw)?; + let body = match version.tag { + TAG_LIVE => Some(body_msgpack( + &self.shape, + self.collection, + key, + version.body, + )?), + TAG_TOMBSTONE => None, + tag => { + return Err(crate::Error::Serialization { + format: "versioned-doc".into(), + detail: format!( + "restore: version {sys_from_ms} of row {key} of '{}' carries tag \ + {tag:#04x}, which no write path records", + self.collection + ), + }); + } + }; + let stamp = VersionStamp { + sys_from_ms: *sys_from_ms, + valid_from_ms: version.valid_from_ms, + valid_until_ms: version.valid_until_ms, + }; + decoded.push((stamp, body)); + } + let first_live = decoded.iter().find_map(|(_, body)| body.as_deref()); + let identity = self.identity(key, first_live)?; + let carried = self.bind(&identity, key)?; + let surrogate = carried.surrogate.as_u32(); + let mut ops = Vec::with_capacity(decoded.len()); + for (stamp, body) in decoded { + ops.push(match body { + Some(value) => document_put( + self.collection, + identity.as_str(), + value, + surrogate, + Some(stamp), + )?, + None => document_tombstone( + self.collection, + identity.as_str(), + surrogate, + stamp.sys_from_ms, + )?, + }); + } + Ok(RowUnit { + ops, + identities: vec![carried], + }) + } +} + +/// Every restored row of `tenant_id`, one unit per row, grouped by +/// collection. `binds` is the backup's primary-key section. +pub(super) fn document_units( + state: &SharedState, + tenant_id: u64, + documents: Vec<(String, Vec)>, + documents_versioned: Vec<(String, Vec)>, + binds: &[SurrogateBindEntry], +) -> crate::Result> { + let grouped = group_rows(tenant_id, documents, documents_versioned)?; + let mut out = Vec::with_capacity(grouped.len()); + for ((db, collection), rows) in grouped { + let database_id = DatabaseId::new(db); + let builder = RowBuilder { + state, + database_id, + tenant: TenantId::new(tenant_id), + collection: &collection, + shape: collection_shape(state, database_id, tenant_id, &collection)?, + binds: binds + .iter() + .filter(|b| b.tenant_id == tenant_id && b.collection == collection) + .map(|b| (b.surrogate, b.pk.as_slice())) + .collect(), + }; + let mut units = Vec::with_capacity(rows.len()); + for (key, row) in &rows { + units.push(match row { + StoredRow::Current(body) => builder.current(*key, body)?, + StoredRow::Versions(versions) => builder.versions(*key, versions)?, + }); + } + out.push(CollectionUnits { + database_id, + collection: collection.clone(), + units, + }); + } + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn keys_split_into_collection_and_storage_key() { + let (db, collection, rest) = split_key("0:7:users:0000002a", 7).unwrap(); + assert_eq!((db, collection, rest), (0, "users", "0000002a")); + assert!(split_key("0:8:users:0000002a", 7).is_err()); + assert!(split_key("0:7:users", 7).is_err()); + } + + #[test] + fn versions_group_under_their_row_in_system_time_order() { + let grouped = group_rows( + 7, + vec![("0:7:plain:00000001".into(), vec![1])], + vec![ + ( + "0:7:ledger:00000002\x0000000000000000000200".into(), + vec![2], + ), + ( + "0:7:ledger:00000002\x0000000000000000000100".into(), + vec![1], + ), + ], + ) + .unwrap(); + let ledger = &grouped[&(0, "ledger".to_string())]; + let key = StorageKey::parse("00000002").unwrap(); + let StoredRow::Versions(versions) = &ledger[&key] else { + panic!("a versioned row groups as versions"); + }; + let order: Vec = versions.iter().map(|(sys, _)| *sys).collect(); + assert_eq!(order, vec![100, 200]); + assert!(matches!( + grouped[&(0, "plain".to_string())][&StorageKey::parse("00000001").unwrap()], + StoredRow::Current(_) + )); + } +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/edges.rs b/nodedb/src/control/backup/restore/redo_reissue/edges.rs new file mode 100644 index 000000000..14cf1bf27 --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/edges.rs @@ -0,0 +1,136 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Restored graph edges as redo units. +//! +//! A backup carries every edge version under its versioned key, +//! `"{collection}\x00{src}\x00{label}\x00{dst}\x00{system_from:020}"`. Each +//! version re-issues at its original `system_from`, so the restored edge keeps +//! its history and its valid-from time. A tombstone version re-issues as a +//! delete at its `system_from`. Every replica updates its CSR index and its +//! node identities as it installs each version. + +use std::collections::BTreeMap; + +use crate::control::state::SharedState; +use crate::control::surrogate::CarriedIdentity; +use crate::engine::graph::edge_store::{ + EdgeValuePayload, is_gdpr_erasure, is_tombstone, parse_versioned_edge_key, +}; +use crate::types::{DatabaseId, TenantId}; +use crate::wal::{EdgeDeleteRedo, EdgePutRedo}; + +use super::sub_record::{edge_delete, edge_put}; +use super::units::{CollectionUnits, RowUnit}; + +fn malformed(key: &str) -> crate::Error { + let prefix: String = key.chars().take(64).collect(); + crate::Error::Serialization { + format: "backup".into(), + detail: format!("restore: edge key '{prefix:?}' is malformed"), + } +} + +/// The identity of node `node_id` in edge collection `collection`, bound +/// through the surrogate assigner as a live edge write binds it. +fn node_identity( + state: &SharedState, + database_id: DatabaseId, + tenant: TenantId, + collection: &str, + node_id: &str, +) -> crate::Result { + let surrogate = + state + .surrogate_assigner + .assign(database_id, tenant, collection, node_id.as_bytes())?; + Ok(CarriedIdentity { + collection: collection.to_string(), + pk_bytes: node_id.as_bytes().to_vec(), + surrogate, + }) +} + +/// One edge version as a unit. +fn edge_unit( + state: &SharedState, + database_id: DatabaseId, + tenant: TenantId, + key: &str, + value: &[u8], +) -> crate::Result { + let (collection, src_id, label, dst_id, system_from) = + parse_versioned_edge_key(key).ok_or_else(|| malformed(key))?; + if is_tombstone(value) { + let op = edge_delete(&EdgeDeleteRedo { + collection: collection.to_string(), + src_id: src_id.to_string(), + label: label.to_string(), + dst_id: dst_id.to_string(), + system_from: Some(system_from), + })?; + return Ok(RowUnit { + ops: vec![op], + identities: Vec::new(), + }); + } + if is_gdpr_erasure(value) { + return Err(crate::Error::Serialization { + format: "backup".into(), + detail: format!( + "restore: edge version {system_from} in '{collection}' is an erasure marker, \ + which no write path records" + ), + }); + } + let payload = EdgeValuePayload::decode(value)?; + let src = node_identity(state, database_id, tenant, collection, src_id)?; + let dst = node_identity(state, database_id, tenant, collection, dst_id)?; + let op = edge_put(&EdgePutRedo { + collection: collection.to_string(), + src_id: src_id.to_string(), + label: label.to_string(), + dst_id: dst_id.to_string(), + properties: payload.properties, + src_surrogate: src.surrogate.as_u32(), + dst_surrogate: dst.surrogate.as_u32(), + system_from: Some(system_from), + })?; + Ok(RowUnit { + ops: vec![op], + identities: vec![src, dst], + }) +} + +/// Every restored edge version of `tenant_id`, one unit per version, grouped +/// by edge collection in key order: each edge's versions in system-time order. +pub(super) fn edge_units( + state: &SharedState, + tenant_id: u64, + edges: Vec<(String, Vec)>, +) -> crate::Result> { + // The edge section carries no database: a tenant backup reads the default + // database's edge store. + let database_id = DatabaseId::DEFAULT; + let tenant = TenantId::new(tenant_id); + let mut by_key: BTreeMap> = BTreeMap::new(); + for (key, value) in edges { + by_key.insert(key, value); + } + let mut grouped: BTreeMap> = BTreeMap::new(); + for (key, value) in &by_key { + let (collection, ..) = parse_versioned_edge_key(key).ok_or_else(|| malformed(key))?; + let unit = edge_unit(state, database_id, tenant, key, value)?; + grouped + .entry(collection.to_string()) + .or_default() + .push(unit); + } + Ok(grouped + .into_iter() + .map(|(collection, units)| CollectionUnits { + database_id, + collection, + units, + }) + .collect()) +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/mod.rs b/nodedb/src/control/backup/restore/redo_reissue/mod.rs new file mode 100644 index 000000000..1a4a1478a --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/mod.rs @@ -0,0 +1,13 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable, replicated RESTORE of document rows, their index entries, and +//! graph edges, as committed redo records. + +mod commit; +mod documents; +mod edges; +mod reissue; +mod sub_record; +mod units; + +pub(in crate::control::backup::restore) use reissue::{RestoredRows, reissue_rows_and_edges}; diff --git a/nodedb/src/control/backup/restore/redo_reissue/reissue.rs b/nodedb/src/control/backup/restore/redo_reissue/reissue.rs new file mode 100644 index 000000000..507b55be1 --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/reissue.rs @@ -0,0 +1,61 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Re-issue every restored document row and graph edge as committed redo. +//! +//! Rows go first, then edges, so an edge's endpoints are in place when it +//! installs. A backup's index entries are not re-issued: every replica +//! derives a row's secondary index entries as it installs the row, exactly as +//! for a committed transaction's row. + +use crate::control::state::SharedState; +use crate::types::{SurrogateBindEntry, TenantId}; + +use super::commit::commit_collection; +use super::documents::document_units; +use super::edges::edge_units; + +/// The backup sections this re-issue consumes. +pub(in crate::control::backup::restore) struct RestoredRows<'a> { + pub documents: Vec<(String, Vec)>, + pub documents_versioned: Vec<(String, Vec)>, + pub edges: Vec<(String, Vec)>, + /// The backup's primary-key section: each row's client identity. + pub binds: &'a [SurrogateBindEntry], +} + +/// What the re-issue committed. +#[derive(Debug, Default, Clone, Copy)] +pub(in crate::control::backup::restore) struct RedoReissueStats { + /// Document sub-records: one per current row, one per version. + pub documents: usize, + /// Edge sub-records: one per edge version. + pub edges: usize, + /// Redo records committed. + pub records: usize, +} + +/// Re-issue `rows` of `tenant_id` durably. The first error fails the restore. +pub(in crate::control::backup::restore) async fn reissue_rows_and_edges( + state: &SharedState, + tenant_id: u64, + rows: RestoredRows<'_>, +) -> crate::Result { + let tenant = TenantId::new(tenant_id); + let mut stats = RedoReissueStats::default(); + let documents = document_units( + state, + tenant_id, + rows.documents, + rows.documents_versioned, + rows.binds, + )?; + for collection in documents { + stats.documents += collection.units.iter().map(|u| u.ops.len()).sum::(); + stats.records += commit_collection(state, tenant, collection).await?; + } + for collection in edge_units(state, tenant_id, rows.edges)? { + stats.edges += collection.units.iter().map(|u| u.ops.len()).sum::(); + stats.records += commit_collection(state, tenant, collection).await?; + } + Ok(stats) +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs b/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs new file mode 100644 index 000000000..84bb6f763 --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs @@ -0,0 +1,148 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Redo sub-record encoders for restored document rows and graph edges. +//! +//! Each encoder writes the exact payload shape the transaction resolver emits +//! and the replay arms decode, so a restored row installs through the same +//! arm a committed transaction's row does: +//! +//! * document put: `(collection, document_id, value, prov, surrogate)`, plus +//! `(sys_from_ms, valid_from_ms, valid_until_ms)` for a version of a +//! `bitemporal=true` collection; +//! * document delete: `(collection, document_id, prov, surrogate, sys_from_ms)`, +//! a tombstone version of a `bitemporal=true` collection; +//! * edge put and delete: [`EdgePutRedo`] and [`EdgeDeleteRedo`]. + +use nodedb_types::sync::wire::SyncProvenance; +use nodedb_wal::record::RecordType; + +use crate::wal::{EdgeDeleteRedo, EdgePutRedo, RedoSubRecord}; + +/// The system and valid time a document version was stored at. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct VersionStamp { + pub sys_from_ms: i64, + pub valid_from_ms: i64, + pub valid_until_ms: i64, +} + +fn encode_error(what: &str, e: impl std::fmt::Display) -> crate::Error { + crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("restore redo: encode {what}: {e}"), + } +} + +/// A document put. `value` is the row as MessagePack. `stamp` is `Some` for a +/// version of a `bitemporal=true` collection, which installs at that exact +/// system time. +pub(super) fn document_put( + collection: &str, + document_id: &str, + value: Vec, + surrogate: u32, + stamp: Option, +) -> crate::Result { + let prov: Option = None; + let payload = match stamp { + Some(s) => zerompk::to_msgpack_vec(&( + collection, + document_id, + value, + prov, + surrogate, + s.sys_from_ms, + s.valid_from_ms, + s.valid_until_ms, + )), + None => zerompk::to_msgpack_vec(&(collection, document_id, value, prov, surrogate)), + } + .map_err(|e| encode_error("document put", e))?; + Ok(RedoSubRecord { + record_type: RecordType::Put as u32, + payload, + }) +} + +/// The tombstone version of a `bitemporal=true` document at `sys_from_ms`. +pub(super) fn document_tombstone( + collection: &str, + document_id: &str, + surrogate: u32, + sys_from_ms: i64, +) -> crate::Result { + let prov: Option = None; + let payload = zerompk::to_msgpack_vec(&(collection, document_id, prov, surrogate, sys_from_ms)) + .map_err(|e| encode_error("document tombstone", e))?; + Ok(RedoSubRecord { + record_type: RecordType::Delete as u32, + payload, + }) +} + +/// One edge version put at its original `system_from`. +pub(super) fn edge_put(put: &EdgePutRedo) -> crate::Result { + Ok(RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(put).map_err(|e| encode_error("edge put", e))?, + }) +} + +/// One edge tombstone at its original `system_from`. +pub(super) fn edge_delete(delete: &EdgeDeleteRedo) -> crate::Result { + Ok(RedoSubRecord { + record_type: RecordType::Delete as u32, + payload: zerompk::to_msgpack_vec(delete).map_err(|e| encode_error("edge delete", e))?, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + type BitemporalPut = ( + String, + String, + Vec, + Option, + u32, + i64, + i64, + i64, + ); + type PlainPut = (String, String, Vec, Option, u32); + type BitemporalDelete = (String, String, Option, u32, i64); + + #[test] + fn a_current_row_encodes_the_plain_put_shape() { + let sub = document_put("users", "u1", vec![0x80], 7, None).unwrap(); + assert_eq!(sub.record_type, RecordType::Put as u32); + let (collection, id, value, prov, surrogate): PlainPut = + zerompk::from_msgpack(&sub.payload).unwrap(); + assert_eq!( + (collection.as_str(), id.as_str(), value, surrogate), + ("users", "u1", vec![0x80], 7) + ); + assert!(prov.is_none()); + assert!(zerompk::from_msgpack::(&sub.payload).is_err()); + } + + #[test] + fn a_version_keeps_its_stamp() { + let stamp = VersionStamp { + sys_from_ms: 1_000, + valid_from_ms: 10, + valid_until_ms: 20, + }; + let sub = document_put("ledger", "e1", vec![0x80], 9, Some(stamp)).unwrap(); + let (_, _, _, _, surrogate, sys, vf, vu): BitemporalPut = + zerompk::from_msgpack(&sub.payload).unwrap(); + assert_eq!((surrogate, sys, vf, vu), (9, 1_000, 10, 20)); + + let tomb = document_tombstone("ledger", "e1", 9, 2_000).unwrap(); + assert_eq!(tomb.record_type, RecordType::Delete as u32); + let (_, id, _, surrogate, sys): BitemporalDelete = + zerompk::from_msgpack(&tomb.payload).unwrap(); + assert_eq!((id.as_str(), surrogate, sys), ("e1", 9, 2_000)); + } +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/units.rs b/nodedb/src/control/backup/restore/redo_reissue/units.rs new file mode 100644 index 000000000..84510118a --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/units.rs @@ -0,0 +1,34 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The unit a restored row or edge re-issues as. + +use crate::control::surrogate::CarriedIdentity; +use crate::types::DatabaseId; +use crate::wal::RedoSubRecord; + +/// One restored row or edge: its sub-records in apply order, and the +/// identities every replica binds before it installs them. A unit never +/// splits across two redo records. +pub(super) struct RowUnit { + pub ops: Vec, + pub identities: Vec, +} + +impl RowUnit { + /// The unit's encoded size, for sizing the records it goes into. + pub(super) fn byte_len(&self) -> usize { + self.ops.iter().map(|op| op.payload.len()).sum::() + + self + .identities + .iter() + .map(|identity| identity.collection.len() + identity.pk_bytes.len()) + .sum::() + } +} + +/// Every unit of one collection. All of them write that collection's vShard. +pub(super) struct CollectionUnits { + pub database_id: DatabaseId, + pub collection: String, + pub units: Vec, +} diff --git a/nodedb/src/control/backup/restore/remote.rs b/nodedb/src/control/backup/restore/remote.rs deleted file mode 100644 index 911d73122..000000000 --- a/nodedb/src/control/backup/restore/remote.rs +++ /dev/null @@ -1,96 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Remote-node dispatch helpers for RESTORE TENANT. - -use std::sync::Arc; -use std::time::Duration; - -use nodedb_cluster::rpc_codec::{ExecuteRequest, ExecuteResponse, RaftRpc, TypedClusterError}; - -use crate::Error; -use crate::bridge::envelope::PhysicalPlan; -use crate::control::state::SharedState; -use crate::types::TraceId; -use nodedb_physical::physical_plan::wire as plan_wire; - -pub(super) const NODE_RESTORE_TIMEOUT: Duration = Duration::from_secs(120); - -pub(super) async fn dispatch_remote( - state: &Arc, - node_id: u64, - tenant_id: u64, - plan: PhysicalPlan, -) -> Result<(), Error> { - let transport = state - .cluster_transport - .as_ref() - .ok_or_else(|| Error::Internal { - detail: format!("restore: cluster_transport unavailable but node {node_id} is remote"), - })?; - let plan_bytes = plan_wire::encode(&plan).map_err(|e| Error::Internal { - detail: format!("restore: plan encode failed: {e}"), - })?; - let req = RaftRpc::ExecuteRequest(ExecuteRequest { - plan_bytes, - tenant_id, - database_id: nodedb_types::id::DatabaseId::DEFAULT.as_u64(), - deadline_remaining_ms: NODE_RESTORE_TIMEOUT.as_millis() as u64, - trace_id: TraceId::generate().0, - descriptor_versions: Vec::new(), - // Restore dispatch is not session-transaction-scoped. - txn_id: None, - }); - let resp = transport - .send_rpc(node_id, req) - .await - .map_err(|e| Error::Internal { - detail: format!("restore RPC to node {node_id} failed: {e}"), - })?; - match resp { - RaftRpc::ExecuteResponse(ExecuteResponse { success: true, .. }) => Ok(()), - RaftRpc::ExecuteResponse(ExecuteResponse { - error: Some(err), .. - }) => Err(map_typed_error(err, node_id)), - RaftRpc::ExecuteResponse(_) => Err(Error::Internal { - detail: format!("restore: empty error response from node {node_id}"), - }), - other => Err(Error::Internal { - detail: format!( - "restore: unexpected RPC response variant from node {node_id}: {other:?}" - ), - }), - } -} - -pub(super) fn map_typed_error(err: TypedClusterError, node_id: u64) -> Error { - match err { - TypedClusterError::Internal { message, .. } => Error::Internal { - detail: format!("restore node {node_id}: {message}"), - }, - TypedClusterError::DeadlineExceeded { elapsed_ms } => Error::Internal { - detail: format!("restore node {node_id}: deadline exceeded after {elapsed_ms}ms"), - }, - TypedClusterError::NotLeader { .. } => Error::Internal { - detail: format!("restore node {node_id}: routed to non-leader"), - }, - TypedClusterError::DescriptorMismatch { collection, .. } => Error::Internal { - detail: format!( - "restore node {node_id}: descriptor mismatch on collection {collection}" - ), - }, - // Keep the shard's verdict typed: a restore refused by the Data Plane - // must not read as a generic internal restore fault. - TypedClusterError::DataPlane { code } => Error::DataPlane(code.into()), - // A constraint verdict keeps its collection and kind, so the client - // reads the SQLSTATE the refusing shard meant. - TypedClusterError::RejectedConstraint { - collection, - constraint, - detail, - } => Error::RejectedConstraint { - collection, - constraint, - detail, - }, - } -} diff --git a/nodedb/src/control/backup/restore/sections.rs b/nodedb/src/control/backup/restore/sections.rs index 7614001b6..6a8621e7c 100644 --- a/nodedb/src/control/backup/restore/sections.rs +++ b/nodedb/src/control/backup/restore/sections.rs @@ -45,6 +45,8 @@ pub(super) fn merge_sections( })?; merged.documents.extend(snap.documents); merged.indexes.extend(snap.indexes); + merged.documents_versioned.extend(snap.documents_versioned); + merged.indexes_versioned.extend(snap.indexes_versioned); merged.edges.extend(snap.edges); merged.vectors.extend(snap.vectors); merged.vector_params.extend(snap.vector_params); @@ -103,23 +105,21 @@ pub(super) fn apply_metadata_sections( for section in &env.sections { match section.origin_node_id { SECTION_ORIGIN_CATALOG_ROWS => { - let Ok(blobs) = zerompk::from_msgpack::>(§ion.body) - else { - tracing::warn!( - tenant_id, - "restore: catalog-rows section failed to decode — skipping" - ); - continue; - }; + let blobs = zerompk::from_msgpack::>(§ion.body) + .map_err(|_| Error::Internal { + detail: "invalid backup format: catalog-rows section is not decodable" + .into(), + })?; for blob in blobs { - let Ok(coll) = zerompk::from_msgpack::(&blob.bytes) else { - tracing::warn!( - tenant_id, - name = %blob.name, - "restore: catalog row failed to decode — skipping" - ); - continue; - }; + let coll = + zerompk::from_msgpack::(&blob.bytes).map_err(|_| { + Error::Internal { + detail: format!( + "invalid backup format: catalog row of '{}' is not decodable", + blob.name + ), + } + })?; // Propose the collection through the metadata Raft // group so every node's applier (`catalog_entry:: // apply::collection::put`) writes the row — mirroring @@ -144,14 +144,12 @@ pub(super) fn apply_metadata_sections( } } SECTION_ORIGIN_SOURCE_TOMBSTONES => { - let Ok(tombs) = zerompk::from_msgpack::>(§ion.body) - else { - tracing::warn!( - tenant_id, - "restore: source-tombstones section failed to decode — skipping" - ); - continue; - }; + let tombs = zerompk::from_msgpack::>(§ion.body) + .map_err(|_| Error::Internal { + detail: "invalid backup format: source-tombstones section is not \ + decodable" + .into(), + })?; for t in tombs { // Replicate via the metadata Raft group so every node's boot WAL // replay barrier matches — a coordinator-local tombstone lets purged diff --git a/nodedb/src/control/backup/restore/timeseries_reissue.rs b/nodedb/src/control/backup/restore/timeseries_reissue.rs index bd4457e3b..2950e23e1 100644 --- a/nodedb/src/control/backup/restore/timeseries_reissue.rs +++ b/nodedb/src/control/backup/restore/timeseries_reissue.rs @@ -10,7 +10,6 @@ //! identity re-derived from tag columns. use std::collections::HashMap; -use std::time::Duration; use nodedb_types::RlsWriteCheck; use nodedb_types::columnar::schema::TS_SYSTEM; @@ -19,20 +18,13 @@ use nodedb_types::value::Value; use crate::Error; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; -use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::state::SharedState; use crate::engine::timeseries::columnar_memtable::{ ColumnData, ColumnType, ColumnarMemtable, ColumnarMemtableConfig, MemtableSnapshot, }; use crate::engine::timeseries::columnar_segment::ColumnarSegmentReader; -use crate::types::{DatabaseId, TenantId, TsFlushedCollectionBlob, VShardId}; +use crate::types::TsFlushedCollectionBlob; use nodedb_physical::physical_plan::TimeseriesOp; -/// Per-collection re-issue dispatch timeout. Generous: a restored collection may -/// carry many flushed partitions' worth of rows in one ingest. -const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); - /// Server-stamped reserved column — re-derived by the ingest path, so it must /// NOT be carried back into the re-issued rows (the ingest handler restamps it). /// `_ts_valid_from` / `_ts_valid_until` ARE client-provided and preserved. @@ -312,68 +304,6 @@ pub fn build_timeseries_ingest_plan( })) } -/// Re-issue a restored timeseries collection's rows durably. -/// -/// Branches identically to a normal write (and to `reissue_columnar_durably`): -/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: append the redo under an outcome-floor window, then -/// `sync_dispatch::dispatch_system`, which closes the window. -pub async fn reissue_timeseries_durably( - state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - collection: &str, - plan: PhysicalPlan, -) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); - - if let Some(proposer) = state.async_raft_proposer() { - let entry = crate::control::wal_replication::to_replicated_entry( - tenant_id, - database_id, - vshard, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or_else(|| Error::Internal { - detail: format!( - "restore reissue: timeseries plan for '{collection}' did not map to a \ - replicated write" - ), - })?; - crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; - return Ok(()); - } - - // Single-node: WAL first (durable for restart replay), then install live. - // The record's outcome-floor window opens before the append and closes - // from the install's outcome. - let owner = RecordOwner { - tenant_id, - database_id, - vshard_id: vshard, - }; - let minted = MintedRecords::open(&state.outcome_floor); - if let Err(error) = minted.append_plan(&state.wal, owner, &plan) { - // Any record appended before the error never reaches a core. - minted.cancel(&state.wal, owner, 0).await?; - return Err(error); - } - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - tenant_id, - database_id, - collection, - plan, - ) - .with_minted(minted), - REISSUE_TIMEOUT, - ) - .await?; - Ok(()) -} - #[cfg(test)] mod tests { use super::*; diff --git a/nodedb/src/control/backup/restore/topology.rs b/nodedb/src/control/backup/restore/topology.rs deleted file mode 100644 index ff6215f63..000000000 --- a/nodedb/src/control/backup/restore/topology.rs +++ /dev/null @@ -1,188 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Topology-aware snapshot bucketing for RESTORE TENANT. - -use std::collections::BTreeMap; - -use nodedb_cluster::routing::{VSHARD_COUNT, vshard_for_collection}; -use nodedb_types::id::DatabaseId; - -use crate::control::backup::snapshot_keys::{ - extract_db_scoped_collection, extract_db_tenant_scoped_collection, -}; -use crate::control::state::SharedState; -use crate::types::TenantDataSnapshot; - -/// Bucketed output from `split_by_current_topology`. -pub(super) struct SplitOutput { - pub buckets: BTreeMap, - pub malformed_keys: usize, - pub route_fallbacks: usize, -} - -enum RouteOutcome { - Routed(u64), - Malformed, - NoLeader, -} - -/// Bucket the merged snapshot per current vshard ownership. -/// -/// Replicated-by-design data (graph edges, CRDT state) goes to every -/// owning node. Single-node mode is the degenerate case: everything to self. -pub(super) fn split_by_current_topology( - state: &SharedState, - tenant_id: u64, - merged: TenantDataSnapshot, -) -> SplitOutput { - let routing = state - .cluster_routing - .as_ref() - .map(|r| r.read().unwrap_or_else(|poisoned| poisoned.into_inner())); - let single_node = routing.is_none() || state.cluster_transport.is_none(); - - if single_node { - let mut out = BTreeMap::new(); - out.insert(state.node_id, merged); - return SplitOutput { - buckets: out, - malformed_keys: 0, - route_fallbacks: 0, - }; - } - let routing = - routing.expect("invariant: single_node is false, so routing.is_some() is guaranteed"); - - let mut all_owners = BTreeMap::::new(); - for vshard in 0..VSHARD_COUNT { - if let Ok(node) = routing.leader_for_vshard(vshard) - && node != 0 - { - all_owners.entry(node).or_default(); - } - } - if all_owners.is_empty() { - all_owners.insert(state.node_id, TenantDataSnapshot::default()); - } - - // Restore today operates on `DatabaseId::DEFAULT`; the snapshot/topology - // wire format gains a database_id alongside tenant_id when multi-database - // restore lands, at which point this binding moves up to a parameter. - let database_id = DatabaseId::DEFAULT; - let route_collection = |coll: &str| -> RouteOutcome { - let v = vshard_for_collection(database_id, coll); - match routing.leader_for_vshard(v) { - Ok(leader) if leader != 0 => RouteOutcome::Routed(leader), - _ => RouteOutcome::NoLeader, - } - }; - // Documents / indexes / vectors / timeseries keys are db-tenant-scoped: - // `"{db}:{tid}:{collection}[:suffix]"` (collection never contains ':'/'\0'). - let route_key = |key: &str| -> RouteOutcome { - match extract_db_tenant_scoped_collection(key, tenant_id) { - Some(coll) => route_collection(coll), - None => RouteOutcome::Malformed, - } - }; - // Columnar / flushed-ts keys are db-scoped: `"{db}:{tid}:{collection}"` - // where the collection is the whole remainder and may itself contain ':'. - let route_db_scoped_key = |key: &str| -> RouteOutcome { - match extract_db_scoped_collection(key, tenant_id) { - Some(coll) => route_collection(coll), - None => RouteOutcome::Malformed, - } - }; - - let mut malformed = 0usize; - let mut fallbacks = 0usize; - let mut resolve = |outcome: RouteOutcome, key: Option<&str>| -> u64 { - match outcome { - RouteOutcome::Routed(node) => node, - RouteOutcome::Malformed => { - malformed += 1; - if let Some(k) = key { - let prefix: String = k.chars().take(64).collect(); - tracing::warn!(tenant_id, key_prefix = %prefix, "restore: malformed key"); - } - state.node_id - } - RouteOutcome::NoLeader => { - fallbacks += 1; - state.node_id - } - } - }; - - for entry in merged.documents { - let node = resolve(route_key(&entry.0), Some(&entry.0)); - all_owners.entry(node).or_default().documents.push(entry); - } - for entry in merged.indexes { - let node = resolve(route_key(&entry.0), Some(&entry.0)); - all_owners.entry(node).or_default().indexes.push(entry); - } - for entry in merged.kv_tables { - let node = resolve(route_collection(&entry.0), Some(&entry.0)); - all_owners.entry(node).or_default().kv_tables.push(entry); - } - for entry in merged.timeseries { - let node = resolve(route_key(&entry.0), Some(&entry.0)); - all_owners.entry(node).or_default().timeseries.push(entry); - } - // Plain-columnar engine state is NOT installed via the snapshot path: the - // snapshot-install lands data in in-memory-only Data Plane maps with no WAL - // record and no Raft entry, so it is lost on restart (single-node) and never - // reaches replicas (cluster). RESTORE re-issues columnar rows as durable - // `ColumnarOp::Insert`s instead (see `columnar_reissue`); `merged.columnar_engines` - // is therefore drained by the caller before this split and never bucketed here. - debug_assert!( - merged.columnar_engines.is_empty(), - "columnar engines must be drained before topology split" - ); - // Vector engine state is likewise NOT installed via the snapshot path: - // RESTORE re-issues each restored vector as a durable `VectorOp::Insert` - // instead (see `vector_reissue`); `merged.vectors` is therefore drained - // by the caller before this split and never bucketed here. - debug_assert!( - merged.vectors.is_empty(), - "vectors must be drained before topology split" - ); - for blob in merged.flushed_ts_segments { - let node = resolve( - route_db_scoped_key(&blob.collection_key), - Some(&blob.collection_key), - ); - all_owners - .entry(node) - .or_default() - .flushed_ts_segments - .push(blob); - } - - // Replicated-by-design: every owning node gets a copy. - for entry in &merged.edges { - for snap in all_owners.values_mut() { - snap.edges.push(entry.clone()); - } - } - // CRDT state is NOT bucketed here: the per-node snapshot fan-out is - // race-prone (skips data groups that have not elected a leader yet) and not - // durable across restart. RESTORE drains the CRDT section before this split - // and re-issues each collection's Loro snapshot durably through Raft to its - // owning data group (see `crdt_reissue`). The vec is therefore empty here by - // contract. - debug_assert!( - merged.crdt_state.is_empty(), - "CRDT state must be drained before topology split" - ); - - SplitOutput { - buckets: all_owners, - malformed_keys: malformed, - route_fallbacks: fallbacks, - } -} - -pub(super) fn is_self(state: &SharedState, node_id: u64) -> bool { - node_id == state.node_id || node_id == 0 || state.cluster_transport.is_none() -} diff --git a/nodedb/src/control/backup/restore/vector_reissue.rs b/nodedb/src/control/backup/restore/vector_reissue.rs index 0a856f0a7..391449082 100644 --- a/nodedb/src/control/backup/restore/vector_reissue.rs +++ b/nodedb/src/control/backup/restore/vector_reissue.rs @@ -7,17 +7,10 @@ //! re-issues each vector as a durable `VectorOp::Insert`: Raft-proposed on //! cluster, WAL-appended + dispatched on single-node. -use std::time::Duration; - use nodedb_types::surrogate::Surrogate; -use crate::Error; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; -use crate::control::server::shared::ddl::sync_dispatch; -use crate::control::state::SharedState; use crate::engine::vector::index_config::{IndexConfig, IndexType}; -use crate::types::{DatabaseId, TenantId, VShardId}; use nodedb_physical::physical_plan::VectorOp; use nodedb_types::vector_distance::DistanceMetric; @@ -107,69 +100,3 @@ pub fn build_vector_set_params_plan( ivf_nprobe: config.ivf_nprobe, }) } - -/// Re-issue a restored vector insert durably. -/// -/// Branches identically to a normal write: -/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: append the redo under an outcome-floor window, then -/// `sync_dispatch::dispatch_system`, which closes the window. -pub async fn reissue_vector_durably( - state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - collection: &str, - plan: PhysicalPlan, -) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); - - if let Some(proposer) = state.async_raft_proposer() { - let entry = crate::control::wal_replication::to_replicated_entry( - tenant_id, - database_id, - vshard, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or_else(|| Error::Internal { - detail: format!( - "restore reissue: vector plan for '{collection}' did not map to a \ - replicated write" - ), - })?; - crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; - return Ok(()); - } - - // Single-node: WAL first (durable for restart replay), then install live. - // The record's outcome-floor window opens before the append and closes - // from the install's outcome. - let owner = RecordOwner { - tenant_id, - database_id, - vshard_id: vshard, - }; - let minted = MintedRecords::open(&state.outcome_floor); - if let Err(error) = minted.append_plan(&state.wal, owner, &plan) { - // Any record appended before the error never reaches a core. - minted.cancel(&state.wal, owner, 0).await?; - return Err(error); - } - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - tenant_id, - database_id, - collection, - plan, - ) - .with_minted(minted), - REISSUE_TIMEOUT, - ) - .await?; - Ok(()) -} - -/// Per-vector re-issue dispatch timeout. Mirrors the columnar/timeseries -/// reissue timeout; a single-vector `Insert` completes far under this. -const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); diff --git a/nodedb/src/control/backup/snapshot_keys.rs b/nodedb/src/control/backup/snapshot_keys.rs index d1d60dcfa..7930cbca1 100644 --- a/nodedb/src/control/backup/snapshot_keys.rs +++ b/nodedb/src/control/backup/snapshot_keys.rs @@ -16,9 +16,8 @@ //! - **collection-name-only** — the key IS the bare collection name (kv tables). //! Routed directly; no extractor needed. //! -//! Both the RESTORE topology splitter and the Raft snapshot SEND builder filter -//! sections by which vshard each entry's collection routes to, so the parsing -//! lives here once and is shared by both — never duplicated ad-hoc. +//! The Raft snapshot SEND builder filters sections by which vshard each +//! entry's collection routes to, so the parsing lives here once. //! //! The backup orchestrator additionally needs to filter a fully-gathered, //! single-tenant [`TenantDataSnapshot`] *in place* to a set of source vshards @@ -85,8 +84,9 @@ pub fn extract_db_scoped_collection(key: &str, tenant_id: u64) -> Option<&str> { /// SEND builder. Every section kind the snapshot carries is classified here so /// adding a section without updating this filter is impossible to miss: /// -/// - db-tenant-scoped keys (`documents`, `indexes`, `vectors`, `timeseries`) -/// via [`extract_db_tenant_scoped_collection`]. +/// - db-tenant-scoped keys (`documents`, `indexes`, `documents_versioned`, +/// `indexes_versioned`, `vectors`, `timeseries`) via +/// [`extract_db_tenant_scoped_collection`]. /// - db-scoped keys (`flushed_ts_segments`, `columnar_engines`) via /// [`extract_db_scoped_collection`]. /// - collection-name-only keys (`kv_tables`) routed directly. @@ -115,6 +115,10 @@ pub fn retain_tenant_data_for_vshards( snap.documents.retain(|(k, _)| in_group_db_tenant_scoped(k)); snap.indexes.retain(|(k, _)| in_group_db_tenant_scoped(k)); + snap.documents_versioned + .retain(|(k, _)| in_group_db_tenant_scoped(k)); + snap.indexes_versioned + .retain(|(k, _)| in_group_db_tenant_scoped(k)); snap.vectors.retain(|(k, _)| in_group_db_tenant_scoped(k)); snap.timeseries .retain(|(k, _)| in_group_db_tenant_scoped(k)); diff --git a/nodedb/src/control/checkpoint_manager.rs b/nodedb/src/control/checkpoint_manager.rs index 38856fcb5..310049ba5 100644 --- a/nodedb/src/control/checkpoint_manager.rs +++ b/nodedb/src/control/checkpoint_manager.rs @@ -129,6 +129,30 @@ pub struct CheckpointCycleInputs<'a> { pub cold_storage: Option>, /// When present, the tombstone set is GC'd to the new truncation point. pub catalog: Option<&'a crate::control::security::catalog::SystemCatalog>, + /// The applied state of every Calvin scheduler on this node. Saved in + /// `catalog` before truncation deletes the applied markers it came from. + pub calvin_mirrors: Option<&'a crate::control::cluster::calvin::scheduler::AppliedMirrors>, +} + +/// The applied state of every mirror, as the catalog stores it. +fn calvin_applied_states( + mirrors: &crate::control::cluster::calvin::scheduler::AppliedMirrors, +) -> Vec { + mirrors.snapshot_all() +} + +/// Save `states` in `catalog`. +fn save_calvin_applied( + states: Vec, + catalog: Option<&crate::control::security::catalog::SystemCatalog>, +) -> crate::Result<()> { + if states.is_empty() { + return Ok(()); + } + let catalog = catalog.ok_or_else(|| crate::Error::Internal { + detail: "checkpoint has no catalog to save the Calvin applied state in".into(), + })?; + catalog.save_calvin_applied(states) } /// Run one checkpoint cycle: dispatch checkpoint to all cores, collect LSNs, @@ -147,6 +171,7 @@ pub async fn run_checkpoint_cycle(inputs: CheckpointCycleInputs<'_>) -> Option) -> Option) -> Option Option { + (self.highest_seen_epoch != NOT_YET_APPLIED_EPOCH).then_some(self.highest_seen_epoch) + } + + /// Whether every position of every epoch at or below `epoch` is applied. + pub fn is_fully_applied_through(&self, epoch: u64) -> bool { + !self.is_sentinel() && self.fully_applied_epoch >= epoch + } + /// Record the delivery of `(epoch, position)` on this vShard, carrying the /// sequencer's per-`(epoch, vShard)` position `count`. /// diff --git a/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs b/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs index 660a93f61..dc84ac9b5 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs @@ -17,6 +17,7 @@ use std::collections::{BTreeSet, HashMap}; use std::sync::{Arc, Mutex}; use super::recovery::NOT_YET_APPLIED_EPOCH; +use crate::control::security::catalog::calvin_applied::StoredCalvinApplied; #[derive(Debug)] struct MirrorState { @@ -67,6 +68,13 @@ impl AppliedMirror { state.tail = state.tail.split_off(&(watermark.saturating_add(1), 0)); } + /// The mirror's state: the fully-applied watermark and the applied + /// positions above it. + pub fn snapshot(&self) -> (u64, BTreeSet<(u64, u32)>) { + let state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + (state.fully_applied_epoch, state.tail.clone()) + } + /// Whether this node's replica applied `(epoch, position)`. pub fn is_applied(&self, epoch: u64, position: u32) -> bool { let state = self.state.lock().unwrap_or_else(|p| p.into_inner()); @@ -98,6 +106,28 @@ impl AppliedMirrors { mirror } + /// Every registered mirror's state, in the shape the catalog stores. + pub fn snapshot_all(&self) -> Vec { + let mirrors: Vec<(u32, Arc)> = self + .by_vshard + .lock() + .unwrap_or_else(|p| p.into_inner()) + .iter() + .map(|(vshard_id, mirror)| (*vshard_id, Arc::clone(mirror))) + .collect(); + mirrors + .into_iter() + .map(|(vshard_id, mirror)| { + let (fully_applied_epoch, tail) = mirror.snapshot(); + StoredCalvinApplied { + vshard_id, + fully_applied_epoch, + tail, + } + }) + .collect() + } + /// The mirror of `vshard_id`, when this node runs its scheduler. pub fn get(&self, vshard_id: u32) -> Option> { self.by_vshard diff --git a/nodedb/src/control/cluster/calvin/scheduler/cut_floor.rs b/nodedb/src/control/cluster/calvin/scheduler/cut_floor.rs new file mode 100644 index 000000000..94bbca1d9 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/cut_floor.rs @@ -0,0 +1,173 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Backup cut markers as one vShard's scheduler receives them. +//! +//! A cut marker arrives between two epoch batches, so every transaction of +//! an epoch at or below the highest epoch delivered before the marker came +//! before it, and every transaction of a later epoch came after it. A +//! transaction after a marker commits above the marker's watermark: a +//! restore of the backup that placed the marker refuses it. A transaction +//! before the marker finishes before the scheduler reports the marker, so the +//! backup holds it. +//! +//! The commit HLC of a transaction is the instant its epoch was created, +//! read once on the sequencer leader and replicated with the batch, raised to +//! the floor of every marker it came after. Replicas of a vShard receive the +//! same inputs in the same order, so they stamp every transaction alike. + +use super::recovery::NOT_YET_APPLIED_EPOCH; + +/// Nanoseconds per millisecond, to express an epoch's millisecond wall time +/// on the nanosecond HLC scale. +const NANOS_PER_MILLI: u64 = 1_000_000; + +/// A marker whose floor applies to the epochs above `through`. +#[derive(Debug, Clone, Copy)] +struct Marker { + through: u64, + floor: u64, +} + +/// A marker the scheduler has not reported yet. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct WaitingCut { + /// The marker's watermark. + pub hlc: u64, + /// The highest epoch delivered before the marker. The marker passes once + /// every epoch at or below it is fully applied. + pub through: u64, +} + +/// The cut markers one scheduler received. +#[derive(Debug, Default)] +pub struct CutFloors { + /// The floor every epoch above the folded markers records at least. + base: u64, + /// Markers not folded into `base`, in arrival order. + markers: Vec, + /// Markers not reported yet. + waiting: Vec, +} + +impl CutFloors { + /// Receive a marker carrying `hlc`, after epochs up to `highest_seen` + /// were delivered (`None` when none was). Returns `true` when every + /// transaction before it finished already: nothing came before it. + pub fn receive(&mut self, hlc: u64, highest_seen: Option) -> bool { + let floor = hlc.saturating_add(1); + match highest_seen { + None => { + self.base = self.base.max(floor); + true + } + Some(through) => { + self.markers.push(Marker { through, floor }); + self.waiting.push(WaitingCut { hlc, through }); + false + } + } + } + + /// The commit HLC of a transaction of `epoch` whose epoch was created at + /// `epoch_system_ms`. + pub fn commit_hlc(&self, epoch: u64, epoch_system_ms: i64) -> u64 { + let created = u64::try_from(epoch_system_ms) + .unwrap_or(0) + .saturating_mul(NANOS_PER_MILLI); + self.markers + .iter() + .filter(|marker| marker.through < epoch) + .map(|marker| marker.floor) + .fold(created.max(self.base), u64::max) + } + + /// Take every waiting marker whose epochs are fully applied, given the + /// test `fully_applied_through`. Returns their watermarks. + pub fn take_passed(&mut self, fully_applied_through: impl Fn(u64) -> bool) -> Vec { + let mut passed = Vec::new(); + self.waiting.retain(|cut| { + if fully_applied_through(cut.through) { + passed.push(cut.hlc); + false + } else { + true + } + }); + passed + } + + /// Fold every marker whose floor covers all epochs above `fully_applied` + /// into the base floor. No transaction of an epoch at or below + /// `fully_applied` commits again, so only the epochs above it need the + /// markers told apart. + pub fn fold(&mut self, fully_applied: u64) { + if fully_applied == NOT_YET_APPLIED_EPOCH { + return; + } + let base = &mut self.base; + self.markers.retain(|marker| { + if marker.through <= fully_applied { + *base = (*base).max(marker.floor); + false + } else { + true + } + }); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_transaction_after_a_marker_commits_above_its_watermark() { + let mut floors = CutFloors::default(); + assert!(!floors.receive(5_000_000_000, Some(7))); + assert_eq!( + floors.commit_hlc(7, 1), + NANOS_PER_MILLI, + "epoch 7 came before" + ); + assert_eq!(floors.commit_hlc(8, 1), 5_000_000_001, "epoch 8 came after"); + assert_eq!( + floors.commit_hlc(8, 9_000), + 9_000 * NANOS_PER_MILLI, + "a later creation instant stands" + ); + } + + #[test] + fn a_marker_passes_once_its_epochs_are_fully_applied() { + let mut floors = CutFloors::default(); + floors.receive(100, Some(3)); + floors.receive(200, Some(5)); + assert_eq!(floors.take_passed(|through| through <= 4), vec![100]); + assert_eq!( + floors.take_passed(|through| through <= 4), + Vec::::new() + ); + assert_eq!(floors.take_passed(|through| through <= 5), vec![200]); + } + + #[test] + fn a_marker_before_any_epoch_passes_at_once_and_raises_every_epoch() { + let mut floors = CutFloors::default(); + assert!(floors.receive(100, None)); + assert_eq!(floors.commit_hlc(0, 0), 101); + assert!(floors.take_passed(|_| false).is_empty()); + } + + #[test] + fn folding_keeps_every_later_stamp() { + let mut floors = CutFloors::default(); + floors.receive(100, Some(3)); + floors.receive(50, Some(6)); + let before: Vec = (4..9).map(|epoch| floors.commit_hlc(epoch, 0)).collect(); + floors.fold(4); + let after: Vec = (5..9).map(|epoch| floors.commit_hlc(epoch, 0)).collect(); + assert_eq!(&before[1..], &after[..]); + floors.fold(NOT_YET_APPLIED_EPOCH); + assert_eq!(floors.commit_hlc(7, 0), 101); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs index 67d1c9d05..48a491c9a 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -38,7 +38,7 @@ impl Scheduler { /// past the bound, halts the scheduler: a skipped flush tears the /// committed txn on this replica. A non-`Ok` drop completes the txn: /// under an abort verdict no replica writes anything. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn finish_resolved_commit( + pub(in crate::control::cluster::calvin::scheduler::driver::core) async fn finish_resolved_commit( &mut self, txn_id: TxnId, response: Response, @@ -74,7 +74,7 @@ impl Scheduler { } let completed = if committed { - self.commit_apply_tail(txn_id, response, redo_lsn) + self.commit_apply_tail(txn_id, response, redo_lsn).await } else { self.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); true @@ -142,7 +142,7 @@ impl Scheduler { /// or the direct-apply dependent/active path, which carries no redo record) /// falls back to appending a `CalvinApplied` marker here, exactly as before /// this record existed. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn commit_apply_tail( + pub(in crate::control::cluster::calvin::scheduler::driver::core) async fn commit_apply_tail( &mut self, txn_id: TxnId, response: Response, @@ -282,6 +282,20 @@ impl Scheduler { // scheduler halted above. return false; }; + // The record at `lsn` is this position's only applied marker. An + // append only buffers it, so it is durable before the mark and the + // ack: a restart that lost it would take the position for unapplied, + // run the transaction again, and never settle its ack. The wait joins + // the WAL group commit, so concurrent Calvin commits share one fsync. + if let Err(e) = self.shared.wal.wait_durable(lsn).await { + self.halt_apply( + txn_id, + HaltReason::WalAppendFailed, + HaltStep::AppliedMarker, + format!("applied marker fsync at lsn {} failed: {e}", lsn.as_u64()), + ); + return false; + } // Control change-stream events are distinct from Data-Plane // WriteEvents. Publish the participant-local logical manifests once, // from the data-group leader, at the authoritative committed LSN. @@ -300,6 +314,9 @@ impl Scheduler { ); } } + // The commit's mark lands before the ack, as a write through the + // funnel records its mark before its response returns. + self.record_calvin_write_mark(txn_id); self.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); true } @@ -343,7 +360,9 @@ mod tests { }, ); - scheduler.finish_resolved_commit(txn_id, internal_error(), true, None); + scheduler + .finish_resolved_commit(txn_id, internal_error(), true, None) + .await; assert!(!scheduler.applied.is_applied(9, 2)); assert!(scheduler.pending.contains_key(&txn_id)); @@ -367,7 +386,9 @@ mod tests { }, ); - scheduler.finish_resolved_commit(txn_id, internal_error(), false, None); + scheduler + .finish_resolved_commit(txn_id, internal_error(), false, None) + .await; assert!(scheduler.applied.is_applied(9, 2)); assert!(!scheduler.pending.contains_key(&txn_id)); @@ -402,7 +423,9 @@ mod tests { pending.flush_scope.sends = 1; scheduler.pending.insert(txn_id, pending); - scheduler.finish_resolved_commit(txn_id, retryable_refusal(), true, None); + scheduler + .finish_resolved_commit(txn_id, retryable_refusal(), true, None) + .await; assert!(!scheduler.is_apply_halted()); assert!(!scheduler.applied.is_applied(9, 2)); @@ -436,7 +459,9 @@ mod tests { pending.flush_scope.sends = MAX_FLUSH_SENDS; } - scheduler.finish_resolved_commit(txn_id, retryable_refusal(), true, None); + scheduler + .finish_resolved_commit(txn_id, retryable_refusal(), true, None) + .await; assert!(scheduler.pending.contains_key(&txn_id)); assert_eq!( diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs index 9fa99a857..02f9eb0fb 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs @@ -22,7 +22,7 @@ impl Scheduler { /// Process a completed executor response (or disconnected channel). /// /// Called from the `completion_rx` arm of the main `select!` loop. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn handle_completion( + pub(in crate::control::cluster::calvin::scheduler::driver::core) async fn handle_completion( &mut self, txn_id: TxnId, request_id: RequestId, @@ -133,7 +133,8 @@ impl Scheduler { committed, redo_lsn, }) => { - self.finish_resolved_commit(txn_id, response, committed, redo_lsn); + self.finish_resolved_commit(txn_id, response, committed, redo_lsn) + .await; return; } Some(CommitState::AwaitingVerdict) => { @@ -191,7 +192,7 @@ impl Scheduler { } // `false` means the commit tail halted the scheduler: the txn stays // pending and unapplied. - if self.commit_apply_tail(txn_id, response, None) { + if self.commit_apply_tail(txn_id, response, None).await { self.metrics.record_completed(); self.on_txn_complete(txn_id); } @@ -223,7 +224,9 @@ mod tests { }, ); - scheduler.handle_completion(txn_id, RequestId::new(9), None); + scheduler + .handle_completion(txn_id, RequestId::new(9), None) + .await; assert!( !scheduler.applied.is_applied(5, 1), @@ -256,8 +259,12 @@ mod tests { pending.commit_state = Some(CommitState::AwaitingRedoResolve); scheduler.pending.insert(second, pending); - scheduler.handle_completion(first, RequestId::new(9), None); - scheduler.handle_completion(second, RequestId::new(10), None); + scheduler + .handle_completion(first, RequestId::new(9), None) + .await; + scheduler + .handle_completion(second, RequestId::new(10), None) + .await; assert!(!scheduler.applied.is_applied(6, 0)); assert!(scheduler.pending.contains_key(&second)); @@ -282,13 +289,15 @@ mod tests { }, ); - scheduler.handle_completion( - txn_id, - RequestId::new(9), - Some(error_response( - crate::bridge::envelope::ErrorCode::OllpRetryRequired, - )), - ); + scheduler + .handle_completion( + txn_id, + RequestId::new(9), + Some(error_response( + crate::bridge::envelope::ErrorCode::OllpRetryRequired, + )), + ) + .await; assert!(!scheduler.applied.is_applied(5, 1)); assert!( diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs new file mode 100644 index 000000000..d93be00c0 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's cut marker on this vShard's scheduler, and the tenant write +//! mark a committed Calvin transaction records. +//! +//! The scheduler reports a marker to the node's cut registry once every +//! transaction delivered before it finished: installed by its flush, or +//! dropped under an abort verdict. The backup waiting on the marker then +//! snapshots a vShard that holds every one of them. Every transaction +//! delivered after the marker records a commit HLC above the marker's +//! watermark, so a restore of that backup refuses it. + +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::security::auth_fence::cluster::group_of_vshard; +use crate::control::state::tenant_marks::MarkSite; + +impl Scheduler { + /// Receive a backup's cut marker carrying `hlc`, in input order. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn receive_cut_marker( + &mut self, + hlc: u64, + ) { + if self + .cut_floors + .receive(hlc, self.applied.highest_seen_epoch()) + { + self.shared.calvin.cuts.note_passed(self.vshard_id, hlc); + } + self.report_passed_cuts(); + } + + /// Report every marker whose earlier transactions all finished, and fold + /// the markers every later epoch is above. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn report_passed_cuts( + &mut self, + ) { + let applied = &self.applied; + let passed = self + .cut_floors + .take_passed(|through| applied.is_fully_applied_through(through)); + for hlc in passed { + self.shared.calvin.cuts.note_passed(self.vshard_id, hlc); + } + self.cut_floors.fold(self.applied.fully_applied_epoch()); + } + + /// Record the commit HLC of the committed transaction `txn_id` on its + /// tenant's observed write high-water, before its `CompletionAck` lets + /// the coordinator acknowledge the COMMIT. + /// + /// Only a slice that carries a primary user data write records it. The + /// implicit edge cleanup that dual-homes alongside one writes no user row + /// of its own. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn record_calvin_write_mark( + &self, + txn_id: TxnId, + ) { + let Some(pending) = self.pending.get(&txn_id) else { + return; + }; + if !pending.has_primary_write { + return; + } + let tx_class = &pending.txn.tx_class; + let tenant_id = tx_class.tenant_id.as_u64(); + let collection = tx_class.write_set.0.first().map(|set| set.collection()); + let commit_hlc = self + .cut_floors + .commit_hlc(pending.txn.epoch, pending.txn.epoch_system_ms); + self.shared + .advance_tenant_write_hlc(tenant_id, commit_hlc, "calvin flush", collection); + + // The durable mark, in the data group that homes this vShard, lands + // before the ack too: RESTORE's guard reads it on every node, after a + // restart as well. + let group_id = match group_of_vshard(&self.shared, self.vshard_id) { + Ok(group_id) => group_id, + Err(error) => { + tracing::error!( + vshard_id = self.vshard_id, + %error, + "calvin: no data group for this vShard; the commit's write mark stays \ + in memory only" + ); + return; + } + }; + let marks = &self.shared.tenant_marks; + marks.raise( + group_id, + tenant_id, + commit_hlc, + MarkSite::CalvinFlush, + collection, + ); + if let Err(error) = marks.persist(self.shared.credentials.catalog()) { + // The mark stays pending; the next persist on this node writes it. + tracing::error!( + vshard_id = self.vshard_id, + %error, + "calvin: the commit's write mark did not persist yet" + ); + } + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs index 22dcbee8c..f87b16717 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs @@ -302,7 +302,9 @@ mod tests { scheduler .pending .insert(held, staged_pending(make_validate_only_txn(3, 0), held)); - scheduler.handle_completion(held, RequestId::new(9), None); + scheduler + .handle_completion(held, RequestId::new(9), None) + .await; assert_eq!(scheduler.intake_closure(), Some(IntakeClosure::ApplyHalted)); let metrics = Arc::clone(&scheduler.metrics); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index a3cf69916..8fd8be522 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -19,6 +19,7 @@ pub mod commit_redo; pub mod commit_resolution_dispatch; pub mod commit_resolve; pub mod completion_route; +mod cut_marker; pub mod deferred; pub mod dispatch; pub mod halt; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs index 7b8fcc395..cf31a06c3 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs @@ -231,7 +231,9 @@ mod tests { scheduler.sequencer_proposer = proposer.clone(); scheduler.registry.seed_expected(cluster_txn_id(txn_id), 2); - scheduler.finish_resolved_commit(txn_id, staged_response(Status::Ok, None), false, None); + scheduler + .finish_resolved_commit(txn_id, staged_response(Status::Ok, None), false, None) + .await; assert!(!scheduler.pending.contains_key(&txn_id)); assert_eq!(proposer.attempt_count(), 1, "the first proposal is refused"); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs index 2f1140b16..9392b872e 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs @@ -53,16 +53,19 @@ impl Scheduler { SchedulerInput::Txn(txn) => self.process_new_txn(txn), SchedulerInput::Reserve { owner, key } => self.install_reservation(owner, key), SchedulerInput::Release { owner, reason } => self.release_reservation(owner, reason), + SchedulerInput::CutMarker { hlc } => self.receive_cut_marker(hlc), } } /// The replicated epoch an input is stamped with — the monotonic logical - /// clock the lease reap advances on. + /// clock the lease reap advances on. A cut marker carries no epoch, so it + /// reports `0`, which never advances the clock. fn input_epoch(input: &SchedulerInput) -> u64 { match input { SchedulerInput::Txn(txn) => txn.epoch, SchedulerInput::Reserve { owner, .. } => owner.epoch, SchedulerInput::Release { owner, .. } => owner.epoch, + SchedulerInput::CutMarker { .. } => 0, } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index 4da6037d3..b9c66e849 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -112,6 +112,10 @@ pub struct Scheduler { /// deterministic threshold below which an orphaned shared reservation is /// released. Purely a function of replicated input order — no wall clock. pub(in crate::control::cluster::calvin::scheduler::driver::core) max_input_epoch: u64, + /// Backup cut markers this scheduler received: the commit HLC floors they + /// set and the markers not yet reported. + pub(in crate::control::cluster::calvin::scheduler::driver::core) cut_floors: + crate::control::cluster::calvin::scheduler::cut_floor::CutFloors, /// Shared mirror of `applied`, read by authorization coverage. pub(in crate::control::cluster::calvin::scheduler::driver::core) applied_mirror: Arc, @@ -241,6 +245,9 @@ impl Scheduler { &applied_tail, ); + // A backup's cut waits on every scheduler this node runs. + shared.calvin.cuts.register(vshard_id); + let capacity_freed = shared .dispatcher .lock() @@ -261,6 +268,7 @@ impl Scheduler { dependent_barrier: BTreeMap::new(), read_result_rx, applied: AppliedGate::new(fully_applied_epoch, applied_tail), + cut_floors: Default::default(), applied_mirror, rebuild_target_epoch, max_input_epoch: 0, @@ -317,7 +325,7 @@ impl Scheduler { /// committed, which would let a session anchor on a torn epoch. `fetch_max` /// keeps it monotonic across all per-vShard schedulers writing the counter. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn publish_watermark( - &self, + &mut self, watermark: u64, ) { self.metrics.update_last_applied_epoch(watermark); @@ -326,6 +334,8 @@ impl Scheduler { .calvin .last_applied_epoch .fetch_max(watermark, std::sync::atomic::Ordering::Release); + // A marker passes once every epoch delivered before it folded. + self.report_passed_cuts(); } /// Spawn a bridge task that awaits a single executor response and forwards @@ -401,7 +411,10 @@ impl Scheduler { maybe_completion = self.completion_rx.recv() => { if let Some((txn_id, request_id, resp_opt)) = maybe_completion { - self.handle_completion(txn_id, request_id, resp_opt); + // Awaited in the arm: the loop takes no other input + // until this completion, its durability wait included, + // is fully handled. + self.handle_completion(txn_id, request_id, resp_opt).await; } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs index 68bbdb80f..1d950ac28 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs @@ -9,7 +9,7 @@ //! proposes it to its sequencer group. use std::collections::BTreeSet; -use std::sync::{Arc, Mutex}; +use std::sync::{Arc, Mutex, Weak}; use tokio::sync::Semaphore; use tracing::debug; @@ -60,18 +60,21 @@ pub(super) fn sequencer_route( pub struct RaftSequencerProposer { node_id: u64, multi_raft: Arc>, - /// Source of the cluster transport and topology for forwards. - shared: Arc, + /// Source of the cluster transport and topology for forwards. Weak: the + /// node's `SharedState` holds this proposer, and a strong handle back + /// would keep the state, and every file it holds open, alive after + /// shutdown. + shared: Weak, /// One permit per forward RPC in flight. forwards: Arc, } impl RaftSequencerProposer { - pub fn new(node_id: u64, multi_raft: Arc>, shared: Arc) -> Self { + pub fn new(node_id: u64, multi_raft: Arc>, shared: &Arc) -> Self { Self { node_id, multi_raft, - shared, + shared: Arc::downgrade(shared), forwards: Arc::new(Semaphore::new(MAX_INFLIGHT_SEQUENCER_FORWARDS)), } } @@ -86,7 +89,10 @@ impl RaftSequencerProposer { leader: u64, bytes: Vec, ) -> Result { - let Some(transport) = self.shared.cluster_transport.as_ref() else { + let Some(shared) = self.shared.upgrade() else { + return Err(SequencerProposeError::ShutDown); + }; + let Some(transport) = shared.cluster_transport.as_ref() else { return Err(SequencerProposeError::NoTransport { leader }); }; let permit = Arc::clone(&self.forwards) @@ -95,7 +101,7 @@ impl RaftSequencerProposer { leader, limit: MAX_INFLIGHT_SEQUENCER_FORWARDS, })?; - register_peers_from_topology(&self.shared, transport, &BTreeSet::from([leader])); + register_peers_from_topology(&shared, transport, &BTreeSet::from([leader])); let transport = Arc::clone(transport); tokio::spawn(async move { let _permit = permit; @@ -230,8 +236,8 @@ mod tests { .unwrap_or_else(|p| p.into_inner()) .last_log_index(SEQUENCER_GROUP_ID) .unwrap_or(0); - let proposer = - RaftSequencerProposer::new(LOCAL_NODE, Arc::clone(&mr), shared_state(dir.path())); + let shared = shared_state(dir.path()); + let proposer = RaftSequencerProposer::new(LOCAL_NODE, Arc::clone(&mr), &shared); let dispatch = proposer.propose(vec![9, 9]).expect("local propose"); @@ -267,8 +273,8 @@ mod tests { assert_eq!(guard.group_leader(SEQUENCER_GROUP_ID), REMOTE_LEADER); guard.last_log_index(SEQUENCER_GROUP_ID).unwrap_or(0) }; - let proposer = - RaftSequencerProposer::new(LOCAL_NODE, Arc::clone(&mr), shared_state(dir.path())); + let shared = shared_state(dir.path()); + let proposer = RaftSequencerProposer::new(LOCAL_NODE, Arc::clone(&mr), &shared); let result = proposer.propose(vec![9, 9]); @@ -288,4 +294,25 @@ mod tests { .unwrap_or(0); assert_eq!(after, before, "a follower appends nothing locally"); } + + /// The node's state holds its proposer. The proposer must not hold the + /// state back: that cycle keeps the catalog open after shutdown, and a + /// reopen on the same path fails to take the catalog lock. + #[test] + fn a_proposer_held_by_the_state_does_not_keep_the_state_alive() { + let dir = tempfile::tempdir().expect("tempdir"); + let mr = multi_raft(dir.path(), vec![LOCAL_NODE]); + let shared = shared_state(dir.path()); + let proposer: Arc = + Arc::new(RaftSequencerProposer::new(LOCAL_NODE, mr, &shared)); + assert!(shared.calvin.sequencer_proposer.set(proposer).is_ok()); + + assert_eq!(Arc::strong_count(&shared), 1); + let weak = Arc::downgrade(&shared); + drop(shared); + assert!( + weak.upgrade().is_none(), + "the state outlived its last owner" + ); + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs index a39513bba..876579c7f 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/seam.rs @@ -28,6 +28,9 @@ pub enum SequencerProposeError { /// again on a later tick. #[error("{limit} sequencer forwards are in flight; the entry for node {leader} waits")] ForwardBusy { leader: u64, limit: usize }, + /// The node shut down: its state is gone. + #[error("the node is shutting down")] + ShutDown, /// The local sequencer group refused the proposal. #[error("sequencer propose: {0}")] Cluster(#[from] ClusterError), diff --git a/nodedb/src/control/cluster/calvin/scheduler/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/mod.rs index 78d9184c7..26126e309 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/mod.rs @@ -2,6 +2,7 @@ pub mod applied_gate; pub mod applied_mirror; +pub mod cut_floor; pub mod driver; pub mod lock; pub mod metrics; @@ -19,4 +20,6 @@ pub use lock::{AcquireOutcome, HotKeyTable, LockKey, LockManager, LockMode, TxnI // keep that path stable via an alias while the module lives under `lock/`. pub use lock as lock_manager; pub use metrics::SchedulerMetrics; -pub use recovery::{AppliedRecovery, NOT_YET_APPLIED_EPOCH, read_applied_recovery}; +pub use recovery::{ + AppliedRecovery, NOT_YET_APPLIED_EPOCH, read_applied_recovery, recover_applied, +}; diff --git a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs index 30c67abfa..f44c06830 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs @@ -123,6 +123,58 @@ pub fn read_applied_recovery(wal: &WalManager, vshard_id: u32) -> crate::Result< }) } +/// This vShard's applied state after a restart: the state the last +/// checkpoint saved in `catalog`, together with the markers the WAL still +/// holds. +/// +/// A checkpoint deletes WAL segments that hold applied markers, while the +/// sequencer log keeps delivering their entries after a restart. Without the +/// saved state the scheduler would take an applied transaction for a new +/// one: its local stage refuses the rows it already wrote, the scheduler +/// halts, and the transaction's completion ack never settles. +pub fn recover_applied( + wal: &WalManager, + catalog: &crate::control::security::catalog::SystemCatalog, + vshard_id: u32, +) -> crate::Result { + let from_wal = read_applied_recovery(wal, vshard_id)?; + let Some(saved) = catalog.load_calvin_applied(vshard_id)? else { + return Ok(from_wal); + }; + Ok(merge_saved(from_wal, saved.fully_applied_epoch, saved.tail)) +} + +/// Fold a saved `(fully_applied_epoch, tail)` into a WAL scan's result. +fn merge_saved( + from_wal: AppliedRecovery, + fully_applied_epoch: u64, + saved_tail: BTreeSet<(u64, u32)>, +) -> AppliedRecovery { + let above_watermark = + |epoch: u64| fully_applied_epoch == NOT_YET_APPLIED_EPOCH || epoch > fully_applied_epoch; + let applied_tail: BTreeSet<(u64, u32)> = from_wal + .applied_tail + .into_iter() + .chain(saved_tail) + .filter(|(epoch, _)| above_watermark(*epoch)) + .collect(); + let mut max_applied_epoch = from_wal.max_applied_epoch; + let candidates = applied_tail + .iter() + .map(|(epoch, _)| *epoch) + .chain((fully_applied_epoch != NOT_YET_APPLIED_EPOCH).then_some(fully_applied_epoch)); + for epoch in candidates { + if max_applied_epoch == NOT_YET_APPLIED_EPOCH || epoch > max_applied_epoch { + max_applied_epoch = epoch; + } + } + AppliedRecovery { + fully_applied_epoch, + applied_tail, + max_applied_epoch, + } +} + /// Decode a WAL record's logical [`RecordType`], stripping the encryption /// flag (bit 31) before comparing. fn record_type_of(record: &WalRecord) -> Option { @@ -298,4 +350,40 @@ mod tests { assert_eq!(rec.applied_tail.len(), 2); assert_eq!(rec.max_applied_epoch, 1); } + + #[test] + fn saved_state_restores_what_a_truncated_wal_lost() { + let dir = TempDir::new().unwrap(); + let wal = open_wal(&dir); + use crate::types::VShardId; + // The WAL still holds only the marker written after the checkpoint. + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_calvin_applied(VShardId::new(1), 6, 0) + .unwrap(); + wal.sync().unwrap(); + let catalog_dir = TempDir::new().unwrap(); + let catalog = crate::control::security::catalog::SystemCatalog::open( + &catalog_dir.path().join("system.redb"), + ) + .unwrap(); + catalog + .save_calvin_applied(vec![ + crate::control::security::catalog::calvin_applied::StoredCalvinApplied { + vshard_id: 1, + fully_applied_epoch: 2, + tail: [(4, 1)].into_iter().collect(), + }, + ]) + .unwrap(); + + let rec = recover_applied(&wal, &catalog, 1).unwrap(); + assert_eq!(rec.fully_applied_epoch, 2); + assert!(rec.applied_tail.contains(&(4, 1))); + assert!(rec.applied_tail.contains(&(6, 0))); + assert_eq!(rec.max_applied_epoch, 6); + + // A vShard with nothing saved keeps the plain WAL scan. + let other = recover_applied(&wal, &catalog, 9).unwrap(); + assert_eq!(other, read_applied_recovery(&wal, 9).unwrap()); + } } diff --git a/nodedb/src/control/cluster/snapshot_applier.rs b/nodedb/src/control/cluster/snapshot_applier.rs index e4e46501c..a4851dc51 100644 --- a/nodedb/src/control/cluster/snapshot_applier.rs +++ b/nodedb/src/control/cluster/snapshot_applier.rs @@ -168,6 +168,18 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { } } + // Take the group's tenant write marks, durably, before this node + // reports the group applied through the snapshot. + if !snap.group_write_marks.is_empty() { + self.shared + .tenant_marks + .raise_group_entries(group_id, &snap.group_write_marks); + self.shared + .tenant_marks + .persist(self.shared.credentials.catalog()) + .map_err(|err| Box::new(err) as Box)?; + } + // The install emitted no per-row events, so the permission cache // reloads before this node reports coverage of the group again. self.shared.authorization_fence.note_snapshot_installed(); diff --git a/nodedb/src/control/cluster/snapshot_builder.rs b/nodedb/src/control/cluster/snapshot_builder.rs index 2647633f2..34af3000e 100644 --- a/nodedb/src/control/cluster/snapshot_builder.rs +++ b/nodedb/src/control/cluster/snapshot_builder.rs @@ -104,7 +104,10 @@ impl DataPlaneSnapshotBuilder { group_vshards: &HashSet, merged: &mut TenantDataSnapshot, ) -> Result<(), Error> { - let plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id }); + let plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id, + cut_watermark: None, + }); let bytes = crate::control::server::shared::ddl::sync_dispatch::dispatch_system( &self.shared, crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( @@ -146,6 +149,16 @@ impl DataPlaneSnapshotBuilder { merged.indexes.push((k, v)); } } + for (k, v) in snap.documents_versioned { + if in_group_db_tenant_scoped(&k) { + merged.documents_versioned.push((k, v)); + } + } + for (k, v) in snap.indexes_versioned { + if in_group_db_tenant_scoped(&k) { + merged.indexes_versioned.push((k, v)); + } + } for (k, v) in snap.vectors { if in_group_db_tenant_scoped(&k) { merged.vectors.push((k, v)); @@ -303,6 +316,10 @@ impl nodedb_cluster::SnapshotBuilder for DataPlaneSnapshotBuilder { .map_err(|e| Box::new(e) as Box)?; } + // The group's tenant write marks travel with its data: the follower + // that installs the snapshot never applies the entries it covers. + merged.group_write_marks = self.shared.tenant_marks.group_entries(group_id); + // Always return a well-formed serialized struct (even when empty) so the // follower-apply unit receives a decodable payload rather than a stub. let out = zerompk::to_msgpack_vec(&merged).map_err(|e| { diff --git a/nodedb/src/control/cluster/start_raft/group_setup.rs b/nodedb/src/control/cluster/start_raft/group_setup.rs index cdb670232..2fa4da5cc 100644 --- a/nodedb/src/control/cluster/start_raft/group_setup.rs +++ b/nodedb/src/control/cluster/start_raft/group_setup.rs @@ -92,11 +92,10 @@ pub(super) fn build_group_setup( // Build the propose tracker and distributed applier. // - // The tracker is wired with the per-group apply watermark - // registry so every `tracker.complete(group_id, idx, _)` call - // also bumps the watcher — coupling the "data applied on this - // node" signal to the single source of truth that proposers - // and cross-node visibility waits both consume. + // The tracker is wired with the per-group apply watermark registry. The + // apply loop bumps it through the tracker once every entry of a group up + // to an index finished, so proposers and cross-node visibility waits read + // one in-order "data applied on this node" signal. let tracker = Arc::new(ProposeTracker::new().with_group_watchers(handle.group_watchers.clone())); let (dist_applier, apply_rx) = create_distributed_applier(tracker.clone()); diff --git a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs index 0d94b8d98..64219b99b 100644 --- a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs +++ b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs @@ -150,9 +150,8 @@ pub(super) fn wire_proposers( // Held weakly for the same cycle-breaking reason as `raft_proposer` above: // the proposer lives on `SharedState`. let state_for_proposer = Arc::downgrade(shared); - let deadline_secs = shared.tuning.network.default_deadline_secs; let async_proposer: Arc = - Arc::new(move |vshard_id, idempotency_key, data| { + Arc::new(move |vshard_id, idempotency_key, data, deadline| { let rl_weak = raft_loop_async.clone(); let tk = tracker_for_proposer.clone(); let state_weak = state_for_proposer.clone(); @@ -160,12 +159,15 @@ pub(super) fn wire_proposers( let rl = rl_weak.upgrade().ok_or_else(|| crate::Error::Internal { detail: "raft propose (async): cluster not running".into(), })?; - let (group_id, log_index) = rl - .propose_via_data_leader(vshard_id, data) - .await - .map_err(|e| crate::Error::Internal { - detail: format!("raft propose (async): {e}"), - })?; + // The attempt gets only what remains of the caller's deadline. + if tokio::time::Instant::now() >= deadline { + return Err(propose_deadline_exceeded()); + } + let (group_id, log_index) = + tokio::time::timeout_at(deadline, rl.propose_via_data_leader(vshard_id, data)) + .await + .map_err(|_| propose_deadline_exceeded())? + .map_err(|e| async_propose_error(vshard_id, e))?; // Register the waiter with the proposer's idempotency // key. The apply path compares against the committed @@ -175,43 +177,42 @@ pub(super) fn wire_proposers( // `RetryableLeaderChange` instead of leaking a // not-our-payload back to the caller. let rx = tk.register(group_id, log_index, idempotency_key); - let applied = - tokio::time::timeout(std::time::Duration::from_secs(deadline_secs), rx) - .await - .map_err(|_| crate::Error::Dispatch { - detail: format!( - "raft commit timeout for group {group_id} index {log_index}" - ), - })? - .map_err(|_| crate::Error::Dispatch { - detail: "propose waiter channel closed".into(), - })? - // Preserve `RetryableLeaderChange` so the gateway - // retry loop can re-propose against the new leader - // — wrapping it in `Dispatch` would hide the - // retryable signal and surface as silent INSERT - // success. Only machinery failures stay wrapped for - // diagnostics; a classified apply verdict keeps its - // client-visible classification. - .map_err(|e| { - if crate::error_classify::is_unclassified_failure(&e) { - crate::Error::Dispatch { - detail: format!("apply error: {e}"), - } - } else { - e - } - }) - // Carry out the write-version the APPLY side stamped, not - // `log_index`. The tracker resolves on the node that applied - // the entry locally, so `write_version` is this replica's own - // post-write `coll_write_lsn` — a WAL LSN, the same domain - // every other feed of that map records in, and the only - // domain the shard-local OCC read validator compares in. The - // raft log index is a per-group counter on a different scale - // entirely; publishing it here made reads validate a WAL LSN - // against a log index. - .map(|applied| (applied.payload, applied.write_version)); + let applied = await_local_apply(LocalApplyWait { + state: &state_weak, + tracker: &tk, + group_id, + log_index, + vshard_id, + deadline, + rx, + }) + .await + // Preserve `RetryableLeaderChange` so the gateway + // retry loop can re-propose against the new leader + // — wrapping it in `Dispatch` would hide the + // retryable signal and surface as silent INSERT + // success. Only machinery failures stay wrapped for + // diagnostics; a classified apply verdict keeps its + // client-visible classification. + .map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + crate::Error::Dispatch { + detail: format!("apply error: {e}"), + } + } else { + e + } + }) + // Carry out the write-version the APPLY side stamped, not + // `log_index`. The tracker resolves on the node that applied + // the entry locally, so `write_version` is this replica's own + // post-write `coll_write_lsn` — a WAL LSN, the same domain + // every other feed of that map records in, and the only + // domain the shard-local OCC read validator compares in. The + // raft log index is a per-group counter on a different scale + // entirely; publishing it here made reads validate a WAL LSN + // against a log index. + .map(|applied| (applied.payload, applied.write_version)); let applied = applied?; // A write to a vShard homing a permission-tree source is // acknowledged only once every lease holder covers it, or its @@ -271,3 +272,135 @@ pub(super) fn wire_proposers( ); Ok(()) } + +/// Where a group's pipeline stands, for a propose waiter that timed out: the +/// Raft commit index, the index handed to the apply loop, and the index the +/// apply loop applied. The first of the three that stops short of the waited +/// index names the stage that stalled. +fn apply_progress(state: Option<&SharedState>, group_id: u64) -> String { + let Some(state) = state else { + return "node is shutting down".to_owned(); + }; + let status = state + .raft_status_fn + .get() + .and_then(|status| status().into_iter().find(|g| g.group_id == group_id)); + let applied = state.applied_index_watcher(group_id).current(); + match status { + Some(group) => format!( + "commit_index={} handed_to_apply_loop={} applied={applied} role={} leader={}", + group.commit_index, group.last_applied, group.role, group.leader_id + ), + None => format!("group not hosted here, applied={applied}"), + } +} + +/// The error an async propose that reached no leader returns. +/// +/// A group with no leader to take the proposal right now accepts the same +/// proposal once it has one, so the proposal is retried: +/// [`crate::Error::NoLeader`]. That covers an election, a leadership transfer +/// in flight, and a leader that stepped down after this node or a forwarding +/// node chose it; a forwarded refusal arrives here with its typed Raft error +/// (`DataProposeResponse::refusal_error`). Every other failure is final here. +fn async_propose_error(vshard_id: u32, error: nodedb_cluster::ClusterError) -> crate::Error { + match error { + nodedb_cluster::ClusterError::Raft( + nodedb_raft::RaftError::LeadershipTransferInProgress + | nodedb_raft::RaftError::NotLeader { .. }, + ) => crate::Error::NoLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + }, + other => crate::Error::Internal { + detail: format!("raft propose (async): {other}"), + }, + } +} + +/// The error a proposal returns once the caller's statement deadline passed. +/// +/// The proposer carries no request id, so the error names request 0. +fn propose_deadline_exceeded() -> crate::Error { + crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + } +} + +/// How often a waiting proposer checks that this node still replicates the +/// entry's group. +const MEMBERSHIP_CHECK: std::time::Duration = std::time::Duration::from_millis(100); + +/// One proposer's wait for this node's apply of its entry. +struct LocalApplyWait<'a> { + state: &'a std::sync::Weak, + tracker: &'a ProposeTracker, + group_id: u64, + log_index: u64, + vshard_id: u32, + deadline: tokio::time::Instant, + rx: tokio::sync::oneshot::Receiver, +} + +/// Wait until this node applied the proposer's entry, and return what the +/// apply produced. +/// +/// The wait ends early when this node leaves the entry's group: a removed +/// replica receives no further entries, so it never applies the index. That +/// ends as [`crate::Error::NotLeader`] naming no leader, which sends the +/// caller to the group's current members. +/// +/// `deadline` is the caller's statement deadline, shared by every attempt. +/// Once it passes, the wait ends as [`crate::Error::DeadlineExceeded`]. The +/// pipeline stage that stalled goes to the log first. +async fn await_local_apply( + wait: LocalApplyWait<'_>, +) -> crate::Result { + let LocalApplyWait { + state, + tracker, + group_id, + log_index, + vshard_id, + deadline, + mut rx, + } = wait; + loop { + let check = tokio::time::sleep( + MEMBERSHIP_CHECK.min(deadline.saturating_duration_since(tokio::time::Instant::now())), + ); + tokio::select! { + received = &mut rx => { + return received.map_err(|_| crate::Error::Dispatch { + detail: "propose waiter channel closed".into(), + })?; + } + _ = check => {} + } + let state = state.upgrade(); + if let Some(state) = state.as_deref() + && !crate::control::security::auth_fence::cluster::hosts_group(state, group_id) + { + tracker.abandon(group_id, log_index); + return Err(crate::Error::NotLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + leader_node: 0, + leader_addr: format!( + "this node left raft group {group_id} before it applied index {log_index}" + ), + }); + } + if tokio::time::Instant::now() >= deadline { + tracker.abandon(group_id, log_index); + tracing::warn!( + group_id, + log_index, + progress = %apply_progress(state.as_deref(), group_id), + oldest_unfinished = %tracker + .applying(group_id) + .map_or_else(|| "nothing".to_owned(), |entry| entry.to_string()), + "raft proposal reached the statement deadline before this node applied it" + ); + return Err(propose_deadline_exceeded()); + } + } +} diff --git a/nodedb/src/control/cluster/start_raft_helpers.rs b/nodedb/src/control/cluster/start_raft_helpers.rs index bd62ca97c..ebcd74c32 100644 --- a/nodedb/src/control/cluster/start_raft_helpers.rs +++ b/nodedb/src/control/cluster/start_raft_helpers.rs @@ -9,7 +9,7 @@ use nodedb_cluster::vshard_handler::{DispatchTarget, dispatch_by_type}; use nodedb_cluster::wire::VShardEnvelope; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; -use crate::control::cluster::calvin::scheduler::read_applied_recovery; +use crate::control::cluster::calvin::scheduler::recover_applied; use crate::control::cluster::calvin::{ RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, SequencerProposer, @@ -159,7 +159,10 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: continue; } - let recovery = read_applied_recovery(&shared.wal, vshard_id)?; + // The applied state the last checkpoint saved, with the markers the + // WAL still holds: a checkpoint deletes the segments that held older + // markers, and the sequencer log delivers their entries again. + let recovery = recover_applied(&shared.wal, shared.credentials.catalog(), vshard_id)?; let (sequenced_tx, sequenced_rx) = tokio::sync::mpsc::channel(scheduler_config.channel_capacity); @@ -331,8 +334,17 @@ pub(super) fn spawn_vshard_schedulers( let sequencer_proposer: Arc = Arc::new(RaftSequencerProposer::new( node_id, Arc::clone(&raft_loop_handle), - Arc::clone(shared), + shared, )); + // A backup's cut proposes its marker through the same proposer. + if shared + .calvin + .sequencer_proposer + .set(Arc::clone(&sequencer_proposer)) + .is_err() + { + tracing::warn!("calvin: the sequencer proposer was set already; keeping the first"); + } // Initial reconcile: schedulers for vShards this node already knows it hosts. reconcile_vshard_schedulers(ReconcileSchedulersParams { diff --git a/nodedb/src/control/crdt_admission.rs b/nodedb/src/control/crdt_admission.rs index 726333486..f210b9cf5 100644 --- a/nodedb/src/control/crdt_admission.rs +++ b/nodedb/src/control/crdt_admission.rs @@ -525,9 +525,7 @@ async fn apply_fenced( vshard_id: workflow.vshard_id, timeout_ms: timeout_ms(workflow.timeout), })??; - workflow - .state - .advance_tenant_write_hlc(workflow.tenant_id.as_u64()); + // This node's apply of the entry recorded its commit HLC. return Ok(CrdtAdmissionOutcome { payload: outcome.0, write_version: outcome.1, @@ -777,12 +775,7 @@ mod tests { })); let seen = Arc::new(Mutex::new(Vec::new())); let policy = RecordingPolicy { seen, reject: true }; - let before_hlc = state - .tenant_write_hlc - .lock() - .expect("hlc lock") - .get(&1) - .copied(); + let before_hlc = state.tenant_write_mark(1); let result = dispatch_crdt_apply_admitted(&state, admission_request(&policy)).await; responder.await.expect("responder completes"); assert!(matches!( @@ -790,12 +783,7 @@ mod tests { Err(crate::Error::CrdtAdmissionCallerFence) )); assert_eq!( - state - .tenant_write_hlc - .lock() - .expect("hlc lock") - .get(&1) - .copied(), + state.tenant_write_mark(1), before_hlc, "policy rejection must not advance the tenant write HLC" ); @@ -827,7 +815,7 @@ mod tests { let fenced = Arc::new(Mutex::new(Vec::new())); let observed = Arc::clone(&fenced); let raw: Arc = - Arc::new(move |_shard, _key, bytes| { + Arc::new(move |_shard, _key, bytes, _deadline| { let observed = Arc::clone(&observed); Box::pin(async move { let entry = @@ -883,7 +871,7 @@ mod tests { let fences = Arc::new(AtomicUsize::new(0)); let count = Arc::clone(&fences); let raw: Arc = - Arc::new(move |_shard, _key, _bytes| { + Arc::new(move |_shard, _key, _bytes, _deadline| { let count = Arc::clone(&count); Box::pin(async move { if count.fetch_add(1, Ordering::SeqCst) == 0 { @@ -947,7 +935,7 @@ mod tests { let attempts = Arc::new(AtomicUsize::new(0)); let raw: Arc = { let attempts = Arc::clone(&attempts); - Arc::new(move |_shard, _key, _bytes| { + Arc::new(move |_shard, _key, _bytes, _deadline| { let attempts = Arc::clone(&attempts); Box::pin(async move { if attempts.fetch_add(1, Ordering::SeqCst) == 0 { @@ -1015,7 +1003,7 @@ mod tests { let fenced_count = Arc::new(AtomicUsize::new(0)); let count = Arc::clone(&fenced_count); let raw: Arc = - Arc::new(move |_shard, _key, _bytes| { + Arc::new(move |_shard, _key, _bytes, _deadline| { let count = Arc::clone(&count); Box::pin(async move { count.fetch_add(1, Ordering::SeqCst); diff --git a/nodedb/src/control/distributed_applier/applier.rs b/nodedb/src/control/distributed_applier/applier.rs index f2a4ffa96..28a61bc68 100644 --- a/nodedb/src/control/distributed_applier/applier.rs +++ b/nodedb/src/control/distributed_applier/applier.rs @@ -129,69 +129,37 @@ impl CommitApplier for DistributedApplier { } let fresh_last = fresh.last().map(|e| e.index).unwrap_or(last_index); - // Empty entries are Raft leader-transition no-ops, not user - // proposals. A waiter registered at (group_id, idx) was - // proposed by a previous leader at index `idx`; when that - // leader stepped down before the entry committed, the new - // leader's election no-op commits at the same index and - // overwrites it. The proposer's data is GONE — silently - // firing `tracker.complete(Ok([]))` here would tell the - // proposer their INSERT succeeded when in fact it was - // truncated, producing the classic "simple_query returned - // Ok but the row never appears" silent data-loss bug. - // - // Surface the truncation as an explicit error so the gateway - // / caller can retry. Idempotent re-propose is safe because - // the encoded payload carries enough identity (collection, - // PK, surrogate) for the apply path to be replayable. - for entry in &fresh { - if entry.data.is_empty() { - tracing::error!( - group_id, - log_index = entry.index, - "leader-change no-op committed at index where a proposer was waiting; \ - surfacing RetryableLeaderChange so the gateway re-proposes" - ); - // applied_key = 0 (no entry payload to derive a key - // from). The slot fires the explicit - // `RetryableLeaderChange` carried in `result`. - self.tracker.complete( - group_id, - entry.index, - 0, - Err(crate::Error::RetryableLeaderChange { - group_id, - log_index: entry.index, - }), - ); - } + // Empty entries are Raft leader-transition no-ops. They go to the apply + // loop with the rest of the batch: the loop resolves a waiter at a + // no-op's index with `RetryableLeaderChange`, and moves the applied + // watermark past the no-op only once every entry before it finished. + let batch: Vec = fresh.iter().map(|e| (*e).clone()).collect(); + let first_index = batch.first().map(|e| e.index).unwrap_or(fresh_last); + let count = batch.len(); + + // A group whose window is full waits: Raft delivers the batch again on + // a later tick. Every other group keeps its own window. + if !self.tracker.window().try_admit(group_id, count) { + debug!( + group_id, + outstanding = self.tracker.window().outstanding(group_id), + "apply window full, entries will be retried on next tick" + ); + self.release_claim(group_id, fresh_last, first_index.saturating_sub(1)); + return 0; } - let real_entries: Vec = fresh - .iter() - .filter(|e| !e.data.is_empty()) - .map(|e| (*e).clone()) - .collect(); - - let Some(first_real_index) = real_entries.first().map(|e| e.index) else { - // Nothing but no-ops, and they are now fully handled. The claim - // already covers them. - return last_index; - }; - // Push to background task. If the channel is full, log a warning // but don't block the tick loop. if let Err(e) = self.apply_tx.try_send(ApplyBatch { group_id, - entries: real_entries, + entries: batch, }) { warn!(group_id, error = %e, "apply queue full, entries will be retried on next tick"); - // Release the part of the claim that was never handed off. The - // watermark may only cover the no-ops strictly BELOW the first - // rejected entry — those were completed above and must not fire a - // second time. Everything from `first_real_index` up is re-collected - // on the next tick. - self.release_claim(group_id, fresh_last, first_real_index.saturating_sub(1)); + self.tracker.window().release(group_id, count); + // Release the claim: every entry of the batch is re-collected on + // the next tick. + self.release_claim(group_id, fresh_last, first_index.saturating_sub(1)); // Don't advance applied index — entries will be re-delivered. return 0; } @@ -270,24 +238,18 @@ mod tests { } #[test] - fn redelivered_leader_change_noop_does_not_resolve_a_waiter_twice() { - let tracker = Arc::new(ProposeTracker::new()); - let (applier, _rx) = create_distributed_applier(tracker.clone()); + fn a_leader_change_noop_is_handed_off_once() { + let (applier, mut rx) = create_distributed_applier(Arc::new(ProposeTracker::new())); let noop = vec![entry(1, b"")]; - let mut waiter = tracker.register(7, 1, 0); applier.apply_committed(7, &noop); - assert!(matches!( - waiter.try_recv(), - Ok(Err(crate::Error::RetryableLeaderChange { .. })) - )); + let batch = rx.try_recv().expect("the no-op reaches the apply loop"); + assert!(batch.entries[0].data.is_empty()); applier.apply_committed(7, &noop); - let mut probe = tracker.register(7, 1, 0); assert!( - probe.try_recv().is_err(), - "a second completion would park an orphan result on an index whose \ - waiter is already gone" + rx.try_recv().is_err(), + "a second hand-off would resolve the no-op's waiter twice" ); } @@ -312,33 +274,51 @@ mod tests { } #[test] - fn noops_below_a_rejected_entry_are_not_completed_twice() { - let tracker = Arc::new(ProposeTracker::new()); + fn a_rejected_batch_is_handed_off_whole_with_its_noops_on_retry() { let (tx, mut rx) = mpsc::channel(1); - let applier = DistributedApplier::new(tx, tracker.clone()); + let applier = DistributedApplier::new(tx, Arc::new(ProposeTracker::new())); let entries = vec![entry(2, b""), entry(3, b"x")]; applier.apply_committed(7, &[entry(1, b"a")]); - let mut waiter = tracker.register(7, 2, 0); assert_eq!(applier.apply_committed(7, &entries), 0); - assert!(matches!( - waiter.try_recv(), - Ok(Err(crate::Error::RetryableLeaderChange { .. })) - )); rx.try_recv().expect("first batch queued"); assert_eq!(applier.apply_committed(7, &entries), 3); + let batch = rx + .try_recv() + .expect("the rejected entries must be re-accepted"); + let indexes: Vec = batch.entries.iter().map(|e| e.index).collect(); + assert_eq!(indexes, vec![2, 3]); + } - let mut probe = tracker.register(7, 2, 0); - assert!( - probe.try_recv().is_err(), - "the no-op below the rejected entry was already handled" + /// A group past its apply window waits for Raft to deliver it again. A + /// different group still hands its entries off. + #[test] + fn a_group_past_its_window_waits_while_another_group_hands_off() { + let tracker = Arc::new(ProposeTracker::new()); + let (applier, mut rx) = create_distributed_applier(Arc::clone(&tracker)); + let limit = crate::control::distributed_applier::APPLY_WINDOW_PER_GROUP as u64; + let full: Vec = (1..=limit).map(|i| entry(i, b"w")).collect(); + + assert_eq!(applier.apply_committed(7, &full), limit); + rx.try_recv().expect("group 7 fills its window"); + assert_eq!( + applier.apply_committed(7, &[entry(limit + 1, b"w")]), + 0, + "a full window must not advance raft's applied index" + ); + assert_eq!(applier.apply_committed(8, &[entry(1, b"w")]), 1); + assert_eq!(rx.try_recv().expect("group 8 hands off").group_id, 8); + + tracker.window().release(7, 1); + assert_eq!( + applier.apply_committed(7, &[entry(limit + 1, b"w")]), + limit + 1 ); let batch = rx .try_recv() - .expect("the rejected entry must be re-accepted"); - let indexes: Vec = batch.entries.iter().map(|e| e.index).collect(); - assert_eq!(indexes, vec![3]); + .expect("group 7 hands off once a slot settles"); + assert_eq!(batch.entries[0].index, limit + 1); } /// Two concurrent deliveries of one group must not both claim the same diff --git a/nodedb/src/control/distributed_applier/apply_loop/array_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/array_dispatch.rs deleted file mode 100644 index a0f9a3b81..000000000 --- a/nodedb/src/control/distributed_applier/apply_loop/array_dispatch.rs +++ /dev/null @@ -1,53 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Array CRDT variants — handled on the Control Plane, bypass the Data Plane. - -use std::sync::Arc; - -use crate::control::array_sync::ArrayOpTarget; -use crate::control::array_sync::raft_apply::{ - AppliedPosition, ArraySchemaPayload, apply_array_op, apply_array_schema, -}; -use crate::control::distributed_applier::propose_tracker::ProposeTracker; -use crate::control::state::SharedState; - -/// Apply a committed `ReplicatedWrite::ArrayOp` entry. -/// -/// Advances the durable prefix only when the op durably applied — same -/// safe-watermark rule as the Data Plane write path, and the same funnel: the -/// op path submits through `submit_write`, so its redo is fsynced before it -/// reports success. A failure breaks the prefix: the entry must stay -/// replayable. -pub(super) async fn apply_array_op_entry( - state: &Arc, - tracker: &Arc, - pos: AppliedPosition, - target: ArrayOpTarget<'_>, - op_bytes: &[u8], - provenance: Option<&[u8]>, -) -> bool { - apply_array_op(state, tracker, pos, target, op_bytes, provenance).await -} - -/// Apply a committed `ReplicatedWrite::ArraySchema` entry. -/// -/// Advances the durable prefix only when the schema snapshot durably -/// imported. -/// -/// This is the one applied branch that mints no WAL redo record, and it needs -/// none: its entire effect is two fsync-committed redb transactions — the -/// schema registry's snapshot row and the array catalog's entry — both -/// written before it reports success. The floor's invariant ("this entry's -/// state survives a restart, so Raft need not redeliver it") is therefore -/// already met by the registries themselves. The cell paths have no such -/// durable store behind them: their state lives in Data-Plane memtables and -/// exists on disk only as the redo record the funnel appends, which is why -/// they must route through `submit_write`. -pub(super) fn apply_array_schema_entry( - state: &Arc, - tracker: &Arc, - pos: AppliedPosition, - payload: ArraySchemaPayload<'_>, -) -> bool { - apply_array_schema(state, tracker, pos, payload) -} diff --git a/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs b/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs index 1b4ff743a..998e8aa74 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/calvin_read_result.rs @@ -44,6 +44,7 @@ pub(super) fn forward_calvin_read_result( group_id, log_index, applied_key, + .. } = pos; let decoded_values: Vec<( diff --git a/nodedb/src/control/distributed_applier/apply_loop/context.rs b/nodedb/src/control/distributed_applier/apply_loop/context.rs new file mode 100644 index 000000000..21b27a800 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/context.rs @@ -0,0 +1,83 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What every entry's apply borrows from the loop, and the futures the loop +//! collects while it prepares later entries. + +use std::collections::BTreeMap; +use std::future::Future; +use std::pin::Pin; +use std::sync::{Arc, Mutex}; + +use tokio::sync::mpsc; + +use crate::control::cluster::calvin::ReadResultEvent; +use crate::control::distributed_applier::propose_tracker::ProposeTracker; +use crate::control::state::SharedState; + +use super::proposal_gate::EntryOutcome; + +/// Senders to each local Calvin scheduler's read-result channel, by vShard. +pub(super) type CalvinReadResultSenders = Arc>>>; + +/// The loop-owned handles an entry's apply borrows. +#[derive(Clone, Copy)] +pub(super) struct ApplyContext<'a> { + pub state: &'a Arc, + pub tracker: &'a Arc, + pub calvin_read_result_senders: &'a CalvinReadResultSenders, +} + +/// An entry whose apply finished: its waiter is resolved, and its outcome +/// waits for every earlier entry of its group before it settles. +pub(super) struct FinishedApply { + pub group_id: u64, + pub log_index: u64, + pub outcome: EntryOutcome, +} + +/// An apply the loop collects while it starts later entries. +pub(super) type ApplyFuture<'a> = Pin + Send + 'a>>; + +/// How a write continues once its enqueue returned. +pub(super) enum Started<'a> { + /// The write is on its core. The apply collects its outcome. + Running(ApplyFuture<'a>), + /// The write concluded without reaching a core. + Concluded(EntryOutcome), +} + +/// A write past its enqueue, the collection its plan named, and whether its +/// plan writes user data: only such a write raises its tenant's write mark, +/// as the write funnel decides for every write it records. +pub(super) struct StartedEntry<'a> { + pub started: Started<'a>, + pub collection: Option, + pub user_write: bool, +} + +impl StartedEntry<'_> { + pub fn concluded(outcome: EntryOutcome) -> Self { + Self { + started: Started::Concluded(outcome), + collection: None, + user_write: false, + } + } +} + +/// A write's enqueue. The next entry of its group starts once it returns. +pub(super) type EnqueueFuture<'a> = Pin> + Send + 'a>>; + +/// What the loop collects: an enqueue that returned, or an apply that +/// finished. +pub(super) enum LoopEvent<'a> { + Enqueued { + group_id: u64, + log_index: u64, + entry: StartedEntry<'a>, + }, + Finished(FinishedApply), +} + +/// A future the loop collects. +pub(super) type LoopFuture<'a> = Pin> + Send + 'a>>; diff --git a/nodedb/src/control/distributed_applier/apply_loop/driver.rs b/nodedb/src/control/distributed_applier/apply_loop/driver.rs index a82e645b3..1d515a0a6 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/driver.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/driver.rs @@ -1,33 +1,23 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Per-batch loop driver: pulls batches off the apply channel, dispatches -//! each entry to the Array CRDT / Calvin fast path or the generic write -//! path, and lands the batch's durable applied floor once every entry in it -//! has been applied. +//! Loop driver: takes batches off the apply channel into the pipeline, and +//! collects the enqueues and applies that finish, until the channel closes +//! and every started entry concluded. use std::sync::Arc; use tokio::sync::mpsc; -use crate::control::array_sync::ArrayOpTarget; -use crate::control::array_sync::raft_apply::{AppliedPosition, ArraySchemaPayload}; use crate::control::cluster::calvin::ReadResultEvent; -use crate::control::distributed_applier::applied_index::AppliedPrefix; use crate::control::distributed_applier::applier::ApplyBatch; -use crate::control::distributed_applier::propose_tracker::ProposeTracker; -use crate::control::state::SharedState; -use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite}; -use crate::types::{DatabaseId, TenantId}; - -use super::array_dispatch::{apply_array_op_entry, apply_array_schema_entry}; -use super::bookkeeping::record_durable_apply; -use super::calvin_read_result::{CalvinReadResultFields, forward_calvin_read_result}; -use super::proposal_gate::{EntryOutcome, ProposalGate}; -use super::transaction_redo::apply_transaction_redo_entry; -use super::write_dispatch::apply_generic_entry; use crate::control::distributed_applier::proposal_ledger::{ PROPOSAL_LEDGER_CAPACITY, ProposalLedger, }; +use crate::control::distributed_applier::propose_tracker::ProposeTracker; +use crate::control::state::SharedState; + +use super::context::ApplyContext; +use super::pipeline::Pipeline; /// Run the background loop that applies committed Raft entries to the local Data Plane. /// @@ -59,232 +49,41 @@ pub async fn run_apply_loop( return; } }; - let mut ledger = ProposalLedger::from_records(&records, PROPOSAL_LEDGER_CAPACITY); + let ledger = ProposalLedger::from_records(&records, PROPOSAL_LEDGER_CAPACITY); drop(records); - while let Some(batch) = apply_rx.recv().await { - // The floor is saved ONCE per batch, after the loop — never per entry. - // `save_applied_index` lands a redb transaction, and redb commits at - // `Durability::Immediate`, so a per-entry save puts one synchronous - // fsync per applied entry directly on the raft apply path. That stalls - // the raft loop hard enough to delay heartbeats and keep elections from - // stabilizing under a multi-node write load. One fsync per batch - // amortizes the cost across every entry in it and keeps the critical - // path free. - // - // `AppliedPrefix` computes WHICH index is safe to save: the highest - // contiguous successfully-applied entry, stopping at the first failure - // and never advancing past it. Every branch below must therefore report - // its outcome — `record` for the ones whose success means a durable - // redo record, `skip` for the ones that apply no durable state at all. - let mut prefix = AppliedPrefix::new(); - let mut gate = ProposalGate { - ledger: &mut ledger, - group_id: batch.group_id, - }; - for entry in &batch.entries { - // Decode once; reused for both the idempotency key and the - // Array/Calvin fast-path match below. Returns 0 for - // unparseable / pre-key entries; the tracker treats 0 as - // "no key" (no mismatch detection). - let replicated_opt = ReplicatedEntry::from_bytes(&entry.data); - let applied_key = replicated_opt - .as_ref() - .map(|e| e.idempotency_key) - .unwrap_or(0); - - // Database scope for the entry, read from the wire. `0` decodes to - // `DatabaseId::DEFAULT` (the pre-`database_id` legacy shape). The - // generic decode path (`from_replicated_entry`) returns only - // `(tenant, vshard, plan, resolved_now_ms)`, so the scope is taken - // from the entry itself — a WAL redo appended under the wrong - // database scope replays into the wrong catalog namespace. - let database_id = replicated_opt - .as_ref() - .map(|e| DatabaseId::new(e.database_id)) - .unwrap_or(DatabaseId::DEFAULT); - // A second committed copy of a proposal this node already applied - // (a re-proposal after a leader change whose first copy also - // committed) resolves its waiter with the first copy's result and - // applies nothing. - if gate.skip_duplicate(&tracker, &mut prefix, entry.index, applied_key) { - continue; + let ctx = ApplyContext { + state: &state, + tracker: &tracker, + calvin_read_result_senders: &calvin_read_result_senders, + }; + let mut pipeline = Pipeline::new(ctx, ledger); + let mut accepting = true; + loop { + if !accepting && !pipeline.has_running() { + // The channel closed and every started entry concluded. The + // pump started every entry that can start, so none is queued. + return; + } + let running = pipeline.has_running(); + tokio::select! { + biased; + Some(event) = pipeline.next_event(), if running => { + pipeline.handle(event); } - - // ── Array CRDT variants — handled on the Control Plane, bypass Data Plane ── - if let Some(replicated) = replicated_opt { - let target_vshard = replicated.vshard_id; - let pos = AppliedPosition { - group_id: batch.group_id, - log_index: entry.index, - applied_key, - }; - match replicated.write { - ReplicatedWrite::ArrayOp { - ref array, - ref op_bytes, - ref provenance, - .. - } => { - let applied_ok = apply_array_op_entry( - &state, - &tracker, - pos, - ArrayOpTarget { - tenant_id: TenantId::new(replicated.tenant_id), - database_id: DatabaseId::new(replicated.database_id), - array, - }, - op_bytes, - provenance.as_deref(), - ) - .await; - // Advance the durable prefix only when the op durably - // applied — same safe-watermark rule as the Data Plane - // write path below, and the same funnel: the op path - // submits through `submit_write`, so its redo is fsynced - // before it reports success. A failure breaks the - // prefix: the entry must stay replayable. - let outcome = EntryOutcome::Applied { - durable: applied_ok, - result: None, - }; - gate.settle(&mut prefix, entry.index, applied_key, outcome); - continue; - } - ReplicatedWrite::ArraySchema { - ref array, - ref snapshot_payload, - schema_hlc_bytes, - } => { - let applied_ok = apply_array_schema_entry( - &state, - &tracker, - pos, - ArraySchemaPayload { - tenant_id: TenantId::new(replicated.tenant_id), - database_id: DatabaseId::new(replicated.database_id), - array, - snapshot_payload, - schema_hlc_bytes, - }, - ); - // Advance the durable prefix only when the schema - // snapshot durably imported. - // - // This is the one applied branch that mints no WAL redo - // record, and it needs none: its entire effect is two - // fsync-committed redb transactions — the schema - // registry's snapshot row and the array catalog's entry - // — both written before it reports success. The floor's - // invariant ("this entry's state survives a restart, so - // Raft need not redeliver it") is therefore already met - // by the registries themselves. The cell paths have no - // such durable store behind them: their state lives in - // Data-Plane memtables and exists on disk only as the - // redo record the funnel appends, which is why they must - // route through `submit_write`. - let outcome = EntryOutcome::Applied { - durable: applied_ok, - result: None, - }; - gate.settle(&mut prefix, entry.index, applied_key, outcome); - continue; - } - ReplicatedWrite::TransactionRedo { .. } => { - let outcome = - apply_transaction_redo_entry(&state, &tracker, pos, &replicated).await; - // Advance the durable prefix when the entry's outcome is - // durable: its keyed redo record fsynced, or a final - // refusal cancelled in the WAL. - gate.settle(&mut prefix, entry.index, applied_key, outcome); - continue; + batch = apply_rx.recv(), if accepting => match batch { + Some(batch) => { + pipeline.accept(batch); + // Take every batch already queued, so one pass starts + // them all and one floor save covers them. + while let Ok(batch) = apply_rx.try_recv() { + pipeline.accept(batch); } - ReplicatedWrite::CalvinReadResult { - epoch, - position, - passive_vshard, - tenant_id, - ref values, - } => { - forward_calvin_read_result( - &tracker, - &calvin_read_result_senders, - pos, - CalvinReadResultFields { - target_vshard, - epoch, - position, - passive_vshard, - tenant_id, - values, - }, - ); - // A read result is forwarded to an in-memory Calvin - // scheduler and writes nothing durable, so it neither - // advances the prefix nor breaks it. Advancing on it - // would assert a redo record that does not exist; - // breaking on it would stall the floor behind an entry - // that a re-delivery could not usefully replay anyway — - // the epoch it belongs to does not survive a restart — - // and force every later write in the batch to be applied - // twice on the next boot. - prefix.skip(); - continue; - } - _ => {} } - } - - let outcome = apply_generic_entry( - &state, - &tracker, - batch.group_id, - entry, - applied_key, - database_id, - ) - .await; - gate.settle(&mut prefix, entry.index, applied_key, outcome); - } - - // One save + one compaction check per batch, against the contiguous - // prefix. Compaction is deliberately driven by the same index the floor - // was just saved at — never the batch's last delivered index — so it can - // never discard an entry the next boot still has to replay. Compacting - // on the raft commit index while the SPSC apply lags would likewise let - // the `SnapshotBuilder` serialize incomplete engine state and corrupt a - // lagging follower's snapshot. - if let Some(applied_index) = prefix.floor() { - record_durable_apply(&state, batch.group_id, applied_index); - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::control::distributed_applier::apply_loop::helpers::deterministic_crdt_fence_noop; - use crate::control::distributed_applier::propose_tracker::AppliedWrite; - - #[test] - fn fenced_frontier_mismatch_completes_retry_and_advances_durable_prefix() { - let result: crate::Result = Err(crate::Error::DataPlane( - crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { - expected: [1; 32], - actual: [2; 32], + None => accepting = false, }, - )); - assert!(deterministic_crdt_fence_noop(&result)); - assert!(matches!( - result, - Err(crate::Error::DataPlane( - crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { .. } - )) - )); - - let mut prefix = AppliedPrefix::new(); - prefix.record(17, deterministic_crdt_fence_noop(&result)); - assert_eq!(prefix.floor(), Some(17)); + } + pipeline.pump(); + pipeline.settle(); } } diff --git a/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs b/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs new file mode 100644 index 000000000..4bb3367d9 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs @@ -0,0 +1,82 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-group state the apply loop keeps across batches: the highest index it +//! applied, to report a second apply of a committed entry, and the backup cut +//! floor, which raises the commit HLC of every entry after a cut barrier. + +use std::collections::HashMap; + +/// Per-group apply state. +#[derive(Debug, Default)] +pub(super) struct GroupWatch { + highest_applied: HashMap, + /// Lowest commit HLC an entry after the group's latest cut barrier + /// records: one above that barrier's watermark. + cut_floor: HashMap, +} + +impl GroupWatch { + /// Note that the apply loop applies `(group_id, log_index)`. Reports an + /// index at or below one it already applied as a second apply. + pub(super) fn note_apply(&mut self, group_id: u64, log_index: u64) { + let highest = self.highest_applied.entry(group_id).or_insert(0); + if log_index <= *highest { + crate::diag::raft_entry_reapplied(group_id, log_index, *highest); + return; + } + *highest = log_index; + } + + /// Raise `group_id`'s cut floor above the watermark `cut_hlc` of a + /// backup's cut barrier. + pub(super) fn raise_cut(&mut self, group_id: u64, cut_hlc: u64) { + let floor = self.cut_floor.entry(group_id).or_insert(0); + *floor = (*floor).max(cut_hlc.saturating_add(1)); + } + + /// The commit HLC an entry of `group_id` stamped `write_hlc` records. + /// + /// An entry the log places after a cut barrier records at least the cut + /// floor, however early its proposer stamped it: the backup that placed + /// the barrier did not contain it, so a restore of that backup refuses + /// it. `0` means the entry carries no stamp; its apply stamps its own + /// append, which already follows every barrier before it. + pub(super) fn commit_hlc(&self, group_id: u64, write_hlc: u64) -> u64 { + if write_hlc == 0 { + return 0; + } + self.cut_floor + .get(&group_id) + .map_or(write_hlc, |floor| write_hlc.max(*floor)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn an_entry_after_a_cut_records_above_the_cut() { + let mut watch = GroupWatch::default(); + assert_eq!(watch.commit_hlc(1, 50), 50); + watch.raise_cut(1, 100); + assert_eq!(watch.commit_hlc(1, 50), 101); + assert_eq!(watch.commit_hlc(1, 200), 200); + assert_eq!(watch.commit_hlc(2, 50), 50, "a cut binds only its group"); + assert_eq!( + watch.commit_hlc(1, 0), + 0, + "an unstamped entry stamps its own append" + ); + } + + #[test] + fn a_second_apply_of_an_index_is_counted() { + let before = crate::diag::raft_entries_reapplied(); + let mut watch = GroupWatch::default(); + watch.note_apply(3, 5); + watch.note_apply(3, 6); + watch.note_apply(3, 6); + assert!(crate::diag::raft_entries_reapplied() > before); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/helpers.rs b/nodedb/src/control/distributed_applier/apply_loop/helpers.rs index 976e9d881..0918a69b6 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/helpers.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/helpers.rs @@ -37,3 +37,24 @@ pub(super) fn deterministic_crdt_fence_noop(result: &crate::Result )) ) } + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::distributed_applier::applied_index::AppliedPrefix; + + #[test] + fn fenced_frontier_mismatch_completes_retry_and_advances_durable_prefix() { + let result: crate::Result = Err(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::CrdtFrontierMismatch { + expected: [1; 32], + actual: [2; 32], + }, + )); + assert!(deterministic_crdt_fence_noop(&result)); + + let mut prefix = AppliedPrefix::new(); + prefix.record(17, deterministic_crdt_fence_noop(&result)); + assert_eq!(prefix.floor(), Some(17)); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/lane.rs b/nodedb/src/control/distributed_applier/apply_loop/lane.rs new file mode 100644 index 000000000..be896f36f --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/lane.rs @@ -0,0 +1,458 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One group's place in the apply pipeline: the entries not yet started, in +//! log order, and the started entries not yet settled, in log order. +//! +//! Entries start in log order, and a write's enqueue returns before the next +//! entry of its group starts, so every core receives a group's writes in the +//! order the log fixed. They finish in any order. They settle in log +//! order: the group's applied index is the highest index with every earlier +//! entry finished, and its durable floor is the highest index with every +//! earlier entry durable. + +use std::collections::VecDeque; + +use nodedb_raft::message::LogEntry; + +use crate::control::distributed_applier::applied_index::AppliedPrefix; +use crate::control::distributed_applier::propose_tracker::{ApplyingEntry, ProposeTracker}; +use crate::control::server::shared::write_admission::plan_writes_user_data; +use crate::control::state::tenant_marks::{MarkSite, TenantMarks}; +use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite, from_replicated_entry}; + +use super::proposal_gate::PrefixStep; + +/// A committed entry handed to the loop and not yet started. +pub(super) struct QueuedEntry { + pub entry: LogEntry, + /// The entry decoded once on arrival. `None` for a leader-change no-op or + /// bytes that do not decode as a replicated entry. + pub decoded: Option, +} + +impl QueuedEntry { + pub fn new(entry: LogEntry) -> Self { + let decoded = if entry.data.is_empty() { + None + } else { + ReplicatedEntry::from_bytes(&entry.data) + }; + Self { entry, decoded } + } + + /// The proposal's idempotency key, `0` when the entry carries none. + pub fn proposal_key(&self) -> u64 { + self.decoded.as_ref().map_or(0, |e| e.idempotency_key) + } + + /// The metadata index the entry's proposer had applied, `0` when the + /// entry carries none. + pub fn metadata_floor(&self) -> u64 { + self.decoded.as_ref().map_or(0, |e| e.metadata_floor) + } + + /// `(tenant_id, write_hlc)` of an entry that writes a tenant's data. A + /// cut barrier and a Calvin read result write nothing, and an entry with + /// no proposer stamp has no commit HLC to record. + pub fn write_stamp(&self) -> Option<(u64, u64)> { + let decoded = self.decoded.as_ref()?; + if decoded.write_hlc == 0 + || matches!( + decoded.write, + ReplicatedWrite::CutBarrier { .. } | ReplicatedWrite::CalvinReadResult { .. } + ) + { + return None; + } + Some((decoded.tenant_id, decoded.write_hlc)) + } + + /// Whether the entry's plan writes user data, as the write funnel decides + /// for the writes it records. Decoded here only for an entry that never + /// reaches the funnel: a second copy of an applied proposal. + pub fn plan_writes_user_data(&self) -> bool { + let Some(decoded) = self.decoded.as_ref() else { + return false; + }; + match decoded.write { + ReplicatedWrite::ArrayOp { .. } + | ReplicatedWrite::ArrayCellPut { .. } + | ReplicatedWrite::ArrayCellDelete { .. } + | ReplicatedWrite::TransactionRedo { .. } => true, + ReplicatedWrite::ArraySchema { .. } + | ReplicatedWrite::CutBarrier { .. } + | ReplicatedWrite::CalvinReadResult { .. } => false, + _ => matches!( + from_replicated_entry(&self.entry.data, None), + Ok(Some((_, _, plan, _))) if plan_writes_user_data(&plan) + ), + } + } + + /// Whether the entry must apply with nothing else of its group in + /// flight. The array paths await their own write inside the apply, so + /// the loop cannot fix their arrival order at the core any other way, + /// and a schema import must follow every earlier entry's apply. + pub fn is_exclusive(&self) -> bool { + self.decoded.as_ref().is_some_and(|e| { + matches!( + e.write, + ReplicatedWrite::ArrayOp { .. } + | ReplicatedWrite::ArraySchema { .. } + | ReplicatedWrite::ArrayCellPut { .. } + | ReplicatedWrite::ArrayCellDelete { .. } + ) + }) + } +} + +/// Where a started entry stands. +pub(super) enum SlotState { + /// The write's enqueue runs. + Starting, + /// The apply runs; its outcome arrives as a finished apply. + Running, + /// A backup's cut barrier. It completes once every earlier entry of its + /// group settled. + Barrier, + /// The entry concluded. + Concluded(PrefixStep), +} + +/// A started entry not yet settled. +pub(super) struct Slot { + pub log_index: u64, + pub proposal_key: u64, + /// The collection the entry writes, when its apply named one. + pub collection: Option, + /// `(tenant_id, commit_hlc)` the entry records on its tenant's mark in + /// this group once it settles, when it carries a proposer stamp. + pub write_mark: Option<(u64, u64)>, + /// Whether the entry's plan writes user data. Only such an entry raises + /// its tenant's mark. + pub user_write: bool, + pub state: SlotState, +} + +/// One group's apply pipeline. +pub(super) struct Lane { + group_id: u64, + pub backlog: VecDeque, + slots: VecDeque, + /// The entry the group's next start waits on: a write whose enqueue + /// runs, or an exclusive entry that runs. + pub blocking: Option, + /// The durable prefix over every entry this process settled for the + /// group. A break holds for the life of the process: an index saved past + /// a non-durable entry would let the next boot skip it. + prefix: AppliedPrefix, + /// The floor last saved for the group. + saved_floor: Option, +} + +impl Lane { + pub fn new(group_id: u64) -> Self { + Self { + group_id, + backlog: VecDeque::new(), + slots: VecDeque::new(), + blocking: None, + prefix: AppliedPrefix::new(), + saved_floor: None, + } + } + + /// Whether any started entry of the group has not concluded. + pub fn has_running(&self) -> bool { + self.slots + .iter() + .any(|slot| matches!(slot.state, SlotState::Starting | SlotState::Running)) + } + + /// Record that the enqueue of the entry at `log_index` returned: its + /// state is `state` now, its apply named `collection`, and `user_write` + /// says whether its plan writes user data. Returns `false` when no entry + /// at that index is starting. + pub fn enqueued( + &mut self, + log_index: u64, + state: SlotState, + collection: Option, + user_write: bool, + ) -> bool { + let Some(slot) = self.slot_mut(log_index) else { + return false; + }; + if !matches!(slot.state, SlotState::Starting) { + return false; + } + slot.state = state; + slot.user_write = user_write; + if collection.is_some() { + slot.collection = collection; + } + if self.blocking == Some(log_index) { + self.blocking = None; + } + true + } + + fn slot_mut(&mut self, log_index: u64) -> Option<&mut Slot> { + let position = self + .slots + .binary_search_by_key(&log_index, |slot| slot.log_index) + .ok()?; + self.slots.get_mut(position) + } + + pub fn push(&mut self, slot: Slot) { + self.slots.push_back(slot); + } + + /// Conclude the running entry at `log_index`: `conclude` receives its + /// proposal key and returns how it moves the prefix. `wrote_rows` says + /// whether the apply wrote the entry's rows; an entry that wrote none + /// raises no tenant write mark. Returns `false` when no running entry + /// has that index. + pub fn conclude( + &mut self, + log_index: u64, + wrote_rows: bool, + conclude: impl FnOnce(u64) -> PrefixStep, + ) -> bool { + let Some(slot) = self.slot_mut(log_index) else { + return false; + }; + if !matches!(slot.state, SlotState::Running) { + return false; + } + slot.user_write &= wrote_rows; + slot.state = SlotState::Concluded(conclude(slot.proposal_key)); + if self.blocking == Some(log_index) { + self.blocking = None; + } + true + } + + /// Settle every concluded entry at the front, in log order: complete a + /// barrier's waiter, raise the tenant's write mark in this group, extend + /// or break the durable prefix, and advance the applied watermark. The + /// mark rises first, so a reader that waited for the applied index sees + /// it. Returns how many entries settled. + pub fn settle(&mut self, tracker: &ProposeTracker, marks: &TenantMarks) -> usize { + let mut settled = 0; + while let Some(front) = self.slots.front() { + let step = match front.state { + SlotState::Starting | SlotState::Running => break, + SlotState::Barrier => { + // Every entry before the barrier finished; a waiting + // backup may snapshot this group now. + tracker.complete( + self.group_id, + front.log_index, + front.proposal_key, + Ok( + crate::control::distributed_applier::AppliedWrite::unversioned( + Vec::new(), + ), + ), + ); + PrefixStep::Neutral + } + SlotState::Concluded(step) => step, + }; + let log_index = front.log_index; + if front.user_write + && let Some((tenant_id, commit_hlc)) = front.write_mark + { + marks.raise( + self.group_id, + tenant_id, + commit_hlc, + MarkSite::ReplicatedApply, + front.collection.as_deref(), + ); + } + self.slots.pop_front(); + match step { + PrefixStep::Neutral => self.prefix.skip(), + PrefixStep::Record(durable) => self.prefix.record(log_index, durable), + } + tracker.note_applied(self.group_id, log_index); + settled += 1; + } + tracker.note_applying( + self.group_id, + self.slots.front().map(|slot| ApplyingEntry { + group_id: self.group_id, + log_index: slot.log_index, + collection: slot.collection.clone(), + }), + ); + settled + } + + /// Whether the durable floor moved past the floor last saved. + pub fn floor_pending(&self) -> bool { + self.prefix + .floor() + .is_some_and(|floor| self.saved_floor.is_none_or(|saved| saved < floor)) + } + + /// The durable floor to save, when it moved past the floor last saved. + pub fn take_floor_to_save(&mut self) -> Option { + let floor = self.prefix.floor()?; + if self.saved_floor.is_some_and(|saved| saved >= floor) { + return None; + } + self.saved_floor = Some(floor); + Some(floor) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn slot(log_index: u64, state: SlotState) -> Slot { + Slot { + log_index, + proposal_key: 0, + collection: None, + write_mark: None, + user_write: false, + state, + } + } + + #[test] + fn entries_settle_in_log_order_whatever_order_they_finish() { + let tracker = ProposeTracker::new(); + let mut lane = Lane::new(1); + lane.push(slot(5, SlotState::Running)); + lane.push(slot(6, SlotState::Running)); + lane.push(slot(7, SlotState::Running)); + + assert!(lane.conclude(7, true, |_| PrefixStep::Record(true))); + assert!(lane.conclude(6, true, |_| PrefixStep::Record(true))); + assert!( + !lane.conclude(6, true, |_| PrefixStep::Record(true)), + "6 concluded already" + ); + assert_eq!( + lane.settle(&tracker, &TenantMarks::default()), + 0, + "entry 5 still runs" + ); + assert_eq!(lane.take_floor_to_save(), None); + assert_eq!( + tracker.applying(1).map(|entry| entry.log_index), + Some(5), + "the timeout diagnostic names the entry the group waits behind" + ); + + assert!(lane.conclude(5, true, |_| PrefixStep::Record(true))); + assert_eq!(lane.settle(&tracker, &TenantMarks::default()), 3); + assert_eq!(lane.take_floor_to_save(), Some(7)); + assert_eq!(lane.take_floor_to_save(), None, "a saved floor saves once"); + assert!(tracker.applying(1).is_none()); + } + + #[test] + fn a_non_durable_entry_holds_the_floor_across_later_settles() { + let tracker = ProposeTracker::new(); + let mut lane = Lane::new(1); + lane.push(slot(1, SlotState::Concluded(PrefixStep::Record(true)))); + lane.push(slot(2, SlotState::Concluded(PrefixStep::Record(false)))); + lane.settle(&tracker, &TenantMarks::default()); + assert_eq!(lane.take_floor_to_save(), Some(1)); + + lane.push(slot(3, SlotState::Concluded(PrefixStep::Record(true)))); + lane.settle(&tracker, &TenantMarks::default()); + assert_eq!( + lane.take_floor_to_save(), + None, + "a floor past entry 2 would let the next boot skip it" + ); + } + + #[test] + fn an_enqueue_that_returns_unblocks_the_next_start() { + let tracker = ProposeTracker::new(); + let mut lane = Lane::new(2); + lane.push(slot(4, SlotState::Starting)); + lane.blocking = Some(4); + assert!(lane.has_running()); + + assert!(lane.enqueued(4, SlotState::Running, Some("docs".to_owned()), true)); + assert_eq!(lane.blocking, None); + lane.settle(&tracker, &TenantMarks::default()); + assert_eq!( + tracker.applying(2).and_then(|entry| entry.collection), + Some("docs".to_owned()) + ); + assert!( + !lane.enqueued(4, SlotState::Running, None, false), + "4 left its enqueue" + ); + } + + #[test] + fn a_user_write_raises_its_mark_before_the_applied_index_moves() { + let tracker = ProposeTracker::new(); + let marks = TenantMarks::default(); + let mut lane = Lane::new(4); + let mut write = slot(1, SlotState::Concluded(PrefixStep::Record(true))); + write.write_mark = Some((7, 500)); + write.user_write = true; + write.collection = Some("docs".to_owned()); + let mut index_change = slot(2, SlotState::Concluded(PrefixStep::Record(true))); + index_change.write_mark = Some((7, 900)); + lane.push(write); + lane.push(index_change); + + lane.settle(&tracker, &marks); + let mark = marks.get(4, 7).expect("the write's mark"); + assert_eq!( + mark.hlc, 500, + "an entry that writes no user data raises no mark" + ); + assert_eq!(mark.collection.as_deref(), Some("docs")); + } + + #[test] + fn a_refused_user_write_raises_no_mark() { + let tracker = ProposeTracker::new(); + let marks = TenantMarks::default(); + let mut lane = Lane::new(4); + let mut refused = slot(1, SlotState::Running); + refused.write_mark = Some((7, 500)); + refused.user_write = true; + refused.collection = Some("docs".to_owned()); + lane.push(refused); + + assert!(lane.conclude(1, false, |_| PrefixStep::Record(true))); + lane.settle(&tracker, &marks); + assert_eq!( + marks.get(4, 7), + None, + "a write that changed no row must not refuse a restore" + ); + } + + #[test] + fn a_barrier_completes_only_after_every_earlier_entry() { + let tracker = ProposeTracker::new(); + let mut lane = Lane::new(3); + lane.push(slot(1, SlotState::Running)); + lane.push(slot(2, SlotState::Barrier)); + let mut barrier = tracker.register(3, 2, 0); + + lane.settle(&tracker, &TenantMarks::default()); + assert!(barrier.try_recv().is_err(), "entry 1 still runs"); + + assert!(lane.conclude(1, true, |_| PrefixStep::Record(true))); + lane.settle(&tracker, &TenantMarks::default()); + assert!(matches!(barrier.try_recv(), Ok(Ok(_)))); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs b/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs new file mode 100644 index 000000000..d182027b2 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs @@ -0,0 +1,130 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Hold a replicated write until this node's catalog reached the one its +//! proposer planned it against. +//! +//! A data group and the metadata group apply on independent loops. Without +//! this hold, a replica can apply a write to a collection before it applied +//! the metadata entries its proposer had already applied: +//! +//! - a same-name collection's purge, whose storage reclaim then removes the +//! write's rows; +//! - the collection's creation, whose registration the write needs. +//! +//! The proposer stamps its applied metadata index on the entry +//! (`ReplicatedEntry::metadata_floor`). The hold runs before the write's +//! enqueue, so the group's later entries wait behind it in log order. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::{METADATA_GROUP_ID, WaitOutcome}; + +use crate::control::distributed_applier::propose_tracker::ProposeTracker; +use crate::control::state::SharedState; + +use super::context::{FinishedApply, StartedEntry}; +use super::proposal_gate::{EntryOutcome, ledger_outcome}; +use super::start::Prepared; + +/// How long one wait slice lasts before the hold logs that it still waits. +const WAIT_SLICE: Duration = Duration::from_secs(5); + +/// The entry a hold belongs to. +#[derive(Debug, Clone, Copy)] +pub(super) struct HeldEntry { + pub group_id: u64, + pub log_index: u64, + pub proposal_key: u64, + /// The metadata index the write waits for. `0` holds nothing. + pub metadata_floor: u64, +} + +/// Put the metadata hold in front of `prepared`'s enqueue or apply. +pub(super) fn hold_for_metadata<'a>( + state: &'a Arc, + tracker: &'a Arc, + held: HeldEntry, + prepared: Prepared<'a>, +) -> Prepared<'a> { + if held.metadata_floor == 0 + || state.applied_index_watcher(METADATA_GROUP_ID).current() >= held.metadata_floor + { + return prepared; + } + match prepared { + Prepared::Enqueue(enqueue) => Prepared::Enqueue(Box::pin(async move { + match await_metadata_floor(state, held).await { + Ok(()) => enqueue.await, + Err(error) => StartedEntry::concluded(conclude_unapplied(tracker, held, error)), + } + })), + Prepared::Exclusive(apply) => Prepared::Exclusive(Box::pin(async move { + match await_metadata_floor(state, held).await { + Ok(()) => apply.await, + Err(error) => FinishedApply { + group_id: held.group_id, + log_index: held.log_index, + outcome: conclude_unapplied(tracker, held, error), + }, + } + })), + other @ (Prepared::Concluded(_) | Prepared::Barrier) => other, + } +} + +/// Wait until this node applied the metadata group through the entry's +/// floor. The wait has no deadline: applying the write earlier breaks the +/// order it holds. It fails only when the metadata group left this node. +async fn await_metadata_floor(state: &SharedState, held: HeldEntry) -> crate::Result<()> { + let watcher = state.applied_index_watcher(METADATA_GROUP_ID); + loop { + let waiting = Arc::clone(&watcher); + let floor = held.metadata_floor; + let outcome = tokio::task::spawn_blocking(move || waiting.wait_for(floor, WAIT_SLICE)) + .await + .map_err(|e| crate::Error::Internal { + detail: format!( + "raft group {} entry {}: the metadata catch-up wait did not finish: {e}", + held.group_id, held.log_index + ), + })?; + match outcome { + WaitOutcome::Reached => return Ok(()), + WaitOutcome::TimedOut => tracing::warn!( + group_id = held.group_id, + log_index = held.log_index, + metadata_floor = floor, + metadata_applied = watcher.current(), + "a replicated write waits for this node's metadata apply to reach the \ + catalog its proposer planned it against" + ), + WaitOutcome::GroupGone => { + return Err(crate::Error::Internal { + detail: format!( + "raft group {} entry {}: the metadata group left this node before it \ + applied index {floor}, the catalog the write was planned against; \ + the write stays unapplied and replays on the next boot", + held.group_id, held.log_index + ), + }); + } + } + } +} + +/// Resolve the entry's waiter with `error`. The entry is not durable, so it +/// holds the group's applied floor and replays on the next boot. +fn conclude_unapplied( + tracker: &ProposeTracker, + held: HeldEntry, + error: crate::Error, +) -> EntryOutcome { + let result = Err(error); + let applied = ledger_outcome(&result); + tracker.complete(held.group_id, held.log_index, held.proposal_key, result); + EntryOutcome::Applied { + durable: false, + result: Some(applied), + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/mod.rs b/nodedb/src/control/distributed_applier/apply_loop/mod.rs index e89d1024a..f6566ef51 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/mod.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/mod.rs @@ -1,35 +1,50 @@ // SPDX-License-Identifier: BUSL-1.1 //! Background apply loop — reads committed Raft entries from the mpsc channel, -//! submits them through the shared Control-Plane write funnel (which appends +//! enqueues each through the shared Control-Plane write funnel (which appends //! each entry's redo record on THIS replica before the enqueue), and resolves //! propose waiters with the result. //! -//! Each batch advances the group's durable applied floor (see -//! [`super::applied_index`]) to its highest contiguous successfully-applied -//! entry, so the next boot replays only above it and no entry is applied by -//! both WAL replay and Raft log replay. +//! Entries of a group start in log order, so every core receives them in the +//! order the log fixed. Their outcomes are collected independently: a write +//! the core parks holds only its own position. Each group's applied index and +//! durable floor (see [`super::applied_index`]) advance in log order, to the +//! highest entry with every earlier entry finished and durable, so the next +//! boot replays only above the floor and no entry is applied by both WAL +//! replay and Raft log replay. //! //! Split by concern: -//! - [`driver`]: the per-batch loop that reads off the apply channel and -//! dispatches each entry, then lands the batch's durable applied floor. -//! - [`array_dispatch`]: the Array CRDT `ArrayOp` / `ArraySchema` apply path. +//! - [`driver`]: takes batches off the apply channel and collects finished +//! applies. +//! - [`pipeline`]: every group's lane and the applies that run. +//! - [`lane`]: one group's queued and started entries, settled in log order. +//! - [`start`]: prepares one entry and routes it to its apply path. +//! - [`context`]: the handles an apply borrows, and the futures the loop +//! collects. //! - [`calvin_read_result`]: forwards a committed `CalvinReadResult` entry to //! the local Calvin scheduler. -//! - [`write_dispatch`]: the generic decode + Data-Plane `submit_write` path. +//! - [`write_dispatch`]: the generic decode + write-funnel enqueue path. //! - [`transaction_redo`]: a committed transaction's redo, stamped with its //! Raft entry and applied through the WAL replay arms. //! - [`proposal_gate`]: skips a second committed copy of an applied proposal -//! and records each applied proposal in the ledger and the WAL. +//! and records each applied proposal in the ledger. +//! - [`group_watch`]: per-group second-apply detection and backup cut floors. //! - [`bookkeeping`]: applied-floor persistence + Raft log compaction trigger. //! - [`helpers`]: shared response/result classification helpers. +//! - [`metadata_floor`]: holds a write until this node's catalog reached the +//! one its proposer planned it against. -mod array_dispatch; mod bookkeeping; mod calvin_read_result; +mod context; mod driver; +mod group_watch; mod helpers; +mod lane; +mod metadata_floor; +mod pipeline; mod proposal_gate; +mod start; mod transaction_redo; mod write_dispatch; diff --git a/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs b/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs new file mode 100644 index 000000000..7f22690fe --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs @@ -0,0 +1,285 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The apply pipeline over every group this node applies. +//! +//! Each group's entries start in log order and settle in log order. A write +//! starts with its enqueue, and the next entry of its group waits for that +//! enqueue to return. Once enqueued, the write runs on its core and finishes +//! whenever its core answers. A write the core parks holds its own position +//! and no other: later entries of its group start and finish, and other +//! groups never wait on it. Only the group's applied index and durable floor +//! wait for it to finish. + +use std::collections::HashMap; + +use futures::FutureExt; +use futures::stream::{FuturesUnordered, StreamExt}; + +use crate::control::distributed_applier::applier::ApplyBatch; +use crate::control::distributed_applier::proposal_ledger::ProposalLedger; + +use super::bookkeeping::record_durable_apply; +use super::context::{ApplyContext, ApplyFuture, LoopEvent, LoopFuture, Started, StartedEntry}; +use super::group_watch::GroupWatch; +use super::lane::{Lane, QueuedEntry, Slot, SlotState}; +use super::metadata_floor::{HeldEntry, hold_for_metadata}; +use super::proposal_gate::{EntryOutcome, ProposalGate}; +use super::start::{Prepared, prepare_entry}; + +/// Every group's lane, and the enqueues and applies that run. +pub(super) struct Pipeline<'a> { + ctx: ApplyContext<'a>, + lanes: HashMap, + gate: ProposalGate, + watch: GroupWatch, + running: FuturesUnordered>, +} + +impl<'a> Pipeline<'a> { + pub fn new(ctx: ApplyContext<'a>, ledger: ProposalLedger) -> Self { + Self { + ctx, + lanes: HashMap::new(), + gate: ProposalGate::new(ledger), + watch: GroupWatch::default(), + running: FuturesUnordered::new(), + } + } + + /// Queue a batch the applier handed off, behind its group's earlier + /// entries. + pub fn accept(&mut self, batch: ApplyBatch) { + let lane = self + .lanes + .entry(batch.group_id) + .or_insert_with(|| Lane::new(batch.group_id)); + lane.backlog + .extend(batch.entries.into_iter().map(QueuedEntry::new)); + } + + /// Whether any enqueue or apply runs. + pub fn has_running(&self) -> bool { + !self.running.is_empty() + } + + /// The next enqueue to return or apply to finish. `None` when nothing + /// runs. + pub async fn next_event(&mut self) -> Option> { + self.running.next().await + } + + /// Handle `event`, then every other event that is ready already. + pub fn handle(&mut self, event: LoopEvent<'a>) { + self.handle_one(event); + while let Some(Some(event)) = self.running.next().now_or_never() { + self.handle_one(event); + } + } + + fn handle_one(&mut self, event: LoopEvent<'a>) { + let (group_id, log_index, handled) = match event { + LoopEvent::Enqueued { + group_id, + log_index, + entry, + } => ( + group_id, + log_index, + self.enqueued(group_id, log_index, entry), + ), + LoopEvent::Finished(finished) => { + let gate = &mut self.gate; + let wrote_rows = finished.outcome.wrote_rows(); + let handled = self.lanes.get_mut(&finished.group_id).is_some_and(|lane| { + lane.conclude(finished.log_index, wrote_rows, |proposal_key| { + gate.conclude(proposal_key, true, finished.outcome) + }) + }); + (finished.group_id, finished.log_index, handled) + } + }; + if !handled { + // Every enqueue and apply the pipeline runs has a slot in its + // group's lane in the matching state: the pump pushes both + // together and only this call moves them on. + tracing::error!( + group_id, + log_index, + "an apply-loop event has no matching entry in its group's lane" + ); + } + } + + fn enqueued(&mut self, group_id: u64, log_index: u64, entry: StartedEntry<'a>) -> bool { + let StartedEntry { + started, + collection, + user_write, + } = entry; + let Some(lane) = self.lanes.get_mut(&group_id) else { + return false; + }; + match started { + Started::Running(apply) => { + self.running.push(finished_event(apply)); + lane.enqueued(log_index, SlotState::Running, collection, user_write) + } + Started::Concluded(outcome) => { + // The write concluded without reaching its core. It leaves + // its enqueue and concludes in one step. + let gate = &mut self.gate; + let wrote_rows = outcome.wrote_rows(); + lane.enqueued(log_index, SlotState::Running, collection, user_write) + && lane.conclude(log_index, wrote_rows, |proposal_key| { + gate.conclude(proposal_key, true, outcome) + }) + } + } + } + + /// Start every entry each group can start now, in log order. + pub fn pump(&mut self) { + let groups: Vec = self + .lanes + .iter() + .filter(|(_, lane)| !lane.backlog.is_empty()) + .map(|(group_id, _)| *group_id) + .collect(); + for group_id in groups { + self.pump_group(group_id); + } + } + + fn pump_group(&mut self, group_id: u64) { + while let Some(queued) = self.next_startable(group_id) { + let log_index = queued.entry.index; + let proposal_key = queued.proposal_key(); + // Stamped before the entry is prepared: a barrier raises the cut + // floor only for the entries after it. + let write_mark = queued.write_stamp().map(|(tenant_id, write_hlc)| { + (tenant_id, self.watch.commit_hlc(group_id, write_hlc)) + }); + // A second copy of an applied proposal never reaches the funnel, + // so its plan is classified here. Its first copy may sit above + // the saved floor, and this copy then carries the mark again. + let repeat_writes = + self.gate.prior_wrote_rows(proposal_key) && queued.plan_writes_user_data(); + let held = HeldEntry { + group_id, + log_index, + proposal_key, + metadata_floor: queued.metadata_floor(), + }; + let prepared = hold_for_metadata( + self.ctx.state, + self.ctx.tracker, + held, + prepare_entry(self.ctx, &mut self.watch, &self.gate, group_id, queued), + ); + let (state, blocks, user_write) = match prepared { + Prepared::Concluded(outcome) => { + let user_write = matches!(outcome, EntryOutcome::Repeat) && repeat_writes; + ( + SlotState::Concluded(self.gate.conclude(proposal_key, false, outcome)), + false, + user_write, + ) + } + Prepared::Barrier => (SlotState::Barrier, false, false), + Prepared::Enqueue(enqueue) => { + self.gate.open(proposal_key); + self.running + .push(Box::pin(enqueue.map(move |entry| LoopEvent::Enqueued { + group_id, + log_index, + entry, + }))); + // The enqueue reports whether the plan writes user data. + (SlotState::Starting, true, false) + } + Prepared::Exclusive(apply) => { + // An array op or cell write: user data. + self.gate.open(proposal_key); + self.running.push(finished_event(apply)); + (SlotState::Running, true, true) + } + }; + let Some(lane) = self.lanes.get_mut(&group_id) else { + return; + }; + if blocks { + lane.blocking = Some(log_index); + } + lane.push(Slot { + log_index, + proposal_key, + collection: None, + write_mark, + user_write, + state, + }); + } + } + + /// Take the next entry of `group_id` when it may start now. + /// + /// It waits while the group's previous write is in its enqueue or an + /// exclusive entry of the group runs, while it is exclusive and an + /// earlier entry of the group has not concluded, and while a copy of its + /// proposal runs: the ledger decides it once that copy concludes. + fn next_startable(&mut self, group_id: u64) -> Option { + let lane = self.lanes.get_mut(&group_id)?; + if lane.blocking.is_some() { + return None; + } + let front = lane.backlog.front()?; + if front.is_exclusive() && lane.has_running() { + return None; + } + if self.gate.in_flight(front.proposal_key()) { + return None; + } + lane.backlog.pop_front() + } + + /// Settle every group's concluded entries in log order, release their + /// window, and save each durable floor that moved. + /// + /// The tenant write marks of the settled entries are persisted before any + /// floor that covers them. An entry above the saved floor is delivered + /// again after a restart and records its mark again, so every committed + /// write keeps a durable mark. + pub fn settle(&mut self) { + let state = self.ctx.state; + let tracker = self.ctx.tracker; + for (group_id, lane) in &mut self.lanes { + let settled = lane.settle(tracker, &state.tenant_marks); + if settled > 0 { + tracker.window().release(*group_id, settled); + } + if !lane.floor_pending() { + continue; + } + // One save per pass that moved the floor, never one per entry: + // each save is an fsync, and the pass coalesces every apply that + // finished before it. + if let Err(error) = state.tenant_marks.persist(state.credentials.catalog()) { + // The floor stays where it is, so every entry above it keeps + // its place in the log. The next pass persists and saves again. + tracing::error!( + group_id = *group_id, + %error, + "apply loop: tenant write marks did not persist; the applied floor waits" + ); + continue; + } + if let Some(floor) = lane.take_floor_to_save() { + record_durable_apply(state, *group_id, floor); + } + } + } +} + +fn finished_event(apply: ApplyFuture<'_>) -> LoopFuture<'_> { + Box::pin(apply.map(LoopEvent::Finished)) +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs b/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs index 1d7f4c0a4..2d1325ee5 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs @@ -3,8 +3,13 @@ //! Per-entry proposal identity gate: skip a second committed copy of a //! proposal, and record every durably applied proposal in the ledger. See //! [`crate::control::distributed_applier::proposal_ledger`]. +//! +//! Entries finish out of order, so the gate also knows which proposals are +//! in flight. A second copy of an in-flight proposal waits until the first +//! copy concludes, then the ledger decides it like any other copy. + +use std::collections::HashMap; -use crate::control::distributed_applier::applied_index::AppliedPrefix; use crate::control::distributed_applier::proposal_ledger::{ AppliedOutcome, PriorApply, ProposalLedger, }; @@ -15,6 +20,10 @@ pub(super) enum EntryOutcome { /// The entry carries no durable state and no proposal to record: it /// neither advances nor breaks the applied prefix. Skipped, + /// A second committed copy of a proposal this node already applied. Its + /// first copy's outcome is durable, so it extends the prefix. The ledger + /// already holds the proposal. + Repeat, /// The entry was applied. `durable` says its outcome survives a restart. /// `result` is what its waiter received, when the apply produced one. Applied { @@ -23,6 +32,36 @@ pub(super) enum EntryOutcome { }, } +impl EntryOutcome { + /// Whether the apply wrote the entry's rows. A refused or failed apply + /// wrote none, so it raises no tenant write mark. A second copy writes + /// nothing itself: its mark follows its first copy (see + /// [`ProposalGate::prior_wrote_rows`]). + pub fn wrote_rows(&self) -> bool { + match self { + Self::Skipped | Self::Repeat => false, + Self::Applied { + result: Some(outcome), + .. + } => outcome.is_ok(), + // An apply with no waiter result reports its success as `durable`. + Self::Applied { + durable, + result: None, + } => *durable, + } + } +} + +/// How a concluded entry moves its group's durable applied prefix. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum PrefixStep { + /// Neither advances nor breaks the prefix. + Neutral, + /// Extends the prefix when `true`, breaks it when `false`. + Record(bool), +} + /// The outcome a waiter received, in the form the ledger keeps. A refusal is /// kept as its typed code; an error with no code keeps its message. pub(super) fn ledger_outcome(result: &crate::Result) -> AppliedOutcome { @@ -35,22 +74,48 @@ pub(super) fn ledger_outcome(result: &crate::Result) -> AppliedOut } } -/// Per-batch state of the proposal gate. -pub(super) struct ProposalGate<'a> { - pub ledger: &'a mut ProposalLedger, - pub group_id: u64, +/// The proposal ledger, plus the proposals whose apply has not concluded. +pub(super) struct ProposalGate { + ledger: ProposalLedger, + /// Keyed proposals started and not concluded, with their copy counts. + in_flight: HashMap, } -impl ProposalGate<'_> { +impl ProposalGate { + pub fn new(ledger: ProposalLedger) -> Self { + Self { + ledger, + in_flight: HashMap::new(), + } + } + + /// Whether a copy of `proposal_key` is started and not concluded. Key `0` + /// names no proposal and is never in flight. + pub fn in_flight(&self, proposal_key: u64) -> bool { + proposal_key != 0 && self.in_flight.contains_key(&proposal_key) + } + + /// Whether the first copy of `proposal_key` applied on this node and + /// wrote its rows. A key recovered from the WAL keeps no outcome, so it + /// counts as written: a mark too high only refuses a restore that FORCE + /// can override, and a mark too low lets a restore overwrite the write. + pub fn prior_wrote_rows(&self, proposal_key: u64) -> bool { + match self.ledger.prior(proposal_key) { + Some(PriorApply::Outcome(outcome)) => outcome.is_ok(), + Some(PriorApply::NoOutcome) => true, + None => false, + } + } + /// When `proposal_key` already applied on this node, resolve the entry's - /// waiter with the first copy's outcome, extend the prefix, and return - /// `true`: the caller skips the entry. The first copy's outcome is - /// durable: the record that carries its key was durable before the floor - /// passed it, or it applied in this process. + /// waiter with the first copy's outcome and return `true`: the caller + /// skips the entry. The first copy's outcome is durable: the record that + /// carries its key was durable before the floor passed it, or it applied + /// in this process. pub fn skip_duplicate( &self, tracker: &ProposeTracker, - prefix: &mut AppliedPrefix, + group_id: u64, log_index: u64, proposal_key: u64, ) -> bool { @@ -63,38 +128,97 @@ impl ProposalGate<'_> { PriorApply::NoOutcome => Ok(AppliedWrite::unversioned(Vec::new())), }; tracing::debug!( - group_id = self.group_id, + group_id, log_index, proposal_key, "skipping a second committed copy of an applied proposal" ); - tracker.complete(self.group_id, log_index, proposal_key, result); - prefix.record(log_index, true); + tracker.complete(group_id, log_index, proposal_key, result); true } - /// Record `outcome` for the entry at `log_index`: extend or break the - /// prefix, and note a durable apply of a keyed proposal in the ledger. + /// Note that a copy of `proposal_key` started and has not concluded. + pub fn open(&mut self, proposal_key: u64) { + if proposal_key != 0 { + *self.in_flight.entry(proposal_key).or_insert(0) += 1; + } + } + + /// Conclude an entry of `proposal_key` with `outcome`: note a durable + /// apply in the ledger, close the copy `open` noted when `opened`, and + /// return how the entry moves its group's prefix. /// /// No separate marker is written: the entry's own records carry its key, /// durable in the same write as its effect. - pub fn settle( + pub fn conclude( &mut self, - prefix: &mut AppliedPrefix, - log_index: u64, proposal_key: u64, + opened: bool, outcome: EntryOutcome, - ) { - let (durable, result) = match outcome { - EntryOutcome::Skipped => { - prefix.skip(); - return; + ) -> PrefixStep { + if opened + && proposal_key != 0 + && let Some(copies) = self.in_flight.get_mut(&proposal_key) + { + *copies -= 1; + if *copies == 0 { + self.in_flight.remove(&proposal_key); } - EntryOutcome::Applied { durable, result } => (durable, result), - }; - if durable { - self.ledger.note(proposal_key, result); } - prefix.record(log_index, durable); + match outcome { + EntryOutcome::Skipped => PrefixStep::Neutral, + EntryOutcome::Repeat => PrefixStep::Record(true), + EntryOutcome::Applied { durable, result } => { + if durable { + self.ledger.note(proposal_key, result); + } + PrefixStep::Record(durable) + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::distributed_applier::proposal_ledger::PROPOSAL_LEDGER_CAPACITY; + + #[test] + fn a_copy_of_an_in_flight_proposal_is_decided_once_the_first_concludes() { + let tracker = ProposeTracker::new(); + let mut gate = ProposalGate::new(ProposalLedger::new(PROPOSAL_LEDGER_CAPACITY)); + gate.open(42); + assert!(gate.in_flight(42)); + assert!(!gate.skip_duplicate(&tracker, 1, 9, 42)); + + let step = gate.conclude( + 42, + true, + EntryOutcome::Applied { + durable: true, + result: Some(Ok(AppliedWrite::unversioned(Vec::new()))), + }, + ); + assert_eq!(step, PrefixStep::Record(true)); + assert!(!gate.in_flight(42)); + assert!(gate.skip_duplicate(&tracker, 1, 9, 42)); + } + + #[test] + fn a_failed_first_copy_leaves_the_second_copy_to_apply() { + let tracker = ProposeTracker::new(); + let mut gate = ProposalGate::new(ProposalLedger::new(PROPOSAL_LEDGER_CAPACITY)); + gate.open(7); + let step = gate.conclude( + 7, + true, + EntryOutcome::Applied { + durable: false, + result: None, + }, + ); + assert_eq!(step, PrefixStep::Record(false)); + assert!(!gate.in_flight(7)); + assert!(!gate.skip_duplicate(&tracker, 1, 9, 7)); } } diff --git a/nodedb/src/control/distributed_applier/apply_loop/start.rs b/nodedb/src/control/distributed_applier/apply_loop/start.rs new file mode 100644 index 000000000..26b3764d1 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/start.rs @@ -0,0 +1,207 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Prepare one committed entry, in log order: note it, skip a second copy of +//! an applied proposal, and route it to its apply path. +//! +//! Preparing never waits. A write leaves here as its enqueue, and the next +//! entry of its group waits for that enqueue to return, so every core +//! receives a group's writes in the order the log fixed. Other groups never +//! wait on it. + +use crate::control::array_sync::ArrayOpTarget; +use crate::control::array_sync::raft_apply::{ + AppliedPosition, ArraySchemaPayload, apply_array_op, apply_array_schema, +}; +use crate::control::wal_replication::ReplicatedWrite; +use crate::types::{DatabaseId, TenantId}; + +use super::calvin_read_result::{CalvinReadResultFields, forward_calvin_read_result}; +use super::context::{ApplyContext, ApplyFuture, EnqueueFuture, FinishedApply}; +use super::group_watch::GroupWatch; +use super::lane::QueuedEntry; +use super::proposal_gate::{EntryOutcome, ProposalGate}; +use super::transaction_redo::prepare_transaction_redo_entry; +use super::write_dispatch::prepare_generic_entry; + +/// How a prepared entry continues. +pub(super) enum Prepared<'a> { + /// The entry concluded while it was prepared. + Concluded(EntryOutcome), + /// A backup's cut barrier: it completes once every earlier entry of its + /// group settled. + Barrier, + /// The write's enqueue. The next entry of the group starts once it + /// returns. + Enqueue(EnqueueFuture<'a>), + /// An apply that awaits its own write. It starts with nothing else of its + /// group running, and the next entry starts once it finishes. + Exclusive(ApplyFuture<'a>), +} + +/// Prepare `queued`, the next entry of `group_id` in log order. +pub(super) fn prepare_entry<'a>( + ctx: ApplyContext<'a>, + watch: &mut GroupWatch, + gate: &ProposalGate, + group_id: u64, + queued: QueuedEntry, +) -> Prepared<'a> { + let QueuedEntry { entry, decoded } = queued; + let log_index = entry.index; + watch.note_apply(group_id, log_index); + + // A leader-change no-op committed where a proposer may wait. The + // proposer's data is gone; firing an empty success would tell it the + // write applied. `RetryableLeaderChange` makes the gateway re-propose. + if entry.data.is_empty() { + tracing::error!( + group_id, + log_index, + "leader-change no-op committed at index where a proposer was waiting; \ + surfacing RetryableLeaderChange so the gateway re-proposes" + ); + ctx.tracker.complete( + group_id, + log_index, + 0, + Err(crate::Error::RetryableLeaderChange { + group_id, + log_index, + }), + ); + return Prepared::Concluded(EntryOutcome::Skipped); + } + + // `0` for unparseable / pre-key entries; the tracker treats 0 as "no key" + // (no mismatch detection). + let applied_key = decoded.as_ref().map_or(0, |e| e.idempotency_key); + // The proposer's commit stamp, raised above any backup cut the log placed + // before this entry. The entry's mark carries it, not the instant this + // replica applies, so a late apply never records a write as newer than a + // backup taken after its ack. + let commit_hlc = watch.commit_hlc(group_id, decoded.as_ref().map_or(0, |e| e.write_hlc)); + // Database scope for the entry, read from the wire. The generic decode + // path returns no scope, so it is taken from the entry itself: a redo + // appended under the wrong scope replays into the wrong namespace. + let database_id = decoded + .as_ref() + .map_or(DatabaseId::DEFAULT, |e| DatabaseId::new(e.database_id)); + + // A second committed copy of a proposal this node already applied (a + // re-proposal after a leader change whose first copy also committed) + // resolves its waiter with the first copy's result and applies nothing. + if gate.skip_duplicate(ctx.tracker, group_id, log_index, applied_key) { + return Prepared::Concluded(EntryOutcome::Repeat); + } + + let pos = AppliedPosition { + group_id, + log_index, + applied_key, + commit_hlc, + }; + let Some(replicated) = decoded else { + return prepare_generic_entry(ctx, pos, entry, database_id, false); + }; + let tenant_id = TenantId::new(replicated.tenant_id); + let entry_database = DatabaseId::new(replicated.database_id); + match replicated.write { + ReplicatedWrite::ArrayOp { + array, + op_bytes, + provenance, + .. + } => { + // The op path submits through the write funnel, so its redo is + // durable before it reports success. A failure breaks the + // prefix: the entry must stay replayable. + Prepared::Exclusive(Box::pin(async move { + let applied_ok = apply_array_op( + ctx.state, + ctx.tracker, + pos, + ArrayOpTarget { + tenant_id, + database_id: entry_database, + array: &array, + }, + &op_bytes, + provenance.as_deref(), + ) + .await; + FinishedApply { + group_id, + log_index, + outcome: EntryOutcome::Applied { + durable: applied_ok, + result: None, + }, + } + })) + } + ReplicatedWrite::ArraySchema { + ref array, + ref snapshot_payload, + schema_hlc_bytes, + } => { + // The one applied branch that mints no WAL redo record, and it + // needs none: its whole effect is two fsync-committed redb + // transactions, the schema registry's snapshot row and the array + // catalog's entry, both written before it reports success. + let applied_ok = apply_array_schema( + ctx.state, + ctx.tracker, + pos, + ArraySchemaPayload { + tenant_id, + database_id: entry_database, + array, + snapshot_payload, + schema_hlc_bytes, + }, + ); + Prepared::Concluded(EntryOutcome::Applied { + durable: applied_ok, + result: None, + }) + } + ReplicatedWrite::ArrayCellPut { .. } | ReplicatedWrite::ArrayCellDelete { .. } => { + prepare_generic_entry(ctx, pos, entry, database_id, true) + } + ReplicatedWrite::TransactionRedo { .. } => { + prepare_transaction_redo_entry(ctx, pos, &replicated) + } + ReplicatedWrite::CutBarrier { hlc } => { + // Every entry after the barrier records above the cut. + watch.raise_cut(group_id, hlc); + Prepared::Barrier + } + ReplicatedWrite::CalvinReadResult { + epoch, + position, + passive_vshard, + tenant_id, + ref values, + } => { + forward_calvin_read_result( + ctx.tracker, + ctx.calvin_read_result_senders, + pos, + CalvinReadResultFields { + target_vshard: replicated.vshard_id, + epoch, + position, + passive_vshard, + tenant_id, + values, + }, + ); + // A read result is forwarded to an in-memory Calvin scheduler and + // writes nothing durable, so it neither advances the prefix nor + // breaks it. The epoch it belongs to does not survive a restart, + // so a re-delivery could not usefully replay it. + Prepared::Concluded(EntryOutcome::Skipped) + } + _ => prepare_generic_entry(ctx, pos, entry, database_id, false), + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs b/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs index e33b63a1b..c33e7d669 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs @@ -1,7 +1,9 @@ +// SPDX-License-Identifier: BUSL-1.1 + //! Apply path for a committed `ReplicatedWrite::TransactionRedo` entry. //! //! Every replica, the proposer included, applies the entry through -//! [`apply_transaction_redo`]: the redo record is appended to this node's WAL, +//! [`enqueue_transaction_redo`]: the redo record is appended to this node's WAL, //! its header carrying the entry's idempotency key, and installed through the //! WAL replay arms. The key makes the record the entry's applied-marker, so an //! entry re-delivered after a restart is recognised by the proposal ledger and @@ -14,30 +16,30 @@ //! refusal as the entry's outcome. It advances the durable prefix like a //! success, because replaying the entry can only refuse it again. -use std::sync::Arc; - use crate::bridge::envelope::Status; use crate::control::array_sync::raft_apply::AppliedPosition; use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; -use crate::control::server::dispatch_utils::refusal_is_final; -use crate::control::state::SharedState; +use crate::control::server::dispatch_utils::{SubmitOutcome, refusal_is_final}; use crate::control::wal_replication::ReplicatedEntry; use crate::control::wal_replication::decode::transaction_redo_payload; -use crate::control::wal_replication::transaction_redo::{RedoTarget, apply_transaction_redo}; +use crate::control::wal_replication::transaction_redo::{RedoTarget, enqueue_transaction_redo}; use crate::types::{DatabaseId, TenantId, VShardId}; +use super::context::{ApplyContext, FinishedApply, Started, StartedEntry}; use super::helpers::committed_response_result; use super::proposal_gate::{EntryOutcome, ledger_outcome}; +use super::start::Prepared; -/// Apply one committed `TransactionRedo` entry and resolve its propose -/// waiter. The outcome says whether the entry's effect is durable on this -/// node, which is what the caller's applied prefix records. -pub(super) async fn apply_transaction_redo_entry( - state: &Arc, - tracker: &Arc, +/// Prepare one committed `TransactionRedo` entry. Its enqueue appends the +/// record and hands it to its core. The apply that follows resolves the +/// propose waiter. Its outcome says whether the entry's effect is durable on +/// this node, which is what the group's applied prefix records. +pub(super) fn prepare_transaction_redo_entry<'a>( + ctx: ApplyContext<'a>, pos: AppliedPosition, entry: &ReplicatedEntry, -) -> EntryOutcome { +) -> Prepared<'a> { + let ApplyContext { state, tracker, .. } = ctx; let payload = match transaction_redo_payload(&entry.write) { Ok(payload) => payload, Err(error) => { @@ -45,10 +47,10 @@ pub(super) async fn apply_transaction_redo_entry( // The entry's own bytes are malformed; a re-delivery decodes the // same bytes and fails the same way, so it holds the floor rather // than skipping a committed transaction. - return EntryOutcome::Applied { + return Prepared::Concluded(EntryOutcome::Applied { durable: false, result: None, - }; + }); } }; let target = RedoTarget { @@ -56,8 +58,42 @@ pub(super) async fn apply_transaction_redo_entry( database_id: DatabaseId::new(entry.database_id), vshard_id: VShardId::new(entry.vshard_id), }; - let submitted = apply_transaction_redo(state, target, &payload, pos.applied_key).await; + Prepared::Enqueue(Box::pin(async move { + let collection = payload.collections.first().cloned(); + let enqueued = enqueue_transaction_redo( + state, + target, + &payload, + pos.applied_key, + pos.carried_commit_hlc(), + ) + .await; + let started = match enqueued { + Ok(pending) => Started::Running(Box::pin(async move { + let submitted = pending.finish(state).await; + FinishedApply { + group_id: pos.group_id, + log_index: pos.log_index, + outcome: conclude_transaction_redo(tracker, pos, submitted), + } + })), + Err(error) => Started::Concluded(conclude_transaction_redo(tracker, pos, Err(error))), + }; + // A committed transaction's redo carries the rows it wrote. + StartedEntry { + started, + collection, + user_write: true, + } + })) +} +/// Resolve a transaction redo's waiter from what the funnel returned. +fn conclude_transaction_redo( + tracker: &ProposeTracker, + pos: AppliedPosition, + submitted: crate::Result, +) -> EntryOutcome { let (result, durable) = match submitted { Ok(outcome) if outcome.response.status == Status::Ok => { (Ok(AppliedWrite::from_response(&outcome.response)), true) diff --git a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs index a80239efa..3b1639139 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs @@ -2,9 +2,10 @@ //! Generic per-entry apply path: decode the replicated entry, route //! Raft-native array cell writes through the array-open bootstrap, and -//! dispatch everything else through the shared Control-Plane write funnel. - -use std::sync::Arc; +//! enqueue everything else through the shared Control-Plane write funnel. +//! +//! The enqueue runs in log order when the entry starts. The outcome is +//! collected by the returned apply, in any order. use tracing::debug; @@ -17,28 +18,70 @@ use crate::control::array_sync::raft_apply::{ }; use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; use crate::control::server::dispatch_utils::{ - ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, error_is_final_refusal, - submit_write, + ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, enqueue_write, + error_is_final_refusal, }; -use crate::control::state::SharedState; use crate::control::wal_replication::from_replicated_entry; use crate::types::{DatabaseId, TraceId}; +use crate::control::server::shared::write_admission::plan_writes_user_data; + +use super::context::{ApplyContext, FinishedApply, Started, StartedEntry}; use super::helpers::{committed_response_result, deterministic_crdt_fence_noop}; use super::proposal_gate::{EntryOutcome, ledger_outcome}; +use super::start::Prepared; -/// Decode `entry` and apply it: Raft-native array cell writes route through -/// the array-open bootstrap and the write funnel; everything else dispatches -/// through the write funnel directly. Returns the outcome the caller records -/// into its applied prefix. -pub(super) async fn apply_generic_entry( - state: &Arc, - tracker: &Arc, - group_id: u64, - entry: &LogEntry, - applied_key: u64, +/// Prepare a generic entry. `exclusive` marks a Raft-native array cell write: +/// its apply awaits the array-open bootstrap and its own write, so it runs +/// with nothing else of its group in flight. Every other entry leaves as its +/// enqueue. +pub(super) fn prepare_generic_entry<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + entry: LogEntry, database_id: DatabaseId, -) -> EntryOutcome { + exclusive: bool, +) -> Prepared<'a> { + if !exclusive { + return Prepared::Enqueue(Box::pin(enqueue_generic_entry( + ctx, + pos, + entry, + database_id, + ))); + } + Prepared::Exclusive(Box::pin(async move { + let outcome = match enqueue_generic_entry(ctx, pos, entry, database_id) + .await + .started + { + Started::Running(apply) => return apply.await, + Started::Concluded(outcome) => outcome, + }; + FinishedApply { + group_id: pos.group_id, + log_index: pos.log_index, + outcome, + } + })) +} + +/// Decode `entry` and enqueue it: Raft-native array cell writes route through +/// the array-open bootstrap and the write funnel, and conclude here; everything +/// else is enqueued through the write funnel directly. +async fn enqueue_generic_entry<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + entry: LogEntry, + database_id: DatabaseId, +) -> StartedEntry<'a> { + let ApplyContext { state, tracker, .. } = ctx; + let AppliedPosition { + group_id, + log_index, + applied_key, + .. + } = pos; let decoded = from_replicated_entry(&entry.data, Some(state.surrogate_assigner.as_ref())); let (tenant_id, vshard_id, plan, resolved_now_ms) = match decoded { Ok(Some(t)) => t, @@ -61,7 +104,7 @@ pub(super) async fn apply_generic_entry( // it buys nothing and costs a double-apply of every later // write in the batch. It applied no state, so it must not // advance the floor either. - return EntryOutcome::Skipped; + return StartedEntry::concluded(EntryOutcome::Skipped); } Err(e) => { tracing::warn!( @@ -83,10 +126,10 @@ pub(super) async fn apply_generic_entry( // state rather than on its own bytes, so a re-delivery can // legitimately succeed. Holding the floor below it is what // keeps it replayable. - return EntryOutcome::Applied { + return StartedEntry::concluded(EntryOutcome::Applied { durable: false, result: None, - }; + }); } }; @@ -97,7 +140,7 @@ pub(super) async fn apply_generic_entry( // write funnel as the generic branch below, which is what gives them // a redo record and the fsync the applied floor asserts. No other // `ReplicatedWrite` variant decodes to a `PhysicalPlan::Array`, so - // this match is exact. + // this match is exact, and the caller runs them as exclusive entries. if matches!( plan, PhysicalPlan::Array(ArrayOp::Put { .. } | ArrayOp::Delete { .. }) @@ -105,11 +148,7 @@ pub(super) async fn apply_generic_entry( let applied_ok = apply_array_cell_write( state, tracker, - AppliedPosition { - group_id, - log_index: entry.index, - applied_key, - }, + pos, ArrayCellTarget { tenant_id, database_id, @@ -119,13 +158,27 @@ pub(super) async fn apply_generic_entry( plan, ) .await; - return EntryOutcome::Applied { + return StartedEntry::concluded(EntryOutcome::Applied { durable: applied_ok, result: None, - }; + }); } - let submitted = submit_write( + let collection = plan + .named_collections() + .first() + .map(|collection| (*collection).to_owned()); + let user_write = plan_writes_user_data(&plan); + debug!( + group_id, + log_index, + tenant_id = tenant_id.as_u64(), + vshard_id = vshard_id.as_u32(), + collection = collection.as_deref().unwrap_or(""), + user_write, + "applying a committed write entry" + ); + let enqueued = enqueue_write( state, SubmitWrite { tenant_id, @@ -154,6 +207,7 @@ pub(super) async fn apply_generic_entry( durability: WalDurability::AppendHere { now_override: resolved_now_ms, apply_key: applied_key, + commit_hlc: pos.carried_commit_hlc(), }, // Raft committed this entry at a fixed log index; every // replica applies it in that order. Re-entering the @@ -168,8 +222,38 @@ pub(super) async fn apply_generic_entry( change_feed: ChangeFeedOwner::Unowned, }, ) - .await - .map(|outcome| outcome.response); + .await; + let started = match enqueued { + Ok(pending) => Started::Running(Box::pin(async move { + let submitted = pending.finish(state).await.map(|outcome| outcome.response); + FinishedApply { + group_id, + log_index, + outcome: conclude_generic_entry(tracker, pos, submitted), + } + })), + Err(error) => Started::Concluded(conclude_generic_entry(tracker, pos, Err(error))), + }; + StartedEntry { + started, + collection, + user_write, + } +} + +/// Resolve a generic entry's waiter from what the funnel returned, and report +/// the outcome its group's prefix records. +fn conclude_generic_entry( + tracker: &ProposeTracker, + pos: AppliedPosition, + submitted: crate::Result, +) -> EntryOutcome { + let AppliedPosition { + group_id, + log_index, + applied_key, + .. + } = pos; // The funnel returns an error-status response as `Ok`; a committed // entry that failed to apply must surface to the propose waiter as a @@ -186,7 +270,7 @@ pub(super) async fn apply_generic_entry( Err(e) => { tracing::warn!( group_id, - index = entry.index, + index = log_index, error = %e, "applying committed write failed" ); @@ -201,11 +285,11 @@ pub(super) async fn apply_generic_entry( || deterministic_crdt_fence_noop(&result) || result.as_ref().is_err_and(error_is_final_refusal); let applied = ledger_outcome(&result); - tracker.complete(group_id, entry.index, applied_key, result); + tracker.complete(group_id, log_index, applied_key, result); - // Extend the batch's durable prefix. On success `submit_write`'s + // Extend the group's durable prefix. On success the funnel's // durable-at-ack barrier has already fsynced this entry's redo, - // which is exactly the fact the floor asserts — `entry.index` is + // which is exactly the fact the floor asserts — `log_index` is // the data-plane applied watermark here, NOT raft's commit index. // On failure the engines did not persist this index, so it is // neither a safe compaction boundary nor a safe restart floor; diff --git a/nodedb/src/control/distributed_applier/apply_window.rs b/nodedb/src/control/distributed_applier/apply_window.rs new file mode 100644 index 000000000..4e68802ca --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_window.rs @@ -0,0 +1,101 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-group bound on committed entries between hand-off and settle. +//! +//! The apply loop enqueues a group's entries in log order and collects each +//! outcome independently, so a parked write holds its own position and no +//! other. The window bounds how many of a group's entries the loop holds at +//! once. The applier refuses a batch that would pass the bound, and Raft +//! delivers it again on a later tick. Only the saturated group waits: every +//! other group keeps its own window. + +use std::collections::HashMap; +use std::sync::Mutex; + +use crate::bridge::dispatch::DATA_PLANE_QUEUE_CAPACITY; + +/// Entries of one group the apply loop may hold between hand-off and settle. +/// +/// One core's request queue. A group's writes spread over the cores that own +/// its vShards, so a window of one queue keeps every one of those cores busy. +/// A larger window only adds writes parked behind a full queue. +pub const APPLY_WINDOW_PER_GROUP: usize = DATA_PLANE_QUEUE_CAPACITY; + +/// Outstanding entry counts per group. +#[derive(Debug)] +pub struct ApplyWindow { + limit: usize, + outstanding: Mutex>, +} + +impl Default for ApplyWindow { + fn default() -> Self { + Self::new(APPLY_WINDOW_PER_GROUP) + } +} + +impl ApplyWindow { + /// A window of `limit` entries per group. + pub fn new(limit: usize) -> Self { + Self { + limit, + outstanding: Mutex::new(HashMap::new()), + } + } + + /// Take `count` entries of `group_id` into the window. Refuses when the + /// group holds entries and `count` more would pass the limit. A group that + /// holds none always takes the batch, so a batch longer than the limit + /// still applies. + pub fn try_admit(&self, group_id: u64, count: usize) -> bool { + let mut outstanding = self.outstanding.lock().unwrap_or_else(|p| p.into_inner()); + let held = outstanding.entry(group_id).or_insert(0); + if *held > 0 && *held + count > self.limit { + return false; + } + *held += count; + true + } + + /// Release `count` entries of `group_id`: settled by the loop, or never + /// handed to it. + pub fn release(&self, group_id: u64, count: usize) { + let mut outstanding = self.outstanding.lock().unwrap_or_else(|p| p.into_inner()); + if let Some(held) = outstanding.get_mut(&group_id) { + *held = held.saturating_sub(count); + } + } + + /// Entries of `group_id` the window holds. + pub fn outstanding(&self, group_id: u64) -> usize { + self.outstanding + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&group_id) + .copied() + .unwrap_or(0) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_full_group_refuses_while_another_group_admits() { + let window = ApplyWindow::new(4); + assert!(window.try_admit(1, 3)); + assert!(!window.try_admit(1, 2), "group 1 would pass its bound"); + assert!(window.try_admit(2, 4), "group 2 keeps its own window"); + window.release(1, 1); + assert!(window.try_admit(1, 2)); + assert_eq!(window.outstanding(1), 4); + } + + #[test] + fn an_empty_group_takes_a_batch_longer_than_the_limit() { + let window = ApplyWindow::new(2); + assert!(window.try_admit(7, 5)); + assert!(!window.try_admit(7, 1)); + } +} diff --git a/nodedb/src/control/distributed_applier/mod.rs b/nodedb/src/control/distributed_applier/mod.rs index 8d9ade3a9..5b808d1fd 100644 --- a/nodedb/src/control/distributed_applier/mod.rs +++ b/nodedb/src/control/distributed_applier/mod.rs @@ -8,11 +8,13 @@ pub mod applied_index; pub mod applier; pub mod apply_loop; +pub mod apply_window; pub mod proposal_ledger; pub mod propose_tracker; pub use applied_index::{AppliedPrefix, save_applied_index}; pub use applier::{ApplyBatch, DistributedApplier, create_distributed_applier}; pub use apply_loop::run_apply_loop; +pub use apply_window::{APPLY_WINDOW_PER_GROUP, ApplyWindow}; pub use proposal_ledger::{AppliedOutcome, PROPOSAL_LEDGER_CAPACITY, PriorApply, ProposalLedger}; -pub use propose_tracker::{AppliedWrite, ProposeResult, ProposeTracker}; +pub use propose_tracker::{AppliedWrite, ApplyingEntry, ProposeResult, ProposeTracker}; diff --git a/nodedb/src/control/distributed_applier/propose_tracker.rs b/nodedb/src/control/distributed_applier/propose_tracker.rs index b4ebb3aa0..8d06a918f 100644 --- a/nodedb/src/control/distributed_applier/propose_tracker.rs +++ b/nodedb/src/control/distributed_applier/propose_tracker.rs @@ -8,6 +8,8 @@ use std::collections::HashMap; use std::collections::hash_map::Entry; use std::sync::{Arc, Mutex}; +use super::apply_window::ApplyWindow; + use tokio::sync::oneshot; use nodedb_cluster::GroupAppliedWatchers; @@ -98,14 +100,41 @@ enum TrackerSlot { /// immediately if `complete()` already fired. pub struct ProposeTracker { slots: Mutex>, - /// Per-Raft-group apply watermark registry. Bumped on every - /// [`Self::complete`] so the watcher reflects "data applied on - /// this node up to index N" — the only semantic that's useful - /// for cross-node visibility waits. Tick-loop bumps cover the - /// metadata group (sync redb apply); this tracker covers data - /// groups (async SPSC dispatch through `run_apply_loop`). - /// `None` only in tests that don't exercise the watcher. + /// Per-Raft-group apply watermark registry. Bumped by + /// [`Self::note_applied`] once every entry of the group up to the index + /// finished, so the watcher reflects "data applied on this node up to + /// index N" — the only semantic that's useful for cross-node visibility + /// waits. Tick-loop bumps cover the metadata group (sync redb apply); + /// this tracker covers data groups (async SPSC dispatch through + /// `run_apply_loop`). `None` only in tests that don't exercise the + /// watcher. group_watchers: Option>, + /// Per group, the oldest committed entry the apply loop has not finished. + applying: Mutex>, + /// Per-group bound on entries between hand-off and settle, shared by the + /// applier that hands entries off and the loop that settles them. + window: Arc, +} + +/// The oldest committed entry of a group the apply loop has not finished: +/// what a propose waiter that timed out names as the entry its group's +/// applied index waits behind. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ApplyingEntry { + pub group_id: u64, + pub log_index: u64, + /// The collection the entry's plan writes, once the apply decoded it. + pub collection: Option, +} + +impl std::fmt::Display for ApplyingEntry { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "group {} index {}", self.group_id, self.log_index)?; + if let Some(collection) = &self.collection { + write!(f, " writing '{collection}'")?; + } + Ok(()) + } } impl Default for ProposeTracker { @@ -119,6 +148,8 @@ impl ProposeTracker { Self { slots: Mutex::new(HashMap::new()), group_watchers: None, + applying: Mutex::new(HashMap::new()), + window: Arc::new(ApplyWindow::default()), } } @@ -129,6 +160,42 @@ impl ProposeTracker { self } + /// The per-group apply window. + pub fn window(&self) -> &Arc { + &self.window + } + + /// Record the oldest entry of `group_id` the apply loop has not finished, + /// or `None` once every entry it holds for the group finished. + pub fn note_applying(&self, group_id: u64, entry: Option) { + let mut applying = self.applying.lock().unwrap_or_else(|p| p.into_inner()); + match entry { + Some(entry) => { + applying.insert(group_id, entry); + } + None => { + applying.remove(&group_id); + } + } + } + + /// The oldest entry of `group_id` the apply loop has not finished, if any. + pub fn applying(&self, group_id: u64) -> Option { + self.applying + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&group_id) + .cloned() + } + + /// Advance `group_id`'s applied watermark to `log_index`. The apply loop + /// calls it once every entry of the group up to `log_index` finished. + pub fn note_applied(&self, group_id: u64, log_index: u64) { + if let Some(w) = &self.group_watchers { + w.bump(group_id, log_index); + } + } + /// Register a waiter for a proposed entry. Returns a receiver that /// resolves when the entry is committed and executed. /// @@ -167,12 +234,27 @@ impl ProposeTracker { rx } + /// Drop the waiter a proposer registered at `(group_id, log_index)` and + /// stopped waiting on: its deadline passed, or this node left the group + /// and will never apply the index. A result already stored there stays. + pub fn abandon(&self, group_id: u64, log_index: u64) { + let mut slots = self.slots.lock().unwrap_or_else(|p| p.into_inner()); + if let Entry::Occupied(e) = slots.entry((group_id, log_index)) + && matches!(e.get(), TrackerSlot::Waiting { .. }) + { + e.remove(); + } + } + /// Complete a waiter after the entry has been committed and executed. /// /// If the proposer has already called `register()`, the result is sent /// immediately. If not, the result is stored so the next `register()` /// call picks it up without waiting. /// + /// Entries of one group complete in any order. The applied watermark + /// moves only through [`Self::note_applied`], in log order. + /// /// Returns true if a live waiter was found and notified, false otherwise. pub fn complete( &self, @@ -181,17 +263,6 @@ impl ProposeTracker { applied_key: u64, result: ProposeResult, ) -> bool { - // Bump the per-group apply watermark. Bumping unconditionally - // (success and error) keeps the watcher monotonic with raft's - // commit progression — a data-plane error means "the entry - // could not be applied" but the entry IS committed and Raft - // has advanced its applied index. Tests waiting on - // visibility care about the success path; liveness on the - // error path requires the bump too. - if let Some(w) = &self.group_watchers { - w.bump(group_id, log_index); - } - let mut slots = self.slots.lock().unwrap_or_else(|p| p.into_inner()); match slots.entry((group_id, log_index)) { Entry::Vacant(e) => { diff --git a/nodedb/src/control/exec_receiver/backup_cut.rs b/nodedb/src/control/exec_receiver/backup_cut.rs new file mode 100644 index 000000000..1a44bb81a --- /dev/null +++ b/nodedb/src/control/exec_receiver/backup_cut.rs @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's consistent cut on a remote source node. +//! +//! The coordinator of a backup picks the envelope watermark `W` and takes the +//! cut on its own replicas. A snapshot it sends to another node carries `W`. +//! That node takes the same cut on its replicas before it snapshots, so every +//! write committed below `W` has its final outcome in the snapshot, and every +//! write at or above `W` refuses a restore of the envelope. + +use nodedb_cluster::rpc_codec::TypedClusterError; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; + +use crate::control::state::SharedState; + +use super::support::execution_error_to_typed; + +/// Take the cut a tenant snapshot plan asks for, then return the plan with +/// the request cleared. Every other plan passes through unchanged. +pub(super) async fn take_backup_cut( + state: &std::sync::Arc, + plan: PhysicalPlan, +) -> Result { + let PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id, + cut_watermark: Some(watermark), + }) = plan + else { + return Ok(plan); + }; + crate::control::backup::cut::cut_at(state, tenant_id, watermark) + .await + .map_err(execution_error_to_typed)?; + Ok(PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id, + cut_watermark: None, + })) +} diff --git a/nodedb/src/control/exec_receiver/executor.rs b/nodedb/src/control/exec_receiver/executor.rs index d10258b14..94d4830e0 100644 --- a/nodedb/src/control/exec_receiver/executor.rs +++ b/nodedb/src/control/exec_receiver/executor.rs @@ -21,6 +21,7 @@ use crate::control::state::SharedState; use crate::control::trace_export::EmitSpanParams; use crate::types::DatabaseId; +use super::backup_cut::take_backup_cut; use super::plan_decode::decode_plan; use super::request_validation::validate_request; use super::support::{PLAN_DECODE_FAILED, SinkOutcome, execution_error_to_typed}; @@ -108,7 +109,7 @@ impl LocalPlanExecutor { /// paths: validate deadline + descriptor versions, decode the plan, reject /// unresolved Exchange nodes. Returns `(plan, database_id, deadline)` on /// success or a typed cluster error to surface to the caller. - fn validate_and_decode( + async fn validate_and_decode( &self, req: &ExecuteRequest, ) -> Result< @@ -121,13 +122,15 @@ impl LocalPlanExecutor { > { let (deadline, database_id) = validate_request(&self.state, req)?; let plan = decode_plan(&self.state, database_id, req.tenant_id, &req.plan_bytes)?; + // A backup's snapshot plan takes the backup's cut on this node first. + let plan = take_backup_cut(&self.state, plan).await?; Ok((plan, database_id, deadline)) } /// One-shot execution: validate + decode, fan across all local cores, /// merge, and return the merged payload. async fn execute_plan_inner(&self, req: ExecuteRequest) -> ExecuteResponse { - let (plan, database_id, deadline) = match self.validate_and_decode(&req) { + let (plan, database_id, deadline) = match self.validate_and_decode(&req).await { Ok(t) => t, Err(e) => return ExecuteResponse::err(e), }; @@ -135,6 +138,22 @@ impl LocalPlanExecutor { let tenant_id = crate::types::TenantId::new(req.tenant_id); let trace_id = nodedb_types::TraceId(req.trace_id); + if let PhysicalPlan::ClusterEvent( + nodedb_physical::physical_plan::ClusterEventOp::TenantWriteMarks { + tenant_id: marks_tenant, + group_ids, + }, + ) = &plan + { + return super::tenant_marks::answer_tenant_marks( + &self.state, + *marks_tenant, + group_ids, + deadline, + ) + .await; + } + if let PhysicalPlan::ClusterEvent( nodedb_physical::physical_plan::ClusterEventOp::PublishTopic { database_id: topic_database_id, @@ -365,7 +384,7 @@ impl LocalPlanExecutor { req: ExecuteRequest, mut sink: impl ChunkSink, ) -> Option { - let (plan, database_id, deadline) = match self.validate_and_decode(&req) { + let (plan, database_id, deadline) = match self.validate_and_decode(&req).await { Ok(t) => t, Err(e) => return Some(e), }; diff --git a/nodedb/src/control/exec_receiver/mod.rs b/nodedb/src/control/exec_receiver/mod.rs index 7f1b62daa..477f8e6b8 100644 --- a/nodedb/src/control/exec_receiver/mod.rs +++ b/nodedb/src/control/exec_receiver/mod.rs @@ -2,9 +2,11 @@ //! Local execution of incoming `ExecuteRequest` / `ExecuteStreamRequest` RPCs. +mod backup_cut; pub mod executor; mod plan_decode; mod request_validation; mod support; +mod tenant_marks; pub use executor::LocalPlanExecutor; diff --git a/nodedb/src/control/exec_receiver/tenant_marks.rs b/nodedb/src/control/exec_receiver/tenant_marks.rs new file mode 100644 index 000000000..49a769d63 --- /dev/null +++ b/nodedb/src/control/exec_receiver/tenant_marks.rs @@ -0,0 +1,63 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Answer a restoring node's request for this node's tenant write marks. +//! +//! The restoring node asks a replica of every data group it does not +//! replicate itself. This node answers only for groups it replicates, and only +//! once it applied every entry the groups committed before the request. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::rpc_codec::{ExecuteResponse, TypedClusterError}; + +use crate::control::backup::restore::guard::{encode_marks, local_tenant_marks}; +use crate::control::security::auth_fence::cluster::hosts_group; +use crate::control::state::SharedState; + +use super::support::execution_error_to_typed; + +/// This node's marks of `tenant_id` in `group_ids`, encoded for the wire. +/// +/// A group this node does not replicate is refused with +/// [`TypedClusterError::NotLeader`] naming the group. The asking node then +/// picks another replica. The refusal carries the replica this node's routing +/// table names for the group, when it names one other than this node. +/// `budget` is what remains of the asking statement's deadline. +pub(super) async fn answer_tenant_marks( + state: &Arc, + tenant_id: u64, + group_ids: &[u64], + budget: Duration, +) -> ExecuteResponse { + if let Some(&group_id) = group_ids.iter().find(|group| !hosts_group(state, **group)) { + return ExecuteResponse::err(TypedClusterError::NotLeader { + group_id, + leader_node_id: replica_hint(state, group_id), + leader_addr: None, + term: 0, + }); + } + let deadline = tokio::time::Instant::now() + budget; + let marks = match local_tenant_marks(state, tenant_id, group_ids, deadline).await { + Ok(marks) => marks, + Err(error) => return ExecuteResponse::err(execution_error_to_typed(error)), + }; + match encode_marks(&marks) { + Ok(payload) => ExecuteResponse::ok(vec![payload], 0, 0), + Err(error) => ExecuteResponse::err(execution_error_to_typed(error)), + } +} + +/// A replica of `group_id` other than this node, from this node's routing +/// table: the leader when it knows one, else the first voter, else the first +/// learner. +fn replica_hint(state: &SharedState, group_id: u64) -> Option { + let routing = state.cluster_routing.as_ref()?; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let info = routing.group_info(group_id)?; + std::iter::once(info.leader) + .chain(info.members.iter().copied()) + .chain(info.learners.iter().copied()) + .find(|&node| node != 0 && node != state.node_id) +} diff --git a/nodedb/src/control/fail_gate.rs b/nodedb/src/control/fail_gate.rs index 1322a9d28..eb7bb6dec 100644 --- a/nodedb/src/control/fail_gate.rs +++ b/nodedb/src/control/fail_gate.rs @@ -36,12 +36,20 @@ pub(crate) async fn wait(name: &str) { /// `funnel::before_dispatch::`, and only a write carrying a WAL /// LSN reaches it. A committed redo parks on the gate of each collection it /// writes. -pub(crate) async fn before_dispatch(plan: &PhysicalPlan, wal_lsn: Option) { +/// +/// `funnel::before_dispatch::node::` parks the write only on +/// node `N`. An in-process cluster test shares one fail-point registry across +/// its nodes, so it names the node to hold one replica's apply. +pub(crate) async fn before_dispatch(node_id: u64, plan: &PhysicalPlan, wal_lsn: Option) { if wal_lsn.is_none() { return; } for collection in plan.named_collections() { wait(&format!("funnel::before_dispatch::{collection}")).await; + wait(&format!( + "funnel::before_dispatch::node{node_id}::{collection}" + )) + .await; } } diff --git a/nodedb/src/control/gateway/core.rs b/nodedb/src/control/gateway/core.rs index 8dd63dac9..506257fcf 100644 --- a/nodedb/src/control/gateway/core.rs +++ b/nodedb/src/control/gateway/core.rs @@ -222,17 +222,6 @@ impl Gateway { status_ok: result.is_ok(), }); - // Advance per-tenant observed write-HLC high-water on any - // successful cluster dispatch (local or remote). Used by - // RESTORE staleness gate. Tracking on success of every - // gateway.execute is intentional: backup captures its - // envelope watermark AFTER its own fan-out, so a fresh - // backup's watermark always dominates the tenant_wm it - // itself advanced. - if result.is_ok() { - shared.advance_tenant_write_hlc(ctx.tenant_id.as_u64()); - } - result } diff --git a/nodedb/src/control/gateway/router.rs b/nodedb/src/control/gateway/router.rs index 52998a6a8..6cfb56bf7 100644 --- a/nodedb/src/control/gateway/router.rs +++ b/nodedb/src/control/gateway/router.rs @@ -513,6 +513,7 @@ mod tests { redo: vec![], collections: vec![], sum_targets: vec![], + origin: nodedb_physical::physical_plan::RedoOrigin::Commit, }), ] { for table in [None, Some(single_node_table())] { diff --git a/nodedb/src/control/otel/receiver.rs b/nodedb/src/control/otel/receiver.rs index bf16480bc..2cb8b75d3 100644 --- a/nodedb/src/control/otel/receiver.rs +++ b/nodedb/src/control/otel/receiver.rs @@ -428,7 +428,7 @@ pub(super) async fn authenticate_otel( crate::control::security::jwt_policy::enforce_stateful_jwt_policy( shared, verified.claims(), - identity.tenant_id, + &identity, ) .map_err(|_| "invalid bearer token".to_owned())?; return admit_transport(shared, identity, peer_addr); diff --git a/nodedb/src/control/planner/rls_injection/array.rs b/nodedb/src/control/planner/rls_injection/array.rs index 10417cd0d..61066c1e7 100644 --- a/nodedb/src/control/planner/rls_injection/array.rs +++ b/nodedb/src/control/planner/rls_injection/array.rs @@ -59,6 +59,8 @@ pub(super) fn inject_cluster_event(_ctx: &RlsCtx<'_>, op: &ClusterEventOp) -> cr // topic publish by topic name — neither names a collection this pass // could resolve a policy against. Access to a stream or topic is // authorized on the stream/topic object itself. - ClusterEventOp::ConsumeStream { .. } | ClusterEventOp::PublishTopic { .. } => Ok(()), + ClusterEventOp::ConsumeStream { .. } + | ClusterEventOp::PublishTopic { .. } + | ClusterEventOp::TenantWriteMarks { .. } => Ok(()), } } diff --git a/nodedb/src/control/planner/rls_injection/meta.rs b/nodedb/src/control/planner/rls_injection/meta.rs index 2f5efdbde..555323526 100644 --- a/nodedb/src/control/planner/rls_injection/meta.rs +++ b/nodedb/src/control/planner/rls_injection/meta.rs @@ -135,7 +135,10 @@ mod tests { #[test] fn tenant_snapshot_is_refused_while_any_policy_applies() { let store = store_with_read_policy("users"); - let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: 1 }); + let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: 1, + cut_watermark: None, + }); assert!(matches!( inject(&mut plan, &store), Err(crate::Error::PlanError { .. }) diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/array.rs b/nodedb/src/control/planner/rls_injection/permission_tree/array.rs index e72634961..dc2fba05d 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/array.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/array.rs @@ -57,6 +57,8 @@ pub(super) fn apply_cluster_event(_ctx: &PermCtx<'_>, op: &ClusterEventOp) -> cr // topic publish by topic name — neither names a collection this pass // could resolve a tree definition against. Access to a stream or topic // is authorized on the stream/topic object itself. - ClusterEventOp::ConsumeStream { .. } | ClusterEventOp::PublishTopic { .. } => Ok(()), + ClusterEventOp::ConsumeStream { .. } + | ClusterEventOp::PublishTopic { .. } + | ClusterEventOp::TenantWriteMarks { .. } => Ok(()), } } diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs b/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs index f56db1f3e..26d6746aa 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs @@ -135,7 +135,10 @@ mod tests { #[test] fn tenant_snapshot_is_refused_while_any_tree_applies() { let cache = cache_with_tree("docs"); - let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: 1 }); + let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: 1, + cut_watermark: None, + }); assert!(matches!( apply(&mut plan, &cache), Err(crate::Error::PlanError { .. }) diff --git a/nodedb/src/control/security/auth_lease/calvin_acks.rs b/nodedb/src/control/security/auth_lease/calvin_acks.rs index 206465ab3..af6596b51 100644 --- a/nodedb/src/control/security/auth_lease/calvin_acks.rs +++ b/nodedb/src/control/security/auth_lease/calvin_acks.rs @@ -113,4 +113,52 @@ mod tests { 1 ); } + + /// A restart loses the in-memory mirror and, after a checkpoint, the WAL + /// markers of transactions applied before it. The mirror rebuilt from the + /// state the checkpoint saved settles every ack the sequencer log replays, + /// so the coverage reaches the sequencer log's applied index. + #[test] + fn a_mirror_rebuilt_from_saved_state_settles_replayed_acks() { + use crate::control::cluster::calvin::scheduler::recover_applied; + use crate::control::security::catalog::SystemCatalog; + use crate::control::security::catalog::calvin_applied::StoredCalvinApplied; + + let wal_dir = tempfile::tempdir().expect("wal dir"); + // The WAL a checkpoint truncated holds no applied marker. + let wal = crate::wal::manager::WalManager::open(wal_dir.path(), false).expect("wal"); + let catalog_dir = tempfile::tempdir().expect("catalog dir"); + let catalog = + SystemCatalog::open(&catalog_dir.path().join("system.redb")).expect("catalog"); + catalog + .save_calvin_applied(vec![StoredCalvinApplied { + vshard_id: 7, + fully_applied_epoch: NOT_YET_APPLIED_EPOCH, + tail: [(0, 0), (3, 1)].into_iter().collect(), + }]) + .expect("save"); + + let recovered = recover_applied(&wal, &catalog, 7).expect("recover"); + let mirrors = AppliedMirrors::default(); + mirrors.register(7, recovered.fully_applied_epoch, &recovered.applied_tail); + + let registry = CalvinCompletionRegistry::new_detached(); + registry.applied_acks.enable(); + registry.applied_acks.record(AppliedCompletionAck { + index: 5, + txn: TxnId::new(0, 0), + vshard_id: 7, + }); + registry.applied_acks.record(AppliedCompletionAck { + index: 9, + txn: TxnId::new(3, 1), + vshard_id: 7, + }); + let coverage = CalvinAckCoverage::default(); + assert_eq!( + coverage.covered_through(®istry, &mirrors, 19, |_| true), + 19, + "every replayed ack settles against the rebuilt mirror" + ); + } } diff --git a/nodedb/src/control/security/auth_lease/mod.rs b/nodedb/src/control/security/auth_lease/mod.rs index f3667f9b4..a6bb95512 100644 --- a/nodedb/src/control/security/auth_lease/mod.rs +++ b/nodedb/src/control/security/auth_lease/mod.rs @@ -10,6 +10,7 @@ pub mod service; pub mod status; pub mod table; pub mod timing; +pub mod withheld_warn; pub use barrier::{ authorization_barrier, await_local_coverage, block_on_barrier, calvin_write_barrier, diff --git a/nodedb/src/control/security/auth_lease/renew_loop.rs b/nodedb/src/control/security/auth_lease/renew_loop.rs index 6c55cb39f..10a9f879e 100644 --- a/nodedb/src/control/security/auth_lease/renew_loop.rs +++ b/nodedb/src/control/security/auth_lease/renew_loop.rs @@ -29,14 +29,11 @@ pub async fn run_renew_loop( ) { let mut confirmed: Vec = Vec::new(); loop { - match confirmed_coverage(&state, &confirmed, timing.renew_every).await { - Ok(coverage) => { - confirmed = coverage; - renew_once(&state, timing, &confirmed).await; - } - Err(error) => { - tracing::warn!(%error, "authorization lease: coverage could not be computed"); - } + // Shutdown ends a renewal in flight too: a coverage read or a renew + // RPC holds the node's state until it returns. + tokio::select! { + _ = renew_round(&state, timing, &mut confirmed) => {} + _ = shutdown.wait_cancelled() => return, } tokio::select! { _ = tokio::time::sleep(timing.renew_every) => {} @@ -45,6 +42,19 @@ pub async fn run_renew_loop( } } +/// Compute this node's confirmed coverage and renew the lease with it. +async fn renew_round(state: &SharedState, timing: LeaseTiming, confirmed: &mut Vec) { + match confirmed_coverage(state, confirmed, timing.renew_every).await { + Ok(coverage) => { + *confirmed = coverage; + renew_once(state, timing, confirmed).await; + } + Err(error) => { + tracing::warn!(%error, "authorization lease: coverage could not be computed"); + } + } +} + /// Send one renewal and install a granted lease. async fn renew_once(state: &SharedState, timing: LeaseTiming, coverage: &[GroupCoverage]) { let Some((leader_id, _)) = metadata_leader(state).filter(|(leader, _)| *leader != 0) else { diff --git a/nodedb/src/control/security/auth_lease/service.rs b/nodedb/src/control/security/auth_lease/service.rs index 3a39085e9..270169f3c 100644 --- a/nodedb/src/control/security/auth_lease/service.rs +++ b/nodedb/src/control/security/auth_lease/service.rs @@ -39,6 +39,7 @@ use crate::control::state::SharedState; use super::leadership::{leader_hint, leading_term}; use super::table::{BarrierState, LeaseTable, RenewDecision}; use super::timing::LeaseTiming; +use super::withheld_warn::WithheldWarnings; /// Answers lease renewals and barriers while this node leads the metadata /// group. @@ -51,6 +52,8 @@ pub struct LeaderLeaseService { changed: Notify, /// One floor load at a time. floors_loading: tokio::sync::Mutex<()>, + /// Rate limit of the withheld-renewal warning. + withheld_warnings: WithheldWarnings, } impl std::fmt::Debug for LeaderLeaseService { @@ -69,6 +72,7 @@ impl LeaderLeaseService { table: Mutex::new(None), changed: Notify::new(), floors_loading: tokio::sync::Mutex::new(()), + withheld_warnings: WithheldWarnings::default(), } } @@ -166,23 +170,53 @@ impl LeaderLeaseService { return Self::not_leader_renewal(Some(&state)); }; if let Err(error) = self.table_ready(&state, term).await { - tracing::debug!(%error, "authorization lease: floors not loaded; renewal withheld"); + if self + .withheld_warnings + .should_warn(req.node_id, Instant::now()) + { + tracing::warn!( + node_id = req.node_id, + %error, + "authorization lease: renewal withheld: the floors of this term are not loaded" + ); + } return AuthLeaseRenewResponse { outcome: AuthLeaseRenewOutcome::Withheld, }; } - let decision = { + let (decision, shortfall) = { let mut table = self.table(); match table.as_mut().filter(|t| t.term() == term) { - Some(table) => table.renew( - req.node_id, - &req.coverage, - Instant::now(), - self.timing.lease, - ), + Some(table) => { + let decision = table.renew( + req.node_id, + &req.coverage, + Instant::now(), + self.timing.lease, + ); + let shortfall = match decision { + RenewDecision::Withheld => table.shortfall(&req.coverage), + RenewDecision::Granted => Vec::new(), + }; + (decision, shortfall) + } None => return Self::not_leader_renewal(Some(&state)), } }; + if decision == RenewDecision::Withheld + && self + .withheld_warnings + .should_warn(req.node_id, Instant::now()) + { + // Each entry: (group_id, floor, reported coverage or None). + tracing::warn!( + node_id = req.node_id, + term, + short_groups = ?shortfall, + "authorization lease: renewal withheld: the node's coverage is below the floor \ + of each group listed as (group_id, floor, reported)" + ); + } self.changed.notify_waiters(); let outcome = match decision { RenewDecision::Withheld => AuthLeaseRenewOutcome::Withheld, diff --git a/nodedb/src/control/security/auth_lease/table.rs b/nodedb/src/control/security/auth_lease/table.rs index 332372978..3b2a53fc1 100644 --- a/nodedb/src/control/security/auth_lease/table.rs +++ b/nodedb/src/control/security/auth_lease/table.rs @@ -125,6 +125,26 @@ impl LeaseTable { RenewDecision::Granted } + /// Every group whose floor `coverage` does not reach, as + /// `(group_id, floor, reported)`. `reported` is `None` for a group the + /// report omits. + pub fn shortfall(&self, coverage: &[GroupCoverage]) -> Vec<(u64, u64, Option)> { + let mut short: Vec<(u64, u64, Option)> = self + .floors + .iter() + .filter_map(|(group_id, floor)| { + let reported = coverage + .iter() + .find(|report| report.group_id == *group_id) + .map(|report| report.through); + (reported.is_none_or(|through| through < *floor)) + .then_some((*group_id, *floor, reported)) + }) + .collect(); + short.sort_unstable(); + short + } + /// Where a barrier on `targets` stands at `now`. pub fn barrier( &self, @@ -189,6 +209,18 @@ mod tests { assert_eq!(table.barrier(&[], now, LEASE), BarrierState::NotReady); } + #[test] + fn the_shortfall_names_each_uncovered_floor() { + let mut table = settled_table(Instant::now()); + table.raise_floors(&[cover(7, 19)]); + assert_eq!( + table.shortfall(&[cover(0, 10), cover(7, 12)]), + vec![(7, 19, Some(12))] + ); + assert_eq!(table.shortfall(&[cover(7, 19)]), vec![(0, 10, None)]); + assert!(table.shortfall(&[cover(0, 10), cover(7, 19)]).is_empty()); + } + #[test] fn a_report_below_a_floor_is_withheld() { let start = Instant::now(); diff --git a/nodedb/src/control/security/auth_lease/withheld_warn.rs b/nodedb/src/control/security/auth_lease/withheld_warn.rs new file mode 100644 index 000000000..bc2ce5b18 --- /dev/null +++ b/nodedb/src/control/security/auth_lease/withheld_warn.rs @@ -0,0 +1,50 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Rate limit for the leader's warning on a withheld lease renewal. +//! +//! A node whose coverage stays below a floor renews every interval, and each +//! renewal is withheld. The warning names the floor and the reported coverage +//! of every short group, once per node per window. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +/// At most one warning per node in this window. +const WARN_WINDOW: Duration = Duration::from_secs(10); + +/// When each node's last warning was logged. +#[derive(Debug, Default)] +pub struct WithheldWarnings { + last: Mutex>, +} + +impl WithheldWarnings { + /// Whether a warning about `node_id` logs at `now`. A `true` starts the + /// node's next window. + pub fn should_warn(&self, node_id: u64, now: Instant) -> bool { + let mut last = self.last.lock().unwrap_or_else(|p| p.into_inner()); + match last.get(&node_id) { + Some(at) if now.duration_since(*at) < WARN_WINDOW => false, + _ => { + last.insert(node_id, now); + true + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn one_warning_per_node_per_window() { + let warnings = WithheldWarnings::default(); + let start = Instant::now(); + assert!(warnings.should_warn(2, start)); + assert!(!warnings.should_warn(2, start + Duration::from_secs(1))); + assert!(warnings.should_warn(3, start + Duration::from_secs(1))); + assert!(warnings.should_warn(2, start + WARN_WINDOW)); + } +} diff --git a/nodedb/src/control/security/catalog/bootstrap_tables.rs b/nodedb/src/control/security/catalog/bootstrap_tables.rs index 441777ef5..86f5ae1fa 100644 --- a/nodedb/src/control/security/catalog/bootstrap_tables.rs +++ b/nodedb/src/control/security/catalog/bootstrap_tables.rs @@ -75,6 +75,8 @@ pub(super) const BOOTSTRAP_TABLES: &[BootstrapTable] = bootstrap_tables![ "collections" => COLLECTIONS, "metadata" => METADATA, "wal_tombstones" => WAL_TOMBSTONES, + "tenant_group_marks" => super::tenant_group_marks::TENANT_GROUP_MARKS, + "calvin_applied" => super::calvin_applied::CALVIN_APPLIED, "l2_cleanup_queue" => L2_CLEANUP_QUEUE, "pending_reclaim" => PENDING_RECLAIM, "column_stats" => COLUMN_STATS, diff --git a/nodedb/src/control/security/catalog/calvin_applied.rs b/nodedb/src/control/security/catalog/calvin_applied.rs new file mode 100644 index 000000000..db18de836 --- /dev/null +++ b/nodedb/src/control/security/catalog/calvin_applied.rs @@ -0,0 +1,181 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Persistent Calvin applied state backing `_system.calvin_applied`. +//! +//! A Calvin scheduler learns at boot which `(epoch, position)` it already +//! applied from the applied markers in the WAL. A checkpoint deletes the WAL +//! segments that hold them, while the sequencer log still holds the entries +//! and delivers them again after a restart. So before each truncation the +//! checkpoint saves every scheduler's applied state here, and boot recovery +//! reads it together with the markers the WAL still holds. + +use std::collections::BTreeSet; + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Table: `vshard_id` -> `(fully_applied_epoch, msgpack of the applied tail)`. +pub(super) const CALVIN_APPLIED: TableDefinition = + TableDefinition::new("_system.calvin_applied"); + +/// Sentinel `fully_applied_epoch`: no epoch is fully applied. +const NONE_FULLY_APPLIED: u64 = u64::MAX; + +/// One vShard's applied state. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct StoredCalvinApplied { + pub vshard_id: u32, + /// Every position of every epoch at or below this is applied. `u64::MAX` + /// means none is. + pub fully_applied_epoch: u64, + /// Applied `(epoch, position)` pairs above the watermark. + pub tail: BTreeSet<(u64, u32)>, +} + +impl StoredCalvinApplied { + /// The union of two states of one vShard: the higher watermark, and the + /// tail entries of both above it. + fn merge(self, other: StoredCalvinApplied) -> StoredCalvinApplied { + let fully_applied_epoch = match (self.fully_applied_epoch, other.fully_applied_epoch) { + (NONE_FULLY_APPLIED, w) | (w, NONE_FULLY_APPLIED) => w, + (a, b) => a.max(b), + }; + let tail = self + .tail + .into_iter() + .chain(other.tail) + .filter(|(epoch, _)| { + fully_applied_epoch == NONE_FULLY_APPLIED || *epoch > fully_applied_epoch + }) + .collect(); + StoredCalvinApplied { + vshard_id: self.vshard_id, + fully_applied_epoch, + tail, + } + } +} + +fn encode_tail(tail: &BTreeSet<(u64, u32)>) -> crate::Result> { + let pairs: Vec<(u64, u32)> = tail.iter().copied().collect(); + zerompk::to_msgpack_vec(&pairs).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("encode calvin applied tail: {e}"), + }) +} + +fn decode_tail(bytes: &[u8]) -> crate::Result> { + let pairs: Vec<(u64, u32)> = + zerompk::from_msgpack(bytes).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("decode calvin applied tail: {e}"), + })?; + Ok(pairs.into_iter().collect()) +} + +impl SystemCatalog { + /// The saved applied state of `vshard_id`, if any. + pub fn load_calvin_applied( + &self, + vshard_id: u32, + ) -> crate::Result> { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("load_calvin_applied read txn", e))?; + let table = read_txn + .open_table(CALVIN_APPLIED) + .map_err(|e| catalog_err("open calvin_applied", e))?; + let Some(row) = table + .get(vshard_id) + .map_err(|e| catalog_err("get calvin_applied", e))? + else { + return Ok(None); + }; + let (fully_applied_epoch, tail) = row.value(); + Ok(Some(StoredCalvinApplied { + vshard_id, + fully_applied_epoch, + tail: decode_tail(tail)?, + })) + } + + /// Save `states` in one transaction. Each is merged with the state + /// already saved for its vShard, so the saved state only grows. + pub fn save_calvin_applied(&self, states: Vec) -> crate::Result<()> { + if states.is_empty() { + return Ok(()); + } + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("save_calvin_applied txn", e))?; + { + let mut table = write_txn + .open_table(CALVIN_APPLIED) + .map_err(|e| catalog_err("open calvin_applied", e))?; + for state in states { + let saved = match table + .get(state.vshard_id) + .map_err(|e| catalog_err("get calvin_applied", e))? + { + Some(row) => { + let (fully_applied_epoch, tail) = row.value(); + Some(StoredCalvinApplied { + vshard_id: state.vshard_id, + fully_applied_epoch, + tail: decode_tail(tail)?, + }) + } + None => None, + }; + let merged = match saved { + Some(saved) => saved.merge(state), + None => state, + }; + let tail = encode_tail(&merged.tail)?; + table + .insert( + merged.vshard_id, + (merged.fully_applied_epoch, tail.as_slice()), + ) + .map_err(|e| catalog_err("insert calvin_applied", e))?; + } + } + write_txn + .commit() + .map_err(|e| catalog_err("commit calvin_applied", e)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn state(vshard_id: u32, w: u64, tail: &[(u64, u32)]) -> StoredCalvinApplied { + StoredCalvinApplied { + vshard_id, + fully_applied_epoch: w, + tail: tail.iter().copied().collect(), + } + } + + #[test] + fn saved_state_reloads_and_only_grows() { + let dir = tempfile::tempdir().expect("tempdir"); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).expect("catalog"); + assert_eq!(catalog.load_calvin_applied(3).expect("load"), None); + + catalog + .save_calvin_applied(vec![state(3, NONE_FULLY_APPLIED, &[(0, 0), (2, 1)])]) + .expect("save"); + catalog + .save_calvin_applied(vec![state(3, 1, &[(4, 0)])]) + .expect("save"); + assert_eq!( + catalog.load_calvin_applied(3).expect("load"), + Some(state(3, 1, &[(2, 1), (4, 0)])) + ); + } +} diff --git a/nodedb/src/control/security/catalog/mod.rs b/nodedb/src/control/security/catalog/mod.rs index d41f6649b..b16453296 100644 --- a/nodedb/src/control/security/catalog/mod.rs +++ b/nodedb/src/control/security/catalog/mod.rs @@ -7,6 +7,7 @@ pub mod auth_types; pub mod auth_users; pub mod blacklist; pub mod bootstrap_tables; +pub mod calvin_applied; pub mod change_streams; pub mod checkpoint; pub mod checkpoints; @@ -63,6 +64,7 @@ pub mod sync_producer; pub mod synonym_groups; pub mod system_catalog; pub mod tables; +pub mod tenant_group_marks; pub mod tenant_id_hwm; pub mod tenant_quotas; pub mod topics; diff --git a/nodedb/src/control/security/catalog/tenant_group_marks.rs b/nodedb/src/control/security/catalog/tenant_group_marks.rs new file mode 100644 index 000000000..e47ad2c8d --- /dev/null +++ b/nodedb/src/control/security/catalog/tenant_group_marks.rs @@ -0,0 +1,95 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Persistent per-group tenant write marks backing +//! `_system.tenant_group_marks`. +//! +//! For each data group this node replicates, the newest commit HLC of any +//! write of each tenant the group applied. The apply loop writes a group's +//! marks before it saves the applied floor that covers them, so every +//! committed entry is either covered by a persisted mark or above the floor, +//! where Raft delivers it again after a restart and the loop derives its mark +//! again. A Calvin commit writes its mark before its install is acknowledged. + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Table: `(group_id, tenant_id)` -> `(commit_hlc, site_code, collection)`. +pub(super) const TENANT_GROUP_MARKS: TableDefinition<(u64, u64), (u64, u8, &str)> = + TableDefinition::new("_system.tenant_group_marks"); + +/// One persisted mark. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct StoredGroupMark { + pub group_id: u64, + pub tenant_id: u64, + /// HLC wall time, in nanoseconds, of the newest write. + pub hlc: u64, + /// Which apply path recorded the write. + pub site: u8, + /// The collection the write named, empty when it named none. + pub collection: String, +} + +impl SystemCatalog { + /// Every persisted mark. + pub fn load_tenant_group_marks(&self) -> crate::Result> { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("load_tenant_group_marks read txn", e))?; + let table = read_txn + .open_table(TENANT_GROUP_MARKS) + .map_err(|e| catalog_err("open tenant_group_marks", e))?; + let mut marks = Vec::new(); + for entry in table + .iter() + .map_err(|e| catalog_err("iterate tenant_group_marks", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("read tenant_group_mark", e))?; + let (group_id, tenant_id) = key.value(); + let (hlc, site, collection) = value.value(); + marks.push(StoredGroupMark { + group_id, + tenant_id, + hlc, + site, + collection: collection.to_owned(), + }); + } + Ok(marks) + } + + /// Raise every mark in `marks` in one transaction. A persisted mark at or + /// above the new one stays. + pub fn raise_tenant_group_marks(&self, marks: &[StoredGroupMark]) -> crate::Result<()> { + if marks.is_empty() { + return Ok(()); + } + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("raise_tenant_group_marks txn", e))?; + { + let mut table = write_txn + .open_table(TENANT_GROUP_MARKS) + .map_err(|e| catalog_err("open tenant_group_marks", e))?; + for mark in marks { + let key = (mark.group_id, mark.tenant_id); + let current = table + .get(key) + .map_err(|e| catalog_err("get tenant_group_mark", e))? + .map(|guard| guard.value().0); + if current.is_some_and(|hlc| hlc >= mark.hlc) { + continue; + } + table + .insert(key, (mark.hlc, mark.site, mark.collection.as_str())) + .map_err(|e| catalog_err("insert tenant_group_mark", e))?; + } + } + write_txn + .commit() + .map_err(|e| catalog_err("commit tenant_group_marks", e)) + } +} diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index 1563539c1..2314f1e07 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -97,6 +97,7 @@ pub async fn dispatch_authorized_autocommit_write( durability: WalDurability::AppendHere { now_override: None, apply_key: 0, + commit_hlc: None, }, }, ) @@ -130,6 +131,7 @@ pub(crate) async fn dispatch_authorized_autocommit_write_with_source( durability: WalDurability::AppendHere { now_override: None, apply_key: 0, + commit_hlc: None, }, }, ) @@ -283,6 +285,7 @@ pub(crate) async fn dispatch_autocommit_write( durability: WalDurability::AppendHere { now_override: None, apply_key: 0, + commit_hlc: None, }, }, ) @@ -916,6 +919,7 @@ mod tests { durability: super::WalDurability::AppendHere { now_override: None, apply_key: KEY, + commit_hlc: None, }, ordering: super::WriteOrdering::AlreadyOrdered, change_feed: super::ChangeFeedOwner::Unowned, diff --git a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs index f92ec8bb0..4c6b021da 100644 --- a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs +++ b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs @@ -167,6 +167,7 @@ mod tests { redo: vec![1], collections: vec!["c".into()], sum_targets: Vec::new(), + origin: nodedb_physical::physical_plan::RedoOrigin::Commit, }); assert_eq!(funnel_minted_redo_engine(&plan), Some("transaction")); } diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index 1f08b6354..40c0cf6be 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -38,7 +38,8 @@ pub(crate) use minted::{ Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, }; pub(crate) use submit_write::{ - ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, + ChangeFeedOwner, PendingWrite, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, + enqueue_write, submit_write, }; pub(crate) use types::{AutocommitWrite, WriteDispatch}; pub(crate) use write_abort::{ diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs index d14e851e7..70bccb67b 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs @@ -56,13 +56,17 @@ pub(super) struct DispatchOutcome { /// the request enters dispatch; the caller observes it on every exit path /// (success, budget over-run, timeout) so the histogram captures the true /// end-to-end shape of the work routed to this vshard. -pub(super) fn dispatch_to_data_plane( +/// +/// `waits_for_capacity` makes a capacity refusal wait for freed capacity and +/// retry, up to the request's deadline. Every other refusal returns at once. +pub(super) async fn dispatch_to_data_plane( shared: &SharedState, ddl_transition: &AuthorizedDdlTransition, target: DispatchTarget, admission_guard: Option, order_guard: Option>, post_apply_pending: bool, + waits_for_capacity: bool, ) -> crate::Result { let dispatch_started = Instant::now(); @@ -90,9 +94,13 @@ pub(super) fn dispatch_to_data_plane( let rx = shared.tracker.register(request_id); - let dispatched = match shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), + let dispatched = if waits_for_capacity { + dispatch_when_capacity_frees(shared, request).await + } else { + match shared.dispatcher.lock() { + Ok(mut d) => d.dispatch(request), + Err(poisoned) => poisoned.into_inner().dispatch(request), + } }; if dispatched.is_err() { // No response will ever arrive for a refused request. @@ -129,3 +137,37 @@ pub(super) fn dispatch_to_data_plane( deferred_guards, }) } + +/// Dispatch `request`, waiting for freed capacity after each capacity +/// refusal, until the request's deadline. Returns the last refusal once the +/// deadline passes. +async fn dispatch_when_capacity_frees(shared: &SharedState, request: Request) -> crate::Result<()> { + let deadline = tokio::time::Instant::from_std(request.deadline); + let capacity_freed = match shared.dispatcher.lock() { + Ok(d) => d.capacity_freed(), + Err(poisoned) => poisoned.into_inner().capacity_freed(), + }; + let mut request = request; + loop { + // Registered before the attempt, so a slot freed between the refusal + // and the wait still wakes it. + let freed = capacity_freed.notified(); + tokio::pin!(freed); + freed.as_mut().enable(); + let attempt = match shared.dispatcher.lock() { + Ok(mut d) => d.try_dispatch(request), + Err(poisoned) => poisoned.into_inner().try_dispatch(request), + }; + let refusal = match attempt { + Ok(()) => return Ok(()), + Err(refusal) => *refusal, + }; + if !matches!(refusal.error, crate::Error::DispatchCapacity { .. }) { + return Err(refusal.error); + } + if tokio::time::timeout_at(deadline, freed).await.is_err() { + return Err(refusal.error); + } + request = refusal.request; + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index 396f5a2bd..65ba3250e 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -1,7 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 -//! The funnel's orchestrator: runs admission, WAL append, dispatch, and -//! response classification in that fixed order for one write. +//! The funnel's enqueue phase: runs admission, WAL append, and dispatch in +//! that fixed order for one write. [`super::pending::PendingWrite::finish`] +//! runs the response phase. use crate::control::server::dispatch_utils::change_events::extract_write_change_set; use crate::control::server::dispatch_utils::durability_barrier::funnel_minted_redo_engine; @@ -11,19 +12,24 @@ use crate::control::server::shared::write_admission::{bare_ok_response, route_wr use crate::control::server::wal_dispatch; use crate::control::state::SharedState; -use super::super::params::{ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability}; +use super::super::params::{ + ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, +}; use super::admission::{AdmissionOutcome, admit_write}; use super::dispatch::{DispatchTarget, dispatch_to_data_plane}; -use super::response::{ResponsePhaseInput, collect_classify_and_finish}; +use super::pending::PendingWrite; +use super::response::{ResponsePhaseInput, UserWriteMark}; use super::wal_append::authorize_and_append; -/// Admit, make durable, enqueue, collect, and publish one write. +/// Admit, make durable, and enqueue one write on its core. /// -/// See [`SubmitOutcome`] for what comes back. -pub(crate) async fn submit_write( +/// The write is on its core's queue when this returns, so a caller that +/// enqueues writes one after another fixes their arrival order at the core. +/// [`PendingWrite::finish`] collects the outcome. +pub(crate) async fn enqueue_write( shared: &SharedState, params: SubmitWrite, -) -> crate::Result { +) -> crate::Result { let SubmitWrite { tenant_id, database_id, @@ -52,6 +58,29 @@ pub(crate) async fn submit_write( .sources() .is_source_collection(collection) }); + // Only a user data write advances the tenant's observed write-HLC, which + // the RESTORE staleness gate compares envelopes against. A schema install + // such as a constraint set writes no row, so it records no mark. The mark + // keeps the path and collection of the write, so a refused restore names it. + let user_write_origin = + crate::control::server::shared::write_admission::plan_writes_user_data(&plan).then(|| { + let site = match &durability { + WalDurability::AppendHere { apply_key: 0, .. } => "write funnel (autocommit)", + WalDurability::AppendHere { .. } => "write funnel (replicated apply)", + WalDurability::CallerSupplied { .. } => "write funnel (caller-appended)", + }; + let collection = plan.named_collections().first().map(|c| (*c).to_owned()); + (site, collection) + }); + // The instant the write committed, which is the value its mark carries: + // - a replicated entry carries its proposer's stamp; + // - a caller that appended upstream committed before this call, so the + // instant this call starts bounds it from above; + // - otherwise the append below is the commit, stamped once it lands. + let upstream_commit_hlc = match &durability { + WalDurability::AppendHere { commit_hlc, .. } => *commit_hlc, + WalDurability::CallerSupplied { .. } => Some(shared.hlc_clock.now().wall_ns), + }; // Records the caller appended for this write, under their outcome-floor // window. Every path below closes the window. let caller_minted = durability.take_minted(); @@ -93,6 +122,10 @@ pub(crate) async fn submit_write( // redelivered copy is never applied. A write no proposal carries has key // `0`. let final_refusal_key = apply_key; + // A write whose order is already final waits out a full dispatcher queue + // rather than failing: a committed entry that fails for local load leaves + // this replica without a write every other replica applied. + let waits_for_capacity = matches!(ordering, WriteOrdering::AlreadyOrdered); // Durable-at-ack obligation, also computed before `plan` moves. `Some` only // for a write whose redo record THIS funnel is required to mint; a caller @@ -131,14 +164,35 @@ pub(crate) async fn submit_write( superseded.finish().await; } let routed = routed?; - return Ok(SubmitOutcome { + return Ok(PendingWrite::done(SubmitOutcome { response: routed .unwrap_or_else(|| bare_ok_response(crate::types::RequestId::new(0))), wal_lsn: None, - }); + })); } }; + // On a server with no Raft groups a user write that mints its own record + // takes its commit stamp now, and its mark is durable before the mint: a + // crash after the record reaches disk keeps the mark. The stamp stays open + // until the mint, so a backup cut at or above it waits for the record. + let local_stamp = match &user_write_origin { + Some((_, collection)) + if appends_here + && caller_minted.is_none() + && upstream_commit_hlc.is_none() + && shared.async_raft_proposer().is_none() => + { + Some(shared.tenant_marks.stamp_local_write( + &shared.hlc_clock, + shared.credentials.catalog(), + tenant_id.as_u64(), + collection.as_deref(), + )?) + } + _ => None, + }; + // A write that mints its own LSN opens its outcome-floor window before the // mint, and appends through it. let minted = match caller_minted { @@ -159,6 +213,19 @@ pub(crate) async fn submit_write( return Err(error); } }; + let commit_hlc = local_stamp + .as_ref() + .map(|stamp| stamp.hlc()) + .or(upstream_commit_hlc) + .unwrap_or_else(|| shared.hlc_clock.now().wall_ns); + // The record is minted: a backup cut now waits for it through the outcome + // floor. + drop(local_stamp); + let user_write = user_write_origin.map(|(site, collection)| UserWriteMark { + site, + collection, + commit_hlc, + }); let ddl_transition = wal_append_outcome.ddl_transition; let plan = wal_append_outcome.plan; let wal_lsn = wal_append_outcome.wal_lsn; @@ -169,7 +236,7 @@ pub(crate) async fn submit_write( // with a higher LSN applies first. The write still holds its own per-key // admission guards, which no write to another key contends on. #[cfg(feature = "failpoints")] - crate::control::fail_gate::before_dispatch(&plan, wal_lsn).await; + crate::control::fail_gate::before_dispatch(shared.node_id, &plan, wal_lsn).await; // Build the wire request and hand it to the Data-Plane dispatcher. let dispatched = dispatch_to_data_plane( @@ -192,7 +259,9 @@ pub(crate) async fn submit_write( admission_guard, order_guard, post_apply.is_some(), - ); + waits_for_capacity, + ) + .await; let dispatch_outcome = match dispatched { Ok(outcome) => { // A core holds the request now. From here the records close @@ -213,12 +282,9 @@ pub(crate) async fn submit_write( } }; - // Collect response(s), classify the outcome, and run the post-apply steps - // a successful write still owes. - let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; - let outcome = collect_classify_and_finish( - shared, - max_result_bytes, + // The response phase collects the outcome and runs the post-apply steps a + // successful write still owes. + Ok(PendingWrite::dispatched( ResponsePhaseInput { request_id: dispatch_outcome.request_id, rx: dispatch_outcome.rx, @@ -237,16 +303,8 @@ pub(crate) async fn submit_write( ddl_transition, deferred_guards: dispatch_outcome.deferred_guards, minted, + user_write, }, - ) - .await?; - if binds_authorization { - crate::control::security::auth_lease::await_local_coverage( - shared, - std::time::Instant::now() - + std::time::Duration::from_secs(shared.tuning.network.default_deadline_secs), - ) - .await?; - } - Ok(outcome) + binds_authorization, + )) } diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs index dcaa1ca98..9e710788a 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs @@ -31,11 +31,14 @@ //! Plane. //! - [`response`]: collecting the response, classifying the outcome, and the //! post-apply steps a successful write still owes. +//! - [`pending`]: the write between its enqueue and its response phase. mod admission; mod dispatch; mod driver; +mod pending; mod response; mod wal_append; -pub(crate) use driver::submit_write; +pub(crate) use driver::enqueue_write; +pub(crate) use pending::{PendingWrite, submit_write}; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs new file mode 100644 index 000000000..5dc04cf15 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs @@ -0,0 +1,85 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A write the funnel enqueued on its core, and the response phase that +//! finishes it. +//! +//! The enqueue and the response phase are separate so a caller can enqueue +//! writes in a fixed order and collect their outcomes in any order. The +//! data-group apply loop does this: it enqueues committed entries in log +//! order and collects each outcome independently, so one parked write never +//! holds back the writes behind it. + +use crate::control::state::SharedState; + +use super::super::params::{SubmitOutcome, SubmitWrite}; +use super::driver::enqueue_write; +use super::response::{ResponsePhaseInput, collect_classify_and_finish}; + +/// A write past its enqueue. +pub(crate) struct PendingWrite { + stage: Stage, +} + +enum Stage { + /// The write has its outcome already: the Calvin scheduler applied it. + Done(SubmitOutcome), + /// A core holds the write. The response phase collects its outcome. + Dispatched { + input: Box, + /// The write changes a permission-tree source on a node with no + /// lease, so its ack waits until the local permission cache holds it. + binds_authorization: bool, + }, +} + +impl PendingWrite { + pub(super) fn done(outcome: SubmitOutcome) -> Self { + Self { + stage: Stage::Done(outcome), + } + } + + pub(super) fn dispatched(input: ResponsePhaseInput, binds_authorization: bool) -> Self { + Self { + stage: Stage::Dispatched { + input: Box::new(input), + binds_authorization, + }, + } + } + + /// Collect the outcome, classify it, and run every step a completed write + /// still owes before it is acknowledged. + /// + /// See [`SubmitOutcome`] for what comes back. + pub(crate) async fn finish(self, shared: &SharedState) -> crate::Result { + let (input, binds_authorization) = match self.stage { + Stage::Done(outcome) => return Ok(outcome), + Stage::Dispatched { + input, + binds_authorization, + } => (input, binds_authorization), + }; + let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; + let outcome = collect_classify_and_finish(shared, max_result_bytes, *input).await?; + if binds_authorization { + crate::control::security::auth_lease::await_local_coverage( + shared, + std::time::Instant::now() + + std::time::Duration::from_secs(shared.tuning.network.default_deadline_secs), + ) + .await?; + } + Ok(outcome) + } +} + +/// Admit, make durable, enqueue, collect, and publish one write. +/// +/// See [`SubmitOutcome`] for what comes back. +pub(crate) async fn submit_write( + shared: &SharedState, + params: SubmitWrite, +) -> crate::Result { + enqueue_write(shared, params).await?.finish(shared).await +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs index 77bb6dc2d..62c05d346 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -51,6 +51,20 @@ pub(super) struct ResponsePhaseInput { pub deferred_guards: super::dispatch::DeferredGuards, /// The records minted for this write, under their outcome-floor window. pub minted: Option, + /// `Some` when the plan writes user data, so a success advances the + /// tenant's observed write-HLC. Reads and system operations never do. + pub user_write: Option, +} + +/// The origin a successful user data write records on the tenant's observed +/// write-HLC. +pub(super) struct UserWriteMark { + /// Which funnel path dispatched the write. + pub site: &'static str, + /// The collection the plan wrote, when it named one. + pub collection: Option, + /// HLC wall time, in nanoseconds, at which the write committed. + pub commit_hlc: u64, } /// Collect the response(s), classify the outcome, and run every step a @@ -86,6 +100,7 @@ pub(super) async fn collect_classify_and_finish( ddl_transition, deferred_guards, minted, + user_write, } = input; let owner = RecordOwner { tenant_id, @@ -222,6 +237,25 @@ pub(super) async fn collect_classify_and_finish( // ops / trigger / staged-write dispatch carry no LSN and skip the barrier; // `durability_barrier` decides which of those skips are legitimate and makes // the rest loud instead of letting them ack a write nothing can recover. + // On a server with no Raft groups the write's mark lives in the catalog. + // It is durable before the ack. A write that minted its own record made it + // durable before the mint, so this costs no catalog commit there. + if response.status == Status::Ok + && let Some(mark) = &user_write + && shared.async_raft_proposer().is_none() + { + rollback_on_err( + shared, + &ddl_transition, + shared.tenant_marks.record_local_write( + shared.credentials.catalog(), + tenant_id.as_u64(), + mark.commit_hlc, + mark.collection.as_deref(), + ), + )?; + } + if response.status == Status::Ok { let durable_target = match (wal_lsn, post_apply_lsn) { (Some(a), Some(b)) => Some(a.max(b)), @@ -252,12 +286,19 @@ pub(super) async fn collect_classify_and_finish( publish_change_set(shared, tenant_id, database_id, change_set, &response); } - // Advance the tenant's observed write-HLC high-water on any successful - // dispatch. Used by the RESTORE staleness gate. Advancing on every - // success (not just writes) is intentionally conservative — - // envelope.watermark is captured AFTER fan-out so it always dominates - // the tenant_wm of a fresh backup. - shared.advance_tenant_write_hlc(tenant_id.as_u64()); + // Record the write's commit HLC on the tenant's observed high-water + // before this response, the ack, returns. The RESTORE staleness gate + // refuses an envelope older than the mark, so the mark is the instant + // the write committed, never the instant this bookkeeping ran: a + // backup taken after the ack then always carries a newer watermark. + if let Some(mark) = &user_write { + shared.advance_tenant_write_hlc( + tenant_id.as_u64(), + mark.commit_hlc, + mark.site, + mark.collection.as_deref(), + ); + } } observe(shared); diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs index b974711d3..ffa055e8b 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs @@ -86,6 +86,7 @@ pub(super) fn authorize_and_append( WalDurability::AppendHere { now_override, apply_key, + .. } => { let outcome = rollback_on_err( shared, diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs b/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs index 7da9ef9a5..e6cdd4c6c 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs @@ -4,7 +4,7 @@ mod ambiguous_ddl; mod funnel; mod params; -pub(crate) use funnel::submit_write; +pub(crate) use funnel::{PendingWrite, enqueue_write, submit_write}; pub(crate) use params::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, }; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs index 1ca820a09..7bdaab306 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs @@ -26,9 +26,15 @@ pub(crate) enum WalDurability { /// write applies, `0` for a write no proposal carries. Every record the /// funnel appends for the write carries it in its header, so the record /// names the proposal it applied in the same durable write. + /// + /// `commit_hlc` is the HLC wall time, in nanoseconds, at which the write + /// committed upstream: the proposer's stamp on a replicated entry. `None` + /// when this append is the commit, so the funnel stamps the instant of the + /// append itself. AppendHere { now_override: Option, apply_key: u64, + commit_hlc: Option, }, /// The caller already recorded this write's durability elsewhere — COMMIT's /// single `Transaction` record, the procedural batch flush, a trigger / diff --git a/nodedb/src/control/server/exchange/all_cores/dispatch.rs b/nodedb/src/control/server/exchange/all_cores/dispatch.rs index a80b18578..7d1ea7905 100644 --- a/nodedb/src/control/server/exchange/all_cores/dispatch.rs +++ b/nodedb/src/control/server/exchange/all_cores/dispatch.rs @@ -195,7 +195,17 @@ pub(crate) async fn execute_plan_all_local_cores( } } -/// Generic gather path: delegate to [`gather_all_cores`] and wrap. +/// Generic gather path. +/// +/// A plan on one collection that is not cluster-partitioned lives wholly on +/// the core that owns the collection's vShard. It runs there alone, and its +/// payload returns verbatim: the shape a single core produces. That shape is +/// not always a msgpack array. A KV point read answers with the stored value +/// itself, and wrapping it as an array element hands the requesting node a +/// different value than a local read returns. +/// +/// Every other plan fans across all local cores, and their row arrays merge +/// into one. async fn generic_gather( state: &SharedState, tenant_id: TenantId, @@ -205,9 +215,31 @@ async fn generic_gather( txn_id: Option, ) -> crate::Result { use crate::control::server::exchange::gather::gather_all_cores; + use crate::control::server::exchange::owning_core::dispatch_single_owning_core; // Forwarded `txn_id`, if any, is stamped on each core's request so a // transactional read honours its staged overlay. Inert when `None`. + if !nodedb_physical::physical_plan::plan_contains_cluster_partitioned_leaf(&plan) + && let Some(collection) = plan.collection() + { + let vshard_id = + crate::types::VShardId::from_collection_in_database(database_id, collection); + let resp = dispatch_single_owning_core( + state, + tenant_id, + database_id, + plan, + vshard_id, + trace_id, + txn_id, + ) + .await?; + return Ok(NodeLevelResult { + payload: resp.payload.to_vec(), + watermark_lsn: resp.watermark_lsn, + read_version_lsn: resp.read_version_lsn, + }); + } let outcome = gather_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id).await?; Ok(NodeLevelResult { payload: outcome.merged_array, diff --git a/nodedb/src/control/server/exchange/all_cores/snapshot.rs b/nodedb/src/control/server/exchange/all_cores/snapshot.rs index ffa8ba8b9..e81804797 100644 --- a/nodedb/src/control/server/exchange/all_cores/snapshot.rs +++ b/nodedb/src/control/server/exchange/all_cores/snapshot.rs @@ -74,6 +74,9 @@ pub(super) async fn fan_tenant_snapshot_all_cores( index_configs, surrogate_pk, tenant_edges, + group_write_marks, + documents_versioned, + indexes_versioned, } = part; merged.documents.extend(documents); merged.indexes.extend(indexes); @@ -89,6 +92,9 @@ pub(super) async fn fan_tenant_snapshot_all_cores( merged.index_configs.extend(index_configs); merged.surrogate_pk.extend(surrogate_pk); merged.tenant_edges.extend(tenant_edges); + merged.group_write_marks.extend(group_write_marks); + merged.documents_versioned.extend(documents_versioned); + merged.indexes_versioned.extend(indexes_versioned); } let payload = zerompk::to_msgpack_vec(&merged).map_err(|e| crate::Error::Serialization { diff --git a/nodedb/src/control/server/exchange/owning_core.rs b/nodedb/src/control/server/exchange/owning_core.rs index 2727e77a0..94fe23547 100644 --- a/nodedb/src/control/server/exchange/owning_core.rs +++ b/nodedb/src/control/server/exchange/owning_core.rs @@ -90,6 +90,40 @@ pub async fn gather_single_owning_core( trace_id: TraceId, txn_id: Option, ) -> crate::Result { + let resp = dispatch_single_owning_core( + state, + tenant_id, + database_id, + plan, + vshard_id, + trace_id, + txn_id, + ) + .await?; + let payload_bytes: &[u8] = resp.payload.as_ref(); + let all_elements = extract_msgpack_elements(payload_bytes); + let merged_array = encode_msgpack_array(&all_elements); + + Ok(GatherOutcome { + raw: payload_bytes.to_vec(), + merged_array, + watermark_lsn: resp.watermark_lsn, + read_version_lsn: resp.read_version_lsn, + shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], + }) +} + +/// Dispatch `plan` to the single Data-Plane core that owns `vshard_id` and +/// return that core's response, its payload in the shape the core produced. +pub async fn dispatch_single_owning_core( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: PhysicalPlan, + vshard_id: VShardId, + trace_id: TraceId, + txn_id: Option, +) -> crate::Result { // `Box::pin` breaks an async-fn recursion cycle: `dispatch_to_data_plane_*` // re-enters `resolve_exchange_in_plan`. The plan handed here is the bare, // Exchange-free child of the resolved Gather, so the re-entrant resolve is a @@ -109,16 +143,5 @@ pub async fn gather_single_owning_core( // validatable) observation; any other error status surfaces with its typed // code rather than being swallowed as an empty success. reject_data_plane_error(&resp)?; - - let payload_bytes: &[u8] = resp.payload.as_ref(); - let all_elements = extract_msgpack_elements(payload_bytes); - let merged_array = encode_msgpack_array(&all_elements); - - Ok(GatherOutcome { - raw: payload_bytes.to_vec(), - merged_array, - watermark_lsn: resp.watermark_lsn, - read_version_lsn: resp.read_version_lsn, - shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], - }) + Ok(resp) } diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs b/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs index 67c5fe2aa..b8ab69fa0 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs @@ -1,7 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Public dispatch entry points, and the write-HLC bookkeeping wrapper around -//! them. +//! Public dispatch entry points. +//! +//! A write records its commit HLC on the tenant's observed high-water where it +//! commits: the write funnel for a local append, the Raft proposer and each +//! replica's apply for a replicated entry. use std::sync::Arc; @@ -27,7 +30,7 @@ impl NodeDbPgHandler { ) -> crate::Result { let mut shard_watermarks = Vec::new(); let mut distributed_reads = Vec::new(); - self.dispatch_task_hlc( + self.dispatch_task_inner( task, user_id, identity, @@ -49,7 +52,7 @@ impl NodeDbPgHandler { let mut shard_watermarks = Vec::new(); let mut distributed_reads = Vec::new(); let resp = self - .dispatch_task_hlc( + .dispatch_task_inner( task, user_id, identity, @@ -59,26 +62,4 @@ impl NodeDbPgHandler { .await?; Ok((resp, shard_watermarks, distributed_reads)) } - - async fn dispatch_task_hlc( - &self, - task: PhysicalTask, - user_id: Option>, - identity: &AuthenticatedIdentity, - shard_watermarks: &mut Vec<(VShardId, Lsn)>, - distributed_reads: &mut Vec, - ) -> crate::Result { - let tenant_id = task.tenant_id; - let result = self - .dispatch_task_inner(task, user_id, identity, shard_watermarks, distributed_reads) - .await; - // Advances per-tenant write-HLC on any successful dispatch; used by RESTORE's - // staleness gate. Backup captures its watermark after fan-out, so it dominates. - if let Ok(ref resp) = result - && resp.status == crate::bridge::envelope::Status::Ok - { - self.state.advance_tenant_write_hlc(tenant_id.as_u64()); - } - result - } } diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs index 62043d722..553ef4a97 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs @@ -29,6 +29,7 @@ impl NodeDbPgHandler { WalDurability::AppendHere { now_override: None, apply_key: 0, + commit_hlc: None, }, ) .await diff --git a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs index 97b4b21a7..11067e398 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs @@ -26,7 +26,7 @@ use nodedb_physical::physical_task::PhysicalTask; use super::super::core::NodeDbPgHandler; use super::super::plan::describe_plan; -use super::gateway_fold::{GatewayFold, GatewayShaping}; +use super::gateway_fold::{GatewayFold, GatewayShaping, plan_produces_rows}; /// Meter one gateway-forwarded task, once its response has already shaped /// successfully — mirrors `calvin_dispatch::meter_calvin_task`, the sibling @@ -99,6 +99,8 @@ impl NodeDbPgHandler { // Resolved once for the whole forwarded task set, before the loop. let redaction = QueryRedaction::for_plans(tenant_id, auth, tasks.iter().map(|t| &t.plan)); let shaping = GatewayShaping { + tenant_id, + database_id, projection, result_formats, redaction: &redaction, @@ -127,6 +129,9 @@ impl NodeDbPgHandler { let mut fold = GatewayFold::with_capacity(tasks.len()); for task in tasks { let plan_kind = describe_plan(&task.plan); + // The task moves into authorization below; a row-producing task + // keeps its plan for shaping the rows it answers with. + let shape_plan = plan_produces_rows(plan_kind).then(|| task.plan.clone()); let counts_toward_tag = plan_counts_toward_statement_tag(&task.plan, has_user_write); let metering_info = PlanMeteringInfo::extract(&task.plan); let emitter = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( @@ -164,6 +169,7 @@ impl NodeDbPgHandler { &mut fold, resp.payload.as_ref(), plan_kind, + shape_plan.as_ref(), counts_toward_tag, &shaping, )?; @@ -201,6 +207,7 @@ impl NodeDbPgHandler { &mut fold, &[], plan_kind, + shape_plan.as_ref(), counts_toward_tag, &shaping, )?; @@ -210,6 +217,7 @@ impl NodeDbPgHandler { &mut fold, payload, plan_kind, + shape_plan.as_ref(), counts_toward_tag, &shaping, )? { diff --git a/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs b/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs index b84981b01..a897a9f2e 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/gateway_fold.rs @@ -8,8 +8,10 @@ use pgwire::api::results::{FieldFormat, Response}; use pgwire::error::PgWireResult; +use crate::bridge::envelope::PhysicalPlan; use crate::control::server::response_shape::compose::{self, ShapeOutcome}; use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::request::MaterializedShapeRequest; use crate::control::server::response_shape::schema::OutputSchema; use crate::control::server::response_shape::types::{ ShapedRows, StatementTag, payload_to_dml_outcome, @@ -24,6 +26,8 @@ use super::super::shape_encode; /// How forwarded rows shape back to the client. pub(super) struct GatewayShaping<'a> { + pub(super) tenant_id: crate::types::TenantId, + pub(super) database_id: crate::types::DatabaseId, pub(super) projection: Option<&'a OutputSchema>, pub(super) result_formats: &'a [FieldFormat], /// Resolved once over the whole forwarded task set. @@ -85,6 +89,15 @@ impl NodeDbPgHandler { /// not answer the statement (`counts_toward_tag` false: a derived /// implicit-edge write beside the user's own) folds as opaque. /// + /// A row-producing payload shapes exactly as a locally dispatched one + /// does, through `shape_response_materialized` with the task's plan. A + /// forwarded Data-Plane payload has the shape a local core produces, and + /// the plan-dependent steps (the KV point-get `{key, value}` row wrap, the + /// vector surrogate-to-PK translation) must run on it too: shaped without + /// them, a KV point read's stored value decodes as a scalar and yields no + /// row. `shape_plan` is the task's plan, which a row-producing kind always + /// carries (see [`plan_produces_rows`]). + /// /// Returns the rows the payload decoded to, `None` for a passthrough /// payload with no row count. pub(super) fn fold_gateway_payload( @@ -92,21 +105,36 @@ impl NodeDbPgHandler { fold: &mut GatewayFold, payload: &[u8], plan_kind: PlanKind, + shape_plan: Option<&PhysicalPlan>, counts_toward_tag: bool, shaping: &GatewayShaping<'_>, ) -> PgWireResult> { - // Gateway forwarding carries no sequence access: a projection with - // Control-Plane computed columns is refused by the shaper rather - // than NULL-filled. - match compose::shape_payload_no_plan( - payload, - plan_kind, - shaping.projection, - Some(shaping.redaction.ctx(&self.state.redaction)), - None, - ) - .map_err(|e| shape_error_to_pg(&e))? - { + let outcome = match (plan_produces_rows(plan_kind), shape_plan) { + (false, _) => ShapeOutcome::Passthrough, + // Gateway forwarding carries no sequence access: a projection + // with Control-Plane computed columns is refused by the shaper + // rather than NULL-filled. + (true, Some(plan)) => compose::shape_response_materialized(MaterializedShapeRequest { + payload, + plan, + plan_kind, + projection: shaping.projection, + state: &self.state, + database_id: shaping.database_id, + tenant_id: shaping.tenant_id, + redaction: Some(shaping.redaction.ctx(&self.state.redaction)), + sequences: None, + }) + .map_err(|e| shape_error_to_pg(&e))?, + (true, None) => { + return Err(error_to_pg(&crate::Error::Internal { + detail: format!( + "gateway fold: a {plan_kind:?} task reached shaping without its plan" + ), + })); + } + }; + match outcome { ShapeOutcome::Rows(shaped) => { let rows = shaped.rows.len() as u64; if matches!(plan_kind, PlanKind::ReturningRows) { @@ -143,3 +171,15 @@ impl NodeDbPgHandler { } } } + +/// Whether a plan of `kind` answers with rows, which shape through the +/// task's plan. Every other kind folds into the command tag. +pub(super) fn plan_produces_rows(kind: PlanKind) -> bool { + match kind { + PlanKind::SingleDocument + | PlanKind::MultiRow + | PlanKind::ArraySlice + | PlanKind::ReturningRows => true, + PlanKind::Execution | PlanKind::DmlResult(_) | PlanKind::DmlResultByOp => false, + } +} diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 055211fd5..553ed1633 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -136,6 +136,16 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::AuthorizationStateBehind { .. } => { ("ERROR", sqlstate::STALE_READ_NOT_LEADER, err.to_string()) } + // Nothing was applied, and a retry succeeds once the group's majority + // is reachable again. + crate::Error::GroupQuorumUnavailable { .. } => { + ("ERROR", sqlstate::LOCK_NOT_AVAILABLE, err.to_string()) + } + // Nothing was restored, and a retry succeeds once a replica of the + // group answers. + crate::Error::GroupMarksUnavailable { .. } => { + ("ERROR", sqlstate::LOCK_NOT_AVAILABLE, err.to_string()) + } crate::Error::ConflictRetry { .. } => { ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) } diff --git a/nodedb/src/control/server/response_shape/compose/materialized.rs b/nodedb/src/control/server/response_shape/compose/materialized.rs index ce00ae8fd..9f4c97ed4 100644 --- a/nodedb/src/control/server/response_shape/compose/materialized.rs +++ b/nodedb/src/control/server/response_shape/compose/materialized.rs @@ -12,8 +12,10 @@ //! encoder; each protocol then encodes those rows in its own wire format //! (pgwire's RowDescription/DataRow, native's MessagePack, http's JSON). //! -//! Producers with no `PhysicalPlan` in scope (ClusterArray, set-op merges, -//! gateway forwarding, clone merges) call [`shape_payload_no_plan`], which +//! Gateway forwarding shapes through this same call with the forwarded task's +//! plan, so a forwarded payload yields the rows a local one does. Producers +//! with no `PhysicalPlan` in scope (ClusterArray, set-op merges, clone merges) +//! call [`shape_payload_no_plan`], which //! skips the plan-dependent `apply_kv_wrap` / `translate_search_response` //! transforms those callers never ran. The pure kernel `shape_decoded_rows` //! is shared with per-batch lazy streaming callers, which have an @@ -102,7 +104,7 @@ pub fn shape_response_materialized( /// Shape a Data-Plane payload with no `PhysicalPlan` in scope. /// /// Producers that never had a plan to KV-wrap or vector-translate -/// (ClusterArray, set-op merges, gateway forwarding, clone merges) call this +/// (ClusterArray, set-op merges, clone merges) call this /// instead of [`shape_response_materialized`]: it applies only the decode + /// scan-envelope unwrap + optional SELECT-list projection steps, skipping the /// plan-dependent `apply_kv_wrap` / `translate_search_response` transforms those diff --git a/nodedb/src/control/server/session_auth/bearer_jwt.rs b/nodedb/src/control/server/session_auth/bearer_jwt.rs index d396ff695..afef1e7b9 100644 --- a/nodedb/src/control/server/session_auth/bearer_jwt.rs +++ b/nodedb/src/control/server/session_auth/bearer_jwt.rs @@ -44,7 +44,7 @@ pub async fn authenticate_bearer_jwt( if let Err(error) = crate::control::security::jwt_policy::enforce_stateful_jwt_policy( state, verified.claims(), - identity.tenant_id, + &identity, ) { debug!(%error, "bearer token refused by auth.jwt policy"); return None; diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs index 61b4859e1..4c848fcdc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs @@ -36,6 +36,7 @@ pub async fn run( ) -> Result { let plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: tenant_id.as_u64(), + cut_watermark: None, }); // Route to the source database: the snapshot reads the tenant's live // data from the database it is being moved out of. diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs index 9dab6a3ab..52bf96302 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs @@ -25,7 +25,6 @@ pub(crate) async fn dispatch_system( task: SystemTask<'_>, timeout: Duration, ) -> crate::Result> { - let tenant_id = task.tenant_id; let resp = dispatch_system_response_with_source(state, task, timeout, crate::event::EventSource::User) .await?; @@ -42,16 +41,9 @@ pub(crate) async fn dispatch_system( return Err(crate::Error::Internal { detail }); } - // Advance the tenant's observed write-HLC high-water. Used by RESTORE to - // reject stale envelopes. Tracking on every dispatch (not just known-write - // ops) is intentional: advance is monotonic, and capturing the backup - // envelope's watermark AFTER its own fan-out ensures envelope.wm >= - // tenant_wm on a fresh backup (so a same-cluster roundtrip passes the - // staleness gate). Reached only after the `resp.status != Ok` early-return - // above, so this point is the "success" branch per the - // advance_tenant_write_hlc contract. - state.advance_tenant_write_hlc(tenant_id.as_u64()); - + // A system task never advances the tenant's observed write-HLC. Its + // `SystemReason` states that no client asked for the work, and RESTORE's + // staleness gate counts only user data writes. Ok(resp.payload.to_vec()) } diff --git a/nodedb/src/control/server/shared/returning/inject.rs b/nodedb/src/control/server/shared/returning/inject.rs index 9c00d0959..edfa0b97c 100644 --- a/nodedb/src/control/server/shared/returning/inject.rs +++ b/nodedb/src/control/server/shared/returning/inject.rs @@ -438,7 +438,9 @@ pub fn inject_returning_spec(plan: &mut PhysicalPlan, spec: ReturningSpec) { | ClusterArrayOp::Delete { .. }, ) | PhysicalPlan::ClusterEvent( - ClusterEventOp::ConsumeStream { .. } | ClusterEventOp::PublishTopic { .. }, + ClusterEventOp::ConsumeStream { .. } + | ClusterEventOp::PublishTopic { .. } + | ClusterEventOp::TenantWriteMarks { .. }, ) => {} } } diff --git a/nodedb/src/control/server/shared/session/commit/single_shard.rs b/nodedb/src/control/server/shared/session/commit/single_shard.rs index 2ffddc7af..bb48c970c 100644 --- a/nodedb/src/control/server/shared/session/commit/single_shard.rs +++ b/nodedb/src/control/server/shared/session/commit/single_shard.rs @@ -166,7 +166,7 @@ async fn commit_redo( } } // No Raft: this node is the only replica, and its apply is the log. - None => match apply_transaction_redo(state, target, payload, 0).await { + None => match apply_transaction_redo(state, target, payload, 0, None).await { Ok(outcome) if outcome.response.status == Status::Ok => None, Ok(outcome) => Some(AbortReason::BatchRejected { code: outcome.response.error_code.as_deref().cloned(), diff --git a/nodedb/src/control/server/shared/write_admission/mod.rs b/nodedb/src/control/server/shared/write_admission/mod.rs index 1fc46d64f..90d073ab9 100644 --- a/nodedb/src/control/server/shared/write_admission/mod.rs +++ b/nodedb/src/control/server/shared/write_admission/mod.rs @@ -11,6 +11,8 @@ pub mod route; pub mod write_order_lock; pub use gate::{WriteAdmission, WriteAdmissionGuard, WriteTarget, admit, cp_routed_to_calvin}; -pub use predicate::{all_writes_bufferable, plan_is_write, plan_requires_txn_buffering}; +pub use predicate::{ + all_writes_bufferable, plan_is_write, plan_requires_txn_buffering, plan_writes_user_data, +}; pub use route::{bare_ok_response, route_write_to_calvin}; pub use write_order_lock::KeyedWriteOrderLock; diff --git a/nodedb/src/control/server/shared/write_admission/predicate/mod.rs b/nodedb/src/control/server/shared/write_admission/predicate/mod.rs index 61323a586..4a2e2474a 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/mod.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/mod.rs @@ -4,6 +4,8 @@ pub mod plan_is_write; pub mod txn_buffering; +pub mod user_data_write; pub use plan_is_write::plan_is_write; pub use txn_buffering::{all_writes_bufferable, plan_requires_txn_buffering}; +pub use user_data_write::plan_writes_user_data; diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index 891b8278c..ef8416f70 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -1780,7 +1780,10 @@ mod tests { schema_json: "{}".into(), source_storage_mode: nodedb_physical::physical_plan::StorageMode::Schemaless, }), - PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: 1 }), + PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: 1, + cut_watermark: None, + }), PhysicalPlan::Meta(MetaOp::RestoreTenantSnapshot { tenant_id: 1, snapshot: Vec::new(), diff --git a/nodedb/src/control/server/shared/write_admission/predicate/user_data_write.rs b/nodedb/src/control/server/shared/write_admission/predicate/user_data_write.rs new file mode 100644 index 000000000..19e98dc53 --- /dev/null +++ b/nodedb/src/control/server/shared/write_admission/predicate/user_data_write.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The user-data write predicate: the writes a tenant's write marks count. +//! +//! RESTORE's staleness guard refuses an envelope older than the tenant's +//! newest user-data write. A write-class plan that installs schema state on a +//! replica changes no row. The node proposes it on its own, for example the +//! constraint reconcile loop on every boot. Counting it would refuse a restore +//! of a backup that already holds every row the tenant wrote. + +use crate::bridge::envelope::PhysicalPlan; +use nodedb_physical::physical_plan::{CrdtOp, KvOp}; + +use super::plan_is_write::plan_is_write; + +/// Whether `plan` writes a tenant's rows: a write-class plan that is not a +/// schema install. +pub fn plan_writes_user_data(plan: &PhysicalPlan) -> bool { + plan_is_write(plan) && !installs_schema(plan) +} + +/// A write-class plan that installs or removes schema state on a replica and +/// changes no row. +fn installs_schema(plan: &PhysicalPlan) -> bool { + matches!( + plan, + PhysicalPlan::Crdt(CrdtOp::SetConstraints { .. } | CrdtOp::DropConstraints { .. }) + | PhysicalPlan::Kv( + KvOp::RegisterIndex { .. } + | KvOp::DropIndex { .. } + | KvOp::RegisterSortedIndex { .. } + | KvOp::DropSortedIndex { .. } + ) + ) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate}; + + fn collection() -> QualifiedCollection { + QualifiedCollection::new(DatabaseId::DEFAULT, "c") + } + + #[test] + fn a_constraint_install_is_not_a_user_data_write() { + let set = PhysicalPlan::Crdt(CrdtOp::SetConstraints { + collection: collection(), + constraint_version: 1, + constraints: Vec::new(), + }); + let drop = PhysicalPlan::Crdt(CrdtOp::DropConstraints { + collection: collection(), + constraint_version: 2, + }); + assert!(plan_is_write(&set), "the install stays write-class"); + assert!(!plan_writes_user_data(&set)); + assert!(!plan_writes_user_data(&drop)); + } + + #[test] + fn a_kv_put_is_a_user_data_write() { + let plan = PhysicalPlan::Kv(KvOp::Put { + collection: collection(), + key: b"k".to_vec(), + value: b"v".to_vec(), + ttl_ms: 0, + surrogate: Surrogate::ZERO, + returning: None, + rls_filters: Vec::new(), + }); + assert!(plan_writes_user_data(&plan)); + } +} diff --git a/nodedb/src/control/server/sync/raft_dispatch/propose.rs b/nodedb/src/control/server/sync/raft_dispatch/propose.rs index 808b1418b..532bde0b5 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/propose.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/propose.rs @@ -30,7 +30,8 @@ use crate::control::wal_replication::{AsyncRaftProposer, ReplicatedEntry}; /// to determine the idempotency gate verdict. /// /// Retries transparently up to five times on [`crate::Error::RetryableLeaderChange`] -/// (leader failover during the propose). Only propose-layer machinery failures +/// (leader failover during the propose). All attempts share one statement +/// deadline, so a retry gets only the time that remains. Only propose-layer machinery failures /// map to [`crate::Error::Dispatch`]; a classified apply verdict passes through. pub(crate) async fn propose_sync_write( state: &SharedState, @@ -40,13 +41,15 @@ pub(crate) async fn propose_sync_write( let idempotency_key = entry.idempotency_key; let data = entry.to_bytes(); let vshard_id = entry.vshard_id; + let deadline = tokio::time::Instant::now() + + Duration::from_secs(state.tuning.network.default_deadline_secs); const BACKOFF_MS: [u64; 5] = [10, 25, 50, 100, 200]; let mut payload: Option> = None; let mut last_err: Option = None; for (attempt, backoff_ms) in BACKOFF_MS.iter().enumerate() { - match proposer(vshard_id, idempotency_key, data.clone()).await { + match proposer(vshard_id, idempotency_key, data.clone(), deadline).await { // The committed log index rides alongside the payload; the sync-ack // path only needs the payload bytes. Ok((p, _committed_version)) => { @@ -70,7 +73,11 @@ pub(crate) async fn propose_sync_write( group_id, log_index, }); - tokio::time::sleep(Duration::from_millis(*backoff_ms)).await; + let backoff = Duration::from_millis(*backoff_ms); + if tokio::time::Instant::now() + backoff >= deadline { + break; + } + tokio::time::sleep(backoff).await; continue; } // Only a machinery failure is re-wrapped. A state-machine verdict diff --git a/nodedb/src/control/server/sync/raft_dispatch/write.rs b/nodedb/src/control/server/sync/raft_dispatch/write.rs index 48d5864ef..a931a1a92 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/write.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/write.rs @@ -74,6 +74,9 @@ pub(crate) async fn dispatch_write_replicated( event_source: EventSource, minted: Option, ) -> crate::Result> { + // The caller appended this write's records before the call, so this + // instant bounds its commit from above and precedes the ack. + let committed_at = state.hlc_clock.now().wall_ns; let task = authorized.into_physical_task(); let tenant_id = task.tenant_id; let database_id = task.database_id; @@ -190,14 +193,30 @@ pub(crate) async fn dispatch_write_replicated( // substring-matching a message. let payload = payload_or_typed_error(resp)?; + // On a node with no Raft groups the write's mark lives in the catalog, and + // it is durable before the ack. + if state.async_raft_proposer().is_none() { + state.tenant_marks.record_local_write( + state.credentials.catalog(), + tenant_id.as_u64(), + committed_at, + Some(collection), + )?; + } + // System-task dispatch bypasses the write funnel's own durable-at-ack barrier — // without this fsync, `kill -9` erases an acked write. if let Some(lsn) = wal_lsn { state.wal.wait_durable(lsn).await?; } - // Mirrors `dispatch_system_with_source`'s success-path write-HLC advance. - state.advance_tenant_write_hlc(tenant_id.as_u64()); + // A device's sync write is user data: RESTORE's staleness gate counts it. + state.advance_tenant_write_hlc( + tenant_id.as_u64(), + committed_at, + "sync write", + Some(collection), + ); Ok(payload) } @@ -405,7 +424,7 @@ mod tests { async fn a_failed_cancel_keeps_the_proposed_result_and_holds_the_window() { let (state, _side, _directory) = fixture(); let raw: Arc = - Arc::new(|_vshard, _key, _data| { + Arc::new(|_vshard, _key, _data, _deadline| { Box::pin(async { Ok((b"applied".to_vec(), crate::types::Lsn::ZERO)) }) }); crate::control::vshard_admission::install_async_raft_proposer(&state, raw) diff --git a/nodedb/src/control/server/wal_dispatch/core.rs b/nodedb/src/control/server/wal_dispatch/core.rs index 667a5bacf..50c419182 100644 --- a/nodedb/src/control/server/wal_dispatch/core.rs +++ b/nodedb/src/control/server/wal_dispatch/core.rs @@ -245,6 +245,7 @@ mod tests { redo: redo.clone(), collections: Vec::new(), sum_targets: Vec::new(), + origin: nodedb_physical::physical_plan::RedoOrigin::Commit, }, ); diff --git a/nodedb/src/control/shutdown/registry.rs b/nodedb/src/control/shutdown/registry.rs index 98d30919b..6818b2ce9 100644 --- a/nodedb/src/control/shutdown/registry.rs +++ b/nodedb/src/control/shutdown/registry.rs @@ -191,18 +191,10 @@ impl LoopRegistry { } = entry; let (mut join, can_abort) = handle.take_handle(); + // Polled even with no budget left: `timeout` polls the handle + // before its timer, so a loop that already exited while an + // earlier one used up the budget is counted clean. let elapsed_budget = deadline.saturating_sub(start.elapsed()); - if elapsed_budget.is_zero() { - // Deadline already consumed — treat anything - // still outstanding as a laggard without - // awaiting. - laggards.push( - abort_and_report_laggard(&mut join, name, registered_at, start, can_abort) - .await, - ); - continue; - } - match tokio::time::timeout(elapsed_budget, &mut join).await { Ok(Ok(())) => exited_clean.push(name), Ok(Err(join_err)) => { @@ -288,11 +280,10 @@ impl LoopRegistry { let (mut join, can_abort) = handle.take_handle(); let elapsed_budget = deadline.saturating_sub(start.elapsed()); - let joined = if elapsed_budget.is_zero() { - None - } else { - tokio::time::timeout(elapsed_budget, &mut join).await.ok() - }; + // Polled even with no budget left: `timeout` polls the handle + // before its timer, so a loop that already exited while an + // earlier one used up the budget is counted clean, not aborted. + let joined = tokio::time::timeout(elapsed_budget, &mut join).await.ok(); match joined { Some(Ok(())) => exited_clean.push(name), Some(Err(join_err)) => { @@ -595,6 +586,44 @@ mod tests { assert!(!report.laggards[0].aborted); } + /// A loop that overruns the budget must not make every loop joined after + /// it a laggard: those exited on the signal, within the budget. + #[tokio::test] + async fn a_slow_loop_does_not_make_later_loops_laggards() { + let watch = Arc::new(ShutdownWatch::new()); + let registry = LoopRegistry::new(); + registry + .register( + "slow", + ShutdownPhase::DrainingControlPlane, + LoopHandle::Async(tokio::spawn(async { + tokio::time::sleep(Duration::from_secs(10)).await; + })), + ) + .expect("register slow loop"); + let mut prompt_rx = watch.subscribe(); + registry + .register( + "prompt", + ShutdownPhase::DrainingControlPlane, + LoopHandle::Async(tokio::spawn( + async move { prompt_rx.wait_cancelled().await }, + )), + ) + .expect("register prompt loop"); + + let report = registry + .shutdown_phase_strict( + &watch, + ShutdownPhase::DrainingControlPlane, + Duration::from_millis(50), + ) + .await; + assert_eq!(report.exited_clean, vec!["prompt"]); + let laggards: Vec<&str> = report.laggards.iter().map(|l| l.name).collect(); + assert_eq!(laggards, vec!["slow"]); + } + #[tokio::test] async fn control_plane_drain_joins_only_its_own_phase() { let watch = Arc::new(ShutdownWatch::new()); diff --git a/nodedb/src/control/state/calvin_cuts.rs b/nodedb/src/control/state/calvin_cuts.rs new file mode 100644 index 000000000..1507a7000 --- /dev/null +++ b/nodedb/src/control/state/calvin_cuts.rs @@ -0,0 +1,98 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The backup cut markers each Calvin scheduler on this node passed. +//! +//! A scheduler passes a marker once every transaction delivered to it before +//! the marker finished. A backup's cut proposes a marker carrying its +//! watermark and waits here until every scheduler this node runs passed it. + +use std::collections::BTreeMap; +use std::sync::Mutex; + +use tokio::sync::Notify; + +/// Highest marker watermark each local scheduler passed, by vShard. +#[derive(Debug, Default)] +pub struct CalvinCuts { + passed: Mutex>, + changed: Notify, +} + +impl CalvinCuts { + /// Register the scheduler of `vshard_id`. A cut waits on it from now on. + pub fn register(&self, vshard_id: u32) { + self.passed + .lock() + .unwrap_or_else(|p| p.into_inner()) + .entry(vshard_id) + .or_insert(0); + } + + /// Record that the scheduler of `vshard_id` passed the marker `hlc`. + pub fn note_passed(&self, vshard_id: u32, hlc: u64) { + { + let mut passed = self.passed.lock().unwrap_or_else(|p| p.into_inner()); + let highest = passed.entry(vshard_id).or_insert(0); + *highest = (*highest).max(hlc); + } + self.changed.notify_waiters(); + } + + /// Whether any scheduler runs on this node. + pub fn is_empty(&self) -> bool { + self.passed + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_empty() + } + + /// The vShards whose scheduler has not passed the marker `hlc`. + pub fn lagging(&self, hlc: u64) -> Vec { + self.passed + .lock() + .unwrap_or_else(|p| p.into_inner()) + .iter() + .filter(|(_, passed)| **passed < hlc) + .map(|(vshard_id, _)| *vshard_id) + .collect() + } + + /// Wait until every scheduler passed the marker `hlc`, or `deadline`. + /// Returns the vShards still lagging, empty once every one passed. + pub async fn await_passed(&self, hlc: u64, deadline: tokio::time::Instant) -> Vec { + loop { + // Registered before the check, so a pass between the check and + // the wait still wakes it. + let changed = self.changed.notified(); + tokio::pin!(changed); + changed.as_mut().enable(); + let lagging = self.lagging(hlc); + if lagging.is_empty() { + return lagging; + } + if tokio::time::timeout_at(deadline, changed).await.is_err() { + return self.lagging(hlc); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn a_cut_waits_for_every_registered_scheduler() { + let cuts = CalvinCuts::default(); + cuts.register(1); + cuts.register(2); + cuts.note_passed(1, 50); + let soon = tokio::time::Instant::now() + std::time::Duration::from_millis(20); + assert_eq!(cuts.await_passed(50, soon).await, vec![2]); + + cuts.note_passed(2, 60); + let later = tokio::time::Instant::now() + std::time::Duration::from_secs(1); + assert!(cuts.await_passed(50, later).await.is_empty()); + assert_eq!(cuts.lagging(55), vec![1]); + } +} diff --git a/nodedb/src/control/state/calvin_local.rs b/nodedb/src/control/state/calvin_local.rs index 2e442b91e..b8efb5081 100644 --- a/nodedb/src/control/state/calvin_local.rs +++ b/nodedb/src/control/state/calvin_local.rs @@ -5,13 +5,15 @@ use std::collections::{BTreeMap, HashMap}; use std::sync::atomic::{AtomicU32, AtomicU64}; -use std::sync::{Arc, Mutex}; +use std::sync::{Arc, Mutex, OnceLock}; +use crate::control::cluster::calvin::scheduler::SequencerProposer; use crate::control::cluster::calvin::scheduler::lock::HotKeyTable; use crate::control::cluster::calvin::scheduler::lock_manager::{LockManager, TxnId}; use super::calvin_apply::CalvinApplyResult; use super::calvin_counters::CalvinCounters; +use super::calvin_cuts::CalvinCuts; /// Per-vShard promotion senders, keyed by vShard id. pub type PromotionSenders = BTreeMap>>; @@ -67,6 +69,11 @@ pub struct CalvinLocalState { /// with [`TxnId::AUTOCOMMIT_EPOCH`] to mint holder identities that never /// collide with a real Calvin `(epoch, position)` schedule position. pub autocommit_lock_seq: AtomicU32, + /// The backup cut markers each local scheduler passed. + pub cuts: CalvinCuts, + /// Hands this node's sequencer entries to the sequencer Raft group. Set + /// once the schedulers start; unset on a node that runs none. + pub sequencer_proposer: OnceLock>, } impl CalvinLocalState { @@ -85,6 +92,8 @@ impl CalvinLocalState { hot_key_table: Arc::new(Mutex::new(HotKeyTable::new())), promotion_senders: Arc::new(Mutex::new(BTreeMap::new())), autocommit_lock_seq: AtomicU32::new(0), + cuts: CalvinCuts::default(), + sequencer_proposer: OnceLock::new(), } } } diff --git a/nodedb/src/control/state/fields.rs b/nodedb/src/control/state/fields.rs index db4535c6b..0fc0267dd 100644 --- a/nodedb/src/control/state/fields.rs +++ b/nodedb/src/control/state/fields.rs @@ -345,7 +345,10 @@ pub struct SharedState { /// Hybrid Logical Clock for metadata descriptor `modification_hlc` stamps. pub hlc_clock: Arc, /// Per-tenant monotonic HLC high-water used by RESTORE for write-order safety. - pub tenant_write_hlc: Arc>>, + pub tenant_write_hlc: Arc>, + /// Durable per-group tenant write marks, the replicated form of the + /// high-water above that RESTORE's staleness guard reads cluster-wide. + pub tenant_marks: super::tenant_marks::TenantMarks, /// Serializes descriptor plan admission with local drain-start installation. /// /// Hold this std mutex while checking drain state and changing lease diff --git a/nodedb/src/control/state/init.rs b/nodedb/src/control/state/init.rs index 5c39a5dae..a58779f45 100644 --- a/nodedb/src/control/state/init.rs +++ b/nodedb/src/control/state/init.rs @@ -328,6 +328,7 @@ impl SharedState { quarantine_storage: Arc::new(object_store::memory::InMemory::new()), hlc_clock: Arc::new(nodedb_types::HlcClock::new()), tenant_write_hlc: Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())), + tenant_marks: super::tenant_marks::TenantMarks::load(test_credentials.catalog())?, lease_admission_gate: Mutex::new(()), lease_grant_gate: Arc::new(Mutex::new(())), lease_drain: Arc::new(crate::control::lease::DescriptorDrainTracker::new()), diff --git a/nodedb/src/control/state/init_prod/open.rs b/nodedb/src/control/state/init_prod/open.rs index e01ad5f7d..76d2e94af 100644 --- a/nodedb/src/control/state/init_prod/open.rs +++ b/nodedb/src/control/state/init_prod/open.rs @@ -322,6 +322,9 @@ impl SharedState { quarantine_storage: Arc::new(object_store::memory::InMemory::new()), hlc_clock: Arc::new(nodedb_types::HlcClock::new()), tenant_write_hlc: Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())), + tenant_marks: crate::control::state::tenant_marks::TenantMarks::load( + credentials.catalog(), + )?, lease_admission_gate: Mutex::new(()), lease_grant_gate: Arc::new(Mutex::new(())), lease_drain: Arc::new(crate::control::lease::DescriptorDrainTracker::new()), diff --git a/nodedb/src/control/state/local_write_stamps.rs b/nodedb/src/control/state/local_write_stamps.rs new file mode 100644 index 000000000..10aa9827b --- /dev/null +++ b/nodedb/src/control/state/local_write_stamps.rs @@ -0,0 +1,134 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Commit stamps of the user writes a server with no Raft groups is minting. +//! +//! Such a write takes its commit stamp before it mints its WAL record, so its +//! mark can be durable first. Between the stamp and the mint, a backup cut +//! would not see the write: the cut waits only for records already minted. +//! So each stamp stays registered here until its record is minted, and the +//! cut waits for every stamp at or below its watermark to go. +//! +//! A stamp is taken under the registry lock, and the cut reads the registry +//! under the same lock after it moves the clock past its watermark. So either +//! the cut sees the stamp, or the stamp reads above the watermark. + +use std::collections::BTreeMap; +use std::sync::Mutex; + +#[derive(Debug, Default)] +struct Registry { + next_id: u64, + /// Open stamp id → its commit HLC wall time, in nanoseconds. + open: BTreeMap, +} + +/// The open commit stamps of local writes. +#[derive(Debug, Default)] +pub struct LocalWriteStamps { + registry: Mutex, + /// Woken each time a stamp closes. + closed: tokio::sync::Notify, +} + +/// One open stamp. Dropping it states the write minted its record, or will +/// mint none. +#[derive(Debug)] +pub struct LocalWriteStamp<'a> { + stamps: &'a LocalWriteStamps, + id: u64, + hlc: u64, +} + +impl LocalWriteStamp<'_> { + /// The write's commit HLC wall time, in nanoseconds. + pub fn hlc(&self) -> u64 { + self.hlc + } +} + +impl Drop for LocalWriteStamp<'_> { + fn drop(&mut self) { + self.stamps + .registry + .lock() + .unwrap_or_else(|p| p.into_inner()) + .open + .remove(&self.id); + self.stamps.closed.notify_waiters(); + } +} + +impl LocalWriteStamps { + /// Open a stamp at the clock's next instant. + pub fn stamp(&self, clock: &nodedb_types::HlcClock) -> LocalWriteStamp<'_> { + let mut registry = self.registry.lock().unwrap_or_else(|p| p.into_inner()); + let hlc = clock.now().wall_ns; + let id = registry.next_id; + registry.next_id += 1; + registry.open.insert(id, hlc); + LocalWriteStamp { + stamps: self, + id, + hlc, + } + } + + fn open_through(&self, watermark: u64) -> bool { + self.registry + .lock() + .unwrap_or_else(|p| p.into_inner()) + .open + .values() + .any(|hlc| *hlc <= watermark) + } + + /// Wait until no stamp at or below `watermark` is open. `false` when + /// `deadline` passes first. + pub async fn await_minted_through( + &self, + watermark: u64, + deadline: tokio::time::Instant, + ) -> bool { + loop { + // Created before the check: `notify_waiters` wakes it from here on. + let closed = self.closed.notified(); + if !self.open_through(watermark) { + return true; + } + if tokio::time::timeout_at(deadline, closed).await.is_err() { + return !self.open_through(watermark); + } + } + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + + #[tokio::test] + async fn the_wait_ends_once_every_stamp_below_the_watermark_closes() { + let clock = nodedb_types::HlcClock::new(); + let stamps = LocalWriteStamps::default(); + let early = stamps.stamp(&clock); + let watermark = early.hlc(); + let late = stamps.stamp(&clock); + assert!(late.hlc() > watermark); + + let short = tokio::time::Instant::now() + Duration::from_millis(20); + assert!( + !stamps.await_minted_through(watermark, short).await, + "an open stamp at the watermark holds the wait" + ); + + drop(early); + let long = tokio::time::Instant::now() + Duration::from_secs(5); + assert!( + stamps.await_minted_through(watermark, long).await, + "a stamp above the watermark does not hold the wait" + ); + drop(late); + } +} diff --git a/nodedb/src/control/state/methods.rs b/nodedb/src/control/state/methods.rs index 3deed2c60..5908fce36 100644 --- a/nodedb/src/control/state/methods.rs +++ b/nodedb/src/control/state/methods.rs @@ -240,43 +240,6 @@ impl SharedState { } } - /// Advance the per-tenant observed write-HLC high-water to the current - /// HLC wall time. Idempotent and monotonic: no-op if a larger value is - /// already recorded. Callers MUST invoke this only after a successful - /// dispatch; "success" is defined as `Response.status == Status::Ok` - /// (and, for `Result` callers, `Result::Ok` as well). A - /// poisoned lock is silently ignored — the high-water is best-effort - /// and the RESTORE staleness gate treats missing entries as zero. - pub fn advance_tenant_write_hlc(&self, tenant_id: u64) { - let wall = self.hlc_clock.now().wall_ns; - // Recover a poisoned lock rather than skipping the advance. The map is - // a plain `HashMap` that a panic elsewhere cannot corrupt, and dropping - // the write would leave the restore staleness gate reading a stale - // high-water mark. - let mut map = self - .tenant_write_hlc - .lock() - .unwrap_or_else(|p| p.into_inner()); - let entry = map.entry(tenant_id).or_insert(0); - if wall > *entry { - *entry = wall; - } - } - - /// Last observed write HLC for `tenant_id`, or `0` when none is recorded. - /// - /// Recovers a poisoned lock. Reporting `0` because the mutex is poisoned - /// would silently disable the restore staleness gate, which compares an - /// envelope's watermark against this value. - pub fn tenant_write_hlc(&self, tenant_id: u64) -> u64 { - self.tenant_write_hlc - .lock() - .unwrap_or_else(|p| p.into_inner()) - .get(&tenant_id) - .copied() - .unwrap_or(0) - } - /// Shared HTTP client reused by every outbound emitter. Cloning the /// Arc is cheap — the client itself owns a connection pool, DNS /// resolver, and TLS session cache that every caller benefits from. diff --git a/nodedb/src/control/state/mod.rs b/nodedb/src/control/state/mod.rs index d8b465379..199bdfe94 100644 --- a/nodedb/src/control/state/mod.rs +++ b/nodedb/src/control/state/mod.rs @@ -3,15 +3,19 @@ mod buses_init; mod calvin_apply; mod calvin_counters; +mod calvin_cuts; mod calvin_local; mod fields; mod init; mod init_prod; mod init_variants; +pub mod local_write_stamps; mod methods; mod methods_audit; mod methods_lease; +pub mod tenant_marks; mod tenant_request; +mod tenant_write; pub mod audit_dml_cache; pub mod collection_to_database; @@ -19,7 +23,9 @@ pub mod idle_timeout_cache; pub use self::calvin_apply::CalvinApplyResult; pub use self::calvin_counters::CalvinCounters; +pub use self::calvin_cuts::CalvinCuts; pub use self::calvin_local::CalvinLocalState; pub use self::fields::SharedState; pub use self::init_prod::DataPlaneHandles; pub use self::tenant_request::TenantRequestGuard; +pub use self::tenant_write::{TenantWriteMark, TenantWriteMarks, TenantWriteOrigin}; diff --git a/nodedb/src/control/state/tenant_marks.rs b/nodedb/src/control/state/tenant_marks.rs new file mode 100644 index 000000000..16e8e221b --- /dev/null +++ b/nodedb/src/control/state/tenant_marks.rs @@ -0,0 +1,373 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-group tenant write marks: for each data group this node replicates, +//! the newest commit HLC of each tenant's writes the group applied. +//! +//! The marks are replicated state: every replica of a group applies the same +//! entries with the same commit stamps, so every replica derives the same +//! marks. They are durable: the apply loop persists a group's marks before +//! the applied floor that covers them, and Raft delivers every entry above +//! the floor again after a restart. RESTORE's staleness guard reads them for +//! every group that holds the tenant's data, so its answer does not depend +//! on which node's memory saw the write. +//! +//! A server with no Raft groups records its writes under +//! [`LOCAL_MARK_GROUP`]. Each such write's mark is durable before its WAL +//! record is minted (see [`TenantMarks::stamp_local_write`]). + +use std::collections::{BTreeSet, HashMap}; +use std::sync::Mutex; + +use crate::control::security::catalog::SystemCatalog; +use crate::control::security::catalog::tenant_group_marks::StoredGroupMark; + +use super::local_write_stamps::{LocalWriteStamp, LocalWriteStamps}; + +/// The pseudo-group a server with no Raft groups records its writes under. +/// No Raft group carries this id. +pub const LOCAL_MARK_GROUP: u64 = u64::MAX; + +/// The apply path that recorded a mark. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MarkSite { + /// A committed data-group entry. + ReplicatedApply, + /// A committed Calvin transaction's install. + CalvinFlush, + /// A write on a server with no Raft groups. + LocalWrite, +} + +impl MarkSite { + /// The name a refused restore reports. + pub fn as_str(self) -> &'static str { + match self { + Self::ReplicatedApply => "replicated apply", + Self::CalvinFlush => "calvin flush", + Self::LocalWrite => "local write", + } + } + + /// The code the persisted and wire forms carry. + pub fn code(self) -> u8 { + match self { + Self::ReplicatedApply => 0, + Self::CalvinFlush => 1, + Self::LocalWrite => 2, + } + } + + /// The site a persisted or wire code names. + pub fn from_code(code: u8) -> Self { + match code { + 1 => Self::CalvinFlush, + 2 => Self::LocalWrite, + _ => Self::ReplicatedApply, + } + } +} + +/// One group's newest write of one tenant. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GroupMark { + /// HLC wall time, in nanoseconds, of the write's commit. + pub hlc: u64, + pub site: MarkSite, + /// The collection the write named, when it named one. + pub collection: Option, +} + +#[derive(Debug, Default)] +struct MarkState { + marks: HashMap<(u64, u64), GroupMark>, + /// Marks raised since the last persist. + dirty: BTreeSet<(u64, u64)>, + /// The highest commit HLC of each mark the catalog holds. + persisted: HashMap<(u64, u64), u64>, +} + +/// This node's per-group tenant write marks. +#[derive(Debug, Default)] +pub struct TenantMarks { + state: Mutex, + /// Held across each persist, so a caller that finds its mark persisted + /// knows the catalog commit that wrote it finished. + persist_lock: Mutex<()>, + /// The local writes minting their records now. + local_stamps: LocalWriteStamps, +} + +impl TenantMarks { + /// The marks persisted in `catalog`. + pub fn load(catalog: &SystemCatalog) -> crate::Result { + let marks = Self::default(); + { + let mut state = marks.state.lock().unwrap_or_else(|p| p.into_inner()); + for stored in catalog.load_tenant_group_marks()? { + state + .persisted + .insert((stored.group_id, stored.tenant_id), stored.hlc); + state.marks.insert( + (stored.group_id, stored.tenant_id), + GroupMark { + hlc: stored.hlc, + site: MarkSite::from_code(stored.site), + collection: (!stored.collection.is_empty()).then_some(stored.collection), + }, + ); + } + } + Ok(marks) + } + + /// Raise `tenant_id`'s mark in `group_id` to `hlc`. A mark at or above it + /// stays. + pub fn raise( + &self, + group_id: u64, + tenant_id: u64, + hlc: u64, + site: MarkSite, + collection: Option<&str>, + ) { + if hlc == 0 { + return; + } + let mut state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + let key = (group_id, tenant_id); + if state.marks.get(&key).is_some_and(|mark| mark.hlc >= hlc) { + return; + } + state.marks.insert( + key, + GroupMark { + hlc, + site, + collection: collection.map(str::to_owned), + }, + ); + state.dirty.insert(key); + } + + /// Persist every mark raised since the last persist. On an error the + /// marks stay pending, so the next persist writes them. + pub fn persist(&self, catalog: &SystemCatalog) -> crate::Result<()> { + let _persisting = self.persist_lock.lock().unwrap_or_else(|p| p.into_inner()); + self.persist_pending(catalog) + } + + /// Persist until the catalog holds `tenant_id`'s mark in `group_id` at or + /// above `hlc`. A persist that already wrote it, or one running now, covers + /// it: concurrent callers share one catalog commit. + pub fn persist_through( + &self, + catalog: &SystemCatalog, + group_id: u64, + tenant_id: u64, + hlc: u64, + ) -> crate::Result<()> { + let _persisting = self.persist_lock.lock().unwrap_or_else(|p| p.into_inner()); + let covered = self + .state + .lock() + .unwrap_or_else(|p| p.into_inner()) + .persisted + .get(&(group_id, tenant_id)) + .is_some_and(|persisted| *persisted >= hlc); + if covered { + return Ok(()); + } + self.persist_pending(catalog) + } + + /// Stamp a user write on a server with no Raft groups, and make its mark + /// durable before the write mints its WAL record. + /// + /// The stamp is the write's commit HLC. It stays open until the returned + /// guard drops, which the caller does once the record is minted. A backup + /// cut waits for every open stamp at or below its watermark (see + /// [`Self::await_local_stamps_minted`]). So a write stamped below the + /// watermark is in the backup, and one stamped above it has a mark above + /// it. A crash after the mint leaves the mark in the catalog, whether or + /// not the record reached disk. + pub fn stamp_local_write( + &self, + clock: &nodedb_types::HlcClock, + catalog: &SystemCatalog, + tenant_id: u64, + collection: Option<&str>, + ) -> crate::Result> { + let stamp = self.local_stamps.stamp(clock); + self.record_local_write(catalog, tenant_id, stamp.hlc(), collection)?; + Ok(stamp) + } + + /// Raise `tenant_id`'s mark under [`LOCAL_MARK_GROUP`] to `hlc` and + /// persist it. A mark the catalog already holds at or above `hlc` costs + /// no catalog commit. + pub fn record_local_write( + &self, + catalog: &SystemCatalog, + tenant_id: u64, + hlc: u64, + collection: Option<&str>, + ) -> crate::Result<()> { + self.raise( + LOCAL_MARK_GROUP, + tenant_id, + hlc, + MarkSite::LocalWrite, + collection, + ); + self.persist_through(catalog, LOCAL_MARK_GROUP, tenant_id, hlc) + } + + /// Wait until every local write stamped at or below `watermark` minted + /// its record. `false` when `deadline` passes first. + pub async fn await_local_stamps_minted( + &self, + watermark: u64, + deadline: tokio::time::Instant, + ) -> bool { + self.local_stamps + .await_minted_through(watermark, deadline) + .await + } + + /// Write every dirty mark. The caller holds `persist_lock`. + fn persist_pending(&self, catalog: &SystemCatalog) -> crate::Result<()> { + let pending: Vec = { + let state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + state + .dirty + .iter() + .filter_map(|key| { + state.marks.get(key).map(|mark| StoredGroupMark { + group_id: key.0, + tenant_id: key.1, + hlc: mark.hlc, + site: mark.site.code(), + collection: mark.collection.clone().unwrap_or_default(), + }) + }) + .collect() + }; + if pending.is_empty() { + return Ok(()); + } + catalog.raise_tenant_group_marks(&pending)?; + let mut state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + for mark in &pending { + let key = (mark.group_id, mark.tenant_id); + // A raise after the snapshot above stays pending. + if state.marks.get(&key).is_some_and(|m| m.hlc == mark.hlc) { + state.dirty.remove(&key); + } + let persisted = state.persisted.entry(key).or_insert(0); + *persisted = (*persisted).max(mark.hlc); + } + Ok(()) + } + + /// Every tenant's mark in `group_id`, in the form a group snapshot carries: + /// `(tenant_id, commit_hlc, site_code, collection)`. + pub fn group_entries(&self, group_id: u64) -> Vec<(u64, u64, u8, String)> { + let state = self.state.lock().unwrap_or_else(|p| p.into_inner()); + let mut entries: Vec<(u64, u64, u8, String)> = state + .marks + .iter() + .filter(|((group, _), _)| *group == group_id) + .map(|((_, tenant_id), mark)| { + ( + *tenant_id, + mark.hlc, + mark.site.code(), + mark.collection.clone().unwrap_or_default(), + ) + }) + .collect(); + entries.sort_unstable(); + entries + } + + /// Raise `group_id`'s marks from the entries of a group snapshot. + pub fn raise_group_entries(&self, group_id: u64, entries: &[(u64, u64, u8, String)]) { + for (tenant_id, hlc, site, collection) in entries { + self.raise( + group_id, + *tenant_id, + *hlc, + MarkSite::from_code(*site), + (!collection.is_empty()).then_some(collection.as_str()), + ); + } + } + + /// `tenant_id`'s mark in `group_id`, if the group applied any write of it. + pub fn get(&self, group_id: u64, tenant_id: u64) -> Option { + self.state + .lock() + .unwrap_or_else(|p| p.into_inner()) + .marks + .get(&(group_id, tenant_id)) + .cloned() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_mark_only_rises() { + let marks = TenantMarks::default(); + marks.raise(1, 7, 100, MarkSite::ReplicatedApply, Some("docs")); + marks.raise(1, 7, 50, MarkSite::CalvinFlush, None); + let mark = marks.get(1, 7).expect("mark"); + assert_eq!(mark.hlc, 100); + assert_eq!(mark.collection.as_deref(), Some("docs")); + assert!(marks.get(2, 7).is_none(), "a mark binds only its group"); + } + + #[test] + fn a_local_write_mark_is_durable_once_stamped() { + let dir = tempfile::tempdir().expect("tempdir"); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).expect("catalog"); + let clock = nodedb_types::HlcClock::new(); + let marks = TenantMarks::default(); + let hlc = { + let stamp = marks + .stamp_local_write(&clock, &catalog, 4, Some("orders")) + .expect("stamp"); + stamp.hlc() + }; + let reloaded = TenantMarks::load(&catalog).expect("reload"); + assert_eq!( + reloaded.get(LOCAL_MARK_GROUP, 4), + Some(GroupMark { + hlc, + site: MarkSite::LocalWrite, + collection: Some("orders".to_owned()), + }) + ); + } + + #[test] + fn persisted_marks_survive_a_reload() { + let dir = tempfile::tempdir().expect("tempdir"); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).expect("catalog"); + let marks = TenantMarks::default(); + marks.raise(3, 1, 900, MarkSite::CalvinFlush, Some("orders")); + marks.persist(&catalog).expect("persist"); + + let reloaded = TenantMarks::load(&catalog).expect("reload"); + assert_eq!( + reloaded.get(3, 1), + Some(GroupMark { + hlc: 900, + site: MarkSite::CalvinFlush, + collection: Some("orders".to_owned()), + }) + ); + } +} diff --git a/nodedb/src/control/state/tenant_write.rs b/nodedb/src/control/state/tenant_write.rs new file mode 100644 index 000000000..53099181f --- /dev/null +++ b/nodedb/src/control/state/tenant_write.rs @@ -0,0 +1,104 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The per-tenant observed write-HLC high-water that RESTORE's staleness gate +//! compares a backup envelope's watermark against. +//! +//! Each mark keeps the write that last raised it: the dispatch site and the +//! collection the plan wrote. A refused restore names that write, so an +//! operator sees which newer data the restore would overwrite. + +use std::collections::HashMap; + +use super::SharedState; + +/// The write that last raised a tenant's high-water. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct TenantWriteOrigin { + /// The dispatch site that recorded the write. + pub site: &'static str, + /// The collection the write's plan named, when it named one. + pub collection: Option, +} + +/// One tenant's observed write high-water. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct TenantWriteMark { + /// HLC wall time, in nanoseconds, of the newest recorded write. + pub hlc: u64, + /// The write that recorded `hlc`. + pub origin: TenantWriteOrigin, +} + +/// Per-tenant marks, keyed by tenant id. +pub type TenantWriteMarks = HashMap; + +impl SharedState { + /// Raise `tenant_id`'s observed write high-water to `commit_hlc`, the HLC + /// wall time in nanoseconds at which a user data write committed, and + /// record `site` and `collection` as its origin. Monotonic: a mark already + /// at or above `commit_hlc` stays. + /// + /// Callers record a write before its ack returns, with the instant the + /// write committed. A backup taken after the ack then always carries a + /// watermark at or above the mark, and a write committed after the backup + /// always carries a newer one. + /// + /// A mark carried from another node folds into this node's clock, so a + /// backup taken here after the write never stamps an older watermark. + pub fn advance_tenant_write_hlc( + &self, + tenant_id: u64, + commit_hlc: u64, + site: &'static str, + collection: Option<&str>, + ) { + if let Err(skew) = self + .hlc_clock + .update_checked(nodedb_types::Hlc::new(commit_hlc, 0)) + { + tracing::warn!( + tenant_id, + commit_hlc, + site, + skew_ns = skew.skew_ns, + "a write's commit HLC runs ahead of this node's clock past the skew bound; \ + RESTORE refuses envelopes older than it until the clock catches up" + ); + } + // Recover a poisoned lock rather than skipping the advance. The map + // is a plain `HashMap` that a panic elsewhere cannot corrupt, and a + // dropped write would leave the staleness gate reading a stale mark. + let mut marks = self + .tenant_write_hlc + .lock() + .unwrap_or_else(|p| p.into_inner()); + if marks + .get(&tenant_id) + .is_some_and(|mark| mark.hlc >= commit_hlc) + { + return; + } + marks.insert( + tenant_id, + TenantWriteMark { + hlc: commit_hlc, + origin: TenantWriteOrigin { + site, + collection: collection.map(str::to_owned), + }, + }, + ); + } + + /// The observed write mark of `tenant_id`, when one is recorded. + /// + /// Recovers a poisoned lock. Reporting no mark because the mutex is + /// poisoned would silently disable the restore staleness gate. + pub fn tenant_write_mark(&self, tenant_id: u64) -> Option { + self.tenant_write_hlc + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&tenant_id) + .cloned() + } +} diff --git a/nodedb/src/control/vshard_admission.rs b/nodedb/src/control/vshard_admission.rs index 9aa1732e3..1001107f0 100644 --- a/nodedb/src/control/vshard_admission.rs +++ b/nodedb/src/control/vshard_admission.rs @@ -110,7 +110,7 @@ fn wrap_async_raft_proposer( sequencer: Arc, raw: Arc, ) -> Arc { - Arc::new(move |vshard_id, idempotency_key, data| { + Arc::new(move |vshard_id, idempotency_key, data, deadline| { let sequencer = Arc::clone(&sequencer); let raw = Arc::clone(&raw); Box::pin(async move { @@ -122,7 +122,7 @@ fn wrap_async_raft_proposer( let vshard_id = VShardId::new(vshard_id); sequencer .run(vshard_id, move || async move { - raw(vshard_id.as_u32(), idempotency_key, data).await + raw(vshard_id.as_u32(), idempotency_key, data, deadline).await }) .await }) @@ -139,6 +139,10 @@ mod tests { use super::*; use crate::types::Lsn; + fn test_deadline() -> tokio::time::Instant { + tokio::time::Instant::now() + std::time::Duration::from_secs(30) + } + fn shard(id: u32) -> VShardId { VShardId::new(id) } @@ -353,7 +357,7 @@ mod tests { let maximum = Arc::clone(&maximum); let release = Arc::clone(&release); let entered = Arc::clone(&entered); - Arc::new(move |_vshard, key, data| { + Arc::new(move |_vshard, key, data, _deadline| { let active = Arc::clone(&active); let maximum = Arc::clone(&maximum); let release = Arc::clone(&release); @@ -371,12 +375,12 @@ mod tests { let wrapped = wrap_async_raft_proposer(Arc::clone(&sequencer), raw); let first = { let wrapped = Arc::clone(&wrapped); - tokio::spawn(async move { wrapped(5, 11, vec![1]).await }) + tokio::spawn(async move { wrapped(5, 11, vec![1], test_deadline()).await }) }; entered.notified().await; let second = { let wrapped = Arc::clone(&wrapped); - tokio::spawn(async move { wrapped(5, 12, vec![2]).await }) + tokio::spawn(async move { wrapped(5, 12, vec![2], test_deadline()).await }) }; while sequencer.slots[5].capacity.available_permits() != 0 { tokio::task::yield_now().await; diff --git a/nodedb/src/control/wal_replication/decode/entry.rs b/nodedb/src/control/wal_replication/decode/entry.rs index 460b52dd9..432751271 100644 --- a/nodedb/src/control/wal_replication/decode/entry.rs +++ b/nodedb/src/control/wal_replication/decode/entry.rs @@ -176,6 +176,9 @@ fn to_physical_plan( ReplicatedWrite::ArraySchema { .. } => Err(crate::Error::Internal { detail: "ArraySchema reached to_physical_plan (should have been intercepted)".into(), }), + ReplicatedWrite::CutBarrier { .. } => Err(crate::Error::Internal { + detail: "CutBarrier reached to_physical_plan (should have been intercepted)".into(), + }), ReplicatedWrite::CalvinReadResult { .. } => Err(crate::Error::Internal { detail: "CalvinReadResult reached to_physical_plan (should have been intercepted)" .into(), diff --git a/nodedb/src/control/wal_replication/decode/transaction_redo.rs b/nodedb/src/control/wal_replication/decode/transaction_redo.rs index 558009313..4237165bc 100644 --- a/nodedb/src/control/wal_replication/decode/transaction_redo.rs +++ b/nodedb/src/control/wal_replication/decode/transaction_redo.rs @@ -21,6 +21,7 @@ pub fn transaction_redo_payload(write: &ReplicatedWrite) -> crate::Result crate::Result TransactionRedoPayload { TransactionRedoPayload { @@ -82,6 +84,7 @@ mod tests { surrogate: Surrogate::new(3), }], event_source: EventSource::Trigger, + origin: RedoOrigin::Restore, } } @@ -107,6 +110,7 @@ mod tests { assert_eq!(decoded.sum_targets, original.sum_targets); assert_eq!(decoded.identities, original.identities); assert_eq!(decoded.event_source, EventSource::Trigger); + assert_eq!(decoded.origin, RedoOrigin::Restore); } #[test] @@ -114,13 +118,14 @@ mod tests { let original = payload(); let plan = original.apply_plan().expect("plan builds"); let nodedb_physical::physical_plan::PhysicalPlan::Meta( - nodedb_physical::physical_plan::MetaOp::ApplyTransactionRedo { redo, .. }, + nodedb_physical::physical_plan::MetaOp::ApplyTransactionRedo { redo, origin, .. }, ) = plan else { panic!("apply plan must be ApplyTransactionRedo"); }; let redo = RedoRecord::from_bytes(&redo).expect("redo decodes"); assert_eq!(redo, original.redo); + assert_eq!(origin, RedoOrigin::Restore); } #[test] diff --git a/nodedb/src/control/wal_replication/encode/transaction_redo.rs b/nodedb/src/control/wal_replication/encode/transaction_redo.rs index bc19a4fa8..1ff3c38f8 100644 --- a/nodedb/src/control/wal_replication/encode/transaction_redo.rs +++ b/nodedb/src/control/wal_replication/encode/transaction_redo.rs @@ -37,6 +37,7 @@ pub fn transaction_redo_entry( }) .collect(), event_source: ReplicatedEventSource::from(payload.event_source), + origin: payload.origin, }, ) } diff --git a/nodedb/src/control/wal_replication/legacy_entry.rs b/nodedb/src/control/wal_replication/legacy_entry.rs index e66916a66..26ad59891 100644 --- a/nodedb/src/control/wal_replication/legacy_entry.rs +++ b/nodedb/src/control/wal_replication/legacy_entry.rs @@ -37,6 +37,8 @@ impl LegacyReplicatedEntry { vshard_id: self.vshard_id, idempotency_key: self.idempotency_key, write: self.write, + write_hlc: 0, + metadata_floor: 0, } } } diff --git a/nodedb/src/control/wal_replication/propose.rs b/nodedb/src/control/wal_replication/propose.rs index 289414b10..ba456bab4 100644 --- a/nodedb/src/control/wal_replication/propose.rs +++ b/nodedb/src/control/wal_replication/propose.rs @@ -12,8 +12,12 @@ use std::sync::Arc; use super::types::{AsyncRaftProposer, ReplicatedEntry}; use crate::control::state::SharedState; -/// Backoff schedule for `RetryableLeaderChange` re-proposals (5 attempts). -const BACKOFF_MS: [u64; 5] = [10, 25, 50, 100, 200]; +/// First backoff before a re-proposal. Each retry doubles it up to +/// [`MAX_BACKOFF`]. +const FIRST_BACKOFF: std::time::Duration = std::time::Duration::from_millis(10); + +/// Longest wait between two re-proposals. +const MAX_BACKOFF: std::time::Duration = std::time::Duration::from_millis(200); /// Propose `entry` via `proposer` and return the Data Plane apply payload bytes /// together with the write's per-collection version (as an @@ -23,49 +27,51 @@ const BACKOFF_MS: [u64; 5] = [10, 25, 50, 100, 200]; /// collection. See [`AsyncRaftProposer`] for why this is a WAL LSN and never the /// Raft log index. /// -/// Retries transparently on [`crate::Error::RetryableLeaderChange`]: the -/// previous leader's entry was overwritten by a new leader's election no-op, so -/// the same write payload is re-proposed against the new leader. The encoded -/// `ReplicatedEntry` carries enough identity (collection, PK, surrogate) to be -/// replayable. Only propose-layer machinery failures map to +/// Re-proposes the same payload until the statement deadline while the group +/// has no leader to take it: +/// - [`crate::Error::RetryableLeaderChange`]: a new leader's election no-op +/// overwrote the previous leader's entry; +/// - [`crate::Error::NoLeader`]: the group is electing, or a leadership +/// transfer is in flight. +/// +/// The encoded `ReplicatedEntry` carries enough identity (collection, PK, +/// surrogate) to be replayable, and its idempotency key makes a copy that +/// committed twice apply once. Only propose-layer machinery failures map to /// [`crate::Error::Dispatch`]; a classified apply verdict passes through. pub(crate) async fn propose_replicated_entry( state: &SharedState, proposer: &Arc, - entry: ReplicatedEntry, + mut entry: ReplicatedEntry, ) -> crate::Result<(Vec, crate::types::Lsn)> { + // The write's commit instant. Stamped once, before the first propose, so + // every re-proposal and every replica's apply carries the same value. + entry.write_hlc = state.hlc_clock.now().wall_ns; + // The catalog this write was planned against: every replica applies it + // only once its own metadata apply reached this index. + entry.metadata_floor = state + .applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID) + .current(); let idempotency_key = entry.idempotency_key; let data = entry.to_bytes(); let vshard_id = entry.vshard_id; - let mut payload = None; - let mut last_err: Option = None; - for (attempt, backoff_ms) in BACKOFF_MS.iter().enumerate() { - match proposer(vshard_id, idempotency_key, data.clone()).await { - Ok(p) => { - payload = Some(p); - break; - } - Err(crate::Error::RetryableLeaderChange { - group_id, - log_index, - }) => { + // The statement deadline. Every attempt, and each attempt's wait for the + // local apply, ends at this one instant. + let deadline = tokio::time::Instant::now() + + std::time::Duration::from_secs(state.tuning.network.default_deadline_secs); + let mut backoff = FIRST_BACKOFF; + let mut attempt: u32 = 0; + let payload = loop { + attempt += 1; + let error = match proposer(vshard_id, idempotency_key, data.clone(), deadline).await { + Ok(p) => break Ok(p), + Err(error @ crate::Error::RetryableLeaderChange { .. }) => { state .raft_propose_leader_change_retries .fetch_add(1, std::sync::atomic::Ordering::Relaxed); - tracing::warn!( - attempt, - group_id, - log_index, - "raft entry overwritten by leader change — re-proposing" - ); - last_err = Some(crate::Error::RetryableLeaderChange { - group_id, - log_index, - }); - tokio::time::sleep(std::time::Duration::from_millis(*backoff_ms)).await; - continue; + error } + Err(error @ crate::Error::NoLeader { .. }) => error, // Only a machinery failure is re-wrapped. A state-machine verdict // (constraint, authz, conflict) carries the client's SQLSTATE. Err(other) if crate::error_classify::is_unclassified_failure(&other) => { @@ -74,11 +80,21 @@ pub(crate) async fn propose_replicated_entry( }); } Err(other) => return Err(other), + }; + if tokio::time::Instant::now() + backoff >= deadline { + break Err(error); } - } - payload.ok_or_else(|| { - last_err.unwrap_or_else(|| crate::Error::Dispatch { - detail: "raft propose retries exhausted".into(), - }) - }) + tracing::warn!( + attempt, + vshard_id, + error = %error, + "raft proposal found no leader to take it; re-proposing" + ); + tokio::time::sleep(backoff).await; + backoff = (backoff * 2).min(MAX_BACKOFF); + }; + // The waiter resolved only once this node's own apply ran the entry, and + // that apply recorded the entry's commit stamp on the tenant's observed + // write high-water before resolving it. + payload } diff --git a/nodedb/src/control/wal_replication/transaction_redo/apply.rs b/nodedb/src/control/wal_replication/transaction_redo/apply.rs index 17103a71c..60cb1a878 100644 --- a/nodedb/src/control/wal_replication/transaction_redo/apply.rs +++ b/nodedb/src/control/wal_replication/transaction_redo/apply.rs @@ -9,7 +9,8 @@ //! it to the owning core, and the fsync completes before the result returns. use crate::control::server::dispatch_utils::{ - ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, + ChangeFeedOwner, PendingWrite, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, + enqueue_write, }; use crate::control::state::SharedState; use crate::control::surrogate::bind_carried_identities; @@ -28,15 +29,32 @@ pub struct RedoTarget { /// Bind the redo's identities, then append and apply it on this node. /// /// `apply_key` is the idempotency key of the Raft entry the redo comes from, -/// `0` on a node with no Raft. The redo record's header carries it. The -/// outcome carries the Data Plane's response verbatim, including an error -/// status. +/// `0` on a node with no Raft. The redo record's header carries it. +/// `commit_hlc` is the entry's commit stamp, `None` on a node with no Raft, +/// where the append here is the commit. The outcome carries the Data Plane's +/// response verbatim, including an error status. pub(crate) async fn apply_transaction_redo( state: &SharedState, target: RedoTarget, payload: &TransactionRedoPayload, apply_key: u64, + commit_hlc: Option, ) -> crate::Result { + enqueue_transaction_redo(state, target, payload, apply_key, commit_hlc) + .await? + .finish(state) + .await +} + +/// Bind the redo's identities, then append it and enqueue it on its core. +/// [`PendingWrite::finish`] collects the outcome. +pub(crate) async fn enqueue_transaction_redo( + state: &SharedState, + target: RedoTarget, + payload: &TransactionRedoPayload, + apply_key: u64, + commit_hlc: Option, +) -> crate::Result { bind_carried_identities( &state.surrogate_assigner, target.database_id, @@ -44,7 +62,7 @@ pub(crate) async fn apply_transaction_redo( &payload.identities, )?; let plan = payload.apply_plan()?; - submit_write( + enqueue_write( state, SubmitWrite { tenant_id: target.tenant_id, @@ -60,6 +78,7 @@ pub(crate) async fn apply_transaction_redo( durability: WalDurability::AppendHere { now_override: None, apply_key, + commit_hlc, }, // Raft fixed the order; a node with no Raft already validated the // commit and holds no gate for it. diff --git a/nodedb/src/control/wal_replication/transaction_redo/mod.rs b/nodedb/src/control/wal_replication/transaction_redo/mod.rs index 4c217a9f2..69de196bb 100644 --- a/nodedb/src/control/wal_replication/transaction_redo/mod.rs +++ b/nodedb/src/control/wal_replication/transaction_redo/mod.rs @@ -13,5 +13,5 @@ pub mod payload; pub mod sum_targets; pub use apply::RedoTarget; -pub(crate) use apply::apply_transaction_redo; +pub(crate) use apply::{apply_transaction_redo, enqueue_transaction_redo}; pub use payload::TransactionRedoPayload; diff --git a/nodedb/src/control/wal_replication/transaction_redo/payload.rs b/nodedb/src/control/wal_replication/transaction_redo/payload.rs index 005496d18..9885faeb6 100644 --- a/nodedb/src/control/wal_replication/transaction_redo/payload.rs +++ b/nodedb/src/control/wal_replication/transaction_redo/payload.rs @@ -3,7 +3,7 @@ //! One committed transaction's redo for one vShard, as a commit hands it to //! the data-group log and as every replica applies it. -use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan, RedoSumTargets}; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan, RedoOrigin, RedoSumTargets}; use crate::control::state::SharedState; use crate::control::surrogate::{CarriedIdentity, collect_plan_identities}; @@ -28,6 +28,8 @@ pub struct TransactionRedoPayload { pub identities: Vec, /// Event source every replica stamps on the writes. pub event_source: EventSource, + /// Which commit-boundary checks every replica's apply runs. + pub origin: RedoOrigin, } impl TransactionRedoPayload { @@ -52,6 +54,7 @@ impl TransactionRedoPayload { plans, )?, event_source, + origin: RedoOrigin::Commit, }) } @@ -61,6 +64,7 @@ impl TransactionRedoPayload { redo: self.redo.to_bytes()?, collections: self.collections.clone(), sum_targets: self.sum_targets.clone(), + origin: self.origin, })) } } diff --git a/nodedb/src/control/wal_replication/types/aliases.rs b/nodedb/src/control/wal_replication/types/aliases.rs index a538be847..359667d08 100644 --- a/nodedb/src/control/wal_replication/types/aliases.rs +++ b/nodedb/src/control/wal_replication/types/aliases.rs @@ -12,7 +12,7 @@ pub type RaftProposer = /// Type alias for the asynchronous Raft propose callback with leader forwarding. /// -/// Takes `(vshard_id, idempotency_key, serialized_entry)` and returns, on +/// Takes `(vshard_id, idempotency_key, serialized_entry, deadline)` and returns, on /// success, the Data Plane apply payload bytes together with the write's /// per-collection version: the written collection's `coll_write_lsn` AFTER the /// write, as the replica that applied the entry recorded it. It is a WAL LSN, @@ -26,10 +26,16 @@ pub type RaftProposer = /// serialized `ReplicatedEntry`; the proposer registers the tracker waiter with /// this key so apply-side mismatch detection can surface `RetryableLeaderChange` /// when a new leader's entry overwrites this one. +/// +/// `deadline` is the caller's absolute statement deadline. A caller that +/// re-proposes passes the same instant to every attempt, so each attempt gets +/// only the time that remains. The proposer never computes a deadline of its +/// own. Past `deadline` it returns [`crate::Error::DeadlineExceeded`]. pub type AsyncRaftProposer = dyn Fn( u32, u64, Vec, + tokio::time::Instant, ) -> std::pin::Pin< Box< dyn std::future::Future< diff --git a/nodedb/src/control/wal_replication/types/replicated_entry.rs b/nodedb/src/control/wal_replication/types/replicated_entry.rs index 42ae3d466..b64175c23 100644 --- a/nodedb/src/control/wal_replication/types/replicated_entry.rs +++ b/nodedb/src/control/wal_replication/types/replicated_entry.rs @@ -42,6 +42,19 @@ pub struct ReplicatedEntry { /// upgrade. pub idempotency_key: u64, pub write: ReplicatedWrite, + /// HLC wall time, in nanoseconds, at which the proposer committed to the + /// write. Stamped once before the first propose, so a re-proposal keeps + /// it. Every replica records it as the write's instant on the tenant's + /// observed write high-water, however late it applies. `0` for an entry + /// proposed without one; its apply stamps the instant of its own append. + pub write_hlc: u64, + /// The metadata-group index the proposer had applied when it proposed + /// the write. Every replica applies the write only once its own metadata + /// apply reached this index, so the write never lands against an older + /// catalog than the one it was planned against: a same-name collection's + /// pending purge has reclaimed its storage, and the collection the write + /// targets is registered. `0` for an entry proposed without one. + pub metadata_floor: u64, } impl ReplicatedEntry { @@ -60,6 +73,8 @@ impl ReplicatedEntry { vshard_id, idempotency_key, write, + write_hlc: 0, + metadata_floor: 0, } } @@ -70,7 +85,7 @@ impl ReplicatedEntry { /// Deserialize from Raft log entry data bytes. /// - /// Tries the current 5-field shape first. If that fails specifically + /// Tries the current 7-field shape first. If that fails specifically /// because the encoded array is the pre-`database_id` 4-element shape /// (an entry proposed by an old leader still mid-upgrade), falls back to /// [`super::legacy_entry::LegacyReplicatedEntry`] and defaults diff --git a/nodedb/src/control/wal_replication/types/replicated_write.rs b/nodedb/src/control/wal_replication/types/replicated_write.rs index 1f5f10210..edceec980 100644 --- a/nodedb/src/control/wal_replication/types/replicated_write.rs +++ b/nodedb/src/control/wal_replication/types/replicated_write.rs @@ -1015,6 +1015,15 @@ pub enum ReplicatedWrite { /// Identities every replica binds before the apply. identities: Vec, event_source: ReplicatedEventSource, + /// Which commit-boundary checks every replica's apply runs. + origin: nodedb_physical::physical_plan::RedoOrigin, + }, + /// A backup's consistent cut through this group's log. It writes no + /// data. Every entry before it applies before the backup snapshots, and + /// every entry after it records a commit HLC above `hlc`, the backup's + /// watermark, so a restore of that backup refuses it. + CutBarrier { + hlc: u64, }, } diff --git a/nodedb/src/data/executor/core_loop/calvin_fence.rs b/nodedb/src/data/executor/core_loop/calvin_fence.rs index 0c04d2558..8ef8141d6 100644 --- a/nodedb/src/data/executor/core_loop/calvin_fence.rs +++ b/nodedb/src/data/executor/core_loop/calvin_fence.rs @@ -190,6 +190,25 @@ impl CoreLoop { ) -> Option { match self.calvin_owner_of(&task) { Some(owner) => { + // Raft fixed this write's order. Parking it holds its data + // group's applied index behind the owner's flush, so the + // capture names the write and its owner. + if matches!( + task.request.admission, + crate::bridge::envelope::Admission::Exempt( + crate::bridge::envelope::ExemptReason::AlreadyOrdered + ) + ) { + crate::diag::replicated_write_parked( + self.core_id, + task.plan() + .named_collections() + .first() + .copied() + .unwrap_or(""), + owner, + ); + } self.calvin.fence.parked.push_back(ParkedWrite { task, owner, diff --git a/nodedb/src/data/executor/dispatch/meta.rs b/nodedb/src/data/executor/dispatch/meta.rs index fc7151019..25ceeb0b1 100644 --- a/nodedb/src/data/executor/dispatch/meta.rs +++ b/nodedb/src/data/executor/dispatch/meta.rs @@ -68,7 +68,7 @@ impl CoreLoop { } } - MetaOp::CreateTenantSnapshot { tenant_id } => { + MetaOp::CreateTenantSnapshot { tenant_id, .. } => { self.execute_create_tenant_snapshot(task, *tenant_id) } @@ -307,7 +307,8 @@ impl CoreLoop { redo, collections, sum_targets, - } => self.execute_apply_transaction_redo( + origin, + } => self.install_redo( task, tid, crate::data::executor::handlers::transaction::redo_apply::CommittedRedo { @@ -315,6 +316,7 @@ impl CoreLoop { collections, sum_targets, }, + *origin, ), MetaOp::StageWrite { plan } => self.execute_stage_write(task, tid, plan), diff --git a/nodedb/src/data/executor/handlers/snapshot/capture_memory.rs b/nodedb/src/data/executor/handlers/snapshot/capture_memory.rs new file mode 100644 index 000000000..d9049bd2e --- /dev/null +++ b/nodedb/src/data/executor/handlers/snapshot/capture_memory.rs @@ -0,0 +1,177 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Tenant snapshot capture of the in-memory engines: vectors and their index +//! config, KV tables, CRDT state and constraints, and timeseries memtables. +//! +//! Every section is all or nothing. A section that fails to export fails the +//! whole snapshot: a snapshot missing it would restore without it. + +use crate::data::executor::core_loop::CoreLoop; +use crate::types::{DatabaseId, TenantDataSnapshot, TenantId}; + +/// The error for a section that failed to export. +fn capture_error(section: &str, key: &str, detail: impl std::fmt::Display) -> crate::Error { + crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("snapshot: {section} '{key}' failed to export: {detail}"), + } +} + +/// `"{db}:{tid}:{collection}"`, the key every in-memory section carries. +fn scoped_key(database_id: DatabaseId, tenant: TenantId, collection: &str) -> String { + format!( + "{}:{}:{}", + database_id.as_u64(), + tenant.as_u64(), + collection + ) +} + +impl CoreLoop { + /// Fill `snapshot`'s in-memory sections with `tenant`'s state. + pub(super) fn capture_memory_engines( + &self, + database_id: DatabaseId, + tenant: TenantId, + snapshot: &mut TenantDataSnapshot, + ) -> crate::Result<()> { + self.capture_vectors(tenant, snapshot)?; + self.capture_kv_tables(tenant, snapshot)?; + self.capture_crdt(database_id, tenant, snapshot)?; + self.capture_timeseries_memtables(tenant, snapshot) + } + + /// Raw vectors, HNSW params and index config per collection. The HNSW + /// graph is rebuilt from the raw vectors on restore. + fn capture_vectors( + &self, + tenant: TenantId, + snapshot: &mut TenantDataSnapshot, + ) -> crate::Result<()> { + for (key, collection) in &self.vector_collections { + if key.1 != tenant { + continue; + } + let key_str = scoped_key(key.0, key.1, &key.2); + let vectors = collection + .export_snapshot() + .map_err(|e| capture_error("vector collection", &key_str, e))?; + let bytes = zerompk::to_msgpack_vec(&vectors) + .map_err(|e| capture_error("vector collection", &key_str, e))?; + snapshot.vectors.push((key_str, bytes)); + } + for (key, params) in &self.vector_params { + if key.1 != tenant { + continue; + } + let key_str = scoped_key(key.0, key.1, &key.2); + let bytes = zerompk::to_msgpack_vec(params) + .map_err(|e| capture_error("vector params", &key_str, e))?; + snapshot.vector_params.push((key_str, bytes)); + } + for (key, cfg) in &self.index_configs { + if key.1 != tenant { + continue; + } + let key_str = scoped_key(key.0, key.1, &key.2); + let bytes = zerompk::to_msgpack_vec(cfg) + .map_err(|e| capture_error("vector index config", &key_str, e))?; + snapshot.index_configs.push((key_str, bytes)); + } + Ok(()) + } + + /// Every live entry of every KV table of `tenant`, keyed by the table's + /// stored collection name. + fn capture_kv_tables( + &self, + tenant: TenantId, + snapshot: &mut TenantDataSnapshot, + ) -> crate::Result<()> { + for (&hash, table) in &self.kv_engine.tables { + let Some(&tid) = self.kv_engine.hash_to_tenant.get(&hash) else { + continue; + }; + if tid != tenant.as_u64() { + continue; + } + let collection_name = self + .kv_engine + .hash_to_collection + .get(&hash) + .cloned() + .ok_or_else(|| crate::Error::Internal { + detail: format!( + "snapshot: KV table {hash} of tenant {tid} has no collection name" + ), + })?; + let bytes = zerompk::to_msgpack_vec(&table.export_entries()) + .map_err(|e| capture_error("KV table", &collection_name, e))?; + snapshot.kv_tables.push((collection_name, bytes)); + } + Ok(()) + } + + /// One Loro export per CRDT collection, plus each collection's installed + /// constraint set and version. + fn capture_crdt( + &self, + database_id: DatabaseId, + tenant: TenantId, + snapshot: &mut TenantDataSnapshot, + ) -> crate::Result<()> { + let Some(crdt) = self.crdt_engines.get(&(database_id, tenant)) else { + return Ok(()); + }; + for (collection, bytes) in crdt.export_all_snapshots()? { + snapshot + .crdt_state + .push((database_id.as_u64(), tenant.as_u64(), collection, bytes)); + } + // A snapshot-installed follower with no constraint set fences every + // peer delta on a constrained collection, so the set travels too. + for collection in crdt.collections_with_constraints() { + let version = crdt.installed_constraint_version(&collection); + if version == 0 { + continue; + } + let constraints = crdt + .constraints_for_collection(&collection) + .iter() + .map(zerompk::to_msgpack_vec) + .collect::, _>>() + .map_err(|e| capture_error("CRDT constraint set", &collection, e))?; + if constraints.is_empty() { + continue; + } + snapshot + .crdt_constraints + .push(crate::types::snapshot::CrdtConstraintEntry { + database_id: database_id.as_u64(), + tenant_id: tenant.as_u64(), + collection, + version, + constraints, + }); + } + Ok(()) + } + + /// The column data of every timeseries memtable of `tenant`. + fn capture_timeseries_memtables( + &self, + tenant: TenantId, + snapshot: &mut TenantDataSnapshot, + ) -> crate::Result<()> { + for ((d, t, coll), mt) in &self.columnar_memtables { + if *t != tenant { + continue; + } + let key_str = scoped_key(*d, *t, coll); + let bytes = zerompk::to_msgpack_vec(&mt.export_snapshot()) + .map_err(|e| capture_error("timeseries memtable", &key_str, e))?; + snapshot.timeseries.push((key_str, bytes)); + } + Ok(()) + } +} diff --git a/nodedb/src/data/executor/handlers/snapshot/create.rs b/nodedb/src/data/executor/handlers/snapshot/create.rs index cd21be88a..7d59fd24d 100644 --- a/nodedb/src/data/executor/handlers/snapshot/create.rs +++ b/nodedb/src/data/executor/handlers/snapshot/create.rs @@ -49,170 +49,65 @@ impl CoreLoop { ); } } - - // 2. Graph edges: scan edge_store by tenant prefix. + // A `bitemporal=true` collection writes only the versioned tables. match self - .edge_store - .scan_edges_for_tenant(database_id, crate::types::TenantId::new(tenant_id)) + .sparse + .scan_versioned_documents_for_tenant(database_id, tenant_id) { - Ok(edges) => snapshot.edges = edges, - Err(e) => warn!(tenant_id, error = %e, "snapshot: edge scan failed, skipping"), - } - - // 3. Vector collections: export raw vectors + doc_id_map. - // The snapshot format stores keys as `"{db}:{tid}:{coll_key}"` strings - // for disk/wire compatibility — convert the tuple key at the boundary. - let tid_obj = crate::types::TenantId::new(tenant_id); - for (key, collection) in &self.vector_collections { - if key.1 != tid_obj { - continue; - } - let vectors = match collection.export_snapshot() { - Ok(v) => v, - Err(e) => { - // Skipping is consistent with the other per-item failures - // here, but omitting vectors from a snapshot is data loss, - // so it is logged at error level rather than warn. - tracing::error!( - key = &key.2, - error = %e, - "snapshot: vector export failed, collection omitted from snapshot" - ); - continue; - } - }; - let key_str = format!("{}:{}:{}", key.0.as_u64(), key.1.as_u64(), key.2); - match zerompk::to_msgpack_vec(&vectors) { - Ok(bytes) => snapshot.vectors.push((key_str, bytes)), - Err(e) => warn!(key = &key.2, error = %e, "snapshot: vector serialization failed"), - } - } - - // 3b. Vector params: export HnswParams per collection. - for (key, params) in &self.vector_params { - if key.1 != tid_obj { - continue; - } - let key_str = format!("{}:{}:{}", key.0.as_u64(), key.1.as_u64(), key.2); - match zerompk::to_msgpack_vec(params) { - Ok(bytes) => snapshot.vector_params.push((key_str, bytes)), - Err(e) => { - warn!(key = &key.2, error = %e, "snapshot: vector_params serialization failed") - } - } - } - - // 3c. Index configs: export IndexConfig per collection. - for (key, cfg) in &self.index_configs { - if key.1 != tid_obj { - continue; - } - let key_str = format!("{}:{}:{}", key.0.as_u64(), key.1.as_u64(), key.2); - match zerompk::to_msgpack_vec(cfg) { - Ok(bytes) => snapshot.index_configs.push((key_str, bytes)), - Err(e) => { - warn!(key = &key.2, error = %e, "snapshot: index_configs serialization failed") - } + Ok(docs) => snapshot.documents_versioned = docs, + Err(e) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("snapshot: versioned document scan failed: {e}"), + }, + ); } } - - // 4. KV tables: export all entries per tenant table. - for (&hash, table) in &self.kv_engine.tables { - let Some(&tid) = self.kv_engine.hash_to_tenant.get(&hash) else { - continue; - }; - if tid != tenant_id { - continue; - } - let collection_name = self - .kv_engine - .hash_to_collection - .get(&hash) - .cloned() - .unwrap_or_else(|| hash.to_string()); - let entries = table.export_entries(); - match zerompk::to_msgpack_vec(&entries) { - Ok(bytes) => snapshot.kv_tables.push((collection_name, bytes)), - Err(e) => warn!(hash, error = %e, "snapshot: kv serialization failed"), + match self + .sparse + .scan_versioned_indexes_for_tenant(database_id, tenant_id) + { + Ok(idx) => snapshot.indexes_versioned = idx, + Err(e) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("snapshot: versioned index scan failed: {e}"), + }, + ); } } - // 5. CRDT state: one Loro export per collection. Each (tenant, - // collection) owns its own doc; entries are carried tenant-explicit and - // collection-tagged so the per-group Raft snapshot builder routes each - // by its single collection's vshard. - if let Some(crdt) = self.crdt_engines.get(&(task.request.database_id, tid_obj)) { - match crdt.export_all_snapshots() { - Ok(per_collection) => { - for (collection, bytes) in per_collection { - snapshot.crdt_state.push(( - task.request.database_id.as_u64(), - tenant_id, - collection, - bytes, - )); - } - } - Err(e) => warn!(tenant_id, error = %e, "snapshot: crdt export failed"), - } - - // 5b. CRDT constraint state: capture the installed constraint set - // + version per collection so a snapshot-installed follower - // reconstructs its validator instead of coming up empty and - // retry-fencing every peer delta on constrained collections. - for collection in crdt.collections_with_constraints() { - let version = crdt.installed_constraint_version(&collection); - if version == 0 { - continue; - } - let constraints = crdt.constraints_for_collection(&collection); - let mut encoded = Vec::with_capacity(constraints.len()); - let mut failed = false; - for constraint in &constraints { - match zerompk::to_msgpack_vec(constraint) { - Ok(bytes) => encoded.push(bytes), - Err(e) => { - warn!( - tenant_id, - collection, - error = %e, - "snapshot: crdt constraint serialization failed" - ); - failed = true; - break; - } - } - } - if failed || encoded.is_empty() { - continue; - } - snapshot - .crdt_constraints - .push(crate::types::snapshot::CrdtConstraintEntry { - database_id: task.request.database_id.as_u64(), - tenant_id, - collection, - version, - constraints: encoded, - }); + // 2. Graph edges: scan edge_store by tenant prefix. A snapshot that + // omitted them would restore every node with no edges. + match self + .edge_store + .scan_edges_for_tenant(database_id, crate::types::TenantId::new(tenant_id)) + { + Ok(edges) => snapshot.edges = edges, + Err(e) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("snapshot: edge scan failed: {e}"), + }, + ); } } - // 6. Timeseries memtables: serialize column data. - // Snapshot format encodes "{database_id}:{tenant_id}:{collection}" keys. + // 3-6. The in-memory engines: vectors, KV, CRDT, timeseries + // memtables. A section that fails to export fails the snapshot: a + // snapshot that left it out would restore without it. let tid_id = crate::types::TenantId::new(tenant_id); - for ((d, t, coll), mt) in &self.columnar_memtables { - if *t != tid_id { - continue; - } - let key_str = format!("{}:{}:{}", d.as_u64(), t.as_u64(), coll); - match zerompk::to_msgpack_vec(&mt.export_snapshot()) { - Ok(bytes) => snapshot.timeseries.push((key_str, bytes)), - Err(e) => { - let key = &key_str; - warn!(key, error = %e, "snapshot: timeseries serialization failed"); - } - } + if let Err(e) = self.capture_memory_engines(task.request.database_id, tid_id, &mut snapshot) + { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("snapshot: in-memory engine capture failed: {e}"), + }, + ); } // 7. Flushed timeseries segments: capture all on-disk partition @@ -246,6 +141,8 @@ impl CoreLoop { tenant_id, documents = snapshot.documents.len(), indexes = snapshot.indexes.len(), + documents_versioned = snapshot.documents_versioned.len(), + indexes_versioned = snapshot.indexes_versioned.len(), edges = snapshot.edges.len(), vectors = snapshot.vectors.len(), kv_tables = snapshot.kv_tables.len(), diff --git a/nodedb/src/data/executor/handlers/snapshot/mod.rs b/nodedb/src/data/executor/handlers/snapshot/mod.rs index 8bcd9e009..429c44f49 100644 --- a/nodedb/src/data/executor/handlers/snapshot/mod.rs +++ b/nodedb/src/data/executor/handlers/snapshot/mod.rs @@ -1,5 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 +mod capture_memory; pub mod create; pub mod restore; mod restore_segments; diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs b/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs index b4c2acf41..39a6aa825 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs @@ -4,36 +4,35 @@ //! `tenant_snapshot::execute_restore_tenant_snapshot` for the sparse/document, //! vector, KV, CRDT, and timeseries engines. -use tracing::warn; - use crate::data::executor::core_loop::CoreLoop; use super::keys::database_id_from_qualified; impl CoreLoop { + /// Install the snapshot's document rows, document versions and index + /// entries under their exported keys. Returns `(documents, indexes)` + /// written. The first entry that fails to install fails the restore: a + /// follower missing it would serve a partial collection. pub(super) fn restore_sparse( &self, - _tenant_id: u64, - documents: &[(String, Vec)], - indexes: &[(String, Vec)], - ) -> (u64, u64) { - let mut docs_written = 0u64; - for (key, value) in documents { - if let Err(e) = self.sparse.put_raw(key, value) { - warn!(key, error = %e, "failed to restore document"); - continue; - } - docs_written += 1; + snap: &crate::types::TenantDataSnapshot, + ) -> crate::Result<(u64, u64)> { + for (key, value) in &snap.documents { + self.sparse.put_raw(key, value)?; + } + for (key, value) in &snap.documents_versioned { + self.sparse.put_versioned_document_raw(key, value)?; + } + for (key, value) in &snap.indexes { + self.sparse.put_index_raw(key, value)?; } - let mut indexes_written = 0u64; - for (key, value) in indexes { - if let Err(e) = self.sparse.put_index_raw(key, value) { - warn!(key, error = %e, "failed to restore index"); - continue; - } - indexes_written += 1; + for (key, value) in &snap.indexes_versioned { + self.sparse.put_versioned_index_raw(key, value)?; } - (docs_written, indexes_written) + Ok(( + (snap.documents.len() + snap.documents_versioned.len()) as u64, + (snap.indexes.len() + snap.indexes_versioned.len()) as u64, + )) } pub(super) fn restore_vector_collection( diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs b/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs index 7151d1a17..cf2a28f8a 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs @@ -5,7 +5,7 @@ use std::sync::Arc; -use tracing::{info, warn}; +use tracing::info; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; @@ -72,8 +72,17 @@ impl CoreLoop { } }; - let (docs_written, indexes_written) = - self.restore_sparse(tenant_id, &snap.documents, &snap.indexes); + let (docs_written, indexes_written) = match self.restore_sparse(&snap) { + Ok(written) => written, + Err(e) => { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("restore: document install failed: {e}"), + }, + ); + } + }; let mut edges_written = 0u64; let mut vectors_written = 0u64; @@ -90,8 +99,12 @@ impl CoreLoop { let database_id = task.request.database_id.as_u64(); for (key, props) in &snap.edges { if let Err(e) = self.edge_store.put_edge_raw(database_id, tid, key, props) { - warn!(key, error = %e, "failed to restore edge"); - continue; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("restore: edge install failed: {e}"), + }, + ); } edges_written += 1; } @@ -107,8 +120,12 @@ impl CoreLoop { .edge_store .put_edge_raw(database_id, edge_tid, key, props) { - warn!(key, error = %e, "failed to restore tenant edge"); - continue; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("restore: tenant edge install failed: {e}"), + }, + ); } edges_written += 1; } @@ -144,8 +161,14 @@ impl CoreLoop { match zerompk::from_msgpack(bytes) { Ok(p) => p, Err(e) => { - warn!(key, error = %e, "failed to decode vector_params snapshot entry"); - continue; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "restore: vector params '{key}' do not decode: {e}" + ), + }, + ); } }; let (vp_db, coll_key) = parse_vector_snapshot_key(key, tenant_id); @@ -164,8 +187,14 @@ impl CoreLoop { match zerompk::from_msgpack(bytes) { Ok(c) => c, Err(e) => { - warn!(key, error = %e, "failed to decode index_configs snapshot entry"); - continue; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "restore: vector index config '{key}' does not decode: {e}" + ), + }, + ); } }; let (ic_db, coll_key) = parse_vector_snapshot_key(key, tenant_id); @@ -186,8 +215,14 @@ impl CoreLoop { match zerompk::from_msgpack(bytes) { Ok(v) => v, Err(e) => { - warn!(key, error = %e, "failed to decode vector snapshot"); - continue; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "restore: vector collection '{key}' does not decode: {e}" + ), + }, + ); } }; let count = vectors.len() as u64; @@ -210,12 +245,9 @@ impl CoreLoop { // own fsynced `.snap` file plus this local checkpoint. Without a // synchronous checkpoint here, a crash before the next periodic // `checkpoint_vector_indexes()` run (every 5 minutes) would lose - // the just-installed vectors. Checkpointing is cheap-idempotent - // (skips empty collections), so this is unconditional rather than - // gated on `replace_mode`: the user-RESTORE path drains vector - // data before it ever reaches here (see - // `control/backup/restore/orchestrate/mod.rs`), so `vectors_written` - // is 0 and the call is a no-op on that path. + // the just-installed vectors. A failed checkpoint fails the + // install, so the snapshot is installed again rather than left + // memory-only. if vectors_written > 0 { match self.checkpoint_vector_indexes() { Ok(outcome) => { @@ -226,21 +258,14 @@ impl CoreLoop { "vector snapshot install checkpointed synchronously" ); } - // The install path has no LSN to clamp — it applies - // Raft-committed vectors with no WAL record behind them, so - // there is no truncation authority to narrow here. What a - // failure does mean is that this core's only local copy of - // the just-installed vectors is still in memory, which the - // operator needs to see rather than have it disappear into - // a discarded count. Err(e) => { - warn!( - core = self.core_id, - tenant_id, - error = %e, - "vector snapshot install checkpoint failed; the installed \ - vectors are memory-only on this core until the next \ - successful checkpoint" + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "restore: vector checkpoint after the install failed: {e}" + ), + }, ); } } @@ -251,8 +276,14 @@ impl CoreLoop { let entries: Vec<(Vec, Vec, u64)> = match zerompk::from_msgpack(bytes) { Ok(e) => e, Err(e) => { - warn!(collection_name, error = %e, "failed to decode kv snapshot"); - continue; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "restore: KV table '{collection_name}' does not decode: {e}" + ), + }, + ); } }; let count = entries.len() as u64; @@ -269,19 +300,22 @@ impl CoreLoop { for (database_raw, tid_raw, collection, bytes) in &snap.crdt_state { if let Err(e) = self.restore_crdt_state(*database_raw, *tid_raw, collection, bytes) { - warn!(tid_raw, %collection, error = %e, "failed to restore crdt state"); - } else { - crdt_written += 1; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "restore: CRDT state of '{collection}' (tenant {tid_raw}) failed: {e}" + ), + }, + ); } + crdt_written += 1; } // Restore CRDT constraint state per collection: reconstructs the // validator's installed constraint set + `installed_constraint_version` // so a snapshot-installed follower does not come up empty and - // retry-fence every peer delta on constrained collections. Fail-safe - // on error — warn and continue, matching the `crdt_state` loop, since - // a failed reconstruction only reverts to the pre-fix (over-rejecting) - // behavior rather than corrupting state. + // retry-fence every peer delta on constrained collections. for entry in &snap.crdt_constraints { if let Err(e) = self.restore_crdt_constraints( entry.database_id, @@ -290,11 +324,17 @@ impl CoreLoop { entry.version, &entry.constraints, ) { - let (tid_raw, collection) = (entry.tenant_id, &entry.collection); - warn!(tid_raw, %collection, error = %e, "failed to restore crdt constraints"); - } else { - crdt_constraints_written += 1; + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "restore: CRDT constraints of '{}' (tenant {}) failed: {e}", + entry.collection, entry.tenant_id + ), + }, + ); } + crdt_constraints_written += 1; } // Restore timeseries memtables and flush each to an on-disk segment diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs index e1e4d718c..bb055f58f 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/entry.rs @@ -31,7 +31,7 @@ //! 5. the fold target rows in `Response::write_set`, so the funnel journals //! them. -use nodedb_physical::physical_plan::RedoSumTargets; +use nodedb_physical::physical_plan::{RedoOrigin, RedoSumTargets}; use nodedb_wal::WalRecord; use nodedb_wal::record::{RecordType, WalRecordArgs}; @@ -57,6 +57,7 @@ pub(in crate::data::executor) struct CommittedRedo<'a> { impl CoreLoop { /// Apply one committed redo record. The request carries the LSN of the /// `TransactionRedo` WAL record the funnel appended for it. + #[cfg(test)] pub(in crate::data::executor) fn execute_apply_transaction_redo( &mut self, task: &ExecutionTask, @@ -75,6 +76,18 @@ impl CoreLoop { task: &ExecutionTask, tid: u64, committed: CommittedRedo<'_>, + ) -> Response { + self.install_redo(task, tid, committed, RedoOrigin::Commit) + } + + /// [`Self::install_committed_redo`] for a record from `origin`, which + /// decides the commit-boundary checks the validate pass runs. + pub(in crate::data::executor) fn install_redo( + &mut self, + task: &ExecutionTask, + tid: u64, + committed: CommittedRedo<'_>, + origin: RedoOrigin, ) -> Response { let Some(lsn) = task.wal_lsn() else { return self.response_error( @@ -98,7 +111,7 @@ impl CoreLoop { let check_scope = RedoApplyScope::new(RedoApplyPass::Validate, committed.sum_targets.to_vec()); if let Err(error) = - self.validate_redo_document_ops(database_id, tid, &doc_ops, &check_scope) + self.validate_redo_document_ops(database_id, tid, &doc_ops, &check_scope, origin) { return self.response_error(task, error); } diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/validate.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/validate.rs index 770e743a2..43ebdd51d 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/validate.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/validate.rs @@ -17,9 +17,15 @@ //! * Stateless PUT / DELETE enforcement — append-only, period lock, state //! transitions, transition checks, retention and legal hold, each against //! the stored pre-image. +//! +//! A [`RedoOrigin::Restore`] record re-installs rows a backup captured. Each +//! row passed BALANCED and the stateless rules when it was first written, and +//! a bitemporal row's earlier versions are history. So a restore runs UNIQUE +//! only. UNIQUE judges the post-state: the last write to each row. use std::collections::{BTreeMap, HashMap, HashSet}; +use nodedb_physical::physical_plan::RedoOrigin; use nodedb_types::Surrogate; use crate::data::executor::core_loop::CoreLoop; @@ -44,6 +50,7 @@ impl CoreLoop { tid: u64, ops: &[RedoDocOp], apply_scope: &super::state::RedoApplyScope, + origin: RedoOrigin, ) -> crate::Result<()> { let mut by_collection: BTreeMap<&str, Vec<&RedoDocOp>> = BTreeMap::new(); for op in ops { @@ -70,7 +77,9 @@ impl CoreLoop { bitemporal, resolved: &resolved, }; - self.check_collection_ops(&scope, &collection_ops)?; + if origin == RedoOrigin::Commit { + self.check_collection_ops(&scope, &collection_ops)?; + } self.check_unique_post_state(&scope, &collection_ops)?; } Ok(()) @@ -177,17 +186,28 @@ impl CoreLoop { // Rows this record rewrites or removes: their stored values do not // count, whatever they are. let touched: HashSet = ops.iter().map(|op| op.surrogate()).collect(); + // The post-state holds each row's last write only. A restored + // bitemporal row writes each of its versions in order, and a value an + // earlier version held is not in the post-state. + let last_write: HashMap = ops + .iter() + .enumerate() + .map(|(index, op)| (op.surrogate(), index)) + .collect(); let doc_engine = DocumentEngine::new(&self.sparse, scope.database_id, scope.tid); for path in unique_paths { // Needle → the surrogate of the record's own row holding it. let mut claimed: HashMap = HashMap::new(); - for op in ops { + for (index, op) in ops.iter().enumerate() { let RedoDocOp::Put { value, surrogate, .. } = op else { continue; }; + if last_write.get(surrogate) != Some(&index) { + continue; + } let Ok(doc) = doc_format::decode_document(value) else { continue; }; diff --git a/nodedb/src/diag/context/mod.rs b/nodedb/src/diag/context/mod.rs index 2bb6ac718..85b8ef86b 100644 --- a/nodedb/src/diag/context/mod.rs +++ b/nodedb/src/diag/context/mod.rs @@ -12,6 +12,7 @@ mod data_plane; mod ingest; mod outcome_floor; mod quota; +mod raft_apply; mod recovery; mod retention; mod vector; @@ -34,6 +35,7 @@ pub(in crate::diag) use quota::{ QuotaRowNotInstalled, QuotaRowWriteFailed, QuotaScopePurgeIncomplete, QuotaScopeReplayAborted, ScopeQuotaNotInstalled, }; +pub(in crate::diag) use raft_apply::{RaftEntryReapplied, ReplicatedWriteParked}; pub(in crate::diag) use recovery::{ReplayRecordUnapplied, WalArchivalFailedTruncationHeld}; pub(in crate::diag) use retention::RetentionAutowireOrphaned; pub(in crate::diag) use vector::VectorIndexNotApplied; diff --git a/nodedb/src/diag/context/raft_apply.rs b/nodedb/src/diag/context/raft_apply.rs new file mode 100644 index 000000000..e3f06c015 --- /dev/null +++ b/nodedb/src/diag/context/raft_apply.rs @@ -0,0 +1,79 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Forensic payloads for the data-group apply path. + +use faultbox::DomainContext; +use faultbox::serde_json::{Value, json}; + +/// The apply loop reached a Raft log index it had already applied. +pub(in crate::diag) struct RaftEntryReapplied { + pub group_id: u64, + pub log_index: u64, + /// The highest index of the group the apply loop applied before. + pub highest_applied: u64, +} + +impl DomainContext for RaftEntryReapplied { + fn domain_kind(&self) -> &'static str { + "nodedb.raft_entry_reapplied" + } + + fn grouping_key(&self) -> String { + // One bug class: a path handed an applied index to the apply loop + // again. Group and index are the occurrence. + "raft_entry_reapplied".to_string() + } + + fn to_json(&self) -> Value { + json!({ + "group_id": self.group_id, + "log_index": self.log_index, + "highest_applied": self.highest_applied, + "why_fatal": "a committed entry applies once per replica. A second apply of an \ + append-shaped write (timeseries, columnar, spatial, predicated \ + update) stores its rows twice, mints a second redo record, fires \ + AFTER triggers twice, and records the write's mark again", + "operator_action": "the entry reached the apply loop twice past the applier's \ + delivery watermark. Find the path that re-queued it: \ + re-delivery after a refused hand-off, snapshot install, or \ + a second applier feeding the same channel", + }) + } +} + +/// A replicated write waited on a core for a staged Calvin transaction that +/// owns its rows. The data group's apply loop awaits it, so every later entry +/// of every group waits too. +pub(in crate::diag) struct ReplicatedWriteParked { + pub core_id: usize, + pub collection: String, + pub owner_epoch: u64, + pub owner_position: u32, + pub owner_vshard: u32, +} + +impl DomainContext for ReplicatedWriteParked { + fn domain_kind(&self) -> &'static str { + "nodedb.replicated_write_parked" + } + + fn grouping_key(&self) -> String { + "replicated_write_parked".to_string() + } + + fn to_json(&self) -> Value { + json!({ + "core_id": self.core_id, + "collection": self.collection, + "owner_epoch": self.owner_epoch, + "owner_position": self.owner_position, + "owner_vshard": self.owner_vshard, + "why_fatal": "the data-group apply loop applies committed entries one at a \ + time. A parked replicated write holds it until the Calvin \ + transaction flushes or drops, so replicated writes to unrelated \ + collections wait behind an external event", + "operator_action": "check the named Calvin transaction's flush; the parked \ + write is refused at its request deadline", + }) + } +} diff --git a/nodedb/src/diag/mod.rs b/nodedb/src/diag/mod.rs index 54dd99a67..fc67e601d 100644 --- a/nodedb/src/diag/mod.rs +++ b/nodedb/src/diag/mod.rs @@ -16,8 +16,9 @@ pub use recording::{ fts_index_update_failed, history_compaction_not_applied, ilp_invalid_utf8_drop, ilp_line_read_drop, metadata_apply_wedged, orphaned_index_entry_after_delete, quota_row_invalid, quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, - quota_scope_replay_aborted, replay_record_unapplied, retention_autowire_orphaned, - scope_quota_not_installed, strict_row_undecodable, synonym_group_not_applied, - vector_index_not_applied, wal_archival_failed_truncation_held, write_acked_without_durability, - write_window_held, write_window_leaked, + quota_scope_replay_aborted, raft_entries_reapplied, raft_entry_reapplied, + replay_record_unapplied, replicated_write_parked, replicated_writes_parked, + retention_autowire_orphaned, scope_quota_not_installed, strict_row_undecodable, + synonym_group_not_applied, vector_index_not_applied, wal_archival_failed_truncation_held, + write_acked_without_durability, write_window_held, write_window_leaked, }; diff --git a/nodedb/src/diag/recording/mod.rs b/nodedb/src/diag/recording/mod.rs index 73afb8d47..8ebe9a52a 100644 --- a/nodedb/src/diag/recording/mod.rs +++ b/nodedb/src/diag/recording/mod.rs @@ -14,6 +14,7 @@ mod data_plane; mod ingest; mod outcome_floor; mod quota; +mod raft_apply; mod recovery; mod retention; mod shared; @@ -34,6 +35,9 @@ pub use quota::{ quota_row_invalid, quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, quota_scope_replay_aborted, scope_quota_not_installed, }; +pub use raft_apply::{ + raft_entries_reapplied, raft_entry_reapplied, replicated_write_parked, replicated_writes_parked, +}; pub use recovery::{ batch_insert_without_surrogates, fts_index_update_failed, orphaned_index_entry_after_delete, replay_record_unapplied, strict_row_undecodable, wal_archival_failed_truncation_held, diff --git a/nodedb/src/diag/recording/raft_apply.rs b/nodedb/src/diag/recording/raft_apply.rs new file mode 100644 index 000000000..55152d53e --- /dev/null +++ b/nodedb/src/diag/recording/raft_apply.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capture sites on the data-group apply path. + +use std::sync::atomic::{AtomicU64, Ordering}; + +use faultbox::{Capture, EventKind}; + +use crate::diag::context; + +/// Count of committed Raft entries the apply loop applied a second time. +static RAFT_ENTRIES_REAPPLIED: AtomicU64 = AtomicU64::new(0); + +/// Count of replicated writes parked behind a staged Calvin transaction. +static REPLICATED_WRITES_PARKED: AtomicU64 = AtomicU64::new(0); + +/// Read the count of committed Raft entries applied twice. Exposed for the +/// metrics exporter and tests. +pub fn raft_entries_reapplied() -> u64 { + RAFT_ENTRIES_REAPPLIED.load(Ordering::Relaxed) +} + +/// Read the count of replicated writes parked behind a staged Calvin +/// transaction. Exposed for the metrics exporter and tests. +pub fn replicated_writes_parked() -> u64 { + REPLICATED_WRITES_PARKED.load(Ordering::Relaxed) +} + +/// Report an apply of a Raft log index the apply loop already applied. +/// Called from the apply loop, the one site that sees every applied index. +pub fn raft_entry_reapplied(group_id: u64, log_index: u64, highest_applied: u64) { + RAFT_ENTRIES_REAPPLIED.fetch_add(1, Ordering::Relaxed); + let ctx = context::RaftEntryReapplied { + group_id, + log_index, + highest_applied, + }; + let _ = Capture::new( + EventKind::InvariantViolation, + "committed Raft entry applied a second time", + ) + .domain(&ctx) + .with_backtrace() + .emit(); +} + +/// Report a replicated write a core parked behind a staged Calvin +/// transaction. Called from the core's Calvin fence when it parks a write +/// whose order Raft already fixed. +pub fn replicated_write_parked(core_id: usize, collection: &str, owner: (u64, u32, u32)) { + REPLICATED_WRITES_PARKED.fetch_add(1, Ordering::Relaxed); + let ctx = context::ReplicatedWriteParked { + core_id, + collection: collection.to_owned(), + owner_epoch: owner.0, + owner_position: owner.1, + owner_vshard: owner.2, + }; + let _ = Capture::new( + EventKind::Error, + "replicated write parked behind a staged Calvin transaction", + ) + .domain(&ctx) + .emit(); +} diff --git a/nodedb/src/engine/bitemporal/enforcement.rs b/nodedb/src/engine/bitemporal/enforcement.rs index 6599043d3..67aff6bce 100644 --- a/nodedb/src/engine/bitemporal/enforcement.rs +++ b/nodedb/src/engine/bitemporal/enforcement.rs @@ -61,7 +61,15 @@ async fn enforcement_loop( mut shutdown: watch::Receiver, tick: Duration, ) { - tokio::time::sleep(Duration::from_secs(STARTUP_DELAY_SECS)).await; + // Shutdown ends the startup delay: a server stopped within it must not + // wait it out. + tokio::select! { + _ = tokio::time::sleep(Duration::from_secs(STARTUP_DELAY_SECS)) => {} + _ = shutdown.wait_for(|stopping| *stopping) => { + info!("bitemporal retention loop shutting down"); + return; + } + } loop { tokio::select! { diff --git a/nodedb/src/engine/sparse/btree_scan.rs b/nodedb/src/engine/sparse/btree_scan.rs index eba450258..91aa836dd 100644 --- a/nodedb/src/engine/sparse/btree_scan.rs +++ b/nodedb/src/engine/sparse/btree_scan.rs @@ -449,7 +449,7 @@ impl SparseEngine { } /// Shared scan logic for any redb table with database/tenant-prefixed keys. - fn scan_table_for_tenant( + pub(in crate::engine::sparse) fn scan_table_for_tenant( &self, table_def: redb::TableDefinition<&str, &[u8]>, database_id: u64, diff --git a/nodedb/src/engine/sparse/btree_versioned/mod.rs b/nodedb/src/engine/sparse/btree_versioned/mod.rs index 31e98c74a..b6f73b0f5 100644 --- a/nodedb/src/engine/sparse/btree_versioned/mod.rs +++ b/nodedb/src/engine/sparse/btree_versioned/mod.rs @@ -13,6 +13,7 @@ pub mod index; pub mod key; pub mod purge; pub mod scan; +pub mod snapshot; pub mod value; pub use doc::VersionedRow; diff --git a/nodedb/src/engine/sparse/btree_versioned/snapshot.rs b/nodedb/src/engine/sparse/btree_versioned/snapshot.rs new file mode 100644 index 000000000..afaf06c82 --- /dev/null +++ b/nodedb/src/engine/sparse/btree_versioned/snapshot.rs @@ -0,0 +1,144 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Tenant export and raw import of the versioned document and index tables. +//! +//! A `bitemporal=true` collection writes only these tables, so a tenant +//! snapshot that reads `DOCUMENTS` and `INDEXES` alone carries none of its +//! rows. Keys and values move verbatim: every version keeps its system time. + +use redb::TableDefinition; + +use super::doc::DOCUMENTS_VERSIONED; +use super::index::INDEXES_VERSIONED; +use crate::engine::sparse::btree::{SparseEngine, redb_err}; + +impl SparseEngine { + /// Every document version of `tenant_id` in `database_id`, as + /// `("{db}:{tid}:{coll}:{doc_id}\x00{sys_from:020}", versioned value)`. + pub fn scan_versioned_documents_for_tenant( + &self, + database_id: u64, + tenant_id: u64, + ) -> crate::Result)>> { + self.scan_table_for_tenant( + DOCUMENTS_VERSIONED, + database_id, + tenant_id, + "versioned document scan", + ) + } + + /// Every versioned index entry of `tenant_id` in `database_id`. + pub fn scan_versioned_indexes_for_tenant( + &self, + database_id: u64, + tenant_id: u64, + ) -> crate::Result)>> { + self.scan_table_for_tenant( + INDEXES_VERSIONED, + database_id, + tenant_id, + "versioned index scan", + ) + } + + /// Install one document version under its exported key. + pub fn put_versioned_document_raw(&self, key: &str, value: &[u8]) -> crate::Result<()> { + self.put_table_raw(DOCUMENTS_VERSIONED, key, value) + } + + /// Install one versioned index entry under its exported key. + pub fn put_versioned_index_raw(&self, key: &str, value: &[u8]) -> crate::Result<()> { + self.put_table_raw(INDEXES_VERSIONED, key, value) + } + + fn put_table_raw( + &self, + table_def: TableDefinition<&str, &[u8]>, + key: &str, + value: &[u8], + ) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| redb_err("write txn", e))?; + { + let mut table = txn + .open_table(table_def) + .map_err(|e| redb_err("open table", e))?; + table + .insert(key, value) + .map_err(|e| redb_err("raw insert", e))?; + } + txn.commit().map_err(|e| redb_err("commit", e))?; + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::{StorageKey, Surrogate}; + + use super::super::value::VersionedPut; + use super::*; + + fn open_temp() -> (SparseEngine, tempfile::TempDir) { + let dir = tempfile::tempdir().unwrap(); + let engine = SparseEngine::open(&dir.path().join("s.redb")).unwrap(); + (engine, dir) + } + + #[test] + fn exported_versions_install_verbatim_on_another_engine() { + let (source, _source_dir) = open_temp(); + let key = StorageKey::for_surrogate(Surrogate::new(7)); + for (sys, body) in [(100, b"v1".as_slice()), (200, b"v2".as_slice())] { + source + .versioned_put(VersionedPut { + database_id: 0, + tenant: 3, + coll: "ledger", + doc_id: &key, + sys_from_ms: sys, + valid_from_ms: 0, + valid_until_ms: i64::MAX, + body, + }) + .unwrap(); + } + source + .versioned_put(VersionedPut { + database_id: 0, + tenant: 4, + coll: "ledger", + doc_id: &key, + sys_from_ms: 100, + valid_from_ms: 0, + valid_until_ms: i64::MAX, + body: b"other tenant", + }) + .unwrap(); + + let exported = source.scan_versioned_documents_for_tenant(0, 3).unwrap(); + assert_eq!(exported.len(), 2, "only tenant 3's versions export"); + + let (target, _target_dir) = open_temp(); + for (key, value) in &exported { + target.put_versioned_document_raw(key, value).unwrap(); + } + assert_eq!( + target + .versioned_get_as_of(0, 3, "ledger", &key, Some(150), None) + .unwrap() + .as_deref(), + Some(b"v1".as_slice()) + ); + assert_eq!( + target + .versioned_get_current(0, 3, "ledger", &key) + .unwrap() + .as_deref(), + Some(b"v2".as_slice()) + ); + } +} diff --git a/nodedb/src/engine/timeseries/retention_policy/enforcement.rs b/nodedb/src/engine/timeseries/retention_policy/enforcement.rs index 23ba09116..ff79dbf7b 100644 --- a/nodedb/src/engine/timeseries/retention_policy/enforcement.rs +++ b/nodedb/src/engine/timeseries/retention_policy/enforcement.rs @@ -43,8 +43,15 @@ async fn enforcement_loop( registry: Arc, mut shutdown: watch::Receiver, ) { - // Start with a short initial delay to let the system warm up. - tokio::time::sleep(Duration::from_secs(10)).await; + // Start with a short initial delay to let the system warm up. Shutdown + // ends the delay: a server stopped within it must not wait it out. + tokio::select! { + _ = tokio::time::sleep(Duration::from_secs(10)) => {} + _ = shutdown.wait_for(|stopping| *stopping) => { + info!("retention enforcement loop shutting down"); + return; + } + } loop { // Find the shortest eval interval among all enabled policies. diff --git a/nodedb/src/error/types.rs b/nodedb/src/error/types.rs index 96ea7fd2f..19f13e65b 100644 --- a/nodedb/src/error/types.rs +++ b/nodedb/src/error/types.rs @@ -358,6 +358,31 @@ pub enum Error { )] RetryableLeaderChange { group_id: u64, log_index: u64 }, + /// A Raft group has no reachable majority of its voters, so nothing can + /// commit in it. The operation that needed it applied nothing. It succeeds + /// once enough of `unreachable` rejoin. + #[error( + "raft group {group_id} has no reachable quorum: voters {voters:?}, unreachable \ + {unreachable:?}; nothing was applied. Retry once a majority of its voters is reachable" + )] + GroupQuorumUnavailable { + group_id: u64, + voters: Vec, + unreachable: Vec, + }, + + /// RESTORE's staleness guard found no replica of a data group that + /// reported the tenant's write marks before the statement deadline. The + /// restore cannot prove it is not stale, so it applied nothing. + /// `refused_by` lists the nodes that answered that they do not replicate + /// the group. + #[error( + "restore: no replica of raft group {group_id} reported the tenant's write marks before \ + the statement deadline (nodes that do not replicate it: {refused_by:?}); nothing was \ + restored. Retry once the group's placement settles" + )] + GroupMarksUnavailable { group_id: u64, refused_by: Vec }, + /// No leader elected on the metadata group yet. Transient — an election /// is in progress — so callers can wait it out instead of failing. #[error("metadata raft group has no elected leader yet; retry needed")] diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index 1add05e5d..afd32e72b 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -186,6 +186,8 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { "metadata raft group has no elected leader yet; retry exhausted".to_string(), ), Error::AuthorizationStateBehind { .. } => NodeDbError::cluster(e.to_string()), + Error::GroupQuorumUnavailable { .. } => NodeDbError::cluster(e.to_string()), + Error::GroupMarksUnavailable { .. } => NodeDbError::cluster(e.to_string()), Error::ExecutionLimitExceeded { detail } => NodeDbError::bad_request(detail), Error::LimitExceeded { limit_name, diff --git a/nodedb/src/event/alert/executor.rs b/nodedb/src/event/alert/executor.rs index a4801d544..1df1d23f7 100644 --- a/nodedb/src/event/alert/executor.rs +++ b/nodedb/src/event/alert/executor.rs @@ -45,8 +45,12 @@ async fn alert_eval_loop( registry: Arc, mut shutdown: watch::Receiver, ) { - // Initial delay to let the system warm up. - tokio::time::sleep(Duration::from_secs(5)).await; + // Initial delay to let the system warm up. Shutdown ends the delay: a + // server stopped within it must not wait it out. + tokio::select! { + _ = tokio::time::sleep(Duration::from_secs(5)) => {} + _ = shutdown.wait_for(|stopping| *stopping) => return, + } // Use the shared hysteresis manager from SharedState so DROP/ALTER handlers // and the eval loop operate on the same state. diff --git a/nodedb/src/main_boot/shutdown_wiring.rs b/nodedb/src/main_boot/shutdown_wiring.rs index 329f50da0..c6a713093 100644 --- a/nodedb/src/main_boot/shutdown_wiring.rs +++ b/nodedb/src/main_boot/shutdown_wiring.rs @@ -60,6 +60,17 @@ pub(crate) fn wire_shutdown_bus( shared.tuning.shutdown.deadline(), ); + // Final WAL fsync. Every append before this phase is buffered until an + // fsync covers it, and a write whose caller never awaited durability + // (a Calvin applied marker, a background maintenance record) has only + // this fsync between it and a graceful exit. Registered at startup so the + // bus cannot pass `WalFsync` without it. + spawn_wal_fsync_barrier(Arc::clone(shared), &shutdown_bus); + + // Close the cluster's QUIC endpoint once the Raft loops stopped, so peers + // see the close at once and the node's UDP port is free for a restart. + spawn_transport_close(Arc::clone(shared), &shutdown_bus); + // Test-only injection: if NODEDB_TEST_SLOW_DRAIN_TASK=1, register a drain // task that sleeps for 2s without calling report_drained, to verify the // offender-abort path in integration tests. This code path is guarded @@ -81,6 +92,53 @@ pub(crate) fn wire_shutdown_bus( (shutdown_rx, shutdown_bus, loop_registry_supervisors) } +/// Register the `WalFsync` participant: once the phase starts, fsync the +/// WAL on a blocking thread, then report drained. +fn spawn_wal_fsync_barrier(shared: Arc, shutdown_bus: &ShutdownBus) { + let mut guard = + shutdown_bus.register_critical_task(ShutdownPhase::WalFsync, "wal::final_fsync"); + tokio::spawn(async move { + guard.await_signal().await; + let wal = Arc::clone(&shared.wal); + match tokio::task::spawn_blocking(move || wal.sync()).await { + Ok(Ok(())) => tracing::info!("shutdown: WAL fsynced"), + Ok(Err(error)) => { + tracing::error!(%error, "shutdown: final WAL fsync failed; restart replays only what reached disk") + } + Err(error) => { + tracing::error!(%error, "shutdown: final WAL fsync task did not complete") + } + } + guard.report_drained(); + }); +} + +/// Register the cluster transport's close at `WalFsync`, after every Raft +/// loop stopped in an earlier phase. A node with no cluster transport +/// registers nothing. +fn spawn_transport_close(shared: Arc, shutdown_bus: &ShutdownBus) { + let Some(transport) = shared.cluster_transport.clone() else { + return; + }; + let mut guard = + shutdown_bus.register_task(ShutdownPhase::WalFsync, "cluster::transport_close", None); + tokio::spawn(async move { + guard.await_signal().await; + if transport + .close(nodedb::control::shutdown::PHASE_BUDGET) + .await + { + tracing::info!("shutdown: cluster transport closed"); + } else { + tracing::warn!( + "shutdown: cluster transport closed, but peers did not acknowledge the close \ + within the phase budget" + ); + } + guard.report_drained(); + }); +} + /// Drain phases that own registry loops, in shutdown order, each with the /// bus task name of its barrier. /// diff --git a/nodedb/src/types/snapshot.rs b/nodedb/src/types/snapshot.rs index fcba2f4e7..98f2e6bf5 100644 --- a/nodedb/src/types/snapshot.rs +++ b/nodedb/src/types/snapshot.rs @@ -157,6 +157,30 @@ pub struct TenantDataSnapshot { #[msgpack(default)] #[serde(default)] pub crdt_constraints: Vec, + + /// The group's tenant write marks, for the per-group Raft snapshot: + /// `[(tenant_id, commit_hlc, site_code, collection), ...]`. A follower + /// caught up by the snapshot never applies the entries it covers, so it + /// takes their marks from here, and RESTORE's staleness guard reads the + /// same marks on it as on every other replica. Empty in a backup: the + /// guard reads the marks of the destination cluster. + #[msgpack(default)] + #[serde(default)] + pub group_write_marks: Vec<(u64, u64, u8, String)>, + + /// Document versions of `bitemporal=true` collections: + /// `[("{db}:{tid}:{collection}:{doc_id}\x00{system_from:020}", versioned_value), ...]`. + /// These collections write only the versioned table, so `documents` + /// carries none of their rows. Every version keeps its system time. + #[msgpack(default)] + #[serde(default)] + pub documents_versioned: Vec<(String, Vec)>, + + /// Versioned secondary-index entries of `bitemporal=true` collections: + /// `[("{db}:{tid}:{collection}:{field}:{value}:{doc_id}\x00{system_from:020}", tag), ...]`. + #[msgpack(default)] + #[serde(default)] + pub indexes_versioned: Vec<(String, Vec)>, } /// One collection's CRDT constraint set plus its installed version, carried diff --git a/nodedb/tests/apply_pipeline_group_independence.rs b/nodedb/tests/apply_pipeline_group_independence.rs new file mode 100644 index 000000000..81ad78d85 --- /dev/null +++ b/nodedb/tests/apply_pipeline_group_independence.rs @@ -0,0 +1,108 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A write parked in one data group never holds back another group. +//! +//! The apply loop applies every data group this node hosts. A write to +//! collection A is parked at the fail gate `funnel::before_dispatch::`, +//! after its record is appended and before its core holds it. A's group waits +//! on it, because every later entry of that group must reach its core after +//! it. A write to collection B, applied in another group, must apply and be +//! acknowledged while A's write is parked. Once released, A's write applies. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +mod crash_harness; + +use std::time::Duration; + +use crash_harness::CrashHarness; +use crash_harness::vshards::{names_in_distinct_data_groups, single_node_data_group}; + +/// How long the parked write must stay unacknowledged. +const PARKED_FOR: Duration = Duration::from_millis(1500); + +/// How long the write in the other group may take. Well under the request +/// deadline a write waiting behind the parked one would wait out. +const OTHER_GROUP_BUDGET: Duration = Duration::from_secs(10); + +#[tokio::test(flavor = "multi_thread")] +async fn a_parked_write_in_one_data_group_never_holds_back_another_group() { + let [parked, other] = names_in_distinct_data_groups(["group_parked", "group_other"]); + assert_ne!( + single_node_data_group(&parked), + single_node_data_group(&other) + ); + let h = CrashHarness::new(); + let release = h.data_dir().join("release-parked-write"); + std::fs::write(&release, b"open").expect("open the gate for the setup"); + let mut h = h.with_env( + "NODEDB_FAILPOINTS", + &format!( + "funnel::before_dispatch::{parked}=wait_file({})", + release.display() + ), + ); + h.spawn(); + h.wait_ready(); + for name in [&parked, &other] { + h.exec(&format!( + "CREATE COLLECTION {name} (k STRING PRIMARY KEY, v STRING) WITH (engine='kv')" + )) + .await; + } + + // Park the write to `parked` between its record and its core. + std::fs::remove_file(&release).expect("close the gate"); + let conn_str = h.pgwire_conn_str(); + let insert = format!("INSERT INTO {parked} (k, v) VALUES ('p', 'held')"); + let parked_write = tokio::spawn(async move { + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .map_err(|e| e.to_string())?; + tokio::spawn(connection); + client + .simple_query(&insert) + .await + .map(|_| ()) + .map_err(|e| e.to_string()) + }); + tokio::time::sleep(PARKED_FOR).await; + assert!( + !parked_write.is_finished(), + "the write to {parked} was not parked at the gate" + ); + + tokio::time::timeout( + OTHER_GROUP_BUDGET, + h.exec(&format!("INSERT INTO {other} (k, v) VALUES ('o', 'x')")), + ) + .await + .unwrap_or_else(|_| { + panic!( + "a write to {other} stalled behind the parked write to {parked}, \ + though the two apply in different data groups" + ) + }); + assert_eq!( + h.query_col_idx(&format!("SELECT v FROM {other} WHERE k = 'o'"), 0) + .await, + vec!["x".to_string()] + ); + assert!( + !parked_write.is_finished(), + "the write to {parked} left the gate before its release" + ); + + std::fs::write(&release, b"release").expect("release the parked write"); + parked_write + .await + .expect("parked write task") + .unwrap_or_else(|e| panic!("parked write: {e}")); + assert_eq!( + h.query_col_idx(&format!("SELECT v FROM {parked} WHERE k = 'p'"), 0) + .await, + vec!["held".to_string()] + ); +} diff --git a/nodedb/tests/calvin_hold_liveness.rs b/nodedb/tests/calvin_hold_liveness.rs new file mode 100644 index 000000000..426b55da0 --- /dev/null +++ b/nodedb/tests/calvin_hold_liveness.rs @@ -0,0 +1,113 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A held Calvin flush never stalls replicated writes to other collections. +//! +//! Transaction A writes two collections on two vShards, so it commits +//! through the Calvin scheduler. Its flush on one vShard is held at a fail +//! point, so A stays staged: its rows are owned until the flush. A replicated +//! autocommit write to a third collection shares no row and no collection +//! with A. It must apply and be acknowledged while A's flush is held. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +mod crash_harness; + +use std::time::{Duration, Instant}; + +use crash_harness::CrashHarness; +use crash_harness::log_fields::{boot_section, count_lines}; +use crash_harness::vshards::names_on_distinct_vshards; + +/// How long the Calvin sequencer may take to elect its leader after a boot. +const CALVIN_READY_TIMEOUT: Duration = Duration::from_secs(30); + +/// How long the test waits for A's flush to be held. +const HOLD_DEADLINE: Duration = Duration::from_secs(30); + +/// How long the unrelated write may take. Well under the 30s request +/// deadline a write parked behind A would wait out. +const UNRELATED_WRITE_BUDGET: Duration = Duration::from_secs(10); + +/// The scheduler's log line for a flush held at the fail point. +const FLUSH_HELD: &str = "calvin: flush held at a fail point"; + +const LOG_DIRECTIVES: &str = "warn,nodedb::control::cluster::calvin::scheduler::driver::core=info"; + +#[tokio::test(flavor = "multi_thread")] +async fn a_replicated_write_to_an_unrelated_collection_applies_while_a_calvin_flush_is_held() { + let [held, peer, other] = names_on_distinct_vshards(["hold_held", "hold_peer", "hold_other"]); + let h = CrashHarness::new(); + let release = h.data_dir().join("release-held-flush"); + let mut h = h.with_env("RUST_LOG", LOG_DIRECTIVES).with_env( + "NODEDB_FAILPOINTS", + &format!( + "calvin::before_flush::{held}=wait_file({})", + release.display() + ), + ); + h.spawn(); + h.wait_ready(); + h.wait_for_calvin_ready(CALVIN_READY_TIMEOUT).await; + for name in [&held, &peer, &other] { + h.exec(&format!( + "CREATE COLLECTION {name} (k STRING PRIMARY KEY, v STRING) WITH (engine='kv')" + )) + .await; + } + + // Transaction A, on its own connection: its COMMIT waits for the held + // flush. + let conn_str = h.pgwire_conn_str(); + let statements = [ + "BEGIN".to_string(), + format!("INSERT INTO {held} (k, v) VALUES ('held', 'a')"), + format!("INSERT INTO {peer} (k, v) VALUES ('peer', 'p')"), + "COMMIT".to_string(), + ]; + let txn = tokio::spawn(async move { + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .map_err(|e| e.to_string())?; + tokio::spawn(connection); + for statement in &statements { + client + .simple_query(statement) + .await + .map_err(|e| format!("{statement}: {e}"))?; + } + Ok::<(), String>(()) + }); + + let deadline = Instant::now() + HOLD_DEADLINE; + while count_lines(&boot_section(&h.server_log(), 1), &[FLUSH_HELD]) == 0 { + assert!( + Instant::now() < deadline && !txn.is_finished(), + "transaction A's flush was never held" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + + tokio::time::timeout( + UNRELATED_WRITE_BUDGET, + h.exec(&format!("INSERT INTO {other} (k, v) VALUES ('o', 'x')")), + ) + .await + .unwrap_or_else(|_| { + panic!( + "a replicated write to {other} stalled behind transaction A's held flush, \ + though it shares no row and no collection with A" + ) + }); + assert_eq!( + h.query_col_idx(&format!("SELECT v FROM {other} WHERE k = 'o'"), 0) + .await, + vec!["x".to_string()] + ); + + std::fs::write(&release, b"release").expect("create the release file"); + txn.await + .expect("transaction A's task") + .unwrap_or_else(|e| panic!("transaction A: {e}")); +} diff --git a/nodedb/tests/crash_harness/mod.rs b/nodedb/tests/crash_harness/mod.rs index 7b67b8b36..45f1f2f3a 100644 --- a/nodedb/tests/crash_harness/mod.rs +++ b/nodedb/tests/crash_harness/mod.rs @@ -32,6 +32,7 @@ mod pgwire; pub use pgwire::{RetryableSchemaChange, Session}; #[path = "../support/mod.rs"] pub mod support; +pub mod vshards; /// Re-exported so a crash test can state its own filesystem precondition /// without pulling the support module in a second time. diff --git a/nodedb/tests/crash_harness/vshards.rs b/nodedb/tests/crash_harness/vshards.rs new file mode 100644 index 000000000..eb411d462 --- /dev/null +++ b/nodedb/tests/crash_harness/vshards.rs @@ -0,0 +1,53 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Collection names that home on distinct vShards. + +use nodedb_types::id::{DatabaseId, VShardId}; + +/// One collection name per prefix, `_`, each on a vShard no other +/// returned name homes on. A transaction that writes two of them spans two +/// vShards, so it commits through the Calvin scheduler. +pub fn names_on_distinct_vshards(prefixes: [&str; N]) -> [String; N] { + let mut taken: Vec = Vec::with_capacity(N); + prefixes.map(|prefix| { + let (name, vshard) = (0..512u32) + .map(|i| { + let name = format!("{prefix}_{i}"); + let vshard = + VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + (name, vshard) + }) + .find(|(_, vshard)| !taken.contains(vshard)) + .unwrap_or_else(|| panic!("no {prefix} name on a free vShard in 512 tries")); + taken.push(vshard); + name + }) +} + +/// Data Raft groups a single-node server runs. The server maps vShard `v` to +/// group `1 + v % SINGLE_NODE_DATA_GROUPS`. +pub const SINGLE_NODE_DATA_GROUPS: u32 = 4; + +/// The data Raft group a single-node server applies writes to `name` in. +pub fn single_node_data_group(name: &str) -> u32 { + let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, name).as_u32(); + 1 + vshard % SINGLE_NODE_DATA_GROUPS +} + +/// One collection name per prefix, `_`, each applied in a data +/// group of a single-node server that no other returned name is applied in. +pub fn names_in_distinct_data_groups(prefixes: [&str; N]) -> [String; N] { + let mut taken: Vec = Vec::with_capacity(N); + prefixes.map(|prefix| { + let (name, group) = (0..512u32) + .map(|i| { + let name = format!("{prefix}_{i}"); + let group = single_node_data_group(&name); + (name, group) + }) + .find(|(_, group)| !taken.contains(group)) + .unwrap_or_else(|| panic!("no {prefix} name in a free data group in 512 tries")); + taken.push(group); + name + }) +} diff --git a/nodedb/tests/crash_replay_stamp_calvin.rs b/nodedb/tests/crash_replay_stamp_calvin.rs index ad521502e..7dc7dd643 100644 --- a/nodedb/tests/crash_replay_stamp_calvin.rs +++ b/nodedb/tests/crash_replay_stamp_calvin.rs @@ -15,6 +15,10 @@ //! higher LSNs until a KV checkpoint's replay stamp names one of them above //! its prefix. That proves the checkpoint was written while A's record was //! appended and not applied. +//! A's COMMIT stays unacknowledged for the whole hold. It either still +//! waits, or it gave up at the server's completion deadline, which is no +//! acknowledgement. A COMMIT that returned success is the bug the hold +//! guards against. //! 3. The test releases the flush. The core installs A's record, and the //! process aborts before the flush response leaves. //! 4. After restart, replay must apply A's record. The Calvin scheduler reads @@ -30,8 +34,8 @@ mod crash_harness; use std::time::{Duration, Instant}; use crash_harness::log_fields::{boot_section, count_lines, log_field}; +use crash_harness::vshards::names_on_distinct_vshards; use crash_harness::{CrashHarness, diagnostics}; -use nodedb_types::id::{DatabaseId, VShardId}; /// How long the test waits for the held flush, and for a checkpoint whose /// stamp names a B: thirty checkpoint cycles at one per second. @@ -47,31 +51,17 @@ const CALVIN_READY_TIMEOUT: Duration = Duration::from_secs(30); /// The scheduler's log line for a flush held at the fail point. const FLUSH_HELD: &str = "calvin: flush held at a fail point"; +/// The error a COMMIT returns once its completion deadline passes with no +/// acknowledgement. The redo record stays appended and the flush stays held. +const COMMIT_DEADLINE: &str = "timed out waiting for Calvin transaction completion"; + const LOG_DIRECTIVES: &str = "warn,nodedb::data::executor::kv_checkpoint=info,\ nodedb::control::cluster::calvin::scheduler::driver::core=info"; -/// Three KV collection names on three different vShards: held, peer and -/// applied. -fn collections_on_distinct_vshards() -> [String; 3] { - let mut taken: Vec = Vec::with_capacity(3); - ["calvin_held", "calvin_peer", "calvin_applied"].map(|prefix| { - let (name, vshard) = (0..512u32) - .map(|i| { - let name = format!("{prefix}_{i}"); - let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); - (name, vshard) - }) - .find(|(_, vshard)| !taken.contains(vshard)) - .unwrap_or_else(|| panic!("no {prefix} name on a free vShard in 512 tries")); - taken.push(vshard); - name - }) -} - #[tokio::test(flavor = "multi_thread")] async fn a_calvin_redo_in_flight_at_a_checkpoint_survives_kill_9() { - let [held, peer, applied] = collections_on_distinct_vshards(); + let [held, peer, applied] = + names_on_distinct_vshards(["calvin_held", "calvin_peer", "calvin_applied"]); let h = CrashHarness::new(); let release = h.data_dir().join("release-held-flush"); let mut h = h @@ -109,10 +99,13 @@ async fn a_calvin_redo_in_flight_at_a_checkpoint_survives_kill_9() { .map_err(|e| e.to_string())?; tokio::spawn(connection); for statement in &statements { - client - .simple_query(statement) - .await - .map_err(|e| format!("{statement}: {e}"))?; + client.simple_query(statement).await.map_err(|e| { + let detail = e + .as_db_error() + .map(|db| format!("{}: {}", db.code().code(), db.message())) + .unwrap_or_else(|| e.to_string()); + format!("{statement}: {detail}") + })?; } Ok::<(), String>(()) }); @@ -156,10 +149,22 @@ async fn a_calvin_redo_in_flight_at_a_checkpoint_survives_kill_9() { diagnostics::log_tail_section(&h.server_log()) ); } - assert!( - !txn_task.is_finished(), - "transaction A finished before its flush was released" - ); + // A COMMIT that gave up at its deadline is unacknowledged, as the hold + // requires. Only a COMMIT that returned success, or failed for another + // reason, breaks the hold. + let txn_task = if txn_task.is_finished() { + let outcome = txn_task.await.expect("transaction A's task panicked"); + match outcome { + Err(error) if error.contains(COMMIT_DEADLINE) => None, + other => panic!( + "transaction A finished before its flush was released: {other:?}.{}\n{}", + h.keep_data_dir_note(), + diagnostics::log_tail_section(&h.server_log()) + ), + } + } else { + Some(txn_task) + }; let read_applied = format!("SELECT v FROM {applied}"); let mut live = h.query_col_idx(&read_applied, 0).await; live.sort(); @@ -176,7 +181,9 @@ async fn a_calvin_redo_in_flight_at_a_checkpoint_survives_kill_9() { diagnostics::log_tail_section(&h.server_log()) ); // A's client lost its connection with the process; its result says nothing. - let _ = txn_task.await; + if let Some(txn_task) = txn_task { + let _ = txn_task.await; + } h.clear_env("NODEDB_FAILPOINTS"); h.reopen(); diff --git a/nodedb/tests/inproc/cases/auth_service_account_scope.rs b/nodedb/tests/inproc/cases/auth_service_account_scope.rs index 374cbc605..b3e08bb78 100644 --- a/nodedb/tests/inproc/cases/auth_service_account_scope.rs +++ b/nodedb/tests/inproc/cases/auth_service_account_scope.rs @@ -116,30 +116,53 @@ async fn alter_service_account_set_databases_unknown_db_rejected() { ); } -/// set_service_account_databases replaces the accessible_databases list. +/// `ALTER SERVICE ACCOUNT ... SET DATABASES` replaces the account's database +/// scope through the replicated user entry: the new database is in scope and +/// the old one is not. #[tokio::test] async fn set_service_account_databases_replaces_list() { - use nodedb::types::TenantId; - let state = make_state(); - - state + let state = make_state_with_catalog(); + let su = superuser(); + ddl_ok(&state, &su, "CREATE DATABASE svc_scope_b").await; + let db_b = state .credentials - .create_service_account( - "svc_replace", - TenantId::new(1), - vec![nodedb::control::security::identity::Role::ReadWrite], - vec![DatabaseId::DEFAULT], - ) - .unwrap(); + .catalog() + .get_database_id_by_name("svc_scope_b") + .unwrap() + .expect("svc_scope_b exists"); + ddl_ok( + &state, + &su, + "CREATE SERVICE ACCOUNT svc_replace FOR DATABASE default", + ) + .await; + let before = state.credentials.get_user("svc_replace").unwrap(); + assert_eq!(before.accessible_databases, vec![DatabaseId::DEFAULT]); - let db2 = DatabaseId::new(2); - state - .credentials - .set_service_account_databases("svc_replace", vec![db2]) - .unwrap(); + ddl_ok( + &state, + &su, + "ALTER SERVICE ACCOUNT svc_replace SET DATABASES svc_scope_b", + ) + .await; let user = state.credentials.get_user("svc_replace").unwrap(); - assert_eq!(user.accessible_databases, vec![db2]); + assert_eq!(user.accessible_databases, vec![db_b]); + assert!( + !user.accessible_databases.contains(&DatabaseId::DEFAULT), + "the replaced database must leave the account's scope" + ); + let stored = state + .credentials + .catalog() + .get_user("svc_replace") + .unwrap() + .expect("the catalog holds the account"); + assert_eq!( + stored.accessible_databases, + vec![db_b.as_u64()], + "the catalog must hold the replaced scope" + ); } /// Service account + key inheritance: key picks up service account's databases. diff --git a/nodedb/tests/wire/cases/backup_support.rs b/nodedb/tests/wire/cases/backup_support.rs new file mode 100644 index 000000000..5db8db111 --- /dev/null +++ b/nodedb/tests/wire/cases/backup_support.rs @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Shared backup and restore steps for the wire backup tests. + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; +use nodedb_types::id::{DatabaseId, VShardId}; + +/// Take a backup of `tenant` over `client` and return the envelope bytes. +pub async fn drain_backup(client: &tokio_postgres::Client, tenant: u64) -> Result, String> { + let stream = client + .copy_out(&format!("COPY (BACKUP TENANT {tenant}) TO STDOUT")) + .await + .map_err(|e| format!("copy_out: {e:?}"))?; + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice(&chunk.map_err(|e| format!("copy_out chunk: {e:?}"))?); + } + Ok(bytes) +} + +/// Restore `envelope` into `tenant` over `client`. The error carries the +/// server's full error text. +pub async fn push_restore( + client: &tokio_postgres::Client, + tenant: u64, + envelope: Vec, +) -> Result<(), String> { + let sink = client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({tenant}) FROM STDIN")) + .await + .map_err(|e| format!("copy_in: {e:?}"))?; + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .map_err(|e| format!("send: {e:?}"))?; + sink.as_mut() + .finish() + .await + .map(|_| ()) + .map_err(|e| format!("{e:?}")) +} + +/// Two collection names, `_a_` and `_b_`, that home on +/// different vShards. A transaction that writes both commits through the +/// Calvin scheduler. +pub fn names_on_two_vshards(prefix: &str) -> (String, String) { + let first = format!("{prefix}_a"); + let first_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &first).as_u32(); + let second = (0..512u32) + .map(|i| format!("{prefix}_b_{i}")) + .find(|name| { + VShardId::from_collection_in_database(DatabaseId::DEFAULT, name).as_u32() + != first_vshard + }) + .unwrap_or_else(|| panic!("no {prefix}_b name on another vShard in 512 tries")); + (first, second) +} diff --git a/nodedb/tests/wire/cases/graph_analytics_authorization.rs b/nodedb/tests/wire/cases/graph_analytics_authorization.rs index b2be18a8a..a2a5e30cc 100644 --- a/nodedb/tests/wire/cases/graph_analytics_authorization.rs +++ b/nodedb/tests/wire/cases/graph_analytics_authorization.rs @@ -43,6 +43,12 @@ async fn seed(server: &TestServer, collection: &str, stranger: &str) { // A custom role confers nothing without an explicit grant, which is what // "no access to this collection" actually looks like — `CREATE USER` // defaults to ReadWrite and `monitor` still confers `Permission::Read`. + // The role is defined and granted nothing: a user may hold only a + // defined role. + server + .exec("CREATE ROLE IF NOT EXISTS analytics_nobody") + .await + .unwrap_or_else(|e| panic!("create role analytics_nobody: {e}")); server .exec(&format!( "CREATE USER {stranger} PASSWORD '{PASSWORD}' ROLE analytics_nobody" diff --git a/nodedb/tests/wire/cases/graph_match_authorization.rs b/nodedb/tests/wire/cases/graph_match_authorization.rs index 144fadbb6..628f16863 100644 --- a/nodedb/tests/wire/cases/graph_match_authorization.rs +++ b/nodedb/tests/wire/cases/graph_match_authorization.rs @@ -45,6 +45,12 @@ async fn seed(server: &TestServer, collection: &str, stranger: &str) { // `Permission::Read`, so neither is an unprivileged principal. A custom // role confers nothing without an explicit grant, which is what "no access // to this collection" actually looks like. + // The role is defined and granted nothing: a user may hold only a + // defined role. + server + .exec("CREATE ROLE IF NOT EXISTS match_nobody") + .await + .unwrap_or_else(|e| panic!("create role match_nobody: {e}")); server .exec(&format!( "CREATE USER {stranger} PASSWORD '{PASSWORD}' ROLE match_nobody" diff --git a/nodedb/tests/wire/cases/graph_traversal_authorization.rs b/nodedb/tests/wire/cases/graph_traversal_authorization.rs index 2a6e3a8e4..cb23d5227 100644 --- a/nodedb/tests/wire/cases/graph_traversal_authorization.rs +++ b/nodedb/tests/wire/cases/graph_traversal_authorization.rs @@ -42,6 +42,12 @@ async fn seed(server: &TestServer, collection: &str, stranger: &str) { // `Permission::Read` (`identity/permission.rs:80`), so neither is an // unprivileged principal. A custom role confers nothing without an explicit // grant, which is what "no access to this collection" actually looks like. + // The role is defined and granted nothing: a user may hold only a + // defined role. + server + .exec("CREATE ROLE IF NOT EXISTS graph_nobody") + .await + .unwrap_or_else(|e| panic!("create role graph_nobody: {e}")); server .exec(&format!( "CREATE USER {stranger} PASSWORD '{PASSWORD}' ROLE graph_nobody" diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 2303a3dbc..e6eecc88c 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 mod aggregate_cache_delete_invalidation; +mod backup_support; mod bitemporal_array_sql; mod bitemporal_asof_scan_parity; mod bitemporal_delete_index_tombstone; @@ -162,8 +163,14 @@ mod spatial_cp_dp_query; mod sql_aggregate_functions; mod sql_alter_after_drop; mod sql_arithmetic_overflow; +mod sql_backup_calvin_cut; +mod sql_backup_consistent_cut; mod sql_backup_restore_columnar; mod sql_backup_restore_columnar_restart; +mod sql_backup_restore_documents; +mod sql_backup_restore_durable_marks; +mod sql_backup_restore_local_marks; +mod sql_backup_restore_staleness; mod sql_backup_restore_timeseries; mod sql_backup_restore_vector_params; mod sql_backup_restore_vector_restart; diff --git a/nodedb/tests/wire/cases/query_function_authorization.rs b/nodedb/tests/wire/cases/query_function_authorization.rs index 719e6b785..eeea8ceb1 100644 --- a/nodedb/tests/wire/cases/query_function_authorization.rs +++ b/nodedb/tests/wire/cases/query_function_authorization.rs @@ -35,6 +35,12 @@ const SUPERUSER_ROLE: &str = "superuser"; /// confers nothing without an explicit grant — that is what "no access to this /// collection" actually looks like. async fn create_stranger(server: &TestServer, user: &str) { + // The role is defined and granted nothing: a user may hold only a + // defined role. + server + .exec("CREATE ROLE IF NOT EXISTS qfn_nobody") + .await + .unwrap_or_else(|e| panic!("create role qfn_nobody: {e}")); server .exec(&format!( "CREATE USER {user} PASSWORD '{PASSWORD}' ROLE qfn_nobody" @@ -45,6 +51,12 @@ async fn create_stranger(server: &TestServer, user: &str) { /// Create a principal that may read its own tenant's collections. async fn create_reader(server: &TestServer, user: &str) { + // The role is defined and granted nothing: a user may hold only a + // defined role. + server + .exec("CREATE ROLE IF NOT EXISTS qfn_nobody") + .await + .unwrap_or_else(|e| panic!("create role qfn_nobody: {e}")); server .exec(&format!( "CREATE USER {user} PASSWORD '{PASSWORD}' ROLE qfn_nobody" diff --git a/nodedb/tests/wire/cases/sorted_index_authorization.rs b/nodedb/tests/wire/cases/sorted_index_authorization.rs index c0c3949ac..636700b7a 100644 --- a/nodedb/tests/wire/cases/sorted_index_authorization.rs +++ b/nodedb/tests/wire/cases/sorted_index_authorization.rs @@ -22,6 +22,12 @@ const PASSWORD: &str = "sidx-secret-42"; /// Create a principal that holds no grant on anything: a custom role confers /// nothing without an explicit grant. async fn create_stranger(server: &TestServer, user: &str) { + // The role is defined and granted nothing: a user may hold only a + // defined role. + server + .exec("CREATE ROLE IF NOT EXISTS sidx_nobody") + .await + .unwrap_or_else(|e| panic!("create role sidx_nobody: {e}")); server .exec(&format!( "CREATE USER {user} PASSWORD '{PASSWORD}' ROLE sidx_nobody" diff --git a/nodedb/tests/wire/cases/sql_backup_calvin_cut.rs b/nodedb/tests/wire/cases/sql_backup_calvin_cut.rs new file mode 100644 index 000000000..2eac7356b --- /dev/null +++ b/nodedb/tests/wire/cases/sql_backup_calvin_cut.rs @@ -0,0 +1,162 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A backup's consistent cut covers Calvin transactions. +//! +//! A transaction that writes two collections on two vShards commits through +//! the Calvin scheduler. Its install records the transaction's commit HLC on +//! the tenant's write mark before the COMMIT is acknowledged, so a Calvin +//! commit after a backup refuses a restore of it. A Calvin transaction held +//! at its flush while a backup starts was sequenced before the backup's cut, +//! so the backup waits for its install and holds its rows. + +use super::backup_support::{drain_backup, names_on_two_vshards, push_restore}; +use crate::harness::TestServer; + +const TENANT: u64 = 1; + +async fn create_kv(server: &TestServer, name: &str) { + server + .exec(&format!( + "CREATE COLLECTION {name} (key STRING PRIMARY KEY, value STRING) WITH (engine='kv')" + )) + .await + .unwrap_or_else(|e| panic!("create {name}: {e}")); +} + +/// The statements of one Calvin transaction writing `key` into both +/// collections. +fn calvin_commit(first: &str, second: &str, key: &str) -> [String; 4] { + [ + "BEGIN".to_owned(), + format!("INSERT INTO {first} (key, value) VALUES ('{key}', 'x')"), + format!("INSERT INTO {second} (key, value) VALUES ('{key}', 'y')"), + "COMMIT".to_owned(), + ] +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_calvin_commit_after_the_backup_refuses_the_restore() { + let server = TestServer::start().await; + let (first, second) = names_on_two_vshards("calvin_mark"); + create_kv(&server, &first).await; + create_kv(&server, &second).await; + for sql in calvin_commit(&first, &second, "before") { + server + .exec(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + let backup = drain_backup(&server.client, TENANT) + .await + .expect("take the backup"); + + for sql in calvin_commit(&first, &second, "after") { + server + .exec(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + + let error = push_restore(&server.client, TENANT, backup) + .await + .expect_err("a restore older than a Calvin commit must be refused"); + assert!( + error.contains("restore refused"), + "expected the staleness refusal, got: {error}" + ); + assert!( + error.contains("calvin flush"), + "the refusal must name the Calvin commit as the newer write, got: {error}" + ); +} + +#[cfg(feature = "failpoints")] +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_calvin_commit_held_at_its_flush_is_in_the_backup_or_refuses_its_restore() { + use std::time::Duration; + + const PARKED_FOR: Duration = Duration::from_millis(1500); + + let (first, second) = names_on_two_vshards("calvin_cut"); + let gate_dir = tempfile::tempdir().expect("gate tempdir"); + let release = gate_dir.path().join("release-flush"); + let server = TestServer::start_with_failpoints(&format!( + "calvin::before_flush::{first}=wait_file({})", + release.display() + )) + .await; + create_kv(&server, &first).await; + create_kv(&server, &second).await; + + // Hold the Calvin transaction at its flush on `first`'s vShard. + let (committer, committer_handle) = server + .connect_as("nodedb", "nodedb") + .await + .expect("connect the committer"); + let statements = calvin_commit(&first, &second, "held"); + let commit = tokio::spawn(async move { + for sql in &statements { + committer + .simple_query(sql) + .await + .map_err(|e| format!("{sql}: {e}"))?; + } + Ok::<(), String>(()) + }); + tokio::time::sleep(PARKED_FOR).await; + assert!( + !commit.is_finished(), + "the COMMIT was not held at its flush" + ); + + // Back up while the transaction is sequenced but not installed. + let (backup_client, backup_handle) = server + .connect_as("nodedb", "nodedb") + .await + .expect("connect the backup client"); + let backup = tokio::spawn(async move { drain_backup(&backup_client, TENANT).await }); + tokio::time::sleep(PARKED_FOR).await; + assert!( + !backup.is_finished(), + "the backup snapshotted before a Calvin transaction sequenced ahead of its cut installed" + ); + + std::fs::write(&release, b"release").expect("release the flush"); + commit + .await + .expect("commit task") + .unwrap_or_else(|e| panic!("commit: {e}")); + let envelope = backup + .await + .expect("backup task") + .unwrap_or_else(|e| panic!("backup: {e}")); + + for name in [&first, &second] { + server + .exec(&format!("DROP COLLECTION {name} PURGE")) + .await + .unwrap_or_else(|e| panic!("purge {name}: {e}")); + } + match push_restore(&server.client, TENANT, envelope).await { + Ok(()) => { + for (name, value) in [(&first, "x"), (&second, "y")] { + let rows = server + .query_text(&format!("SELECT value FROM {name} WHERE key = 'held'")) + .await + .unwrap_or_else(|e| panic!("read {name}: {e}")); + assert_eq!( + rows, + vec![value.to_owned()], + "the restore succeeded, so the held Calvin commit must be in the backup" + ); + } + } + Err(error) => assert!( + error.contains("restore refused"), + "a restore may only fail by refusing the newer write, got: {error}" + ), + } + + committer_handle.abort(); + backup_handle.abort(); +} diff --git a/nodedb/tests/wire/cases/sql_backup_consistent_cut.rs b/nodedb/tests/wire/cases/sql_backup_consistent_cut.rs new file mode 100644 index 000000000..9c8bb43c1 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_backup_consistent_cut.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Compiled only with `--features failpoints`. +//! +//! A backup is a consistent cut: a write in flight when the backup starts is +//! either in the backup or refuses a restore of it. It is never silently +//! lost. +//! +//! The fail gate `funnel::before_dispatch::` parks a write after +//! its record is minted and before a core applies it. The test parks an +//! INSERT there, starts a backup, releases the INSERT, and restores the +//! backup over a purge of the collection. The restore either refuses the +//! newer write or brings the row back. + +#[cfg(feature = "failpoints")] +use std::time::Duration; + +#[cfg(feature = "failpoints")] +use super::backup_support::{drain_backup, push_restore}; +#[cfg(feature = "failpoints")] +use crate::harness::TestServer; + +#[cfg(feature = "failpoints")] +const TENANT: u64 = 1; + +#[cfg(feature = "failpoints")] +const COLLECTION: &str = "cut_docs"; + +/// How long the parked INSERT and the backup must stay unfinished. +#[cfg(feature = "failpoints")] +const PARKED_FOR: Duration = Duration::from_millis(1500); + +#[cfg(feature = "failpoints")] +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_write_in_flight_at_a_backup_is_in_the_backup_or_refuses_its_restore() { + let gate_dir = tempfile::tempdir().expect("gate tempdir"); + let release = gate_dir.path().join("release-insert"); + std::fs::write(&release, b"").expect("open the gate"); + let server = TestServer::start_with_failpoints(&format!( + "funnel::before_dispatch::{COLLECTION}=wait_file({})", + release.display() + )) + .await; + server + .exec(&format!( + "CREATE COLLECTION {COLLECTION} (key STRING PRIMARY KEY, value STRING) WITH (engine='kv')" + )) + .await + .expect("create the collection"); + + // Park the INSERT between its record and its apply. + std::fs::remove_file(&release).expect("close the gate"); + let (writer, writer_handle) = server + .connect_as("nodedb", "nodedb") + .await + .expect("connect the writer"); + let insert = tokio::spawn(async move { + writer + .simple_query(&format!( + "INSERT INTO {COLLECTION} (key, value) VALUES ('in_flight', 'x')" + )) + .await + .map(|_| ()) + .map_err(|e| e.to_string()) + }); + tokio::time::sleep(PARKED_FOR).await; + assert!( + !insert.is_finished(), + "the INSERT was not parked at the gate" + ); + + // Back up while the INSERT is in flight. + let (backup_client, backup_handle) = server + .connect_as("nodedb", "nodedb") + .await + .expect("connect the backup client"); + let backup = tokio::spawn(async move { drain_backup(&backup_client, TENANT).await }); + tokio::time::sleep(PARKED_FOR).await; + assert!( + !insert.is_finished(), + "the INSERT left the gate before its release" + ); + assert!( + !backup.is_finished(), + "the backup snapshotted while a write below its watermark had no outcome" + ); + + std::fs::write(&release, b"").expect("release the INSERT"); + insert + .await + .expect("insert task") + .unwrap_or_else(|e| panic!("insert: {e}")); + let envelope = backup + .await + .expect("backup task") + .unwrap_or_else(|e| panic!("backup: {e}")); + + server + .exec(&format!("DROP COLLECTION {COLLECTION} PURGE")) + .await + .expect("purge the collection"); + match push_restore(&server.client, TENANT, envelope).await { + Ok(()) => { + let rows = server + .query_text(&format!( + "SELECT value FROM {COLLECTION} WHERE key = 'in_flight'" + )) + .await + .expect("read the restored row"); + assert_eq!( + rows, + vec!["x".to_string()], + "the restore succeeded, so the in-flight write must be in the backup" + ); + } + Err(error) => { + assert!( + error.contains("restore refused"), + "a restore may only fail by refusing the newer write, got: {error}" + ); + } + } + + writer_handle.abort(); + backup_handle.abort(); +} diff --git a/nodedb/tests/wire/cases/sql_backup_restore_documents.rs b/nodedb/tests/wire/cases/sql_backup_restore_documents.rs new file mode 100644 index 000000000..502cd979b --- /dev/null +++ b/nodedb/tests/wire/cases/sql_backup_restore_documents.rs @@ -0,0 +1,157 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE re-issues document rows, their index entries, and graph edges as +//! durable writes. +//! +//! The backup holds a schemaless collection with a secondary index and a +//! unique index, a strict collection, a strict `bitemporal=true` collection +//! with two versions of one row, and two edges. A fresh server restores it, +//! then restarts. Before and after the restart every row reads back, the +//! index lookups answer, the unique index refuses a duplicate, the bitemporal +//! row keeps both versions, and the edges traverse. +//! +//! Both single-node modes run it: the default server, whose data groups are +//! Raft groups of one, and a standalone server with no Raft groups. + +use super::backup_support::{drain_backup, push_restore}; +use crate::harness::TestServer; + +const TENANT: u64 = 1; + +const SETUP: &[&str] = &[ + "CREATE COLLECTION rd_people (id STRING PRIMARY KEY, city STRING, email STRING) \ + WITH (engine='document_schemaless')", + "CREATE INDEX ON rd_people (city)", + "CREATE UNIQUE INDEX rd_people_email ON rd_people (email)", + "INSERT INTO rd_people (id, city, email) VALUES ('alice', 'paris', 'a@x')", + "INSERT INTO rd_people (id, city, email) VALUES ('bob', 'rome', 'b@x')", + "INSERT INTO rd_people (id, city, email) VALUES ('carol', 'paris', 'c@x')", + "GRAPH INSERT EDGE IN 'rd_people' FROM 'alice' TO 'bob' TYPE 'knows'", + "GRAPH INSERT EDGE IN 'rd_people' FROM 'bob' TO 'carol' TYPE 'knows'", + "CREATE COLLECTION rd_accounts (id STRING PRIMARY KEY, owner STRING, balance INT) \ + WITH (engine='document_strict')", + "INSERT INTO rd_accounts (id, owner, balance) VALUES ('acc1', 'alice', 10)", + "INSERT INTO rd_accounts (id, owner, balance) VALUES ('acc2', 'bob', 20)", + "CREATE COLLECTION rd_ledger (id STRING PRIMARY KEY, value STRING) \ + WITH (engine='document_strict', bitemporal=true)", + "INSERT INTO rd_ledger (id, value) VALUES ('e1', 'draft')", + "UPDATE rd_ledger SET value = 'final' WHERE id = 'e1'", +]; + +async fn column(server: &TestServer, sql: &str) -> Vec { + let mut values: Vec = server + .query_rows(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")) + .into_iter() + .filter_map(|row| row.into_iter().next()) + .collect(); + values.sort(); + values +} + +async fn neighbors(server: &TestServer, node: &str) -> String { + server + .query_text(&format!( + "GRAPH NEIGHBORS IN 'rd_people' OF '{node}' LABEL 'knows' DIRECTION out" + )) + .await + .unwrap_or_else(|e| panic!("neighbors of {node}: {e}")) + .join("") +} + +/// Every restored row, index entry, version and edge reads back on `server`. +async fn assert_restored(server: &TestServer, stage: &str) { + assert_eq!( + column(server, "SELECT id FROM rd_people WHERE city = 'paris'").await, + vec!["alice", "carol"], + "{stage}: the secondary index lookup" + ); + assert_eq!( + column(server, "SELECT id FROM rd_people WHERE email = 'b@x'").await, + vec!["bob"], + "{stage}: the unique index lookup" + ); + let duplicate = server + .exec("INSERT INTO rd_people (id, city, email) VALUES ('dave', 'oslo', 'a@x')") + .await; + assert!( + duplicate.is_err(), + "{stage}: the unique index must refuse a restored row's email" + ); + assert_eq!( + column(server, "SELECT balance FROM rd_accounts WHERE id = 'acc2'").await, + vec!["20"], + "{stage}: the strict row by primary key" + ); + assert_eq!( + column(server, "SELECT owner FROM rd_accounts").await, + vec!["alice", "bob"], + "{stage}: every strict row" + ); + assert_eq!( + column(server, "SELECT value FROM rd_ledger WHERE id = 'e1'").await, + vec!["final"], + "{stage}: the bitemporal row's current version" + ); + assert_eq!( + column(server, "SELECT value FROM rd_ledger AS OF SYSTEM TIME NULL").await, + vec!["draft", "final"], + "{stage}: the bitemporal row keeps both versions" + ); + let from_alice = neighbors(server, "alice").await; + assert!( + from_alice.contains("bob"), + "{stage}: the edge alice -> bob, got {from_alice}" + ); + let from_bob = neighbors(server, "bob").await; + assert!( + from_bob.contains("carol"), + "{stage}: the edge bob -> carol, got {from_bob}" + ); +} + +async fn source_backup() -> Vec { + let source = TestServer::start().await; + for sql in SETUP { + source + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + drain_backup(&source.client, TENANT) + .await + .expect("take the backup") +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn restored_documents_indexes_and_edges_survive_a_restart() { + let backup = source_backup().await; + + let target = TestServer::start().await; + push_restore(&target.client, TENANT, backup) + .await + .unwrap_or_else(|e| panic!("restore: {e}")); + assert_restored(&target, "after the restore").await; + + let (target, dir) = target.take_dir(); + target.graceful_shutdown().await; + let (target, _dir) = TestServer::open_on_path(dir).await; + assert_restored(&target, "after a restart").await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn restored_documents_indexes_and_edges_survive_a_standalone_restart() { + let backup = source_backup().await; + + let target = TestServer::start_standalone().await; + push_restore(&target.client, TENANT, backup) + .await + .unwrap_or_else(|e| panic!("restore: {e}")); + assert_restored(&target, "after the restore").await; + + let (target, dir) = target.take_dir(); + target.graceful_shutdown().await; + let (target, _dir) = TestServer::open_on_path_standalone(dir).await; + assert_restored(&target, "after a restart").await; +} diff --git a/nodedb/tests/wire/cases/sql_backup_restore_durable_marks.rs b/nodedb/tests/wire/cases/sql_backup_restore_durable_marks.rs new file mode 100644 index 000000000..1f5f8847d --- /dev/null +++ b/nodedb/tests/wire/cases/sql_backup_restore_durable_marks.rs @@ -0,0 +1,109 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's staleness guard reads durable, replicated write marks. +//! +//! A server restart clears every in-memory mark. A write committed after a +//! backup still refuses a restore of it after the restart, because each data +//! group persists the marks of the writes it applied. A Calvin commit +//! persists its mark before its COMMIT is acknowledged. + +use super::backup_support::{drain_backup, names_on_two_vshards, push_restore}; +use crate::harness::TestServer; + +const TENANT: u64 = 1; + +async fn exec_all(server: &TestServer, statements: &[String]) { + for sql in statements { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } +} + +/// Shut `server` down and reopen it on the same data directory. The +/// directory must outlive the reopened server. +async fn restart(server: TestServer) -> (TestServer, impl Sized) { + let (server, dir) = server.take_dir(); + server.graceful_shutdown().await; + TestServer::open_on_path(dir).await +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_write_after_the_backup_refuses_the_restore_after_a_restart() { + let server = TestServer::start().await; + exec_all( + &server, + &[ + "CREATE COLLECTION durable_mark_kv (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')" + .to_owned(), + "INSERT INTO durable_mark_kv (key, value) VALUES ('a', '1')".to_owned(), + ], + ) + .await; + let backup = drain_backup(&server.client, TENANT) + .await + .expect("take the backup"); + exec_all( + &server, + &["INSERT INTO durable_mark_kv (key, value) VALUES ('b', '2')".to_owned()], + ) + .await; + + let (server, _dir) = restart(server).await; + + let error = push_restore(&server.client, TENANT, backup) + .await + .expect_err("a write after the backup must refuse the restore after a restart"); + assert!( + error.contains("restore refused"), + "expected the staleness refusal, got: {error}" + ); + assert!( + error.contains("durable_mark_kv"), + "the refusal must name the collection of the newer write, got: {error}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_calvin_commit_after_the_backup_refuses_the_restore_after_a_restart() { + let server = TestServer::start().await; + let (first, second) = names_on_two_vshards("durable_calvin"); + for name in [&first, &second] { + server + .exec(&format!( + "CREATE COLLECTION {name} (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')" + )) + .await + .unwrap_or_else(|e| panic!("create {name}: {e}")); + } + let backup = drain_backup(&server.client, TENANT) + .await + .expect("take the backup"); + exec_all( + &server, + &[ + "BEGIN".to_owned(), + format!("INSERT INTO {first} (key, value) VALUES ('k', 'x')"), + format!("INSERT INTO {second} (key, value) VALUES ('k', 'y')"), + "COMMIT".to_owned(), + ], + ) + .await; + + let (server, _dir) = restart(server).await; + + let error = push_restore(&server.client, TENANT, backup) + .await + .expect_err("a Calvin commit after the backup must refuse the restore after a restart"); + assert!( + error.contains("restore refused"), + "expected the staleness refusal, got: {error}" + ); + assert!( + error.contains("calvin flush"), + "the refusal must name the Calvin commit as the newer write, got: {error}" + ); +} diff --git a/nodedb/tests/wire/cases/sql_backup_restore_local_marks.rs b/nodedb/tests/wire/cases/sql_backup_restore_local_marks.rs new file mode 100644 index 000000000..bd1508f28 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_backup_restore_local_marks.rs @@ -0,0 +1,124 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's staleness guard on a server with no Raft groups. +//! +//! Such a server has no data group to carry its write marks, so it keeps +//! them in its catalog under a local pseudo-group, durable before each +//! write's WAL record. A write after a backup still refuses a restore of it +//! after a restart. + +use super::backup_support::{drain_backup, push_restore}; +use crate::harness::TestServer; + +const TENANT: u64 = 1; + +/// Shut `server` down and reopen it, still standalone, on the same data +/// directory. The directory must outlive the reopened server. +async fn restart(server: TestServer) -> (TestServer, impl Sized) { + let (server, dir) = server.take_dir(); + server.graceful_shutdown().await; + TestServer::open_on_path_standalone(dir).await +} + +async fn exec_all(server: &TestServer, statements: &[&str]) { + for sql in statements { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } +} + +async fn assert_refused(server: &TestServer, backup: Vec, collection: &str) { + let error = push_restore(&server.client, TENANT, backup) + .await + .expect_err("a write after the backup must refuse the restore after a restart"); + assert!( + error.contains("restore refused"), + "expected the staleness refusal, got: {error}" + ); + assert!( + error.contains("local write"), + "the refusal must come from the durable local mark, got: {error}" + ); + assert!( + error.contains(collection), + "the refusal must name the collection of the newer write, got: {error}" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_standalone_write_after_the_backup_refuses_the_restore_after_a_restart() { + let server = TestServer::start_standalone().await; + exec_all( + &server, + &[ + "CREATE COLLECTION local_mark_docs (id STRING PRIMARY KEY, value STRING) \ + WITH (engine='document_strict')", + "INSERT INTO local_mark_docs (id, value) VALUES ('a', '1')", + ], + ) + .await; + let backup = drain_backup(&server.client, TENANT) + .await + .expect("take the backup"); + exec_all( + &server, + &["INSERT INTO local_mark_docs (id, value) VALUES ('b', '2')"], + ) + .await; + + let (server, _dir) = restart(server).await; + assert_refused(&server, backup, "local_mark_docs").await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_standalone_commit_after_the_backup_refuses_the_restore_after_a_restart() { + let server = TestServer::start_standalone().await; + exec_all( + &server, + &[ + "CREATE COLLECTION local_mark_kv (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')", + ], + ) + .await; + let backup = drain_backup(&server.client, TENANT) + .await + .expect("take the backup"); + exec_all( + &server, + &[ + "BEGIN", + "INSERT INTO local_mark_kv (key, value) VALUES ('k1', 'x')", + "INSERT INTO local_mark_kv (key, value) VALUES ('k2', 'y')", + "COMMIT", + ], + ) + .await; + + let (server, _dir) = restart(server).await; + assert_refused(&server, backup, "local_mark_kv").await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_standalone_backup_after_the_last_write_restores_after_a_restart() { + let server = TestServer::start_standalone().await; + exec_all( + &server, + &[ + "CREATE COLLECTION local_mark_fresh (id STRING PRIMARY KEY, value STRING) \ + WITH (engine='document_strict')", + "INSERT INTO local_mark_fresh (id, value) VALUES ('a', '1')", + ], + ) + .await; + let backup = drain_backup(&server.client, TENANT) + .await + .expect("take the backup"); + + let (server, _dir) = restart(server).await; + push_restore(&server.client, TENANT, backup) + .await + .unwrap_or_else(|e| panic!("a backup newer than every write must restore: {e}")); +} diff --git a/nodedb/tests/wire/cases/sql_backup_restore_staleness.rs b/nodedb/tests/wire/cases/sql_backup_restore_staleness.rs new file mode 100644 index 000000000..031a48c77 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_backup_restore_staleness.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RESTORE's staleness guard. +//! +//! A restore is refused when the tenant took a user data write after the +//! envelope's watermark: restoring would silently overwrite that write. Reads +//! and system bookkeeping after the backup are not writes, and never refuse +//! the restore. + +use crate::harness::TestServer; + +const TENANT: u64 = 1; + +async fn drain_backup(server: &TestServer) -> Vec { + super::backup_support::drain_backup(&server.client, TENANT) + .await + .expect("BACKUP TENANT") +} + +async fn push_restore(server: &TestServer, bytes: Vec) -> Result<(), String> { + super::backup_support::push_restore(&server.client, TENANT, bytes).await +} + +async fn create_with_row(server: &TestServer, name: &str) { + for sql in [ + format!( + "CREATE COLLECTION {name} (id TEXT PRIMARY KEY, v INT) WITH (engine='document_strict')" + ), + format!("INSERT INTO {name} (id, v) VALUES ('a', 1)"), + ] { + server + .exec(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } +} + +#[tokio::test] +async fn a_user_write_after_the_backup_refuses_the_restore() { + let server = TestServer::start().await; + create_with_row(&server, "stale_guard_docs").await; + let backup = drain_backup(&server).await; + + server + .exec("INSERT INTO stale_guard_docs (id, v) VALUES ('b', 2)") + .await + .expect("a newer user write"); + + let error = push_restore(&server, backup) + .await + .expect_err("a restore older than a user write must be refused"); + assert!( + error.contains("restore refused"), + "expected the staleness refusal, got: {error}" + ); + assert!( + error.contains("stale_guard_docs"), + "the refusal must name the collection of the newer write, got: {error}" + ); +} + +#[tokio::test] +async fn reads_and_ddl_after_the_backup_do_not_refuse_the_restore() { + let server = TestServer::start().await; + create_with_row(&server, "stale_guard_reads").await; + let backup = drain_backup(&server).await; + + for _ in 0..3 { + server + .query_text("SELECT id FROM stale_guard_reads") + .await + .expect("a read after the backup"); + } + server + .exec("DROP COLLECTION stale_guard_reads PURGE") + .await + .expect("purge before the restore"); + + push_restore(&server, backup) + .await + .expect("reads and system bookkeeping after the backup must not refuse the restore"); + let rows = server + .query_text("SELECT id FROM stale_guard_reads") + .await + .expect("read the restored collection"); + assert_eq!(rows, vec!["a".to_string()]); +} diff --git a/nodedb/tests/wire/cases/version_history_authorization.rs b/nodedb/tests/wire/cases/version_history_authorization.rs index 3bc158ea7..29fd0ffba 100644 --- a/nodedb/tests/wire/cases/version_history_authorization.rs +++ b/nodedb/tests/wire/cases/version_history_authorization.rs @@ -35,6 +35,12 @@ async fn seed(server: &TestServer, collection: &str, stranger: &str) { )) .await .unwrap_or_else(|e| panic!("seed doc-1: {e}")); + // The role is defined and granted nothing: a user may hold only a + // defined role. + server + .exec("CREATE ROLE IF NOT EXISTS version_nobody") + .await + .unwrap_or_else(|e| panic!("create role version_nobody: {e}")); server .exec(&format!( "CREATE USER {stranger} PASSWORD '{PASSWORD}' ROLE version_nobody" diff --git a/nodedb/tests/wire/harness/lifecycle.rs b/nodedb/tests/wire/harness/lifecycle.rs index 0928801e6..aad6330d7 100644 --- a/nodedb/tests/wire/harness/lifecycle.rs +++ b/nodedb/tests/wire/harness/lifecycle.rs @@ -141,6 +141,20 @@ impl TestServer { (server, dir) } + /// [`Self::open_on_path`] for a server booted with + /// [`Self::start_standalone`]: no cluster topology, so no Raft groups. + pub async fn open_on_path_standalone(dir: TestDataDir) -> (Self, TestDataDir) { + let spawned = process::spawn( + dir.path(), + AuthMode::Trust, + TuningOverrides::standalone(), + 1, + ); + let placeholder = tempfile::tempdir().expect("placeholder tempdir"); + let server = Self::connect_and_build(spawned, placeholder, AuthMode::Trust).await; + (server, dir) + } + /// Open a server backed by an existing data directory with a custom /// `columnar_flush_threshold`. Pass the same value the original server /// used to keep flush behaviour consistent across the restart boundary. From aad2507932e7376e0f5e54fee1a654f6f4a7f609 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 11:09:13 +0800 Subject: [PATCH 36/64] feat(raft): bound queued proposals by their caller deadline A proposal admitted into the per-shard sequencer's queue used to wait for its turn with no time limit, so a slow or stuck holder could stall every proposal behind it past its caller's own deadline. Add run_until, which reserves a queue place like run but times the lock wait out at a caller-supplied deadline: a waiter still queued at its deadline leaves without running and fails with DeadlineExceeded, and never reaches the raw proposer. Waiters behind it keep their order and still run. wrap_async_raft_proposer now calls run_until with the proposal's deadline instead of run. Factor the permit acquisition shared by run and run_until into a reserve helper, and cover the new deadline path with a test that expires one queued proposal mid-queue and checks the two behind it still propose in order. --- nodedb/src/control/vshard_admission.rs | 160 +++++++++++++++++++++++-- 1 file changed, 153 insertions(+), 7 deletions(-) diff --git a/nodedb/src/control/vshard_admission.rs b/nodedb/src/control/vshard_admission.rs index 1001107f0..b309e8dc3 100644 --- a/nodedb/src/control/vshard_admission.rs +++ b/nodedb/src/control/vshard_admission.rs @@ -65,15 +65,56 @@ impl VShardAdmissionSequencer { Fut: Future>, { let slot = self.slot(vshard_id)?; - let _queued = Arc::clone(&slot.capacity) - .try_acquire_owned() - .map_err(|_| crate::Error::VShardAdmissionCapacityExceeded { - vshard_id, - capacity: self.capacity, - })?; + let _queued = self.reserve(slot, vshard_id)?; let _active = slot.active.lock().await; operation().await } + + /// Run one admission operation like [`Self::run`], but wait in the queue + /// only until `deadline`. + /// + /// A queued operation still waiting at `deadline` leaves the queue and + /// returns [`crate::Error::DeadlineExceeded`]. Its factory is never + /// called. Waiters behind it keep their order. + /// + /// `timeout_at` polls the lock before the timer. A lock granted in the + /// same poll the deadline fires wins, and the operation runs. A lock future + /// dropped on timeout hands any turn it was granted to the next waiter. So + /// each operation either runs once or leaves without running. + pub async fn run_until( + &self, + vshard_id: VShardId, + deadline: tokio::time::Instant, + operation: F, + ) -> crate::Result + where + F: FnOnce() -> Fut, + Fut: Future>, + { + let slot = self.slot(vshard_id)?; + let _queued = self.reserve(slot, vshard_id)?; + let _active = tokio::time::timeout_at(deadline, slot.active.lock()) + .await + .map_err(|_| crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + })?; + operation().await + } + + /// Take one of the vShard's queue places, or fail at once when all are + /// taken. + fn reserve( + &self, + slot: &VShardAdmissionSlot, + vshard_id: VShardId, + ) -> crate::Result { + Arc::clone(&slot.capacity).try_acquire_owned().map_err(|_| { + crate::Error::VShardAdmissionCapacityExceeded { + vshard_id, + capacity: self.capacity, + } + }) + } } impl Default for VShardAdmissionSequencer { @@ -120,8 +161,10 @@ fn wrap_async_raft_proposer( }); } let vshard_id = VShardId::new(vshard_id); + // The queue wait ends at the caller's deadline. A proposal that + // leaves the queue then never reaches `raw`. sequencer - .run(vshard_id, move || async move { + .run_until(vshard_id, deadline, move || async move { raw(vshard_id.as_u32(), idempotency_key, data, deadline).await }) .await @@ -399,4 +442,107 @@ mod tests { ); assert_eq!(maximum.load(Ordering::SeqCst), 1); } + + /// Records the key of every proposal that reaches the raw proposer. + fn recording_proposer() -> (Arc, Arc>>) { + let proposed = Arc::new(std::sync::Mutex::new(Vec::new())); + let raw: Arc = { + let proposed = Arc::clone(&proposed); + Arc::new(move |_vshard, key, data, _deadline| { + let proposed = Arc::clone(&proposed); + Box::pin(async move { + proposed.lock().expect("proposed log").push(key); + Ok((data, Lsn::new(key))) + }) + }) + }; + (raw, proposed) + } + + /// Wait until `taken` places of `vshard`'s queue are held. + async fn until_queued(sequencer: &VShardAdmissionSequencer, vshard: usize, taken: usize) { + while sequencer.capacity - sequencer.slots[vshard].capacity.available_permits() != taken { + tokio::task::yield_now().await; + } + } + + #[tokio::test(start_paused = true)] + async fn a_queued_proposal_past_its_deadline_leaves_the_queue_and_never_proposes() { + let sequencer = Arc::new(VShardAdmissionSequencer::with_capacity(8)); + let (raw, proposed) = recording_proposer(); + let wrapped = wrap_async_raft_proposer(Arc::clone(&sequencer), raw); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let holder = { + let sequencer = Arc::clone(&sequencer); + let entered = Arc::clone(&entered); + let release = Arc::clone(&release); + tokio::spawn(async move { + sequencer + .run(shard(6), move || async move { + entered.notify_one(); + release.notified().await; + Ok(()) + }) + .await + }) + }; + entered.notified().await; + + let start = tokio::time::Instant::now(); + let expiring = tokio::spawn(wrapped( + 6, + 21, + vec![21], + start + std::time::Duration::from_secs(1), + )); + until_queued(&sequencer, 6, 2).await; + let second = tokio::spawn(wrapped( + 6, + 22, + vec![22], + start + std::time::Duration::from_secs(60), + )); + until_queued(&sequencer, 6, 3).await; + let third = tokio::spawn(wrapped( + 6, + 23, + vec![23], + start + std::time::Duration::from_secs(60), + )); + until_queued(&sequencer, 6, 4).await; + + let expired = expiring.await.expect("expiring joins"); + assert!( + matches!(expired, Err(crate::Error::DeadlineExceeded { .. })), + "a proposal still queued at its deadline fails with the deadline error, got {expired:?}" + ); + assert!( + proposed.lock().expect("proposed log").is_empty(), + "a proposal that left the queue never reaches the raw proposer" + ); + until_queued(&sequencer, 6, 3).await; + + release.notify_one(); + holder + .await + .expect("holder joins") + .expect("holder succeeds"); + assert_eq!( + second + .await + .expect("second joins") + .expect("second proposes"), + (vec![22], Lsn::new(22)) + ); + assert_eq!( + third.await.expect("third joins").expect("third proposes"), + (vec![23], Lsn::new(23)) + ); + assert_eq!( + *proposed.lock().expect("proposed log"), + vec![22, 23], + "the waiters behind the expired proposal propose in their queue order" + ); + } } From 40261d008fbf01f1116baef10589db14637a9dd5 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 12:35:17 +0800 Subject: [PATCH 37/64] feat(events): tag restored rows so triggers do not fire twice A RESTORE re-issues rows whose AFTER triggers already fired when they were first written. Every write event, replicated Raft entry, and transaction redo now carries EventSource::Restore end to end, and the ASYNC/DEFERRED trigger dispatcher, the batched-trigger collector, DML audit, and CRDT delta packaging all treat it as a non-firing source alongside Trigger and RaftFollower. CdcEvent gains a source field (JSON and MessagePack) so downstream consumers can tell a restored row from a client write. Trigger dispatch now decodes each event's row image through a fallible path instead of silently treating an undecodable row as absent, and KV write events carry the same {key, value} row shape a KV read returns so trigger bindings and WAL replay see a decodable row. KV batch puts and multi-key deletes now emit one write event per row instead of none. --- nodedb-test-support/src/lib.rs | 1 + .../control/backup/restore/crdt_reissue.rs | 5 +- nodedb/src/control/backup/restore/durable.rs | 7 +- .../backup/restore/redo_reissue/commit.rs | 27 ++- nodedb/src/control/crdt_admission.rs | 5 +- .../distributed_applier/apply_loop/start.rs | 17 +- .../apply_loop/write_dispatch.rs | 43 ++--- nodedb/src/control/event_trigger.rs | 12 +- .../shared/ddl/sync_dispatch/dispatch.rs | 5 +- .../shared/ddl/sync_dispatch/system_task.rs | 42 +++++ .../server/sync/raft_dispatch/response.rs | 1 + .../server/sync/raft_dispatch/write.rs | 1 + nodedb/src/control/trigger/batch/collector.rs | 13 +- .../encode/transaction_redo.rs | 1 + .../control/wal_replication/legacy_entry.rs | 3 + .../wal_replication/types/replicated_entry.rs | 29 +++- .../types/transaction_redo_wire.rs | 8 +- .../src/data/executor/core_loop/deferred.rs | 72 +++++++- .../src/data/executor/core_loop/event_emit.rs | 59 +++++++ .../src/data/executor/handlers/kv/atomic.rs | 103 ++++++++++-- nodedb/src/data/executor/handlers/kv/batch.rs | 27 +++ .../data/executor/handlers/kv/crud/delete.rs | 69 ++++---- .../executor/handlers/kv/crud/write_basic.rs | 20 +-- .../executor/handlers/kv/crud/write_upsert.rs | 10 +- nodedb/src/data/executor/handlers/kv/field.rs | 13 ++ .../executor/handlers/kv/predicate/apply.rs | 5 +- .../executor/handlers/kv/resolve/apply.rs | 10 +- .../src/data/executor/handlers/kv/transfer.rs | 20 +-- .../transaction/redo_apply/document.rs | 4 + .../handlers/transaction/redo_apply/events.rs | 45 +++-- .../handlers/transaction/redo_apply/state.rs | 3 + nodedb/src/event/audit_dml/consumer.rs | 8 +- nodedb/src/event/cdc/buffer.rs | 1 + nodedb/src/event/cdc/compaction.rs | 1 + nodedb/src/event/cdc/consume.rs | 3 + nodedb/src/event/cdc/event.rs | 55 +++++- nodedb/src/event/cdc/redaction.rs | 1 + nodedb/src/event/cdc/router.rs | 2 + nodedb/src/event/crdt_sync/packager.rs | 27 ++- nodedb/src/event/streaming_mv/processor.rs | 2 + nodedb/src/event/topic/types.rs | 1 + nodedb/src/event/trigger/dispatcher/single.rs | 90 ++++++++-- nodedb/src/event/types.rs | 66 +++++++- nodedb/src/event/wal_replay.rs | 5 + nodedb/src/event/wal_replay_parse.rs | 6 +- nodedb/tests/inproc/cases/cdc_arc_fanout.rs | 2 + nodedb/tests/inproc/cases/event_cdc.rs | 2 + nodedb/tests/inproc/cases/event_topics.rs | 2 + .../inproc/cases/system_task_call_sites.rs | 11 +- nodedb/tests/wire/cases/mod.rs | 1 + .../wire/cases/sql_backup_restore_triggers.rs | 159 ++++++++++++++++++ 51 files changed, 928 insertions(+), 197 deletions(-) create mode 100644 nodedb/tests/wire/cases/sql_backup_restore_triggers.rs diff --git a/nodedb-test-support/src/lib.rs b/nodedb-test-support/src/lib.rs index 240cbfbec..6a79b6b09 100644 --- a/nodedb-test-support/src/lib.rs +++ b/nodedb-test-support/src/lib.rs @@ -54,5 +54,6 @@ pub fn make_cdc_event( field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: nodedb::event::EventSource::User, } } diff --git a/nodedb/src/control/backup/restore/crdt_reissue.rs b/nodedb/src/control/backup/restore/crdt_reissue.rs index 3dbbf83e6..514aad4ef 100644 --- a/nodedb/src/control/backup/restore/crdt_reissue.rs +++ b/nodedb/src/control/backup/restore/crdt_reissue.rs @@ -53,7 +53,8 @@ async fn reissue_crdt_collection( )? .ok_or_else(|| Error::Internal { detail: "restore reissue: crdt import did not map to a replicated write".into(), - })?; + })? + .with_event_source(EventSource::Restore); crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; return Ok(()); } @@ -74,7 +75,7 @@ async fn reissue_crdt_collection( vshard_id: vshard, plan, trace_id: crate::types::TraceId::ZERO, - event_source: EventSource::CrdtSync, + event_source: EventSource::Restore, txn_id: None, }, ), diff --git a/nodedb/src/control/backup/restore/durable.rs b/nodedb/src/control/backup/restore/durable.rs index 54a5963f9..f110a6b81 100644 --- a/nodedb/src/control/backup/restore/durable.rs +++ b/nodedb/src/control/backup/restore/durable.rs @@ -26,6 +26,8 @@ const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); /// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. /// - Single-node: append the redo under an outcome-floor window, then /// `sync_dispatch::dispatch_system`, which closes the window. +/// +/// Both branches give the write's events [`crate::event::EventSource::Restore`]. pub async fn reissue_plan_durably( state: &SharedState, tenant_id: TenantId, @@ -47,7 +49,10 @@ pub async fn reissue_plan_durably( "restore reissue: the plan restored into '{collection}' did not map to a \ replicated write" ), - })?; + })? + // Every replica applies the write as restored: AFTER triggers do not + // fire again for it. + .with_event_source(crate::event::EventSource::Restore); let (_, write_version) = crate::control::wal_replication::propose_replicated_entry(state, proposer, entry) .await?; diff --git a/nodedb/src/control/backup/restore/redo_reissue/commit.rs b/nodedb/src/control/backup/restore/redo_reissue/commit.rs index 0e7f9aab3..dabb04be9 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/commit.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/commit.rs @@ -79,7 +79,9 @@ fn batch_payload(collection: &str, batch: Vec) -> TransactionRedoPayloa // The backup holds every target row with its total already folded in. sum_targets: Vec::new(), identities, - event_source: EventSource::User, + // Every replica applies the rows as restored: AFTER triggers fired + // when the rows were first written, and do not fire again. + event_source: EventSource::Restore, origin: RedoOrigin::Restore, } } @@ -203,4 +205,27 @@ mod tests { assert!(payload.sum_targets.is_empty()); assert_eq!(payload.origin, RedoOrigin::Restore); } + + #[test] + fn a_restored_record_carries_the_restore_source_to_every_replica() { + let payload = batch_payload("c", vec![unit(1, 1, "a")]); + assert_eq!(payload.event_source, EventSource::Restore); + let entry = transaction_redo_entry( + TenantId::new(1), + crate::types::DatabaseId::DEFAULT, + VShardId::new(0), + &payload, + ); + assert_eq!( + crate::event::EventSource::from(entry.event_source), + EventSource::Restore + ); + match entry.write { + crate::control::wal_replication::ReplicatedWrite::TransactionRedo { + event_source, + .. + } => assert_eq!(EventSource::from(event_source), EventSource::Restore), + other => panic!("expected a transaction redo, got {other:?}"), + } + } } diff --git a/nodedb/src/control/crdt_admission.rs b/nodedb/src/control/crdt_admission.rs index f210b9cf5..4914410a6 100644 --- a/nodedb/src/control/crdt_admission.rs +++ b/nodedb/src/control/crdt_admission.rs @@ -515,7 +515,10 @@ async fn apply_fenced( )? .ok_or(crate::Error::CrdtAdmissionInvalidPlan { reason: "admitted CRDT Apply has no replicated form", - })?; + })? + // Every replica gives the write the source this node dispatches it + // with. + .with_event_source(workflow.event_source); let outcome = tokio::time::timeout( workflow.timeout, crate::control::wal_replication::propose_replicated_entry(workflow.state, raw, entry), diff --git a/nodedb/src/control/distributed_applier/apply_loop/start.rs b/nodedb/src/control/distributed_applier/apply_loop/start.rs index 26b3764d1..2edb3efe2 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/start.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/start.rs @@ -21,7 +21,7 @@ use super::group_watch::GroupWatch; use super::lane::QueuedEntry; use super::proposal_gate::{EntryOutcome, ProposalGate}; use super::transaction_redo::prepare_transaction_redo_entry; -use super::write_dispatch::prepare_generic_entry; +use super::write_dispatch::{EntryScope, prepare_generic_entry}; /// How a prepared entry continues. pub(super) enum Prepared<'a> { @@ -86,6 +86,15 @@ pub(super) fn prepare_entry<'a>( let database_id = decoded .as_ref() .map_or(DatabaseId::DEFAULT, |e| DatabaseId::new(e.database_id)); + // The source the proposer stamped. An entry that does not decode applies + // nothing, so its source is never read. + let event_source = decoded + .as_ref() + .map_or(crate::event::EventSource::User, |e| e.event_source.into()); + let scope = EntryScope { + database_id, + event_source, + }; // A second committed copy of a proposal this node already applied (a // re-proposal after a leader change whose first copy also committed) @@ -101,7 +110,7 @@ pub(super) fn prepare_entry<'a>( commit_hlc, }; let Some(replicated) = decoded else { - return prepare_generic_entry(ctx, pos, entry, database_id, false); + return prepare_generic_entry(ctx, pos, entry, scope, false); }; let tenant_id = TenantId::new(replicated.tenant_id); let entry_database = DatabaseId::new(replicated.database_id); @@ -166,7 +175,7 @@ pub(super) fn prepare_entry<'a>( }) } ReplicatedWrite::ArrayCellPut { .. } | ReplicatedWrite::ArrayCellDelete { .. } => { - prepare_generic_entry(ctx, pos, entry, database_id, true) + prepare_generic_entry(ctx, pos, entry, scope, true) } ReplicatedWrite::TransactionRedo { .. } => { prepare_transaction_redo_entry(ctx, pos, &replicated) @@ -202,6 +211,6 @@ pub(super) fn prepare_entry<'a>( // so a re-delivery could not usefully replay it. Prepared::Concluded(EntryOutcome::Skipped) } - _ => prepare_generic_entry(ctx, pos, entry, database_id, false), + _ => prepare_generic_entry(ctx, pos, entry, scope, false), } } diff --git a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs index 3b1639139..4f55e6686 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs @@ -31,6 +31,16 @@ use super::helpers::{committed_response_result, deterministic_crdt_fence_noop}; use super::proposal_gate::{EntryOutcome, ledger_outcome}; use super::start::Prepared; +/// What a generic entry's apply takes from its decoded envelope. +#[derive(Debug, Clone, Copy)] +pub(super) struct EntryScope { + /// Database scope of the entry. + pub database_id: DatabaseId, + /// The source the proposer stamped. Every replica gives the write's + /// events this source. + pub event_source: crate::event::EventSource, +} + /// Prepare a generic entry. `exclusive` marks a Raft-native array cell write: /// its apply awaits the array-open bootstrap and its own write, so it runs /// with nothing else of its group in flight. Every other entry leaves as its @@ -39,22 +49,14 @@ pub(super) fn prepare_generic_entry<'a>( ctx: ApplyContext<'a>, pos: AppliedPosition, entry: LogEntry, - database_id: DatabaseId, + scope: EntryScope, exclusive: bool, ) -> Prepared<'a> { if !exclusive { - return Prepared::Enqueue(Box::pin(enqueue_generic_entry( - ctx, - pos, - entry, - database_id, - ))); + return Prepared::Enqueue(Box::pin(enqueue_generic_entry(ctx, pos, entry, scope))); } Prepared::Exclusive(Box::pin(async move { - let outcome = match enqueue_generic_entry(ctx, pos, entry, database_id) - .await - .started - { + let outcome = match enqueue_generic_entry(ctx, pos, entry, scope).await.started { Started::Running(apply) => return apply.await, Started::Concluded(outcome) => outcome, }; @@ -73,8 +75,12 @@ async fn enqueue_generic_entry<'a>( ctx: ApplyContext<'a>, pos: AppliedPosition, entry: LogEntry, - database_id: DatabaseId, + scope: EntryScope, ) -> StartedEntry<'a> { + let EntryScope { + database_id, + event_source, + } = scope; let ApplyContext { state, tracker, .. } = ctx; let AppliedPosition { group_id, @@ -186,13 +192,12 @@ async fn enqueue_generic_entry<'a>( vshard_id, plan, trace_id: TraceId::generate(), - // Cluster mode has exactly ONE write-apply path — this loop; - // the proposing node does not execute locally before commit - // either. Tagging these `RaftFollower` would mean AFTER - // triggers, DML audit, and CRDT packaging never fire anywhere - // in cluster mode, so the committed write keeps the `User` - // source its proposer had. - event_source: crate::event::EventSource::User, + // Cluster mode has exactly ONE write-apply path: this loop. The + // proposing node does not execute locally before commit either. + // So the committed write keeps the source its proposer stamped + // on the entry. A client write stays `User`, and a restored row + // stays `Restore`, on every replica. + event_source, txn_id: None, // Auth ran on the proposing node before the entry was // proposed; the committed entry carries no session user. diff --git a/nodedb/src/control/event_trigger.rs b/nodedb/src/control/event_trigger.rs index e49e47f9d..e96e511d9 100644 --- a/nodedb/src/control/event_trigger.rs +++ b/nodedb/src/control/event_trigger.rs @@ -37,8 +37,16 @@ pub async fn process_write_event( // An action's own writes come back through the Event Plane. Firing event // definitions on them lets an action that writes to the collection it // watches re-trigger itself without bound, so only the same sources that - // fire triggers fire event definitions. - if !matches!(event.source, EventSource::User | EventSource::Deferred) { + // fire triggers fire event definitions. A restored row fired its event + // definitions when it was first written. + let fires = match event.source { + EventSource::User | EventSource::Deferred => true, + EventSource::Trigger + | EventSource::RaftFollower + | EventSource::CrdtSync + | EventSource::Restore => false, + }; + if !fires { trace!( source = %event.source, collection = %event.collection, diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs index 52bf96302..c85f42694 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs @@ -25,9 +25,8 @@ pub(crate) async fn dispatch_system( task: SystemTask<'_>, timeout: Duration, ) -> crate::Result> { - let resp = - dispatch_system_response_with_source(state, task, timeout, crate::event::EventSource::User) - .await?; + let event_source = task.reason.event_source(); + let resp = dispatch_system_response_with_source(state, task, timeout, event_source).await?; if resp.status != Status::Ok { // DDL/DSL callers receive the flattened message form. Callers that need diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs index 5745cc66f..8a0977df9 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs @@ -63,6 +63,24 @@ impl SystemReason { Self::AdmittedContinuation => "admitted_continuation", } } + + /// The source the task's write events carry into the Event Plane. + /// + /// A restore re-issues rows whose AFTER triggers fired when they were + /// first written, so its writes carry `Restore`. Every other reason + /// carries `User`. + pub(crate) fn event_source(self) -> crate::event::EventSource { + match self { + Self::BackupRestore => crate::event::EventSource::Restore, + Self::RetentionEnforcement + | Self::ClusterSnapshot + | Self::DdlApply + | Self::CatalogMaintenance + | Self::TenantLifecycle + | Self::EventPlane + | Self::AdmittedContinuation => crate::event::EventSource::User, + } + } } /// A Data-Plane dispatch with no user identity behind it. @@ -107,3 +125,27 @@ impl<'a> SystemTask<'a> { self } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn only_a_restore_carries_the_restore_source() { + assert_eq!( + SystemReason::BackupRestore.event_source(), + crate::event::EventSource::Restore + ); + for reason in [ + SystemReason::RetentionEnforcement, + SystemReason::ClusterSnapshot, + SystemReason::DdlApply, + SystemReason::CatalogMaintenance, + SystemReason::TenantLifecycle, + SystemReason::EventPlane, + SystemReason::AdmittedContinuation, + ] { + assert_eq!(reason.event_source(), crate::event::EventSource::User); + } + } +} diff --git a/nodedb/src/control/server/sync/raft_dispatch/response.rs b/nodedb/src/control/server/sync/raft_dispatch/response.rs index cf6d10a00..2ffc930c3 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/response.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/response.rs @@ -144,6 +144,7 @@ async fn dispatch_sync_response_inner( Some(proposer) => { let entry = ReplicableWrite::decide_for_replication(&plan).and_then(|replicable| { to_replicated_entry(tenant_id, database_id, vshard_id, &replicable) + .map(|entry| entry.map(|entry| entry.with_event_source(event_source))) }); match entry { Ok(entry) => entry.map(|entry| (proposer, entry)), diff --git a/nodedb/src/control/server/sync/raft_dispatch/write.rs b/nodedb/src/control/server/sync/raft_dispatch/write.rs index a931a1a92..258babb9c 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/write.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/write.rs @@ -111,6 +111,7 @@ pub(crate) async fn dispatch_write_replicated( if let Some(proposer) = state.async_raft_proposer() { let entry = ReplicableWrite::decide_for_replication(&plan).and_then(|replicable| { to_replicated_entry(tenant_id, database_id, vshard_id, &replicable) + .map(|entry| entry.map(|entry| entry.with_event_source(event_source))) }); let entry = match entry { Ok(entry) => entry, diff --git a/nodedb/src/control/trigger/batch/collector.rs b/nodedb/src/control/trigger/batch/collector.rs index 7f3484a91..1bc1c4068 100644 --- a/nodedb/src/control/trigger/batch/collector.rs +++ b/nodedb/src/control/trigger/batch/collector.rs @@ -310,9 +310,16 @@ pub fn push_write_event( ) -> Option { use crate::event::types::{EventSource, WriteOp}; - // Only User-originated events fire triggers. - if !matches!(event.source, EventSource::User) { - return None; + // Only User-originated events fire batched AFTER triggers. Deferred + // events fire through the deferred dispatcher. A restored row fired its + // triggers when it was first written. + match event.source { + EventSource::User => {} + EventSource::Trigger + | EventSource::RaftFollower + | EventSource::CrdtSync + | EventSource::Deferred + | EventSource::Restore => return None, } let op_str = match event.op { diff --git a/nodedb/src/control/wal_replication/encode/transaction_redo.rs b/nodedb/src/control/wal_replication/encode/transaction_redo.rs index 1ff3c38f8..a0bd0e8cb 100644 --- a/nodedb/src/control/wal_replication/encode/transaction_redo.rs +++ b/nodedb/src/control/wal_replication/encode/transaction_redo.rs @@ -40,4 +40,5 @@ pub fn transaction_redo_entry( origin: payload.origin, }, ) + .with_event_source(payload.event_source) } diff --git a/nodedb/src/control/wal_replication/legacy_entry.rs b/nodedb/src/control/wal_replication/legacy_entry.rs index 26ad59891..ff327acc0 100644 --- a/nodedb/src/control/wal_replication/legacy_entry.rs +++ b/nodedb/src/control/wal_replication/legacy_entry.rs @@ -39,6 +39,9 @@ impl LegacyReplicatedEntry { write: self.write, write_hlc: 0, metadata_floor: 0, + // The old shape carries no source. Its replicas applied it as a + // client write. + event_source: super::types::ReplicatedEventSource::User, } } } diff --git a/nodedb/src/control/wal_replication/types/replicated_entry.rs b/nodedb/src/control/wal_replication/types/replicated_entry.rs index b64175c23..cd8da003b 100644 --- a/nodedb/src/control/wal_replication/types/replicated_entry.rs +++ b/nodedb/src/control/wal_replication/types/replicated_entry.rs @@ -4,6 +4,7 @@ //! [`super::ReplicatedWrite`] on the Raft log, plus its (de)serialization. use super::replicated_write::ReplicatedWrite; +use super::transaction_redo_wire::ReplicatedEventSource; /// Metadata carried alongside the write for routing on the receiving node. #[derive( @@ -55,6 +56,9 @@ pub struct ReplicatedEntry { /// pending purge has reclaimed its storage, and the collection the write /// targets is registered. `0` for an entry proposed without one. pub metadata_floor: u64, + /// The source every replica stamps on the write's events. It decides + /// whether AFTER triggers fire, on every replica alike. + pub event_source: ReplicatedEventSource, } impl ReplicatedEntry { @@ -75,9 +79,16 @@ impl ReplicatedEntry { write, write_hlc: 0, metadata_floor: 0, + event_source: ReplicatedEventSource::User, } } + /// Stamp the source every replica gives the write's events. + pub fn with_event_source(mut self, source: crate::event::EventSource) -> Self { + self.event_source = ReplicatedEventSource::from(source); + self + } + /// Serialize to bytes for Raft log entry data. pub fn to_bytes(&self) -> Vec { zerompk::to_msgpack_vec(self).expect("ReplicatedEntry serialization cannot fail") @@ -85,7 +96,7 @@ impl ReplicatedEntry { /// Deserialize from Raft log entry data bytes. /// - /// Tries the current 7-field shape first. If that fails specifically + /// Tries the current 8-field shape first. If that fails specifically /// because the encoded array is the pre-`database_id` 4-element shape /// (an entry proposed by an old leader still mid-upgrade), falls back to /// [`super::legacy_entry::LegacyReplicatedEntry`] and defaults @@ -184,6 +195,22 @@ mod tests { } } + #[test] + fn the_event_source_roundtrips_and_defaults_to_user() { + let write = ReplicatedWrite::CutBarrier { hlc: 7 }; + let plain = ReplicatedEntry::new(1, 0, 3, write.clone()); + assert_eq!(plain.event_source, ReplicatedEventSource::User); + + let restored = ReplicatedEntry::new(1, 0, 3, write) + .with_event_source(crate::event::EventSource::Restore); + let decoded = ReplicatedEntry::from_bytes(&restored.to_bytes()).expect("decode"); + assert_eq!(decoded.event_source, ReplicatedEventSource::Restore); + assert_eq!( + crate::event::EventSource::from(decoded.event_source), + crate::event::EventSource::Restore + ); + } + #[test] fn non_default_database_id_roundtrips_through_encode_decode() { let tenant = TenantId::new(1); diff --git a/nodedb/src/control/wal_replication/types/transaction_redo_wire.rs b/nodedb/src/control/wal_replication/types/transaction_redo_wire.rs index 3aad56ff5..3ab27fa95 100644 --- a/nodedb/src/control/wal_replication/types/transaction_redo_wire.rs +++ b/nodedb/src/control/wal_replication/types/transaction_redo_wire.rs @@ -27,10 +27,11 @@ pub struct ReplicatedIdentity { pub surrogate: u32, } -/// The source a committed transaction's writes carry into the Event Plane. +/// The source a committed entry's writes carry into the Event Plane. /// /// Every replica stamps the same source, so a transaction a trigger issued -/// does not re-fire that trigger on any replica. +/// does not re-fire that trigger on any replica, and a restored row fires no +/// AFTER trigger on any replica. #[derive( Debug, Clone, @@ -48,6 +49,7 @@ pub enum ReplicatedEventSource { RaftFollower, CrdtSync, Deferred, + Restore, } impl From for ReplicatedEventSource { @@ -58,6 +60,7 @@ impl From for ReplicatedEventSource { EventSource::RaftFollower => Self::RaftFollower, EventSource::CrdtSync => Self::CrdtSync, EventSource::Deferred => Self::Deferred, + EventSource::Restore => Self::Restore, } } } @@ -70,6 +73,7 @@ impl From for EventSource { ReplicatedEventSource::RaftFollower => Self::RaftFollower, ReplicatedEventSource::CrdtSync => Self::CrdtSync, ReplicatedEventSource::Deferred => Self::Deferred, + ReplicatedEventSource::Restore => Self::Restore, } } } diff --git a/nodedb/src/data/executor/core_loop/deferred.rs b/nodedb/src/data/executor/core_loop/deferred.rs index be71017e4..d42bfb9ae 100644 --- a/nodedb/src/data/executor/core_loop/deferred.rs +++ b/nodedb/src/data/executor/core_loop/deferred.rs @@ -3,8 +3,11 @@ //! Deferred trigger events of a committed transaction. //! //! The redo install collects the document writes of a committed record and, -//! once the record settled, emits them as WriteEvents with -//! `EventSource::Deferred` so the Event Plane fires DEFERRED-mode triggers. +//! once the record settled, emits them as WriteEvents. A client transaction's +//! rows carry `EventSource::Deferred`, so the Event Plane fires DEFERRED-mode +//! triggers. Rows of any other source keep that source, so a trigger's own +//! transaction and a restore fire no DEFERRED trigger (see +//! [`committed_row_source`]). use std::sync::Arc; @@ -12,6 +15,25 @@ use super::CoreLoop; use crate::engine::document::store::RowIdentity; use crate::event::types::{EventSource, RowId, WriteEvent, WriteOp}; +/// The source a committed record's document-row events carry. +/// +/// A client transaction's rows fire DEFERRED-mode triggers, so they carry +/// `Deferred`. Every other source keeps its own: a trigger's transaction +/// does not re-fire triggers, and a restored row fired its triggers when it +/// was first written. +pub(in crate::data::executor) const fn committed_row_source( + record_source: EventSource, +) -> EventSource { + match record_source { + EventSource::User => EventSource::Deferred, + EventSource::Trigger => EventSource::Trigger, + EventSource::RaftFollower => EventSource::RaftFollower, + EventSource::CrdtSync => EventSource::CrdtSync, + EventSource::Deferred => EventSource::Deferred, + EventSource::Restore => EventSource::Restore, + } +} + /// A write that occurred during a transaction, pending deferred trigger emission. pub(in crate::data::executor) struct DeferredWrite { pub collection: String, @@ -22,19 +44,20 @@ pub(in crate::data::executor) struct DeferredWrite { } impl CoreLoop { - /// Emit deferred trigger events for a committed transaction. + /// Emit the document-row events of a committed transaction. /// - /// Called after a committed redo record installed and settled. - /// Each write in the transaction is emitted as a WriteEvent with - /// `EventSource::Deferred`, which the Event Plane consumer routes - /// to DEFERRED-mode triggers. + /// Called after a committed redo record installed and settled. Each + /// write is emitted as a WriteEvent whose source is + /// [`committed_row_source`] of the record's `record_source`. pub(in crate::data::executor) fn emit_deferred_events( &mut self, writes: Vec, + record_source: EventSource, database_id: crate::types::DatabaseId, tenant_id: crate::types::TenantId, vshard_id: crate::types::VShardId, ) { + let source = committed_row_source(record_source); let producer = match self.event_producer.as_mut() { Some(p) => p, None => return, @@ -58,7 +81,7 @@ impl CoreLoop { database_id, tenant_id, vshard_id, - source: EventSource::Deferred, + source, new_value: write.new_value.map(|v| Arc::from(v.as_slice())), old_value: write.old_value.map(|v| Arc::from(v.as_slice())), system_time_ms, @@ -71,3 +94,36 @@ impl CoreLoop { } } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn only_a_client_transaction_fires_deferred_triggers() { + assert_eq!( + committed_row_source(EventSource::User), + EventSource::Deferred + ); + assert_eq!( + committed_row_source(EventSource::Restore), + EventSource::Restore + ); + assert_eq!( + committed_row_source(EventSource::Trigger), + EventSource::Trigger + ); + assert_eq!( + committed_row_source(EventSource::RaftFollower), + EventSource::RaftFollower + ); + assert_eq!( + committed_row_source(EventSource::CrdtSync), + EventSource::CrdtSync + ); + assert_eq!( + committed_row_source(EventSource::Deferred), + EventSource::Deferred + ); + } +} diff --git a/nodedb/src/data/executor/core_loop/event_emit.rs b/nodedb/src/data/executor/core_loop/event_emit.rs index fe39e111f..ff911745b 100644 --- a/nodedb/src/data/executor/core_loop/event_emit.rs +++ b/nodedb/src/data/executor/core_loop/event_emit.rs @@ -46,6 +46,65 @@ impl CoreLoop { } } + /// Emit a KV write event. Both row images are shaped into the + /// `{key, value}` row every KV read returns (`msgpack_scan::kv_row_msgpack`), + /// so the Event Plane decodes a KV row the same way as any other row. + /// `new_stored` and `old_stored` are the bodies as the engine stores them. + pub(in crate::data::executor) fn emit_kv_write_event( + &mut self, + task: &super::super::task::ExecutionTask, + collection: &str, + op: crate::event::WriteOp, + key: &[u8], + new_stored: Option<&[u8]>, + old_stored: Option<&[u8]>, + ) { + let key_str = String::from_utf8_lossy(key); + let new_row = new_stored.map(|body| msgpack_scan::kv_row_msgpack(&key_str, body)); + let old_row = old_stored.map(|body| msgpack_scan::kv_row_msgpack(&key_str, body)); + self.emit_write_event( + task, + collection, + op, + crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + new_row.as_deref(), + old_row.as_deref(), + ); + } + + /// A stored document row as the Event Plane reads it. A strict Binary + /// Tuple becomes MessagePack. A schemaless body gains its `id`. + pub(in crate::data::executor) fn stored_event_image( + &self, + database_id: u64, + tid: u64, + collection: &str, + identity: &str, + stored: &[u8], + ) -> Vec { + match self.resolve_event_payload(database_id, tid, collection, stored) { + Some(converted) => converted, + None => self.body_event_image(database_id, tid, collection, identity, stored), + } + } + + /// A MessagePack document body as the Event Plane reads it. A schemaless + /// body gains its `id`, the identity every read injects. + pub(in crate::data::executor) fn body_event_image( + &self, + database_id: u64, + tid: u64, + collection: &str, + identity: &str, + body: &[u8], + ) -> Vec { + if self.is_schemaless_document_collection(database_id, tid, collection) { + msgpack_scan::inject_str_field(body, "id", identity) + } else { + body.to_vec() + } + } + /// Whether `collection` is a schemaless document collection. /// /// A schemaless body carries no storage key of its own, so its `id` field diff --git a/nodedb/src/data/executor/handlers/kv/atomic.rs b/nodedb/src/data/executor/handlers/kv/atomic.rs index 0102ba4e1..81e3575dc 100644 --- a/nodedb/src/data/executor/handlers/kv/atomic.rs +++ b/nodedb/src/data/executor/handlers/kv/atomic.rs @@ -114,12 +114,11 @@ impl CoreLoop { } // The event carries the bytes the engine stored: the whole row // for a typed row, the decimal text for a raw body. - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Update, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, Some(written.as_slice()), None, ); @@ -186,12 +185,11 @@ impl CoreLoop { } // The event carries the bytes the engine stored: the whole row // for a typed row, the decimal text for a raw body. - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Update, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, Some(written.as_slice()), None, ); @@ -261,12 +259,11 @@ impl CoreLoop { if let Some(ref m) = self.metrics { m.record_kv_put(); } - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Update, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, Some(written.as_slice()), None, ); @@ -343,12 +340,11 @@ impl CoreLoop { if let Some(ref m) = self.metrics { m.record_kv_put(); } - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Update, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, Some(written.as_slice()), old.as_deref(), ); @@ -500,6 +496,11 @@ mod tests { nodedb_types::value_to_msgpack(&Value::Object(map)).expect("encode row") } + /// The `{key, value}` row a KV write event carries for `stored`. + fn kv_row(key: &str, stored: &[u8]) -> Vec { + nodedb_query::msgpack_scan::kv_row_msgpack(key, stored) + } + fn columns(bytes: &[u8]) -> HashMap { match nodedb_types::value_from_msgpack(bytes).expect("decode row") { Value::Object(map) => map, @@ -547,10 +548,11 @@ mod tests { let new_value = event.new_value.expect("the event carries the new row"); assert_eq!( new_value.as_ref(), - stored(&h.core, b"player").as_slice(), - "the event carries exactly the bytes the engine stored" + kv_row("player", &stored(&h.core, b"player")).as_slice(), + "the event carries the stored row, shaped as the KV read row" ); let row = columns(&new_value); + assert_eq!(row.get("key"), Some(&Value::String("player".into()))); assert_eq!(row.get("n"), Some(&Value::Integer(8))); assert_eq!(row.get("label"), Some(&Value::String("gold".into()))); } @@ -576,7 +578,10 @@ mod tests { let event = h.events.try_recv().expect("INCR_FLOAT emits a write event"); let new_value = event.new_value.expect("the event carries the new row"); - assert_eq!(new_value.as_ref(), stored(&h.core, b"player").as_slice()); + assert_eq!( + new_value.as_ref(), + kv_row("player", &stored(&h.core, b"player")).as_slice() + ); let row = columns(&new_value); assert_eq!(row.get("score"), Some(&Value::Float(2.5))); assert_eq!(row.get("label"), Some(&Value::String("gold".into()))); @@ -595,7 +600,12 @@ mod tests { assert_eq!(resp.status, Status::Ok, "{:?}", resp.error_code); let event = h.events.try_recv().expect("INCR emits a write event"); - assert_eq!(event.new_value.as_deref(), Some(b"42".as_slice())); + assert_eq!( + event.new_value.as_deref(), + Some(kv_row("hits", b"42").as_slice()) + ); + let row = columns(event.new_value.as_deref().expect("the new row")); + assert_eq!(row.get("value"), Some(&Value::String("42".into()))); assert_eq!(stored(&h.core, b"hits"), b"42".to_vec()); } @@ -623,4 +633,65 @@ mod tests { ); assert_eq!(stored(&h.core, b"name"), b"abc".to_vec()); } + + #[test] + fn a_put_emits_the_key_value_row_as_its_new_image() { + let mut h = make_core(); + let t = task(); + let resp = h.core.execute_kv_put( + &t, + crate::data::executor::handlers::kv::crud::KvWriteParams { + did: did(), + tid: TID, + collection: COLLECTION, + key: b"k", + value: b"x", + ttl_ms: 0, + surrogate: Surrogate::new(1), + returning: None, + rls_filters: &[], + }, + ); + assert_eq!(resp.status, Status::Ok, "{:?}", resp.error_code); + + let event = h.events.try_recv().expect("a put emits a write event"); + assert_eq!(event.op, WriteOp::Insert); + let row = columns(event.new_value.as_deref().expect("the new row")); + assert_eq!(row.get("key"), Some(&Value::String("k".into()))); + assert_eq!(row.get("value"), Some(&Value::String("x".into()))); + assert!(event.old_value.is_none(), "a fresh key has no old row"); + } + + #[test] + fn a_delete_emits_the_removed_row_as_its_old_image() { + let mut h = make_core(); + seed(&mut h.core, b"k", b"x"); + let t = task(); + let check = RlsWriteCheck::already_decided_elsewhere(); + let keys = vec![b"k".to_vec(), b"absent".to_vec()]; + let resp = h.core.execute_kv_delete( + &t, + crate::data::executor::handlers::kv::crud::KvDeleteParams { + did: did(), + tid: TID, + collection: COLLECTION, + keys: &keys, + rls_write_check: &check, + returning: None, + rls_filters: &[], + }, + ); + assert_eq!(resp.status, Status::Ok, "{:?}", resp.error_code); + + let event = h.events.try_recv().expect("a delete emits a write event"); + assert_eq!(event.op, WriteOp::Delete); + assert!(event.new_value.is_none()); + let row = columns(event.old_value.as_deref().expect("the removed row")); + assert_eq!(row.get("key"), Some(&Value::String("k".into()))); + assert_eq!(row.get("value"), Some(&Value::String("x".into()))); + assert!( + h.events.try_recv().is_none(), + "an absent key removes no row and emits no event" + ); + } } diff --git a/nodedb/src/data/executor/handlers/kv/batch.rs b/nodedb/src/data/executor/handlers/kv/batch.rs index 0217cc740..0ef8eee11 100644 --- a/nodedb/src/data/executor/handlers/kv/batch.rs +++ b/nodedb/src/data/executor/handlers/kv/batch.rs @@ -110,6 +110,23 @@ impl CoreLoop { debug!(core = self.core_id, %collection, count = entries.len(), "kv batch put"); // See `CoreLoop::kv_ttl_now_ms` for the precedence this resolves. let now_ms: u64 = self.kv_ttl_now_ms(task); + // Each entry's pre-image, in `entries` order, for its write event. A + // key the batch names twice sees the batch's own earlier value. + let priors: Vec>> = { + let mut written: std::collections::HashMap<&[u8], &[u8]> = + std::collections::HashMap::with_capacity(entries.len()); + entries + .iter() + .map(|(key, value)| { + let prior = match written.get(key.as_slice()) { + Some(earlier) => Some(earlier.to_vec()), + None => self.kv_engine.get(did, tid, collection, key, now_ms), + }; + written.insert(key.as_slice(), value.as_slice()); + prior + }) + .collect() + }; let new_count = self.kv_engine.batch_put(KvBatchPutParams { database_id: did, tenant_id: tid, @@ -119,6 +136,16 @@ impl CoreLoop { now_ms, surrogates, }); + // One write event per entry: a batch INSERT fires the same row + // triggers as the single-row form. + for ((key, value), prior) in entries.iter().zip(&priors) { + let op = if prior.is_some() { + crate::event::WriteOp::Update + } else { + crate::event::WriteOp::Insert + }; + self.emit_kv_write_event(task, collection, op, key, Some(value), prior.as_deref()); + } // One WAL record covers the whole batch; record every written key's // version against that single LSN. if task.wal_lsn().is_some() { diff --git a/nodedb/src/data/executor/handlers/kv/crud/delete.rs b/nodedb/src/data/executor/handlers/kv/crud/delete.rs index 81f29f645..7691a8611 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/delete.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/delete.rs @@ -14,10 +14,10 @@ use crate::engine::kv::current_ms; impl CoreLoop { /// `rls_write_check` is the compiled RLS write policy. The row a delete - /// removes is the image the policy decides, and a delete otherwise reads - /// nothing at all — so a non-empty check, or a `RETURNING` clause, is what - /// makes the pre-image be read in the first place, and one rejected key - /// fails the whole statement before any key is removed. + /// removes is the image the policy decides, and one rejected key fails + /// the whole statement before any key is removed. Every removed row's + /// pre-image is read: the policy decides it, `RETURNING` projects it, and + /// the delete event carries it as the OLD row. pub(in crate::data::executor) fn execute_kv_delete( &mut self, task: &ExecutionTask, @@ -40,27 +40,27 @@ impl CoreLoop { nodedb_types::WriteGateDecision::AdmitAll ); // Pre-images of the keys that exist, in `keys` order: the rows the - // write gate decides and the rows `RETURNING` projects. An absent key - // removes no row, so it has no image and counts as not-deleted. + // write gate decides, the rows `RETURNING` projects, and the OLD rows + // the delete events carry. An absent key removes no row, so it has no + // image, counts as not-deleted, and emits no event. A key named twice + // removes one row. let mut pre_images: Vec<(&[u8], Vec)> = Vec::with_capacity(keys.len()); - if gated || returning.is_some() { - for key in keys { - let Some(body) = self.kv_engine.get(did, tid, collection, key, now_ms) else { - continue; - }; - if gated - && let Err(e) = super::super::rls::admit_kv_row( - rls_write_check, - &body, - key, - tid, - collection, - ) - { - return self.response_error(task, e); - } - pre_images.push((key.as_slice(), body)); + let mut seen: std::collections::HashSet<&[u8]> = + std::collections::HashSet::with_capacity(keys.len()); + for key in keys { + if !seen.insert(key.as_slice()) { + continue; } + let Some(body) = self.kv_engine.get(did, tid, collection, key, now_ms) else { + continue; + }; + if gated + && let Err(e) = + super::super::rls::admit_kv_row(rls_write_check, &body, key, tid, collection) + { + return self.response_error(task, e); + } + pre_images.push((key.as_slice(), body)); } let count = self.kv_engine.delete(did, tid, collection, keys, now_ms); @@ -68,19 +68,16 @@ impl CoreLoop { m.record_kv_delete(); } - // Emit delete events to Event Plane (one per deleted key). - if count > 0 { - for key in keys { - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( - task, - collection, - crate::event::WriteOp::Delete, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), - None, - None, - ); - } + // One delete event per removed row, carrying the row it removed. + for (key, body) in &pre_images { + self.emit_kv_write_event( + task, + collection, + crate::event::WriteOp::Delete, + key, + None, + Some(body.as_slice()), + ); } if let Some(spec) = returning { diff --git a/nodedb/src/data/executor/handlers/kv/crud/write_basic.rs b/nodedb/src/data/executor/handlers/kv/crud/write_basic.rs index 4d0059475..1137c82b5 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/write_basic.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/write_basic.rs @@ -59,19 +59,11 @@ impl CoreLoop { // replaced an existing row and the Event Plane must see an Update // event with both sides populated. Otherwise it was a fresh // Insert. - let key_str = String::from_utf8_lossy(key); let (op, old_slice): (_, Option<&[u8]>) = match old.as_deref() { Some(o) => (crate::event::WriteOp::Update, Some(o)), None => (crate::event::WriteOp::Insert, None), }; - self.emit_write_event( - task, - collection, - op, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), - Some(value), - old_slice, - ); + self.emit_kv_write_event(task, collection, op, key, Some(value), old_slice); self.note_kv_write_lsn(task, did, tid, collection, key); if let Some(spec) = returning { @@ -153,12 +145,11 @@ impl CoreLoop { m.record_kv_put(); } - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Insert, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, Some(value), None, ); @@ -234,12 +225,11 @@ impl CoreLoop { m.record_kv_put(); } - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Insert, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, Some(value), None, ); diff --git a/nodedb/src/data/executor/handlers/kv/crud/write_upsert.rs b/nodedb/src/data/executor/handlers/kv/crud/write_upsert.rs index 37f76c9aa..ecdfe7345 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/write_upsert.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/write_upsert.rs @@ -90,19 +90,11 @@ impl CoreLoop { // every downstream consumer's perspective; a fresh key with no // prior value is an Insert. The pre-write `existing_bytes` probe // above is the source of truth. - let key_str = String::from_utf8_lossy(key); let (op, old_slice): (_, Option<&[u8]>) = match existing_bytes.as_deref() { Some(o) => (crate::event::WriteOp::Update, Some(o)), None => (crate::event::WriteOp::Insert, None), }; - self.emit_write_event( - task, - collection, - op, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), - Some(&stored_bytes), - old_slice, - ); + self.emit_kv_write_event(task, collection, op, key, Some(&stored_bytes), old_slice); if let Some(spec) = returning { // The MERGED body, not the caller's: on a conflict the submitted diff --git a/nodedb/src/data/executor/handlers/kv/field.rs b/nodedb/src/data/executor/handlers/kv/field.rs index 6a29f001f..d250e24f6 100644 --- a/nodedb/src/data/executor/handlers/kv/field.rs +++ b/nodedb/src/data/executor/handlers/kv/field.rs @@ -200,6 +200,19 @@ impl CoreLoop { now_ms, surrogate, }); + let op = if current.is_some() { + crate::event::WriteOp::Update + } else { + crate::event::WriteOp::Insert + }; + self.emit_kv_write_event( + task, + collection, + op, + key, + Some(computed.new_value.as_slice()), + current.as_deref(), + ); self.note_kv_write_lsn(task, did, tid, collection, key); if let Some(spec) = returning { // `computed.new_value` IS the stored body: the merge is persisted diff --git a/nodedb/src/data/executor/handlers/kv/predicate/apply.rs b/nodedb/src/data/executor/handlers/kv/predicate/apply.rs index d11e9318b..fd641e0dc 100644 --- a/nodedb/src/data/executor/handlers/kv/predicate/apply.rs +++ b/nodedb/src/data/executor/handlers/kv/predicate/apply.rs @@ -92,12 +92,11 @@ impl CoreLoop { if let Some(ref m) = self.metrics { m.record_kv_put(); } - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Update, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, Some(new_value), Some(old_body), ); diff --git a/nodedb/src/data/executor/handlers/kv/resolve/apply.rs b/nodedb/src/data/executor/handlers/kv/resolve/apply.rs index faf014da2..74acd176f 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/apply.rs @@ -116,12 +116,11 @@ impl CoreLoop { Some(_) => crate::event::WriteOp::Update, None => crate::event::WriteOp::Insert, }; - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection.as_str(), op, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, Some(value), precondition.as_deref(), ); @@ -137,12 +136,11 @@ impl CoreLoop { if let Some(ref m) = self.metrics { m.record_kv_delete(); } - let key_str = String::from_utf8_lossy(key); - self.emit_write_event( + self.emit_kv_write_event( task, collection.as_str(), crate::event::WriteOp::Delete, - crate::engine::document::store::RowIdentity::from_user_key(key_str.as_ref()), + key, None, precondition.as_deref(), ); diff --git a/nodedb/src/data/executor/handlers/kv/transfer.rs b/nodedb/src/data/executor/handlers/kv/transfer.rs index ec93a7efb..a557e6bb5 100644 --- a/nodedb/src/data/executor/handlers/kv/transfer.rs +++ b/nodedb/src/data/executor/handlers/kv/transfer.rs @@ -192,21 +192,19 @@ impl CoreLoop { } // Emit CDC events. - let src_str = String::from_utf8_lossy(source_key); - let dst_str = String::from_utf8_lossy(dest_key); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Update, - crate::engine::document::store::RowIdentity::from_user_key(src_str.as_ref()), + source_key, Some(&new_source), Some(&source_bytes), ); - self.emit_write_event( + self.emit_kv_write_event( task, collection, crate::event::WriteOp::Update, - crate::engine::document::store::RowIdentity::from_user_key(dst_str.as_ref()), + dest_key, Some(&new_dest), if dest_bytes.is_empty() { None @@ -215,6 +213,8 @@ impl CoreLoop { }, ); + let src_str = String::from_utf8_lossy(source_key); + let dst_str = String::from_utf8_lossy(dest_key); match response_codec::encode_json_as_msgpack(&serde_json::json!({ "source_key": src_str, "dest_key": dst_str, @@ -311,19 +311,19 @@ impl CoreLoop { // Emit CDC events. let item_str = String::from_utf8_lossy(item_key); let dest_str = String::from_utf8_lossy(dest_key); - self.emit_write_event( + self.emit_kv_write_event( task, source_collection, crate::event::WriteOp::Delete, - crate::engine::document::store::RowIdentity::from_user_key(item_str.as_ref()), + item_key, None, Some(&item_data), ); - self.emit_write_event( + self.emit_kv_write_event( task, dest_collection, crate::event::WriteOp::Insert, - crate::engine::document::store::RowIdentity::from_user_key(dest_str.as_ref()), + dest_key, Some(&item_data), None, ); diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs index b631d8b89..c77e03b52 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/document.rs @@ -187,6 +187,9 @@ impl CoreLoop { WriteOp::Insert }, old_value: outcome.prior_value.clone(), + // The submitted body, before any hash-chain wrapping: the image a + // client wrote, as the materialized-sum fold reads it too. + new_body: Some(value.to_vec()), index_tuples, }); if self.recording_redo_undo() { @@ -284,6 +287,7 @@ impl CoreLoop { identity: RowIdentity::from_user_key(row.document_id), op: WriteOp::Delete, old_value: Some(old_value), + new_body: None, index_tuples, }); self.record_committed_targets(target_writes); diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs index cb865600f..a8d88b7e9 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/events.rs @@ -5,8 +5,10 @@ //! The transaction batch emitted three kinds of events, and the apply emits //! the same set: //! -//! * one `Deferred` event per document row written (source rows and -//! materialized-sum targets), carrying the pre-image, for DEFERRED triggers; +//! * one event per document row written (source rows and materialized-sum +//! targets), carrying the pre-image. A client transaction's rows carry +//! `Deferred`, for DEFERRED triggers. Rows of other sources keep their +//! source; //! * one event per KV key written, from the KV write handlers; //! * one event per graph node-label delta, on the label stream. //! @@ -17,7 +19,6 @@ use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::core_loop::deferred::DeferredWrite; use crate::data::executor::task::ExecutionTask; -use crate::engine::document::store::RowIdentity; use crate::event::WriteOp; use super::state::AppliedDocWrite; @@ -96,7 +97,6 @@ impl CoreLoop { } for image in kv_images { - let identity = RowIdentity::from_user_key(String::from_utf8_lossy(&image.key).as_ref()); match image.new_value { Some(new_value) => { let op = if image.prior.is_some() { @@ -104,24 +104,24 @@ impl CoreLoop { } else { WriteOp::Insert }; - self.emit_write_event( + self.emit_kv_write_event( task, &image.collection, op, - identity, + &image.key, Some(new_value.as_slice()), image.prior.as_deref(), ); } // A delete of an absent key removed nothing. None if image.prior.is_some() => { - self.emit_write_event( + self.emit_kv_write_event( task, &image.collection, WriteOp::Delete, - identity, - None, + &image.key, None, + image.prior.as_deref(), ); } None => {} @@ -137,19 +137,34 @@ impl CoreLoop { self.emit_graph_label_event(task, &label.node_id, &label.labels, op); } + // Each row carries its post-image and pre-image as the Event Plane + // reads them, the same as an autocommit write's event: an AFTER INSERT + // or UPDATE trigger binds NEW, and CDC and views read the new row. + let database_id = task.request.database_id.as_u64(); + let tid = task.request.tenant_id.as_u64(); let deferred: Vec = doc_writes .into_iter() - .map(|write| DeferredWrite { - collection: write.collection, - op: write.op, - identity: write.identity, - new_value: None, - old_value: write.old_value, + .map(|write| { + let id = write.identity.as_str(); + let new_value = write.new_body.as_deref().map(|body| { + self.body_event_image(database_id, tid, &write.collection, id, body) + }); + let old_value = write.old_value.as_deref().map(|stored| { + self.stored_event_image(database_id, tid, &write.collection, id, stored) + }); + DeferredWrite { + new_value, + old_value, + collection: write.collection, + op: write.op, + identity: write.identity, + } }) .collect(); if !deferred.is_empty() { self.emit_deferred_events( deferred, + task.request.event_source, task.request.database_id, task.request.tenant_id, task.request.vshard_id, diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs index 88868c8fc..258c61bdd 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/state.rs @@ -91,6 +91,8 @@ pub(in crate::data::executor) struct AppliedDocWrite { pub op: WriteOp, /// The row as stored before this write, when the write read it. pub old_value: Option>, + /// The MessagePack body this write installed. `None` for a delete. + pub new_body: Option>, /// Every `(field, value)` index entry the write added, removed, or /// versioned. pub index_tuples: Vec<(String, String)>, @@ -236,6 +238,7 @@ pub(in crate::data::executor) fn target_doc_write(target: &TargetWrite) -> Appli WriteOp::Insert }, old_value: outcome.prior_value.clone(), + new_body: Some(target.body.clone()), index_tuples, } } diff --git a/nodedb/src/event/audit_dml/consumer.rs b/nodedb/src/event/audit_dml/consumer.rs index d4e767014..79777879d 100644 --- a/nodedb/src/event/audit_dml/consumer.rs +++ b/nodedb/src/event/audit_dml/consumer.rs @@ -45,10 +45,13 @@ pub fn audit_dml_event( // Only User-sourced writes are subject to DML auditing. match event.source { EventSource::User => {} + // A restored row is not client DML. The RESTORE statement itself is + // the audited action. EventSource::Trigger | EventSource::RaftFollower | EventSource::CrdtSync - | EventSource::Deferred => return, + | EventSource::Deferred + | EventSource::Restore => return, } // Only data-modifying ops (not Heartbeat). @@ -152,7 +155,8 @@ mod tests { EventSource::Trigger | EventSource::RaftFollower | EventSource::CrdtSync - | EventSource::Deferred => {} + | EventSource::Deferred + | EventSource::Restore => {} } } diff --git a/nodedb/src/event/cdc/buffer.rs b/nodedb/src/event/cdc/buffer.rs index 0cabe7304..ec17121d1 100644 --- a/nodedb/src/event/cdc/buffer.rs +++ b/nodedb/src/event/cdc/buffer.rs @@ -344,6 +344,7 @@ mod tests { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: crate::event::EventSource::User, } } diff --git a/nodedb/src/event/cdc/compaction.rs b/nodedb/src/event/cdc/compaction.rs index 1be1c077a..7457b6e97 100644 --- a/nodedb/src/event/cdc/compaction.rs +++ b/nodedb/src/event/cdc/compaction.rs @@ -120,6 +120,7 @@ mod tests { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: crate::event::EventSource::User, } } diff --git a/nodedb/src/event/cdc/consume.rs b/nodedb/src/event/cdc/consume.rs index 8cc10df76..4d5e58ead 100644 --- a/nodedb/src/event/cdc/consume.rs +++ b/nodedb/src/event/cdc/consume.rs @@ -670,6 +670,7 @@ mod tests { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: crate::event::EventSource::User, }); } let params = ConsumeParams { @@ -781,6 +782,7 @@ mod tests { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: crate::event::EventSource::User, }); state .cdc_router @@ -802,6 +804,7 @@ mod tests { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: crate::event::EventSource::User, }); let params = ConsumeParams { diff --git a/nodedb/src/event/cdc/event.rs b/nodedb/src/event/cdc/event.rs index 8e6a95bb0..0bcc5d5d4 100644 --- a/nodedb/src/event/cdc/event.rs +++ b/nodedb/src/event/cdc/event.rs @@ -36,6 +36,9 @@ pub struct CdcEvent { pub database_id: DatabaseId, /// Tenant ID. pub tenant_id: u64, + /// Where the write came from, such as `user` or `restore`. A consumer + /// tells a restored row from a client write by it. + pub source: crate::event::EventSource, /// New row value (for INSERT and UPDATE). JSON bytes. #[serde(skip_serializing_if = "Option::is_none")] pub new_value: Option, @@ -93,7 +96,7 @@ impl Serialize for CdcEvent { where S: serde::Serializer, { - let field_count = 11 + let field_count = 12 + usize::from(self.new_value.is_some()) + usize::from(self.old_value.is_some()) + usize::from(self.field_diffs.is_some()) @@ -110,6 +113,7 @@ impl Serialize for CdcEvent { state.serialize_field("offset", &self.offset_token())?; state.serialize_field("database_id", &self.database_id)?; state.serialize_field("tenant_id", &self.tenant_id)?; + state.serialize_field("source", &self.source)?; state.serialize_field("schema_version", &self.schema_version)?; if let Some(value) = &self.new_value { state.serialize_field("new_value", value)?; @@ -134,7 +138,7 @@ impl Serialize for CdcEvent { impl zerompk::ToMessagePack for CdcEvent { fn write(&self, writer: &mut W) -> zerompk::Result<()> { - let field_count = 10 + let field_count = 11 + usize::from(self.new_value.is_some()) + usize::from(self.old_value.is_some()) + usize::from(self.field_diffs.is_some()) @@ -159,6 +163,8 @@ impl zerompk::ToMessagePack for CdcEvent { writer.write_u64(self.tenant_id)?; writer.write_string("database_id")?; writer.write_u64(self.database_id.as_u64())?; + writer.write_string("source")?; + writer.write_string(self.source.as_str())?; writer.write_string("schema_version")?; writer.write_u64(self.schema_version)?; if let Some(ref v) = self.new_value { @@ -197,6 +203,7 @@ impl<'a> zerompk::FromMessagePack<'a> for CdcEvent { let mut lsn: u64 = 0; let mut tenant_id: u64 = 0; let mut database_id = DatabaseId::DEFAULT; + let mut source: Option = None; let mut schema_version: u64 = 0; let mut new_value: Option = None; let mut old_value: Option = None; @@ -215,6 +222,13 @@ impl<'a> zerompk::FromMessagePack<'a> for CdcEvent { "lsn" => lsn = reader.read_u64()?, "tenant_id" => tenant_id = reader.read_u64()?, "database_id" => database_id = DatabaseId::new(reader.read_u64()?), + "source" => { + let name = reader.read_string()?; + source = Some( + crate::event::EventSource::from_name(&name) + .ok_or(zerompk::Error::InvalidMarker(0))?, + ); + } "schema_version" => schema_version = reader.read_u64()?, "new_value" => new_value = Some(JsonValue::read(reader)?.0), "old_value" => old_value = Some(JsonValue::read(reader)?.0), @@ -228,6 +242,9 @@ impl<'a> zerompk::FromMessagePack<'a> for CdcEvent { } } } + // Every encoder writes the source, so an event without one is not + // a CDC event. + let source = source.ok_or(zerompk::Error::InvalidMarker(0))?; Ok(CdcEvent { sequence, partition, @@ -237,6 +254,7 @@ impl<'a> zerompk::FromMessagePack<'a> for CdcEvent { event_time, lsn, tenant_id, + source, database_id, schema_version, new_value, @@ -263,6 +281,7 @@ mod tests { event_time: 1700000000000, lsn: 100, tenant_id: 1, + source: crate::event::EventSource::User, database_id: DatabaseId::DEFAULT, new_value: Some(serde_json::json!({"id": 1, "total": 99.99})), old_value: None, @@ -293,6 +312,7 @@ mod tests { event_time: 1700000001000, lsn: 200, tenant_id: 1, + source: crate::event::EventSource::User, database_id: DatabaseId::DEFAULT, new_value: Some(serde_json::json!({"name": "Alice"})), old_value: Some(serde_json::json!({"name": "Bob"})), @@ -323,6 +343,7 @@ mod tests { event_time: 1700000002000, lsn: 300, tenant_id: 1, + source: crate::event::EventSource::User, database_id: DatabaseId::DEFAULT, new_value: Some(serde_json::json!({"name": "Carol"})), old_value: None, @@ -342,4 +363,34 @@ mod tests { assert_eq!(parsed_mp.system_time_ms, Some(1_700_000_000_000)); assert_eq!(parsed_mp.valid_time_ms, Some(1_500_000_000_000)); } + + #[test] + fn a_restored_row_is_tagged_restore_in_json_and_msgpack() { + let event = CdcEvent { + sequence: 4, + partition: 0, + collection: "users".into(), + op: "INSERT".into(), + row_id: "u-2".into(), + event_time: 1700000003000, + lsn: 400, + tenant_id: 1, + source: crate::event::EventSource::Restore, + database_id: DatabaseId::DEFAULT, + new_value: Some(serde_json::json!({"name": "Dana"})), + old_value: None, + schema_version: 0, + field_diffs: None, + system_time_ms: None, + valid_time_ms: None, + }; + + let json: serde_json::Value = sonic_rs::from_slice(&event.to_json_bytes()).unwrap(); + assert_eq!(json["source"], "restore"); + let parsed_json: CdcEvent = serde_json::from_slice(&event.to_json_bytes()).unwrap(); + assert_eq!(parsed_json.source, crate::event::EventSource::Restore); + + let parsed_mp: CdcEvent = zerompk::from_msgpack(&event.to_msgpack_bytes()).unwrap(); + assert_eq!(parsed_mp.source, crate::event::EventSource::Restore); + } } diff --git a/nodedb/src/event/cdc/redaction.rs b/nodedb/src/event/cdc/redaction.rs index d301cca75..61788ac18 100644 --- a/nodedb/src/event/cdc/redaction.rs +++ b/nodedb/src/event/cdc/redaction.rs @@ -400,6 +400,7 @@ mod tests { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: crate::event::EventSource::User, }) } diff --git a/nodedb/src/event/cdc/router.rs b/nodedb/src/event/cdc/router.rs index 6e360d6fe..128a8188d 100644 --- a/nodedb/src/event/cdc/router.rs +++ b/nodedb/src/event/cdc/router.rs @@ -128,6 +128,7 @@ impl CdcRouter { field_diffs, system_time_ms: event.system_time_ms, valid_time_ms: event.valid_time_ms, + source: event.source, }); // RECOMPUTE correction (only built if any stream actually needs it). @@ -192,6 +193,7 @@ impl CdcRouter { field_diffs: None, system_time_ms: event.system_time_ms, valid_time_ms: event.valid_time_ms, + source: event.source, }) }) .clone(); diff --git a/nodedb/src/event/crdt_sync/packager.rs b/nodedb/src/event/crdt_sync/packager.rs index 535653cc6..b5f5f51d5 100644 --- a/nodedb/src/event/crdt_sync/packager.rs +++ b/nodedb/src/event/crdt_sync/packager.rs @@ -93,12 +93,16 @@ impl DeltaPackager { ledger: Option<&CrdtLedger>, delivery: &CrdtSyncDelivery, ) -> bool { - // Only package User-originated writes. - // CrdtSync events are inbound FROM Lite — don't echo back. - // Trigger/RaftFollower events are derivative — the original User - // event already covers the data change. - if event.source != EventSource::User { - return false; + // User writes and restored rows change the base data Lite peers hold, + // so both are packaged. CrdtSync events came FROM Lite and are not + // echoed back. Trigger, RaftFollower and Deferred events are derived + // from a User event that already carries the data change. + match event.source { + EventSource::User | EventSource::Restore => {} + EventSource::Trigger + | EventSource::RaftFollower + | EventSource::CrdtSync + | EventSource::Deferred => return false, } // Check if any connected Lite session cares about this collection. @@ -226,6 +230,17 @@ mod tests { assert!(!packager.package_and_enqueue(&trigger_event, None, None, &delivery)); } + #[test] + fn a_restored_row_passes_the_source_gate() { + let packager = DeltaPackager::new(); + let delivery = CrdtSyncDelivery::new(); + // No session subscribes, so the event stops at the subscriber check. + // The skip counter moves only for an event past the source gate. + let restored = make_event(EventSource::Restore, WriteOp::Insert); + assert!(!packager.package_and_enqueue(&restored, None, None, &delivery)); + assert_eq!(packager.deltas_skipped.load(Ordering::Relaxed), 1); + } + #[test] fn skips_heartbeats() { let packager = DeltaPackager::new(); diff --git a/nodedb/src/event/streaming_mv/processor.rs b/nodedb/src/event/streaming_mv/processor.rs index 966c2e37d..0cb087f3e 100644 --- a/nodedb/src/event/streaming_mv/processor.rs +++ b/nodedb/src/event/streaming_mv/processor.rs @@ -54,6 +54,7 @@ pub fn process_write_event_for_mvs(event: &WriteEvent, registry: &MvRegistry, st field_diffs: None, system_time_ms: event.system_time_ms, valid_time_ms: event.valid_time_ms, + source: event.source, }; for mv_state in &mv_states { @@ -210,6 +211,7 @@ mod tests { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: crate::event::EventSource::User, } } diff --git a/nodedb/src/event/topic/types.rs b/nodedb/src/event/topic/types.rs index db0a9ea6c..1a55c61c5 100644 --- a/nodedb/src/event/topic/types.rs +++ b/nodedb/src/event/topic/types.rs @@ -96,6 +96,7 @@ impl TopicMessage { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: crate::event::EventSource::User, } } } diff --git a/nodedb/src/event/trigger/dispatcher/single.rs b/nodedb/src/event/trigger/dispatcher/single.rs index 68bd6d169..8f555ad9d 100644 --- a/nodedb/src/event/trigger/dispatcher/single.rs +++ b/nodedb/src/event/trigger/dispatcher/single.rs @@ -35,9 +35,11 @@ use super::identity::trigger_identity; /// Dispatch a `WriteEvent` to matching AFTER triggers. /// -/// Skips events not from `EventSource::User` / `Deferred` (cascade -/// prevention). Trigger failures never propagate to the caller — they are -/// queued for retry and, once out of attempts, routed to the DLQ. +/// Fires only for `EventSource::User` / `Deferred`. Trigger and replicated +/// writes are skipped to prevent cascades. Restored rows are skipped because +/// their triggers fired at original write time. Trigger failures never +/// propagate to the caller. They are queued for retry and, once out of +/// attempts, routed to the DLQ. pub async fn dispatch_triggers( event: &WriteEvent, state: &Arc, @@ -46,7 +48,10 @@ pub async fn dispatch_triggers( let mode_filter = match event.source { EventSource::User => Some(TriggerExecutionMode::Async), EventSource::Deferred => Some(TriggerExecutionMode::Deferred), - _ => { + EventSource::Trigger + | EventSource::RaftFollower + | EventSource::CrdtSync + | EventSource::Restore => { trace!( source = %event.source, collection = %event.collection, @@ -56,9 +61,6 @@ pub async fn dispatch_triggers( } }; - let new_fields = row_fields(event.new_value.as_deref(), event.row_id.as_str()); - let old_fields = row_fields(event.old_value.as_deref(), event.row_id.as_str()); - let identity = trigger_identity(event.tenant_id); let op_str = event.op.to_string(); let source = ActionSource { @@ -84,7 +86,26 @@ pub async fn dispatch_triggers( WriteOp::BulkInsert { .. } | WriteOp::BulkDelete { .. } ); - if !is_bulk { + // A row image that does not decode refuses the ROW pass: firing with a + // missing NEW or OLD row would skip every trigger without a trace. + let fields = if is_bulk { + None + } else { + match event_row_fields(event) { + Ok(fields) => Some(fields), + Err(error) => { + record_row_failures( + &source, + FireReport::from_precondition(error), + None, + None, + queue, + ); + None + } + } + }; + if let Some((new_fields, old_fields)) = fields { let report = fire_for_operation(FireForOperationParams { operation: &op_str, state, @@ -138,18 +159,35 @@ pub async fn dispatch_triggers( record_statement_failures(&source, report, queue); } -/// Decode one side of an event payload into trigger row bindings. -fn row_fields( - payload: Option<&[u8]>, - row_id: &str, -) -> Option> { - let map = deserialize_event_payload(payload?)?; +/// The NEW and OLD row fields of `event`, with the row identity injected. +/// +/// Every engine emits its row images as the decoded row map a read returns. +/// A KV row is the `{key, value}` row the Data Plane shapes at emission. +fn event_row_fields(event: &WriteEvent) -> crate::Result<(RowFields, RowFields)> { + let row_id = event.row_id.as_str(); + Ok(( + row_fields(event.new_value.as_deref(), row_id)?, + row_fields(event.old_value.as_deref(), row_id)?, + )) +} + +type RowFields = Option>; + +/// One row image as fields. `None` when the event carries no image. An image +/// that is not a row map is an error. +fn row_fields(payload: Option<&[u8]>, row_id: &str) -> crate::Result { + let Some(payload) = payload else { + return Ok(None); + }; + let map = deserialize_event_payload(payload).ok_or_else(|| crate::Error::Internal { + detail: format!("trigger dispatch: the row image of '{row_id}' is not a row map"), + })?; let mut fields: std::collections::HashMap = map .into_iter() .map(|(k, v)| (k, nodedb_types::Value::from(v))) .collect(); inject_row_identity(&mut fields, row_id); - Some(fields) + Ok(Some(fields)) } /// The statement-level DML event a write op represents, if any. @@ -307,4 +345,26 @@ mod tests { let bytes = serde_json::to_vec(&serde_json::json!([1, 2, 3])).unwrap(); assert!(deserialize_event_payload(&bytes).is_none()); } + + #[test] + fn a_shaped_kv_row_binds_its_key_and_value() { + let row = nodedb_query::msgpack_scan::kv_row_msgpack("a", b"x"); + let fields = super::row_fields(Some(&row), "a") + .expect("a KV row decodes") + .expect("an image is present"); + assert_eq!( + fields.get("key"), + Some(&nodedb_types::Value::String("a".into())) + ); + assert_eq!( + fields.get("value"), + Some(&nodedb_types::Value::String("x".into())) + ); + } + + #[test] + fn a_document_image_that_is_not_a_map_is_an_error() { + assert!(super::row_fields(Some(b"x"), "a").is_err()); + assert!(super::row_fields(None, "a").expect("no image").is_none()); + } } diff --git a/nodedb/src/event/types.rs b/nodedb/src/event/types.rs index e321abb4e..a80b1f4b3 100644 --- a/nodedb/src/event/types.rs +++ b/nodedb/src/event/types.rs @@ -247,7 +247,11 @@ impl std::fmt::Display for WriteOp { /// Source of a write event. The Event Plane uses this to decide whether /// to fire AFTER triggers and other side effects. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] +/// +/// Serde names match [`EventSource::as_str`]. CDC events carry the source +/// under those names. +#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] +#[serde(rename_all = "snake_case")] pub enum EventSource { /// User-originated DML. AFTER triggers should fire. User, @@ -260,20 +264,46 @@ pub enum EventSource { /// Deferred trigger write. The Event Plane fires DEFERRED-mode triggers /// for these events (post-commit from transaction batch). Deferred, + /// A row a RESTORE re-issued from a backup. AFTER triggers do not fire: + /// they fired when the row was first written. CDC streams deliver the + /// event tagged `restore`. Consumers that keep derived state in step + /// with the base data process it. + Restore, } -impl std::fmt::Display for EventSource { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { +impl EventSource { + /// The source's stable name, as CDC events and logs show it. + pub const fn as_str(self) -> &'static str { match self { - Self::User => write!(f, "user"), - Self::Trigger => write!(f, "trigger"), - Self::RaftFollower => write!(f, "raft_follower"), - Self::CrdtSync => write!(f, "crdt_sync"), - Self::Deferred => write!(f, "deferred"), + Self::User => "user", + Self::Trigger => "trigger", + Self::RaftFollower => "raft_follower", + Self::CrdtSync => "crdt_sync", + Self::Deferred => "deferred", + Self::Restore => "restore", + } + } + + /// The source named `name`, as [`Self::as_str`] spells it. + pub fn from_name(name: &str) -> Option { + match name { + "user" => Some(Self::User), + "trigger" => Some(Self::Trigger), + "raft_follower" => Some(Self::RaftFollower), + "crdt_sync" => Some(Self::CrdtSync), + "deferred" => Some(Self::Deferred), + "restore" => Some(Self::Restore), + _ => None, } } } +impl std::fmt::Display for EventSource { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(self.as_str()) + } +} + /// Deserialize a MessagePack or JSON payload into a [`serde_json::Map`]. /// /// WriteEvent payloads are stored in the same format as the WAL payload @@ -345,6 +375,26 @@ mod tests { fn event_source_display() { assert_eq!(EventSource::User.to_string(), "user"); assert_eq!(EventSource::RaftFollower.to_string(), "raft_follower"); + assert_eq!(EventSource::Restore.to_string(), "restore"); + } + + #[test] + fn every_event_source_name_round_trips_through_serde_and_from_name() { + for source in [ + EventSource::User, + EventSource::Trigger, + EventSource::RaftFollower, + EventSource::CrdtSync, + EventSource::Deferred, + EventSource::Restore, + ] { + assert_eq!(EventSource::from_name(source.as_str()), Some(source)); + let json = sonic_rs::to_string(&source).expect("encode source"); + assert_eq!(json, format!("\"{}\"", source.as_str())); + let decoded: EventSource = sonic_rs::from_str(&json).expect("decode source"); + assert_eq!(decoded, source); + } + assert_eq!(EventSource::from_name("unknown"), None); } #[test] diff --git a/nodedb/src/event/wal_replay.rs b/nodedb/src/event/wal_replay.rs index c59a84757..483731d40 100644 --- a/nodedb/src/event/wal_replay.rs +++ b/nodedb/src/event/wal_replay.rs @@ -392,6 +392,11 @@ mod tests { let event = one_event(&record, &mut seq); assert_eq!(event.collection.as_ref(), "cache"); assert_eq!(event.op, WriteOp::Insert); + // The same `{key, value}` row image a live KV write event carries. + assert_eq!( + event.new_value.as_deref(), + Some(nodedb_query::msgpack_scan::kv_row_msgpack("key1", b"val1").as_slice()) + ); } #[test] diff --git a/nodedb/src/event/wal_replay_parse.rs b/nodedb/src/event/wal_replay_parse.rs index fd45ccb84..a6848870a 100644 --- a/nodedb/src/event/wal_replay_parse.rs +++ b/nodedb/src/event/wal_replay_parse.rs @@ -98,8 +98,10 @@ pub(super) fn parse_put_record( if let Some((collection, key, value)) = decode_kv_put_event_fields(payload) { *sequence += 1; let key_str = String::from_utf8_lossy(&key); + // The same `{key, value}` row image the live KV write event carries. + let row = nodedb_query::msgpack_scan::kv_row_msgpack(&key_str, &value); let (system_time_ms, valid_time_ms) = - crate::event::bitemporal_extract::extract_stamps(Some(&value)); + crate::event::bitemporal_extract::extract_stamps(Some(&row)); // AUDIT_DML rows replayed from WAL after a crash carry user_id = None and // statement_digest = None; pre-crash audit rows are durable in the catalog. // Widening the WAL record format to carry these fields is tracked separately. @@ -114,7 +116,7 @@ pub(super) fn parse_put_record( tenant_id, vshard_id, source: EventSource::User, - new_value: Some(Arc::from(value.as_slice())), + new_value: Some(Arc::from(row.as_slice())), old_value: None, system_time_ms, valid_time_ms, diff --git a/nodedb/tests/inproc/cases/cdc_arc_fanout.rs b/nodedb/tests/inproc/cases/cdc_arc_fanout.rs index c9de3bbed..7c8bc9194 100644 --- a/nodedb/tests/inproc/cases/cdc_arc_fanout.rs +++ b/nodedb/tests/inproc/cases/cdc_arc_fanout.rs @@ -127,6 +127,7 @@ fn buffer_composite_read_shares_event_allocation_across_polls() { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: nodedb::event::EventSource::User, }; buf.push(ev); @@ -173,6 +174,7 @@ fn buffer_partition_read_shares_event_allocation() { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: nodedb::event::EventSource::User, }; buf.push(ev); diff --git a/nodedb/tests/inproc/cases/event_cdc.rs b/nodedb/tests/inproc/cases/event_cdc.rs index 8b356a42e..3dd1525ed 100644 --- a/nodedb/tests/inproc/cases/event_cdc.rs +++ b/nodedb/tests/inproc/cases/event_cdc.rs @@ -153,6 +153,7 @@ fn log_compaction_keeps_latest_per_key() { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: nodedb::event::EventSource::User, }); buf.push(CdcEvent { sequence: 2, @@ -170,6 +171,7 @@ fn log_compaction_keeps_latest_per_key() { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: nodedb::event::EventSource::User, }); // Before compaction: both events present. diff --git a/nodedb/tests/inproc/cases/event_topics.rs b/nodedb/tests/inproc/cases/event_topics.rs index 09c4894c9..76668781a 100644 --- a/nodedb/tests/inproc/cases/event_topics.rs +++ b/nodedb/tests/inproc/cases/event_topics.rs @@ -66,6 +66,7 @@ fn topic_buffer_publish_and_consume() { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: nodedb::event::EventSource::User, }); } @@ -106,6 +107,7 @@ fn topic_retention_eviction() { field_diffs: None, system_time_ms: None, valid_time_ms: None, + source: nodedb::event::EventSource::User, }); } diff --git a/nodedb/tests/inproc/cases/system_task_call_sites.rs b/nodedb/tests/inproc/cases/system_task_call_sites.rs index 62a4779da..785e467e7 100644 --- a/nodedb/tests/inproc/cases/system_task_call_sites.rs +++ b/nodedb/tests/inproc/cases/system_task_call_sites.rs @@ -30,12 +30,13 @@ const ALLOWED: &[&str] = &[ "engine/timeseries/retention_policy/autowire.rs", "engine/timeseries/retention_policy/enforcement.rs", "engine/bitemporal/enforcement.rs", - // Backup capture and restore reissue. + // Backup capture and restore reissue. The COPY handler authorizes the + // statement once, against the tenant's BACKUP permission. The rows it then + // captures or re-issues are the whole tenant, not one user's view: a + // per-collection grant or an RLS write policy must not drop a restored + // row. `durable.rs` is the single-node re-issue of every non-redo engine. "control/backup/orchestrator.rs", - "control/backup/restore/orchestrate/restore.rs", - "control/backup/restore/columnar_reissue.rs", - "control/backup/restore/timeseries_reissue.rs", - "control/backup/restore/vector_reissue.rs", + "control/backup/restore/durable.rs", // Cluster snapshot transfer. "control/cluster/snapshot_builder.rs", "control/cluster/snapshot_applier.rs", diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index e6eecc88c..681b4bc07 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -172,6 +172,7 @@ mod sql_backup_restore_durable_marks; mod sql_backup_restore_local_marks; mod sql_backup_restore_staleness; mod sql_backup_restore_timeseries; +mod sql_backup_restore_triggers; mod sql_backup_restore_vector_params; mod sql_backup_restore_vector_restart; mod sql_backup_restore_wire; diff --git a/nodedb/tests/wire/cases/sql_backup_restore_triggers.rs b/nodedb/tests/wire/cases/sql_backup_restore_triggers.rs new file mode 100644 index 000000000..888980b78 --- /dev/null +++ b/nodedb/tests/wire/cases/sql_backup_restore_triggers.rs @@ -0,0 +1,159 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A RESTORE re-issues rows without firing AFTER triggers. +//! +//! A trigger fired when its row was first written. The restored row carries +//! the `restore` event source, so neither an ASYNC nor a DEFERRED trigger +//! fires for it again. +//! +//! Each test ends with a sentinel write on another collection, whose own +//! trigger writes a `sentinel` marker. The server runs one core, and the +//! Event Plane handles that core's events in order. So once the sentinel +//! marker lands, any trigger the restore fired has landed too. + +use std::time::Duration; + +use crate::harness::TestServer; + +const TENANT: u64 = 1; + +/// How long to wait for an asynchronous trigger to land. +const MARKER_TIMEOUT: Duration = Duration::from_secs(30); + +/// Poll `audit` until it holds `marker` at least once. Fails once +/// `MARKER_TIMEOUT` passes. +async fn wait_for_marker(server: &TestServer, marker: &str) { + let deadline = tokio::time::Instant::now() + MARKER_TIMEOUT; + loop { + let rows = markers(server, marker).await; + if !rows.is_empty() { + return; + } + assert!( + tokio::time::Instant::now() < deadline, + "timed out waiting for the '{marker}' marker in audit" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } +} + +/// Every `audit` row carrying `marker`. +async fn markers(server: &TestServer, marker: &str) -> Vec { + server + .query_text(&format!( + "SELECT marker FROM audit WHERE marker = '{marker}'" + )) + .await + .unwrap_or_else(|e| panic!("read audit markers: {e}")) +} + +async fn exec_all(server: &TestServer, statements: &[&str]) { + for sql in statements { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } +} + +async fn backup_and_restore(server: &TestServer) { + let backup = super::backup_support::drain_backup(&server.client, TENANT) + .await + .expect("BACKUP TENANT"); + super::backup_support::push_restore(&server.client, TENANT, backup) + .await + .expect("RESTORE with no write after the backup"); +} + +/// A KV row restores through the durable re-issue path. Its ASYNC AFTER +/// trigger does not fire again. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_restored_kv_row_fires_no_async_after_trigger() { + let server = TestServer::start().await; + exec_all( + &server, + &[ + "CREATE COLLECTION audit", + "CREATE COLLECTION kv_src (key STRING PRIMARY KEY, value STRING) WITH (engine='kv')", + "CREATE COLLECTION kv_sentinel (key STRING PRIMARY KEY, value STRING) \ + WITH (engine='kv')", + "CREATE TRIGGER kv_src_audit AFTER INSERT OR UPDATE ON kv_src FOR EACH ROW \ + BEGIN INSERT INTO audit (marker) VALUES ('fired'); END;", + "CREATE TRIGGER kv_sentinel_audit AFTER INSERT ON kv_sentinel FOR EACH ROW \ + BEGIN INSERT INTO audit (marker) VALUES ('sentinel'); END;", + "INSERT INTO kv_src (key, value) VALUES ('a', 'x')", + ], + ) + .await; + wait_for_marker(&server, "fired").await; + + backup_and_restore(&server).await; + + exec_all( + &server, + &["INSERT INTO kv_sentinel (key, value) VALUES ('s', 'x')"], + ) + .await; + wait_for_marker(&server, "sentinel").await; + + assert_eq!( + markers(&server, "fired").await.len(), + 1, + "only the original insert fires the trigger, the restore does not" + ); + let restored = server + .query_text("SELECT key FROM kv_src") + .await + .expect("read the restored rows"); + assert_eq!(restored, vec!["a".to_string()]); +} + +/// A document row restores through the redo re-issue path. Its DEFERRED +/// AFTER trigger does not fire again. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_restored_document_row_fires_no_deferred_after_trigger() { + let server = TestServer::start().await; + exec_all( + &server, + &[ + "CREATE COLLECTION audit", + "CREATE COLLECTION doc_src (id TEXT PRIMARY KEY, v INT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION doc_sentinel (id TEXT PRIMARY KEY, v INT) \ + WITH (engine='document_strict')", + "CREATE DEFERRED TRIGGER doc_src_audit AFTER INSERT OR UPDATE ON doc_src \ + FOR EACH ROW BEGIN INSERT INTO audit (marker) VALUES ('fired'); END;", + "CREATE DEFERRED TRIGGER doc_sentinel_audit AFTER INSERT ON doc_sentinel \ + FOR EACH ROW BEGIN INSERT INTO audit (marker) VALUES ('sentinel'); END;", + "BEGIN", + "INSERT INTO doc_src (id, v) VALUES ('a', 1)", + "COMMIT", + ], + ) + .await; + wait_for_marker(&server, "fired").await; + + backup_and_restore(&server).await; + + exec_all( + &server, + &[ + "BEGIN", + "INSERT INTO doc_sentinel (id, v) VALUES ('s', 1)", + "COMMIT", + ], + ) + .await; + wait_for_marker(&server, "sentinel").await; + + assert_eq!( + markers(&server, "fired").await.len(), + 1, + "only the original insert fires the trigger, the restore does not" + ); + let restored = server + .query_text("SELECT id FROM doc_src") + .await + .expect("read the restored rows"); + assert_eq!(restored, vec!["a".to_string()]); +} From 41850d3a30bcccbbab921c8197793b3c5d561b52 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 12:55:14 +0800 Subject: [PATCH 38/64] feat(events): add REMOVE EVENT and read definitions off the Event Plane REMOVE EVENT ON drops a DEFINE EVENT definition, replicated and transaction-staged the same way DEFINE EVENT is; an undefined name errors with SQLSTATE 42704. The Event Plane previously read a collection's event definitions from redb on every write event. It now reads an in-memory index that the catalog keeps in step with every committed collection write (put, put-if-absent, delete, migration), so the Event Plane never touches storage. --- nodedb/src/control/event_trigger.rs | 15 +- .../control/security/catalog/collections.rs | 81 +++++- .../security/catalog/event_defs_index.rs | 238 ++++++++++++++++++ nodedb/src/control/security/catalog/mod.rs | 1 + .../security/catalog/system_catalog.rs | 7 + .../server/shared/ddl/neutral/field_def.rs | 75 ++++++ .../ddl/neutral/router/string_engine_ops.rs | 8 +- nodedb/tests/wire/cases/mod.rs | 1 + .../wire/cases/sql_define_event_lifecycle.rs | 103 ++++++++ 9 files changed, 514 insertions(+), 15 deletions(-) create mode 100644 nodedb/src/control/security/catalog/event_defs_index.rs create mode 100644 nodedb/tests/wire/cases/sql_define_event_lifecycle.rs diff --git a/nodedb/src/control/event_trigger.rs b/nodedb/src/control/event_trigger.rs index e96e511d9..717f1198c 100644 --- a/nodedb/src/control/event_trigger.rs +++ b/nodedb/src/control/event_trigger.rs @@ -55,22 +55,17 @@ pub async fn process_write_event( return; } - let catalog = shared.credentials.catalog(); - let coll = match catalog.get_collection( + // The committed definitions, from memory: the Event Plane reads no redb. + let Some(event_defs) = shared.credentials.catalog().event_definitions( event.database_id, event.tenant_id.as_u64(), &event.collection, - ) { - Ok(Some(collection)) => collection, - _ => return, - }; - - if coll.event_defs.is_empty() { + ) else { return; - } + }; let op_str = event_operation(event.op); - for (index, event_def) in coll.event_defs.iter().enumerate() { + for (index, event_def) in event_defs.iter().enumerate() { let when_upper = event_def.when_condition.to_uppercase(); let matches = match when_upper.as_str() { "INSERT" => matches!(event.op, WriteOp::Insert | WriteOp::BulkInsert { .. }), diff --git a/nodedb/src/control/security/catalog/collections.rs b/nodedb/src/control/security/catalog/collections.rs index 426b92c62..0cfe2bbd9 100644 --- a/nodedb/src/control/security/catalog/collections.rs +++ b/nodedb/src/control/security/catalog/collections.rs @@ -119,7 +119,9 @@ impl SystemCatalog { .insert((database_id.as_u64(), inner_key.as_str()), bytes.as_slice()) .map_err(|e| catalog_err("insert collection", e))?; } - write_txn.commit().map_err(|e| catalog_err("commit", e)) + write_txn.commit().map_err(|e| catalog_err("commit", e))?; + self.event_defs.install(database_id, coll); + Ok(()) } /// Insert a collection only when its catalog key is absent. @@ -157,6 +159,9 @@ impl SystemCatalog { } }; write_txn.commit().map_err(|e| catalog_err("commit", e))?; + if inserted { + self.event_defs.install(database_id, coll); + } Ok(inserted) } @@ -254,6 +259,9 @@ impl SystemCatalog { .is_some(); } write_txn.commit().map_err(|e| catalog_err("commit", e))?; + if removed { + self.event_defs.remove(database_id, tenant_id, name); + } Ok(removed) } @@ -398,7 +406,9 @@ impl SystemCatalog { } write_txn .commit() - .map_err(|e| catalog_err("migrate_collections commit", e)) + .map_err(|e| catalog_err("migrate_collections commit", e))?; + // The migration wrote rows outside `put_collection`. + self.reload_event_definitions() } } @@ -493,6 +503,73 @@ mod tests { assert_eq!(fetched.unwrap().name, "users"); } + fn with_event(tenant_id: u64, name: &str) -> StoredCollection { + let mut c = make_coll(tenant_id, name); + c.event_defs = vec![super::super::collection_constraints::EventDefinition { + name: "ev".into(), + collection: name.into(), + when_condition: "INSERT".into(), + then_action: "SELECT 1".into(), + }]; + c + } + + #[test] + fn committed_writes_keep_the_event_index_in_step() { + let (_dir, cat) = open_catalog(); + let db = DatabaseId::DEFAULT; + cat.put_collection(db, &with_event(1, "orders")).unwrap(); + assert_eq!( + cat.event_definitions(db, 1, "orders").map(|d| d.len()), + Some(1) + ); + + cat.put_collection(db, &make_coll(1, "orders")).unwrap(); + assert!(cat.event_definitions(db, 1, "orders").is_none()); + + cat.put_collection(db, &with_event(1, "orders")).unwrap(); + assert!(cat.delete_collection(db, 1, "orders").unwrap()); + assert!(cat.event_definitions(db, 1, "orders").is_none()); + } + + #[test] + fn a_skipped_insert_leaves_the_event_index_unchanged() { + let (_dir, cat) = open_catalog(); + let db = DatabaseId::DEFAULT; + cat.put_collection(db, &make_coll(1, "orders")).unwrap(); + assert!( + !cat.put_collection_if_absent(db, &with_event(1, "orders")) + .unwrap() + ); + assert!(cat.event_definitions(db, 1, "orders").is_none()); + } + + #[test] + fn a_failed_write_leaves_the_event_index_unchanged() { + let (_dir, cat) = open_catalog(); + let db = DatabaseId::DEFAULT; + cat.fail_next_collection_write_for_test(); + assert!(cat.put_collection(db, &with_event(1, "orders")).is_err()); + assert!(cat.event_definitions(db, 1, "orders").is_none()); + } + + #[test] + fn reopening_the_catalog_loads_the_event_index() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + { + let cat = SystemCatalog::open(&path).unwrap(); + cat.put_collection(DatabaseId::DEFAULT, &with_event(1, "orders")) + .unwrap(); + } + let cat = SystemCatalog::open(&path).unwrap(); + assert_eq!( + cat.event_definitions(DatabaseId::DEFAULT, 1, "orders") + .map(|d| d.len()), + Some(1) + ); + } + #[test] fn missing_returns_none() { let (_dir, cat) = open_catalog(); diff --git a/nodedb/src/control/security/catalog/event_defs_index.rs b/nodedb/src/control/security/catalog/event_defs_index.rs new file mode 100644 index 000000000..35116e546 --- /dev/null +++ b/nodedb/src/control/security/catalog/event_defs_index.rs @@ -0,0 +1,238 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! In-memory index of each collection's DEFINE EVENT definitions. +//! +//! The Event Plane reads a collection's event definitions for every write +//! event. It must not read redb, so it reads this index instead. +//! +//! [`SystemCatalog`](super::system_catalog::SystemCatalog) owns the index and +//! keeps it in step with the committed `COLLECTIONS` table: +//! - `open` and `migrate_collections` rebuild it from every committed +//! collection row (`reload_event_definitions`); +//! - `put_collection` and `put_collection_if_absent` install a row after the +//! redb commit succeeds; +//! - `delete_collection` removes a row after the redb commit succeeds. +//! +//! Every collection writer goes through those functions: the replicated +//! catalog apply, the single-node fallback, transaction COMMIT, catalog +//! restore, and maintenance writers. DDL a transaction stages lives in the +//! catalog overlay and never reaches the table before COMMIT, so the index +//! holds committed definitions only. + +use std::collections::HashMap; +use std::sync::{Arc, RwLock}; + +use nodedb_types::DatabaseId; + +use redb::{ReadableDatabase, ReadableTable}; + +use super::collection::StoredCollection; +use super::collection_constraints::EventDefinition; +use super::system_catalog::SystemCatalog; +use super::tables::COLLECTIONS; +use super::types::catalog_err; + +/// `(database, tenant, collection)`. +type IndexKey = (DatabaseId, u64, String); + +/// Committed event definitions by collection. Send + Sync. A reader clones +/// the definitions out and holds no lock afterwards. +#[derive(Debug, Default)] +pub struct EventDefsIndex { + by_collection: RwLock>>, +} + +impl EventDefsIndex { + pub fn new() -> Self { + Self::default() + } + + /// Replace the whole index with the definitions of `rows`, each keyed + /// under the database its table row is stored in. + pub fn load_all(&self, rows: &[(DatabaseId, StoredCollection)]) { + let mut map = HashMap::new(); + for (database_id, row) in rows { + if let Some(defs) = active_defs(row) { + map.insert(key_of(*database_id, row), defs); + } + } + *self.write() = map; + } + + /// Record `row`, committed under `database_id`. An inactive row, or a + /// row with no event definitions, removes the collection's entry. + pub fn install(&self, database_id: DatabaseId, row: &StoredCollection) { + let key = key_of(database_id, row); + let mut map = self.write(); + match active_defs(row) { + Some(defs) => { + map.insert(key, defs); + } + None => { + map.remove(&key); + } + } + } + + /// Forget a collection whose row was deleted. + pub fn remove(&self, database_id: DatabaseId, tenant_id: u64, collection: &str) { + self.write() + .remove(&(database_id, tenant_id, collection.to_owned())); + } + + /// The committed event definitions of a collection. `None` when it has + /// none. + pub fn get( + &self, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, + ) -> Option> { + let map = self + .by_collection + .read() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + map.get(&(database_id, tenant_id, collection.to_owned())) + .cloned() + } + + fn write(&self) -> std::sync::RwLockWriteGuard<'_, HashMap>> { + self.by_collection + .write() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + } +} + +impl SystemCatalog { + /// Rebuild the event-definition index from every committed collection + /// row, keyed by the database each row is stored under. + pub fn reload_event_definitions(&self) -> crate::Result<()> { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("read txn", e))?; + let table = read_txn + .open_table(COLLECTIONS) + .map_err(|e| catalog_err("open collections", e))?; + let mut rows = Vec::new(); + for entry in table + .iter() + .map_err(|e| catalog_err("iterate collections", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("read collection", e))?; + let (database_id, _) = key.value(); + let row: StoredCollection = zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("deser collection", e))?; + rows.push((DatabaseId::new(database_id), row)); + } + self.event_defs.load_all(&rows); + Ok(()) + } + + /// The committed DEFINE EVENT definitions of a collection. Reads memory + /// only. `None` when the collection has none. + pub fn event_definitions( + &self, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, + ) -> Option> { + self.event_defs.get(database_id, tenant_id, collection) + } +} + +fn key_of(database_id: DatabaseId, row: &StoredCollection) -> IndexKey { + (database_id, row.tenant_id, row.name.clone()) +} + +/// The definitions an active row carries. `None` for an inactive row or an +/// empty list. +fn active_defs(row: &StoredCollection) -> Option> { + (row.is_active && !row.event_defs.is_empty()).then(|| Arc::from(row.event_defs.as_slice())) +} + +#[cfg(test)] +mod tests { + use super::*; + + const DB: DatabaseId = DatabaseId::DEFAULT; + + fn def(name: &str) -> EventDefinition { + EventDefinition { + name: name.into(), + collection: "orders".into(), + when_condition: "INSERT".into(), + then_action: "SELECT 1".into(), + } + } + + fn row(defs: Vec) -> StoredCollection { + let mut row = StoredCollection::new(7, "orders", "admin"); + row.event_defs = defs; + row + } + + fn names(index: &EventDefsIndex, row: &StoredCollection) -> Vec { + index + .get(DB, row.tenant_id, &row.name) + .map(|defs| defs.iter().map(|d| d.name.clone()).collect()) + .unwrap_or_default() + } + + #[test] + fn install_records_the_definitions() { + let index = EventDefsIndex::new(); + let r = row(vec![def("a")]); + index.install(DB, &r); + assert_eq!(names(&index, &r), vec!["a".to_string()]); + } + + #[test] + fn install_replaces_the_previous_definitions() { + let index = EventDefsIndex::new(); + index.install(DB, &row(vec![def("a")])); + let r = row(vec![def("b"), def("c")]); + index.install(DB, &r); + assert_eq!(names(&index, &r), vec!["b".to_string(), "c".to_string()]); + } + + #[test] + fn a_row_with_no_definitions_removes_the_entry() { + let index = EventDefsIndex::new(); + index.install(DB, &row(vec![def("a")])); + let r = row(Vec::new()); + index.install(DB, &r); + assert!(index.get(DB, r.tenant_id, &r.name).is_none()); + } + + #[test] + fn a_dropped_collection_fires_no_event() { + let index = EventDefsIndex::new(); + index.install(DB, &row(vec![def("a")])); + let mut dropped = row(vec![def("a")]); + dropped.is_active = false; + index.install(DB, &dropped); + assert!(index.get(DB, dropped.tenant_id, &dropped.name).is_none()); + } + + #[test] + fn remove_forgets_a_purged_collection() { + let index = EventDefsIndex::new(); + let r = row(vec![def("a")]); + index.install(DB, &r); + index.remove(DB, r.tenant_id, &r.name); + assert!(index.get(DB, r.tenant_id, &r.name).is_none()); + } + + #[test] + fn load_all_skips_inactive_rows() { + let index = EventDefsIndex::new(); + let live = row(vec![def("a")]); + let mut gone = StoredCollection::new(7, "gone", "admin"); + gone.event_defs = vec![def("x")]; + gone.is_active = false; + index.load_all(&[(DB, live.clone()), (DB, gone.clone())]); + assert_eq!(names(&index, &live), vec!["a".to_string()]); + assert!(index.get(DB, gone.tenant_id, &gone.name).is_none()); + } +} diff --git a/nodedb/src/control/security/catalog/mod.rs b/nodedb/src/control/security/catalog/mod.rs index b16453296..e316a06a8 100644 --- a/nodedb/src/control/security/catalog/mod.rs +++ b/nodedb/src/control/security/catalog/mod.rs @@ -28,6 +28,7 @@ pub mod database_grants; pub mod database_quotas; pub mod database_types; pub mod dependencies; +pub mod event_defs_index; pub mod function_types; pub mod functions; pub mod index_record; diff --git a/nodedb/src/control/security/catalog/system_catalog.rs b/nodedb/src/control/security/catalog/system_catalog.rs index 45525a49b..efa13ba3a 100644 --- a/nodedb/src/control/security/catalog/system_catalog.rs +++ b/nodedb/src/control/security/catalog/system_catalog.rs @@ -21,6 +21,9 @@ use super::types::*; pub struct SystemCatalog { pub(super) db: Arc, pub(super) crdt_signing_root: Arc>>, + /// Committed DEFINE EVENT definitions, for readers that must not read + /// redb. + pub(super) event_defs: Arc, #[cfg(test)] pub(super) fail_next_user_counter_write: Arc, #[cfg(test)] @@ -47,6 +50,7 @@ impl SystemCatalog { let catalog = Self { db: Arc::new(db), crdt_signing_root: Arc::new(std::sync::RwLock::new(None)), + event_defs: Arc::new(super::event_defs_index::EventDefsIndex::new()), #[cfg(test)] fail_next_user_counter_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), #[cfg(test)] @@ -55,6 +59,7 @@ impl SystemCatalog { fail_next_collection_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), }; catalog.bootstrap_default_database()?; + catalog.reload_event_definitions()?; Ok(catalog) } @@ -69,6 +74,7 @@ impl SystemCatalog { let catalog = Self { db: Arc::new(db), crdt_signing_root: Arc::new(std::sync::RwLock::new(None)), + event_defs: Arc::new(super::event_defs_index::EventDefsIndex::new()), #[cfg(test)] fail_next_user_counter_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), #[cfg(test)] @@ -77,6 +83,7 @@ impl SystemCatalog { fail_next_collection_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), }; catalog.bootstrap_default_database()?; + catalog.reload_event_definitions()?; Ok(catalog) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/field_def.rs b/nodedb/src/control/server/shared/ddl/neutral/field_def.rs index c16f5947e..2be5dd838 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/field_def.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/field_def.rs @@ -8,6 +8,8 @@ //! reads (VALUE computed fields). //! - `DEFINE EVENT ON WHEN THEN ` — //! stores an event definition in the catalog. +//! - `REMOVE EVENT ON ` — removes an event definition from +//! the catalog. //! //! Handlers build [`DdlResult`] directly and carry no pgwire wire types. @@ -251,3 +253,76 @@ pub fn define_event( rows_affected: None, }]) } + +/// Parse and apply a REMOVE EVENT statement. +/// +/// Syntax: REMOVE EVENT ON +/// +/// The collection descriptor is replicated without the definition, the same +/// way DEFINE EVENT replicates it with one. Inside a transaction the change +/// is held for COMMIT, as DEFINE EVENT's is. An undefined name is an error +/// with SQLSTATE 42704. +pub fn remove_event( + state: &SharedState, + identity: &AuthenticatedIdentity, + database_id: DatabaseId, + sql: &str, +) -> Result, DdlError> { + let parts: Vec<&str> = sql + .trim() + .trim_end_matches(';') + .split_whitespace() + .collect(); + if parts.len() != 5 || !parts[3].eq_ignore_ascii_case("ON") { + return Err(err("42601", "syntax: REMOVE EVENT ON ")); + } + let event_name = parse_ident_token(parts[2])?; + let collection = parse_ident_token(parts[4])?; + let tenant_id = identity.tenant_id; + + let audit = ArcAuditEmitter(std::sync::Arc::clone(&state.audit)); + authorize_collection( + identity, + database_id, + &collection, + Permission::Alter, + &state.permissions, + &state.roles, + &audit, + ) + .map_err(|error| err("42501", &format!("permission denied: {}", error.resource())))?; + + let catalog = state.credentials.catalog(); + let mut coll = match catalog.get_collection(database_id, tenant_id.as_u64(), &collection) { + Ok(Some(coll)) => coll, + Ok(None) => { + return Err(err( + "42P01", + &format!("collection '{collection}' does not exist"), + )); + } + Err(e) => return Err(err("XX000", &format!("read collection: {e}"))), + }; + let before = coll.event_defs.len(); + coll.event_defs.retain(|e| e.name != event_name); + if coll.event_defs.len() == before { + return Err(err( + "42704", + &format!("event '{event_name}' on '{collection}' does not exist"), + )); + } + crate::control::catalog_entry::persist_collection_replicated(state, database_id, &coll) + .map_err(|e| err("XX000", &format!("save collection: {e}")))?; + + state.audit_record( + crate::control::security::audit::AuditEvent::AdminAction, + Some(tenant_id), + &identity.username, + &format!("removed event '{event_name}' from '{collection}'"), + ); + + Ok(vec![DdlResult::Status { + command: "REMOVE EVENT".to_string(), + rows_affected: None, + }]) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs index afca89e3f..9b08071a7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs @@ -330,15 +330,17 @@ pub(super) async fn try_string( return Some(estimate_count::estimate_count(state, identity, database_id, sql).await); } - // `DEFINE FIELD …` / `DEFINE EVENT …` — string-recognized (no typed DDL - // variant); the pgwire schema string router dispatched both from the raw - // SQL. Replicate that exactly here, before the parse gate. + // `DEFINE FIELD …` / `DEFINE EVENT …` / `REMOVE EVENT …` — + // string-recognized (no typed DDL variant), before the parse gate. if upper.starts_with("DEFINE FIELD ") { return Some(field_def::define_field(state, identity, database_id, sql)); } if upper.starts_with("DEFINE EVENT ") { return Some(field_def::define_event(state, identity, database_id, sql)); } + if upper.starts_with("REMOVE EVENT ") { + return Some(field_def::remove_event(state, identity, database_id, sql)); + } // `EXPLAIN TIERS ON [RANGE …]` — string-recognized (no typed // DDL variant); the pgwire admin string router dispatched it from the raw diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 681b4bc07..137ff2da7 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -189,6 +189,7 @@ mod sql_default_declared_types; mod sql_default_expressions; mod sql_default_vector_primary; mod sql_default_volatility; +mod sql_define_event_lifecycle; mod sql_division_by_zero; mod sql_division_by_zero_composite; mod sql_dml_affected_counts; diff --git a/nodedb/tests/wire/cases/sql_define_event_lifecycle.rs b/nodedb/tests/wire/cases/sql_define_event_lifecycle.rs new file mode 100644 index 000000000..c1f4c6f6e --- /dev/null +++ b/nodedb/tests/wire/cases/sql_define_event_lifecycle.rs @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! DEFINE EVENT and REMOVE EVENT, end to end. +//! +//! A defined event's THEN action runs for each matching write. After REMOVE +//! EVENT, a write runs no action. Removing an undefined event is an error +//! with SQLSTATE 42704. +//! +//! The server runs one core, and the Event Plane handles that core's events +//! in order. A sentinel write, whose own event logs `sentinel`, lands after +//! any action an earlier write started. + +use std::time::Duration; + +use crate::harness::TestServer; + +/// How long to wait for an event action to land. +const ACTION_TIMEOUT: Duration = Duration::from_secs(30); + +/// Poll `ev_log` until it holds a row with id `id`. Fails once +/// `ACTION_TIMEOUT` passes. +async fn wait_for_log(server: &TestServer, id: &str) { + let deadline = tokio::time::Instant::now() + ACTION_TIMEOUT; + loop { + if logged(server, id).await { + return; + } + assert!( + tokio::time::Instant::now() < deadline, + "timed out waiting for '{id}' in ev_log" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } +} + +async fn logged(server: &TestServer, id: &str) -> bool { + let rows = server + .query_text(&format!("SELECT id FROM ev_log WHERE id = '{id}'")) + .await + .unwrap_or_else(|e| panic!("read ev_log: {e}")); + !rows.is_empty() +} + +async fn exec_all(server: &TestServer, statements: &[&str]) { + for sql in statements { + server + .exec(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_removed_event_runs_no_action() { + let server = TestServer::start().await; + exec_all( + &server, + &[ + "CREATE COLLECTION ev_src", + "CREATE COLLECTION ev_sentinel", + "CREATE COLLECTION ev_log", + "DEFINE EVENT log_insert ON ev_src WHEN INSERT \ + THEN INSERT INTO ev_log (id) VALUES ($document_id)", + "DEFINE EVENT log_sentinel ON ev_sentinel WHEN INSERT \ + THEN INSERT INTO ev_log (id) VALUES ('sentinel')", + "INSERT INTO ev_src (id, v) VALUES ('a', 1)", + ], + ) + .await; + wait_for_log(&server, "a").await; + + exec_all( + &server, + &[ + "REMOVE EVENT log_insert ON ev_src", + "INSERT INTO ev_src (id, v) VALUES ('b', 2)", + "INSERT INTO ev_sentinel (id, v) VALUES ('s', 1)", + ], + ) + .await; + wait_for_log(&server, "sentinel").await; + + assert!( + !logged(&server, "b").await, + "a write after REMOVE EVENT must run no action" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn removing_an_undefined_event_is_42704() { + let server = TestServer::start().await; + exec_all(&server, &["CREATE COLLECTION ev_none"]).await; + let error = server + .client + .simple_query("REMOVE EVENT missing ON ev_none") + .await + .expect_err("an undefined event cannot be removed"); + let code = error + .as_db_error() + .map(|db| db.code().code().to_owned()) + .unwrap_or_default(); + assert_eq!(code, "42704", "unexpected error: {error}"); +} From 1d7dd58972df8ca407fec0addb53ac0e4e8b42ff Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 14:34:45 +0800 Subject: [PATCH 39/64] feat(events): stamp WAL row writes with their event source The WAL record header grows to 55 bytes (format v2) with a one-byte event source field alongside the existing apply key, so a row-write record stores the source its write ran with instead of relying on callers to reconstruct it at replay time. WalRecord::new_stamped and RecordStamp replace new_keyed as the way callers set both the apply key and event source together. WalAppender gains with_event_source and refuses to append a row-write record (Put, Delete, TransactionRedo, graph node-label) with no source named, so no row write is ever stored without one. All WAL dispatch, executor replay, calvin scheduler, backup restore and DDL call sites now name the event source of the write they append. WAL-driven event replay (replay_wal_to_events) rebuilds each event with the source its record carries instead of a fixed default, covering raw Put/Delete records and TransactionRedo's per-row-kind sources through the new ReplayScope/RowSources types. A restored, trigger, CRDT-sync or deferred row therefore keeps its source across a WAL replay the same way it does on its live write. --- nodedb-wal/src/lazy_reader.rs | 3 +- nodedb-wal/src/lib.rs | 6 +- nodedb-wal/src/mmap_reader/reader.rs | 2 + nodedb-wal/src/reader.rs | 1 + nodedb-wal/src/record/header.rs | 54 ++- nodedb-wal/src/record/mod.rs | 5 +- nodedb-wal/src/record/wal_record.rs | 80 ++++- nodedb-wal/src/segmented.rs | 4 +- nodedb-wal/src/writer/core.rs | 12 +- nodedb/src/control/backup/restore/durable.rs | 7 +- .../post_apply/async_dispatch/vector.rs | 9 +- .../scheduler/driver/core/commit_redo.rs | 1 + .../scheduler/driver/core/redo_window.rs | 1 + .../calvin/scheduler/driver/core/request.rs | 8 +- .../cluster/calvin/scheduler/recovery.rs | 2 + .../distributed_applier/proposal_ledger.rs | 4 + nodedb/src/control/orchestrated_write.rs | 6 +- .../procedural/executor/core/dispatch.rs | 20 +- .../control/server/dispatch_utils/dispatch.rs | 1 + .../server/dispatch_utils/minted/owned.rs | 1 + .../server/dispatch_utils/minted/records.rs | 8 +- .../server/dispatch_utils/minted/resolve.rs | 1 + .../submit_write/funnel/driver.rs | 27 +- .../submit_write/funnel/response.rs | 9 +- .../submit_write/funnel/wal_append.rs | 2 + .../ddl/neutral/collection/index/teardown.rs | 8 +- .../shared/ddl/neutral/graph_ops/edge.rs | 8 +- .../ddl/neutral/materialized_view/refresh.rs | 8 +- .../ddl/neutral/tree_ops/create_index.rs | 8 +- .../src/control/server/wal_dispatch/core.rs | 15 +- .../src/control/server/wal_dispatch/graph.rs | 12 +- .../control/server/wal_dispatch/timeseries.rs | 2 + .../server/wal_dispatch/write_set_redo.rs | 12 +- .../src/data/executor/core_loop/deferred.rs | 60 +--- nodedb/src/data/executor/wal_replay/kv_put.rs | 1 + .../src/data/executor/wal_replay_kv_expiry.rs | 2 + .../src/data/executor/wal_replay_kv_incr.rs | 1 + .../executor/wal_replay_kv_insert_conflict.rs | 1 + nodedb/src/data/executor/wal_replay_kv_ttl.rs | 2 + nodedb/src/event/consumer/run.rs | 2 + nodedb/src/event/mod.rs | 1 + nodedb/src/event/types.rs | 82 +++++ nodedb/src/event/wal_replay.rs | 331 ++++++++++++------ nodedb/src/event/wal_replay_parse.rs | 71 ++-- nodedb/src/event/wal_replay_scope.rs | 69 ++++ nodedb/src/wal/manager/append.rs | 12 +- nodedb/src/wal/manager/append_index.rs | 4 +- nodedb/src/wal/manager/append_transaction.rs | 4 +- nodedb/src/wal/manager/appender.rs | 91 ++++- nodedb/src/wal/manager/durable_commit.rs | 3 + nodedb/src/wal/manager/encryption.rs | 4 + nodedb/src/wal/manager/ops.rs | 5 + .../inproc/cases/event_wal_replay_source.rs | 62 ++++ nodedb/tests/inproc/cases/mod.rs | 1 + 54 files changed, 879 insertions(+), 277 deletions(-) create mode 100644 nodedb/src/event/wal_replay_scope.rs create mode 100644 nodedb/tests/inproc/cases/event_wal_replay_source.rs diff --git a/nodedb-wal/src/lazy_reader.rs b/nodedb-wal/src/lazy_reader.rs index 72d916ca4..bb4316480 100644 --- a/nodedb-wal/src/lazy_reader.rs +++ b/nodedb-wal/src/lazy_reader.rs @@ -121,7 +121,7 @@ impl LazyWalReader { Ok(None) } - /// Read the next record header (54 bytes) without reading the payload. + /// Read the next record header without reading the payload. /// /// Returns `None` at EOF or first corruption. After this call, use /// either `read_payload()` to get the payload or `skip_payload()` to @@ -461,6 +461,7 @@ mod tests { payload_len: (MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, apply_key: 0, + event_source: crate::record::NO_EVENT_SOURCE, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/lib.rs b/nodedb-wal/src/lib.rs index 8ba0ccaa0..bb41def0d 100644 --- a/nodedb-wal/src/lib.rs +++ b/nodedb-wal/src/lib.rs @@ -59,9 +59,9 @@ pub use preamble::{ }; pub use reader::{StopReason, WalReader}; pub use record::{ - CalvinAppliedPayload, FtsDeletePayload, FtsIndexPayload, RecordHeader, RecordTarget, - RecordType, SpatialDeletePayload, SpatialPutPayload, WalRecord, WalRecordArgs, - WriteAbortedPayload, + CalvinAppliedPayload, FtsDeletePayload, FtsIndexPayload, NO_EVENT_SOURCE, RecordHeader, + RecordStamp, RecordTarget, RecordType, SpatialDeletePayload, SpatialPutPayload, WalRecord, + WalRecordArgs, WriteAbortedPayload, }; pub use recovery::{RecoveryInfo, recover}; pub use replay::{ diff --git a/nodedb-wal/src/mmap_reader/reader.rs b/nodedb-wal/src/mmap_reader/reader.rs index 893495bb0..6a6041aa0 100644 --- a/nodedb-wal/src/mmap_reader/reader.rs +++ b/nodedb-wal/src/mmap_reader/reader.rs @@ -480,6 +480,7 @@ mod tests { payload_len: 1, database_id: 0, apply_key: 0, + event_source: crate::record::NO_EVENT_SOURCE, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); @@ -502,6 +503,7 @@ mod tests { payload_len: (crate::record::MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, apply_key: 0, + event_source: crate::record::NO_EVENT_SOURCE, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/reader.rs b/nodedb-wal/src/reader.rs index ff015be4c..f63b03205 100644 --- a/nodedb-wal/src/reader.rs +++ b/nodedb-wal/src/reader.rs @@ -472,6 +472,7 @@ mod tests { payload_len: (crate::record::MAX_WAL_PAYLOAD_SIZE + 1) as u32, database_id: 0, apply_key: 0, + event_source: crate::record::NO_EVENT_SOURCE, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/record/header.rs b/nodedb-wal/src/record/header.rs index 5c950a7cf..499d2cb24 100644 --- a/nodedb-wal/src/record/header.rs +++ b/nodedb-wal/src/record/header.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: Apache-2.0 -//! WAL record header: fixed 54-byte prefix + constants. +//! WAL record header: fixed 55-byte prefix + constants. use crate::error::{Result, WalError}; @@ -22,7 +22,10 @@ pub const WAL_MAGIC: u32 = 0x5359_4E57; // "SYNW" /// /// v1 is the initial shipped format with 54-byte headers (u64 tenant_id, /// u16 vshard_id, u32 payload_len, u16 reserved, u32 crc32c). -pub const WAL_FORMAT_VERSION: u16 = 1; +/// +/// v2 adds the one-byte event source at offset 50 and grows the header to +/// 55 bytes. A v1 record does not open. +pub const WAL_FORMAT_VERSION: u16 = 2; /// Maximum WAL record payload size (64 MiB). Distinct from cluster RPC's limit. pub const MAX_WAL_PAYLOAD_SIZE: usize = 64 * 1024 * 1024; @@ -31,13 +34,19 @@ pub const MAX_WAL_PAYLOAD_SIZE: usize = 64 * 1024 * 1024; /// /// Layout (all little-endian): /// magic(4) | format_version(2) | record_type(4) | lsn(8) | tenant_id(8) -/// | vshard_id(4) | payload_len(4) | database_id(8) | reserved(8) | crc32c(4) +/// | vshard_id(4) | payload_len(4) | database_id(8) | apply_key(8) +/// | event_source(1) | crc32c(4) /// /// `database_id` occupies bytes 34–41 (previously part of the 16-byte reserved /// field). `apply_key` occupies bytes 42–49. Bytes 34–41 were zero-filled in /// prior records, so `database_id == 0` maps to `DatabaseId(0)` (the default /// database), preserving backward compatibility without a format-version bump. -pub const HEADER_SIZE: usize = 54; +pub const HEADER_SIZE: usize = 55; + +/// The event source of a record that carries no row write. Replay rebuilds +/// no write event from it. A write record carries the code of the source its +/// write ran with; the WAL stores the code and does not interpret it. +pub const NO_EVENT_SOURCE: u8 = 0; /// Bit 14 in `record_type` signals the payload is AES-256-GCM encrypted. /// Separate from bit 15 (required flag). Both bits keep their positions; @@ -48,7 +57,7 @@ pub const ENCRYPTED_FLAG: u32 = 0x0000_4000; /// must not be silently skipped. pub const REQUIRED_FLAG: u32 = 0x0000_8000; -/// WAL record header (fixed 54 bytes). +/// WAL record header (fixed 55 bytes). #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct RecordHeader { pub magic: u32, @@ -70,6 +79,10 @@ pub struct RecordHeader { /// it applied from the records themselves. Covered by CRC32C. Occupies /// bytes 42–49. pub apply_key: u64, + /// The event source of the row write this record carries, as the writer's + /// code. [`NO_EVENT_SOURCE`] for a record that carries no row write. + /// Covered by CRC32C. Occupies byte 50. + pub event_source: u8, pub crc32c: u32, } @@ -85,7 +98,8 @@ impl RecordHeader { buf[30..34].copy_from_slice(&self.payload_len.to_le_bytes()); buf[34..42].copy_from_slice(&self.database_id.to_le_bytes()); buf[42..50].copy_from_slice(&self.apply_key.to_le_bytes()); - buf[50..54].copy_from_slice(&self.crc32c.to_le_bytes()); + buf[50] = self.event_source; + buf[51..55].copy_from_slice(&self.crc32c.to_le_bytes()); buf } @@ -108,7 +122,8 @@ impl RecordHeader { apply_key: u64::from_le_bytes([ buf[42], buf[43], buf[44], buf[45], buf[46], buf[47], buf[48], buf[49], ]), - crc32c: u32::from_le_bytes([buf[50], buf[51], buf[52], buf[53]]), + event_source: buf[50], + crc32c: u32::from_le_bytes([buf[51], buf[52], buf[53], buf[54]]), } } @@ -172,6 +187,7 @@ mod tests { payload_len: 100, database_id: 0, apply_key: 0, + event_source: NO_EVENT_SOURCE, crc32c: 0xDEAD_BEEF, } } @@ -184,11 +200,11 @@ mod tests { } #[test] - fn header_golden_54_bytes_exact_offsets() { + fn header_golden_55_bytes_exact_offsets() { // magic at 0..4, format_version at 4..6, record_type at 6..10, // lsn at 10..18, tenant_id at 18..26, vshard_id at 26..30, // payload_len at 30..34, database_id at 34..42, apply_key at 42..50, - // crc32c at 50..54. + // event_source at 50, crc32c at 51..55. let header = RecordHeader { magic: WAL_MAGIC, format_version: WAL_FORMAT_VERSION, @@ -199,10 +215,11 @@ mod tests { payload_len: 256, database_id: 0xABCD_0000_1234_5678, apply_key: 0, + event_source: NO_EVENT_SOURCE, crc32c: 0x1234_5678, }; let b = header.to_bytes(); - assert_eq!(b.len(), 54); + assert_eq!(b.len(), 55); // magic assert_eq!(&b[0..4], &WAL_MAGIC.to_le_bytes()); // format_version @@ -221,8 +238,10 @@ mod tests { assert_eq!(&b[34..42], &0xABCD_0000_1234_5678u64.to_le_bytes()); // apply_key — zero assert_eq!(&b[42..50], &[0u8; 8]); + // event_source + assert_eq!(b[50], NO_EVENT_SOURCE); // crc32c - assert_eq!(&b[50..54], &0x1234_5678u32.to_le_bytes()); + assert_eq!(&b[51..55], &0x1234_5678u32.to_le_bytes()); } #[test] @@ -238,6 +257,7 @@ mod tests { payload_len: 0, database_id: 7, apply_key: 0, + event_source: NO_EVENT_SOURCE, crc32c: 0, }; let bytes = header.to_bytes(); @@ -272,6 +292,7 @@ mod tests { payload_len: 0, database_id: 0, apply_key: 0, + event_source: NO_EVENT_SOURCE, crc32c: 0, }; let bytes = header.to_bytes(); @@ -356,4 +377,15 @@ mod tests { ); assert_eq!(decoded2.logical_record_type(), 0x0001_0001 | REQUIRED_FLAG); } + + #[test] + fn every_event_source_code_roundtrips() { + for code in 0..=u8::MAX { + let mut header = make_header(1, 0); + header.event_source = code; + let decoded = RecordHeader::from_bytes(&header.to_bytes()); + assert_eq!(decoded.event_source, code); + assert_eq!(decoded, header); + } + } } diff --git a/nodedb-wal/src/record/mod.rs b/nodedb-wal/src/record/mod.rs index b050277fd..df2d857f4 100644 --- a/nodedb-wal/src/record/mod.rs +++ b/nodedb-wal/src/record/mod.rs @@ -16,11 +16,12 @@ pub use anchor::{ANCHOR_PAYLOAD_SIZE, LsnMsAnchorPayload}; pub use calvin::CalvinAppliedPayload; pub use fts_spatial::{FtsDeletePayload, FtsIndexPayload, SpatialDeletePayload, SpatialPutPayload}; pub use header::{ - ENCRYPTED_FLAG, HEADER_SIZE, MAX_WAL_PAYLOAD_SIZE, RecordHeader, WAL_FORMAT_VERSION, WAL_MAGIC, + ENCRYPTED_FLAG, HEADER_SIZE, MAX_WAL_PAYLOAD_SIZE, NO_EVENT_SOURCE, RecordHeader, + WAL_FORMAT_VERSION, WAL_MAGIC, }; pub(crate) use padding::pad_buffer_to_alignment; pub use padding::{MIN_PADDING_RECORD_SIZE, padding_record, padding_span}; pub use surrogate::{SURROGATE_PAYLOAD_SIZE, SurrogateAllocPayload, SurrogateBindPayload}; pub use sync_seq::{SYNC_SEQ_ADVANCE_PAYLOAD_SIZE, SyncSeqAdvancePayload}; pub use types::RecordType; -pub use wal_record::{RecordTarget, WalRecord, WalRecordArgs}; +pub use wal_record::{RecordStamp, RecordTarget, WalRecord, WalRecordArgs}; diff --git a/nodedb-wal/src/record/wal_record.rs b/nodedb-wal/src/record/wal_record.rs index 85652fedc..e34facdc4 100644 --- a/nodedb-wal/src/record/wal_record.rs +++ b/nodedb-wal/src/record/wal_record.rs @@ -3,7 +3,8 @@ //! `WalRecord` — header + payload with encryption + checksum helpers. use super::header::{ - ENCRYPTED_FLAG, HEADER_SIZE, MAX_WAL_PAYLOAD_SIZE, RecordHeader, WAL_FORMAT_VERSION, WAL_MAGIC, + ENCRYPTED_FLAG, HEADER_SIZE, MAX_WAL_PAYLOAD_SIZE, NO_EVENT_SOURCE, RecordHeader, + WAL_FORMAT_VERSION, WAL_MAGIC, }; use crate::error::{Result, WalError}; use crate::preamble::PREAMBLE_SIZE; @@ -15,14 +16,35 @@ pub struct WalRecord { pub payload: Vec, } -/// The header fields of a record its caller decides: its type and scope. -/// The writer assigns the LSN. +/// The header fields of a record its caller decides: its type, scope and +/// event source. The writer assigns the LSN. #[derive(Debug, Clone, Copy)] pub struct RecordTarget { pub record_type: u32, pub tenant_id: u64, pub vshard_id: u32, pub database_id: u64, + /// The event source code of the row write the record carries. + /// [`NO_EVENT_SOURCE`] for a record that carries no row write. + pub event_source: u8, +} + +/// The header fields that tie a record to the write that appended it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RecordStamp { + /// The idempotency key of the replicated proposal whose apply appended + /// the record. `0` when no proposal apply appended it. + pub apply_key: u64, + /// The event source code of the row write the record carries. + pub event_source: u8, +} + +impl RecordStamp { + /// No proposal key and no row write. + pub const NONE: Self = Self { + apply_key: 0, + event_source: NO_EVENT_SOURCE, + }; } /// Parameters for [`WalRecord::new`]. @@ -53,13 +75,17 @@ impl WalRecord { /// zero-filled). Pre-existing records with zeros decode to `DatabaseId(0)` /// (the default database), preserving backward compatibility. pub fn new(args: WalRecordArgs<'_>) -> Result { - Self::new_keyed(args, 0) + Self::new_stamped(args, RecordStamp::NONE) } - /// [`Self::new`] for a record appended by the apply of the replicated - /// proposal `apply_key`. The key rides the header, inside the CRC and the - /// encryption AAD, so the record and the key are durable together. - pub fn new_keyed(args: WalRecordArgs<'_>, apply_key: u64) -> Result { + /// [`Self::new`] with the proposal key and event source of `stamp`. Both + /// ride the header, inside the CRC and the encryption AAD, so the record + /// and its stamp are durable together. + pub fn new_stamped(args: WalRecordArgs<'_>, stamp: RecordStamp) -> Result { + let RecordStamp { + apply_key, + event_source, + } = stamp; let WalRecordArgs { record_type, lsn, @@ -88,6 +114,7 @@ impl WalRecord { payload_len: 0, database_id, apply_key, + event_source, crc32c: 0, }; let header_bytes = temp_header.to_bytes(); @@ -116,6 +143,7 @@ impl WalRecord { payload_len: final_payload.len() as u32, database_id, apply_key, + event_source, crc32c: 0, }; @@ -133,6 +161,12 @@ impl WalRecord { self.header.apply_key } + /// The event source code of the row write this record carries. + /// [`NO_EVENT_SOURCE`] for a record that carries no row write. + pub fn event_source(&self) -> u8 { + self.header.event_source + } + /// Decrypt the payload if the record is encrypted. /// /// `epoch` must come from the on-disk segment preamble, not from the @@ -363,4 +397,34 @@ mod tests { let decoded = LsnMsAnchorPayload::from_bytes(&record.payload).unwrap(); assert_eq!(decoded, anchor); } + + #[test] + fn a_stamped_record_keeps_its_event_source_and_checksum() { + for code in [NO_EVENT_SOURCE, 1, 2, 3, 4, 5, 6, u8::MAX] { + let record = WalRecord::new_stamped( + WalRecordArgs { + record_type: 1, + lsn: 9, + tenant_id: 1, + vshard_id: 0, + database_id: 0, + payload: b"row".to_vec(), + encryption_key: None, + preamble_bytes: None, + }, + RecordStamp { + apply_key: 7, + event_source: code, + }, + ) + .expect("record"); + assert_eq!(record.event_source(), code); + assert_eq!(record.apply_key(), 7); + record + .verify_checksum() + .expect("checksum covers the source"); + let decoded = RecordHeader::from_bytes(&record.header.to_bytes()); + assert_eq!(decoded.event_source, code); + } + } } diff --git a/nodedb-wal/src/segmented.rs b/nodedb-wal/src/segmented.rs index 74bbfa1c3..4a31e212e 100644 --- a/nodedb-wal/src/segmented.rs +++ b/nodedb-wal/src/segmented.rs @@ -197,6 +197,7 @@ impl SegmentedWal { tenant_id, vshard_id, database_id, + event_source: crate::record::NO_EVENT_SOURCE, }, payload, 0, @@ -204,7 +205,8 @@ impl SegmentedWal { } /// [`Self::append`] for a record appended by the apply of the replicated - /// proposal `apply_key` (see [`crate::WalRecord::new_keyed`]). + /// proposal `apply_key`, carrying the event source in `target` (see + /// [`crate::WalRecord::new_stamped`]). pub fn append_keyed( &mut self, target: RecordTarget, diff --git a/nodedb-wal/src/writer/core.rs b/nodedb-wal/src/writer/core.rs index 680de4b2f..a311adef2 100644 --- a/nodedb-wal/src/writer/core.rs +++ b/nodedb-wal/src/writer/core.rs @@ -260,6 +260,7 @@ impl WalWriter { tenant_id, vshard_id, database_id, + event_source: crate::record::NO_EVENT_SOURCE, }, payload, 0, @@ -267,7 +268,8 @@ impl WalWriter { } /// [`Self::append`] for a record appended by the apply of the replicated - /// proposal `apply_key` (see [`WalRecord::new_keyed`]). + /// proposal `apply_key`, carrying the event source in `target` (see + /// [`WalRecord::new_stamped`]). pub fn append_keyed( &mut self, target: RecordTarget, @@ -279,6 +281,7 @@ impl WalWriter { tenant_id, vshard_id, database_id, + event_source, } = target; if self.sealed { return Err(WalError::Sealed); @@ -287,7 +290,7 @@ impl WalWriter { let lsn = self.next_lsn.load(Ordering::Relaxed); let preamble_bytes = self.segment_preamble.as_ref().map(|p| p.to_bytes()); - let record = WalRecord::new_keyed( + let record = WalRecord::new_stamped( WalRecordArgs { record_type, lsn, @@ -298,7 +301,10 @@ impl WalWriter { encryption_key: self.encryption_ring.as_ref().map(|r| r.current()), preamble_bytes: preamble_bytes.as_ref(), }, - apply_key, + crate::record::RecordStamp { + apply_key, + event_source, + }, )?; let header_bytes = record.header.to_bytes(); diff --git a/nodedb/src/control/backup/restore/durable.rs b/nodedb/src/control/backup/restore/durable.rs index f110a6b81..f4309799f 100644 --- a/nodedb/src/control/backup/restore/durable.rs +++ b/nodedb/src/control/backup/restore/durable.rs @@ -74,7 +74,12 @@ pub async fn reissue_plan_durably( vshard_id: vshard, }; let minted = MintedRecords::open(&state.outcome_floor); - if let Err(error) = minted.append_plan(&state.wal, owner, &plan) { + if let Err(error) = minted.append_plan( + &state.wal, + owner, + &plan, + sync_dispatch::SystemReason::BackupRestore.event_source(), + ) { // Any record appended before the error never reaches a core. minted.cancel(&state.wal, owner, 0).await?; return Err(error); diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs index f4ed74724..b0da81747 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs @@ -345,7 +345,14 @@ fn append_redo( database_id, vshard_id: VShardId::from_collection_in_database(database_id, target.collection), }; - let outcome = minted.append_plan(&shared.wal, owner, plan)?; + let outcome = minted.append_plan( + &shared.wal, + owner, + plan, + // A vector index change writes no row; its records carry no row + // image, and the source names the committed DDL that ran it. + crate::event::EventSource::User, + )?; Ok(outcome.lsn) } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index 5f08146b9..11c773826 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -106,6 +106,7 @@ impl Scheduler { let records = MintedRecords::open(&self.shared.outcome_floor); let appended = records .appender(&self.shared.wal, crate::wal::manager::NO_APPLY_KEY) + .with_event_source(super::request::CALVIN_EVENT_SOURCE) .append_transaction_redo( tenant_id, VShardId::new(self.vshard_id), diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs index e91945f96..04dab5e8b 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/redo_window.rs @@ -54,6 +54,7 @@ mod tests { let records = MintedRecords::open(&scheduler.shared.outcome_floor); let lsn = records .appender(&scheduler.shared.wal, crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs index f967e7da4..44bd7b658 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs @@ -9,6 +9,12 @@ use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request}; use crate::types::{DatabaseId, Lsn, ReadConsistency, RequestId, TenantId, VShardId}; use nodedb_physical::physical_plan::PhysicalPlan; +/// The event source every Calvin sub-operation runs with. The redo record a +/// committed Calvin transaction appends carries the same source, so WAL +/// replay rebuilds the events its flush emits. +pub(in crate::control::cluster::calvin::scheduler::driver::core) const CALVIN_EVENT_SOURCE: + crate::event::EventSource = crate::event::EventSource::User; + impl Scheduler { /// Builds a `Request` for an already-sequenced Calvin sub-operation. /// @@ -37,7 +43,7 @@ impl Scheduler { trace_id: nodedb_types::TraceId([0u8; 16]), consistency: ReadConsistency::Strong, idempotency_key: None, - event_source: crate::event::EventSource::User, + event_source: CALVIN_EVENT_SOURCE, user_roles: Vec::new(), user_id: None, statement_digest: None, diff --git a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs index f44c06830..4d61ae75e 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs @@ -309,6 +309,7 @@ mod tests { }), }; wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_transaction_redo( TenantId::new(0), VShardId::new(vshard), @@ -328,6 +329,7 @@ mod tests { calvin_stamp: None, }; wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_transaction_redo( TenantId::new(0), VShardId::new(vshard), diff --git a/nodedb/src/control/distributed_applier/proposal_ledger.rs b/nodedb/src/control/distributed_applier/proposal_ledger.rs index 5d9311937..85374ffea 100644 --- a/nodedb/src/control/distributed_applier/proposal_ledger.rs +++ b/nodedb/src/control/distributed_applier/proposal_ledger.rs @@ -189,9 +189,11 @@ mod tests { let wal = open_wal(&dir); let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); wal.appender(0xAB) + .with_event_source(crate::event::EventSource::User) .append_put(tid, vs, db, b"keyed") .expect("append keyed put"); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put(tid, vs, db, b"unkeyed") .expect("append unkeyed put"); wal.sync().expect("sync wal"); @@ -211,6 +213,7 @@ mod tests { let (tid, vs, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); let forward = wal .appender(0xAB) + .with_event_source(crate::event::EventSource::User) .append_put(tid, vs, db, b"refused") .expect("append keyed put"); wal.appender(NO_APPLY_KEY) @@ -218,6 +221,7 @@ mod tests { .expect("append unkeyed abort"); let final_forward = wal .appender(0xCD) + .with_event_source(crate::event::EventSource::User) .append_put(tid, vs, db, b"refused for good") .expect("append keyed put"); wal.appender(0xCD) diff --git a/nodedb/src/control/orchestrated_write.rs b/nodedb/src/control/orchestrated_write.rs index a8acb1917..30d038ea9 100644 --- a/nodedb/src/control/orchestrated_write.rs +++ b/nodedb/src/control/orchestrated_write.rs @@ -43,7 +43,11 @@ pub(crate) async fn apply_orchestrated_write( // WAL-only restart rebuilds the index from pre-write records. No-op // on a target with no write-set. crate::control::server::wal_dispatch::mint_dispatch_local_redo( - state.wal.appender(crate::wal::manager::NO_APPLY_KEY), + state + .wal + .appender(crate::wal::manager::NO_APPLY_KEY) + // `dispatch_local` runs the write as a client write. + .with_event_source(crate::event::EventSource::User), tenant_id, database_id, collection, diff --git a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs index cca6675aa..875a3abc4 100644 --- a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs +++ b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs @@ -181,15 +181,17 @@ impl<'a> StatementExecutor<'a> { vshard_id: task.vshard_id, }; let minted = MintedRecords::open(&self.state.outcome_floor); - let outcome = match minted.append_plan(&self.state.wal, owner, &task.plan) { - Ok(outcome) => outcome, - Err(error) => { - // Any record appended before the error never reaches - // a core. - minted.cancel(&self.state.wal, owner, 0).await?; - return Err(error); - } - }; + let outcome = + match minted.append_plan(&self.state.wal, owner, &task.plan, self.event_source) + { + Ok(outcome) => outcome, + Err(error) => { + // Any record appended before the error never reaches + // a core. + minted.cancel(&self.state.wal, owner, 0).await?; + return Err(error); + } + }; crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( self.state, diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index 2314f1e07..29b7f4f31 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -607,6 +607,7 @@ mod tests { let minted = super::MintedRecords::open(&state.outcome_floor); let lsn = minted .appender(&state.wal, crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), diff --git a/nodedb/src/control/server/dispatch_utils/minted/owned.rs b/nodedb/src/control/server/dispatch_utils/minted/owned.rs index e221f2825..31787545c 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/owned.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/owned.rs @@ -157,6 +157,7 @@ mod tests { let minted = MintedRecords::open(floor); let lsn = minted .appender(wal, NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), diff --git a/nodedb/src/control/server/dispatch_utils/minted/records.rs b/nodedb/src/control/server/dispatch_utils/minted/records.rs index af54d63e3..d38746087 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/records.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/records.rs @@ -113,15 +113,18 @@ impl MintedRecords { wal.recording_appender(apply_key, self) } - /// Append `plan`'s redo records under this window. + /// Append `plan`'s redo records under this window. Row-write records + /// carry `event_source`, the source the write is dispatched with. pub(crate) fn append_plan( &self, wal: &Arc, owner: RecordOwner, plan: &PhysicalPlan, + event_source: crate::event::EventSource, ) -> crate::Result { wal_append(WalAppendRequest { wal: self.appender(wal, NO_APPLY_KEY), + event_source, tenant_id: owner.tenant_id, vshard_id: owner.vshard_id, database_id: owner.database_id, @@ -374,6 +377,7 @@ mod tests { fn append(wal: &Arc, minted: &MintedRecords, body: &[u8]) -> Lsn { minted .appender(wal, NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -427,6 +431,7 @@ mod tests { let floor = OutcomeFloor::new(); let lsn = wal .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -528,6 +533,7 @@ mod tests { let floor = OutcomeFloor::new(); let lsn = wal .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), diff --git a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs index ae74f51d3..7bbc7104e 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs @@ -142,6 +142,7 @@ mod tests { let minted = MintedRecords::open(floor); let lsn = minted .appender(wal, NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index 65ba3250e..7f18bc5dc 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -202,17 +202,23 @@ pub(crate) async fn enqueue_write( // Array DDL authorization + durability, under the admission guard, // immediately before the enqueue below. - let wal_append_outcome = - match authorize_and_append(shared, owner, plan, durability, minted.as_ref()) { - Ok(outcome) => outcome, - Err(error) => { - // No record of this write reaches a core. - if let Some(minted) = minted { - minted.cancel(&shared.wal, owner, 0).await?; - } - return Err(error); + let wal_append_outcome = match authorize_and_append( + shared, + owner, + plan, + durability, + minted.as_ref(), + event_source, + ) { + Ok(outcome) => outcome, + Err(error) => { + // No record of this write reaches a core. + if let Some(minted) = minted { + minted.cancel(&shared.wal, owner, 0).await?; } - }; + return Err(error); + } + }; let commit_hlc = local_stamp .as_ref() .map(|stamp| stamp.hlc()) @@ -297,6 +303,7 @@ pub(crate) async fn enqueue_write( appends_here, final_refusal_key, apply_key, + event_source, post_apply, funnel_redo_engine, change_set, diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs index 62c05d346..7d3eab08b 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -40,6 +40,9 @@ pub(super) struct ResponsePhaseInput { /// The idempotency key every record this write appends carries (see /// `WalDurability::AppendHere`). pub apply_key: u64, + /// The event source the write runs with. Its post-apply redo records + /// carry it. + pub event_source: crate::event::EventSource, /// The key a final refusal's abort marker carries, `0` when this write's /// refusals are not final. A final refusal is the proposal's outcome: the /// proposal ledger rebuilt at boot counts the key as applied. @@ -93,6 +96,7 @@ pub(super) async fn collect_classify_and_finish( wal_lsn, appends_here, apply_key, + event_source, final_refusal_key, post_apply, funnel_redo_engine, @@ -206,7 +210,10 @@ pub(super) async fn collect_classify_and_finish( shared, &ddl_transition, wal_dispatch::append_write_set_redo( - shared.wal.appender(apply_key), + shared + .wal + .appender(apply_key) + .with_event_source(event_source), tenant_id, vshard_id, database_id, diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs index ffa055e8b..0f3900c94 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs @@ -69,6 +69,7 @@ pub(super) fn authorize_and_append( mut plan: PhysicalPlan, durability: WalDurability, minted: Option<&MintedRecords>, + event_source: crate::event::EventSource, ) -> crate::Result { let RecordOwner { tenant_id, @@ -96,6 +97,7 @@ pub(super) fn authorize_and_append( Some(minted) => minted.appender(&shared.wal, apply_key), None => shared.wal.appender(apply_key), }, + event_source, tenant_id, vshard_id, database_id, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs index 4ab02d78d..cde9fccd0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs @@ -190,7 +190,13 @@ async fn vector( vshard_id: vshard, }; let minted = MintedRecords::open(&state.outcome_floor); - let appended = match minted.append_plan(&state.wal, owner, &plan) { + let appended = match minted.append_plan( + &state.wal, + owner, + &plan, + // The drop is dispatched as a client statement. + crate::event::EventSource::User, + ) { Ok(appended) => appended, Err(e) => { // Any record appended before the error never reaches a core. diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs index f6c4b78fe..35d568487 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs @@ -442,7 +442,13 @@ pub async fn set_node_labels( vshard_id, }; let minted = crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); - if let Err(e) = minted.append_plan(&state.wal, owner, &plan) { + if let Err(e) = minted.append_plan( + &state.wal, + owner, + &plan, + // The same source the edge write is dispatched with below. + crate::event::EventSource::User, + ) { // Any record appended before the error never reaches a core. minted .cancel(&state.wal, owner, 0) diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs index c94404772..1dfc83369 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs @@ -315,7 +315,13 @@ async fn dispatch_sql( }; let minted = crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); - if let Err(e) = minted.append_plan(&state.wal, owner, checked.plan()) { + if let Err(e) = minted.append_plan( + &state.wal, + owner, + checked.plan(), + // The refresh is dispatched as a client write. + crate::event::EventSource::User, + ) { // Any record appended before the error never reaches a core. minted .cancel(&state.wal, owner, 0) diff --git a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs index 43323d4f6..c2b426a27 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs @@ -292,7 +292,13 @@ async fn append_edge_batch( ) -> crate::Result { let owner = edge_batch_owner(tenant_id, shard); let minted = crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); - match minted.append_plan(&state.wal, owner, plan) { + match minted.append_plan( + &state.wal, + owner, + plan, + // The same source the index write is dispatched with. + crate::event::EventSource::User, + ) { Ok(_) => Ok(minted), Err(e) => { // Any record appended before the error never reaches a core. diff --git a/nodedb/src/control/server/wal_dispatch/core.rs b/nodedb/src/control/server/wal_dispatch/core.rs index 50c419182..bf5499567 100644 --- a/nodedb/src/control/server/wal_dispatch/core.rs +++ b/nodedb/src/control/server/wal_dispatch/core.rs @@ -41,6 +41,10 @@ pub struct WalAppendOutcome { pub struct WalAppendRequest<'a> { /// The appender, which names the apply key every appended record carries. pub wal: WalAppender<'a>, + /// The event source the write runs with. Every row-write record it + /// appends carries it, so WAL replay rebuilds the event the live write + /// emits. + pub event_source: crate::event::EventSource, pub tenant_id: TenantId, pub vshard_id: VShardId, pub database_id: DatabaseId, @@ -58,12 +62,13 @@ pub struct WalAppendRequest<'a> { pub now_override: Option, } -/// Append a write operation to the WAL for single-node durability. +/// Append a client autocommit write to the WAL for single-node durability. /// /// Serializes the write as MessagePack and appends to the appropriate /// WAL record type. Read operations are no-ops (return Ok immediately). -/// The records carry no apply key: no replicated proposal owns them. A -/// proposal's apply goes through [`wal_append`] with a keyed appender. +/// The records carry no apply key: no replicated proposal owns them. Row-write +/// records carry `EventSource::User`. Any other write goes through +/// [`wal_append`] and names its own source. /// /// Returns the WAL LSN allocated for writes it appended (`Some`), or `None` /// for reads / control ops that need no WAL record. The caller stamps the @@ -93,6 +98,8 @@ pub fn wal_append_if_write_with_creds( ) -> crate::Result { wal_append(WalAppendRequest { wal: wal.appender(NO_APPLY_KEY), + // This entry point appends a client autocommit write. + event_source: crate::event::EventSource::User, tenant_id, vshard_id, database_id, @@ -110,6 +117,7 @@ pub fn wal_append_if_write_with_creds( pub fn wal_append(req: WalAppendRequest<'_>) -> crate::Result { let WalAppendRequest { wal, + event_source, tenant_id, vshard_id, database_id, @@ -117,6 +125,7 @@ pub fn wal_append(req: WalAppendRequest<'_>) -> crate::Result credentials, now_override, } = req; + let wal = wal.with_event_source(event_source); let mut resolved_now_ms: Option = None; // Every engine routes through one exhaustive per-engine match (no `_` // catch-all anywhere, enforced by `deny(wildcard_enum_match_arm)`), so a diff --git a/nodedb/src/control/server/wal_dispatch/graph.rs b/nodedb/src/control/server/wal_dispatch/graph.rs index 569eb7430..97c4d3a1c 100644 --- a/nodedb/src/control/server/wal_dispatch/graph.rs +++ b/nodedb/src/control/server/wal_dispatch/graph.rs @@ -178,7 +178,8 @@ mod tests { ]; let lsn = wal_append_graph_edge_put_batch( - wal.appender(NO_APPLY_KEY), + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User), TenantId::new(7), VShardId::new(0), DatabaseId::DEFAULT, @@ -224,7 +225,8 @@ mod tests { ]; let lsn = wal_append_graph_edge_delete_batch( - wal.appender(NO_APPLY_KEY), + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User), TenantId::new(7), VShardId::new(0), DatabaseId::DEFAULT, @@ -264,7 +266,8 @@ mod tests { let wal = open_wal(dir.path()); let lsn = wal_append_graph_edge_put_batch( - wal.appender(NO_APPLY_KEY), + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User), TenantId::new(7), VShardId::new(0), DatabaseId::DEFAULT, @@ -281,7 +284,8 @@ mod tests { let wal = open_wal(dir.path()); let lsn = wal_append_graph_edge_delete_batch( - wal.appender(NO_APPLY_KEY), + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User), TenantId::new(7), VShardId::new(0), DatabaseId::DEFAULT, diff --git a/nodedb/src/control/server/wal_dispatch/timeseries.rs b/nodedb/src/control/server/wal_dispatch/timeseries.rs index 48fe4b47b..f4999b736 100644 --- a/nodedb/src/control/server/wal_dispatch/timeseries.rs +++ b/nodedb/src/control/server/wal_dispatch/timeseries.rs @@ -464,6 +464,7 @@ mod tests { let outcome = super::super::wal_append(super::super::WalAppendRequest { wal: wal.appender(NO_APPLY_KEY), + event_source: crate::event::EventSource::User, tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), database_id: DatabaseId::DEFAULT, @@ -519,6 +520,7 @@ mod tests { let append = |apply_key: u64| { super::super::wal_append(super::super::WalAppendRequest { wal: wal.appender(apply_key), + event_source: crate::event::EventSource::User, tenant_id: TenantId::new(1), vshard_id: VShardId::new(0), database_id: DatabaseId::DEFAULT, diff --git a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs index 2da8ea86e..f5e33123c 100644 --- a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs +++ b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs @@ -262,7 +262,8 @@ mod tests { }]; let lsn = append_write_set_redo( - wal.appender(NO_APPLY_KEY), + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User), TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT, @@ -304,7 +305,8 @@ mod tests { }]; append_write_set_redo( - wal.appender(NO_APPLY_KEY), + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User), TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT, @@ -338,7 +340,8 @@ mod tests { }]; append_write_set_redo( - wal.appender(NO_APPLY_KEY), + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User), TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT, @@ -365,7 +368,8 @@ mod tests { let dir = tempfile::tempdir().expect("tempdir"); let wal = open_wal(dir.path()); let lsn = append_write_set_redo( - wal.appender(NO_APPLY_KEY), + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User), TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT, diff --git a/nodedb/src/data/executor/core_loop/deferred.rs b/nodedb/src/data/executor/core_loop/deferred.rs index d42bfb9ae..5d41a94d9 100644 --- a/nodedb/src/data/executor/core_loop/deferred.rs +++ b/nodedb/src/data/executor/core_loop/deferred.rs @@ -7,7 +7,7 @@ //! rows carry `EventSource::Deferred`, so the Event Plane fires DEFERRED-mode //! triggers. Rows of any other source keep that source, so a trigger's own //! transaction and a restore fire no DEFERRED trigger (see -//! [`committed_row_source`]). +//! [`EventSource::committed_row_source`]). use std::sync::Arc; @@ -15,25 +15,6 @@ use super::CoreLoop; use crate::engine::document::store::RowIdentity; use crate::event::types::{EventSource, RowId, WriteEvent, WriteOp}; -/// The source a committed record's document-row events carry. -/// -/// A client transaction's rows fire DEFERRED-mode triggers, so they carry -/// `Deferred`. Every other source keeps its own: a trigger's transaction -/// does not re-fire triggers, and a restored row fired its triggers when it -/// was first written. -pub(in crate::data::executor) const fn committed_row_source( - record_source: EventSource, -) -> EventSource { - match record_source { - EventSource::User => EventSource::Deferred, - EventSource::Trigger => EventSource::Trigger, - EventSource::RaftFollower => EventSource::RaftFollower, - EventSource::CrdtSync => EventSource::CrdtSync, - EventSource::Deferred => EventSource::Deferred, - EventSource::Restore => EventSource::Restore, - } -} - /// A write that occurred during a transaction, pending deferred trigger emission. pub(in crate::data::executor) struct DeferredWrite { pub collection: String, @@ -47,8 +28,8 @@ impl CoreLoop { /// Emit the document-row events of a committed transaction. /// /// Called after a committed redo record installed and settled. Each - /// write is emitted as a WriteEvent whose source is - /// [`committed_row_source`] of the record's `record_source`. + /// write is emitted as a WriteEvent whose source is the + /// [`EventSource::committed_row_source`] of the record's `record_source`. pub(in crate::data::executor) fn emit_deferred_events( &mut self, writes: Vec, @@ -57,7 +38,7 @@ impl CoreLoop { tenant_id: crate::types::TenantId, vshard_id: crate::types::VShardId, ) { - let source = committed_row_source(record_source); + let source = record_source.committed_row_source(); let producer = match self.event_producer.as_mut() { Some(p) => p, None => return, @@ -94,36 +75,3 @@ impl CoreLoop { } } } - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn only_a_client_transaction_fires_deferred_triggers() { - assert_eq!( - committed_row_source(EventSource::User), - EventSource::Deferred - ); - assert_eq!( - committed_row_source(EventSource::Restore), - EventSource::Restore - ); - assert_eq!( - committed_row_source(EventSource::Trigger), - EventSource::Trigger - ); - assert_eq!( - committed_row_source(EventSource::RaftFollower), - EventSource::RaftFollower - ); - assert_eq!( - committed_row_source(EventSource::CrdtSync), - EventSource::CrdtSync - ); - assert_eq!( - committed_row_source(EventSource::Deferred), - EventSource::Deferred - ); - } -} diff --git a/nodedb/src/data/executor/wal_replay/kv_put.rs b/nodedb/src/data/executor/wal_replay/kv_put.rs index fb86829da..f3b9b7c02 100644 --- a/nodedb/src/data/executor/wal_replay/kv_put.rs +++ b/nodedb/src/data/executor/wal_replay/kv_put.rs @@ -399,6 +399,7 @@ mod tests { let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); for payload in payloads { wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TID), VShardId::new(0), diff --git a/nodedb/src/data/executor/wal_replay_kv_expiry.rs b/nodedb/src/data/executor/wal_replay_kv_expiry.rs index 2dd6fd604..434d183c7 100644 --- a/nodedb/src/data/executor/wal_replay_kv_expiry.rs +++ b/nodedb/src/data/executor/wal_replay_kv_expiry.rs @@ -289,6 +289,7 @@ mod tests { ) .expect("wal append seed put"); wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TID), VShardId::new(0), @@ -424,6 +425,7 @@ mod tests { let dir = tempfile::tempdir().expect("wal tempdir"); let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TID), VShardId::new(0), diff --git a/nodedb/src/data/executor/wal_replay_kv_incr.rs b/nodedb/src/data/executor/wal_replay_kv_incr.rs index a40887d5f..35096d31a 100644 --- a/nodedb/src/data/executor/wal_replay_kv_incr.rs +++ b/nodedb/src/data/executor/wal_replay_kv_incr.rs @@ -414,6 +414,7 @@ mod tests { ) .expect("wal append seed put"); wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TID), VShardId::new(0), diff --git a/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs b/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs index d077ee72c..424cdfd92 100644 --- a/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs +++ b/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs @@ -492,6 +492,7 @@ mod tests { ) .expect("wal append seed put"); wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TID), VShardId::new(0), diff --git a/nodedb/src/data/executor/wal_replay_kv_ttl.rs b/nodedb/src/data/executor/wal_replay_kv_ttl.rs index 19642397f..e5a4e9611 100644 --- a/nodedb/src/data/executor/wal_replay_kv_ttl.rs +++ b/nodedb/src/data/executor/wal_replay_kv_ttl.rs @@ -93,6 +93,7 @@ mod tests { let dir = tempfile::tempdir().expect("wal tempdir"); let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TID), VShardId::new(0), @@ -132,6 +133,7 @@ mod tests { let dir = tempfile::tempdir().expect("wal tempdir"); let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); wal.appender(crate::wal::manager::NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TID), VShardId::new(0), diff --git a/nodedb/src/event/consumer/run.rs b/nodedb/src/event/consumer/run.rs index 3c66dd93c..d6c675554 100644 --- a/nodedb/src/event/consumer/run.rs +++ b/nodedb/src/event/consumer/run.rs @@ -539,6 +539,7 @@ mod tests { .expect("put payload"); let lsn = wal .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TENANT), VShardId::new(0), @@ -679,6 +680,7 @@ mod tests { let lsn = node .wal .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(TENANT), VShardId::new(0), diff --git a/nodedb/src/event/mod.rs b/nodedb/src/event/mod.rs index 363695114..248765ff4 100644 --- a/nodedb/src/event/mod.rs +++ b/nodedb/src/event/mod.rs @@ -30,6 +30,7 @@ pub mod trigger; pub mod types; pub mod wal_replay; pub mod wal_replay_parse; +pub mod wal_replay_scope; pub mod watermark; pub mod watermark_tracker; pub mod webhook; diff --git a/nodedb/src/event/types.rs b/nodedb/src/event/types.rs index a80b1f4b3..e5871eb2f 100644 --- a/nodedb/src/event/types.rs +++ b/nodedb/src/event/types.rs @@ -284,6 +284,52 @@ impl EventSource { } } + /// The source a committed transaction's document rows carry, for a + /// record whose writes ran with `self`. + /// + /// A client transaction's rows fire DEFERRED-mode triggers, so they carry + /// `Deferred`. Every other source keeps its own: a trigger's transaction + /// does not re-fire triggers, and a restored row fired its triggers when + /// it was first written. The live apply and WAL replay both use this. + pub const fn committed_row_source(self) -> Self { + match self { + Self::User => Self::Deferred, + Self::Trigger => Self::Trigger, + Self::RaftFollower => Self::RaftFollower, + Self::CrdtSync => Self::CrdtSync, + Self::Deferred => Self::Deferred, + Self::Restore => Self::Restore, + } + } + + /// The source's code in a WAL record header. Code `0` is + /// `nodedb_wal::NO_EVENT_SOURCE`, a record with no row write, so no source + /// maps to it. + pub const fn wal_code(self) -> u8 { + match self { + Self::User => 1, + Self::Trigger => 2, + Self::RaftFollower => 3, + Self::CrdtSync => 4, + Self::Deferred => 5, + Self::Restore => 6, + } + } + + /// The source a WAL record header code names. `None` for + /// `nodedb_wal::NO_EVENT_SOURCE` and for a code no source uses. + pub const fn from_wal_code(code: u8) -> Option { + match code { + 1 => Some(Self::User), + 2 => Some(Self::Trigger), + 3 => Some(Self::RaftFollower), + 4 => Some(Self::CrdtSync), + 5 => Some(Self::Deferred), + 6 => Some(Self::Restore), + _ => None, + } + } + /// The source named `name`, as [`Self::as_str`] spells it. pub fn from_name(name: &str) -> Option { match name { @@ -397,6 +443,42 @@ mod tests { assert_eq!(EventSource::from_name("unknown"), None); } + #[test] + fn every_event_source_round_trips_through_its_wal_code() { + for source in [ + EventSource::User, + EventSource::Trigger, + EventSource::RaftFollower, + EventSource::CrdtSync, + EventSource::Deferred, + EventSource::Restore, + ] { + assert_ne!(source.wal_code(), nodedb_wal::NO_EVENT_SOURCE); + assert_eq!(EventSource::from_wal_code(source.wal_code()), Some(source)); + } + assert_eq!( + EventSource::from_wal_code(nodedb_wal::NO_EVENT_SOURCE), + None + ); + } + + #[test] + fn only_a_client_transaction_fires_deferred_triggers() { + assert_eq!( + EventSource::User.committed_row_source(), + EventSource::Deferred + ); + for source in [ + EventSource::Trigger, + EventSource::RaftFollower, + EventSource::CrdtSync, + EventSource::Deferred, + EventSource::Restore, + ] { + assert_eq!(source.committed_row_source(), source); + } + } + #[test] fn write_event_construction() { let event = WriteEvent { diff --git a/nodedb/src/event/wal_replay.rs b/nodedb/src/event/wal_replay.rs index 483731d40..a382339ad 100644 --- a/nodedb/src/event/wal_replay.rs +++ b/nodedb/src/event/wal_replay.rs @@ -28,13 +28,14 @@ //! and are not yet emitted as WriteEvents. use nodedb_wal::WalRecord; -use nodedb_wal::record::{RecordType, WalRecordArgs}; -use tracing::{trace, warn}; +use nodedb_wal::record::RecordType; +use tracing::{error, trace, warn}; -use crate::event::types::WriteEvent; +use crate::event::types::{EventSource, WriteEvent}; use crate::event::wal_replay_parse::{ parse_delete_record, parse_graph_node_label_record, parse_put_record, }; +use crate::event::wal_replay_scope::{ReplayScope, RowSources}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; use crate::wal::WalManager; @@ -139,70 +140,41 @@ fn convert_records_to_events( /// CDC, and change streams fire on restart exactly as they did on the forward /// path. fn record_to_events(record: &WalRecord, sequence: &mut u64) -> Vec { - let logical_type = record.logical_record_type(); - let Some(record_type) = RecordType::from_raw(logical_type) else { + let Some(record_type) = RecordType::from_raw(record.logical_record_type()) else { return Vec::new(); }; + match row_kind(record_type) { + Some(kind) => row_record_events(record, kind, sequence), + None => Vec::new(), + } +} - let tenant_id = TenantId::new(record.header.tenant_id); - // `database_id` is part of the WAL envelope. A zero value is the named - // pre-database-header compatibility encoding and maps to the built-in - // default database; current WAL writers always stamp the request database. - let database_id = DatabaseId::new(record.header.database_id); - let vshard_id = VShardId::new(record.header.vshard_id); - let lsn = Lsn::new(record.header.lsn); - +/// The row-write kind of a record type. `None` for a type that carries no +/// row write the forward path emits an event for. +fn row_kind(record_type: RecordType) -> Option { match record_type { - RecordType::Put => { - parse_put_record(&record.payload, database_id, tenant_id, vshard_id, lsn, sequence) - .into_iter() - .collect() - } - RecordType::Delete => { - parse_delete_record(&record.payload, database_id, tenant_id, vshard_id, lsn, sequence) - .into_iter() - .collect() - } - // A Calvin cross-shard commit is durable as a `TransactionRedo` whose - // sub-ops carry each engine's own per-op payload. Decompose it into the - // same WriteEvents the forward path emitted, so the effect (triggers/CDC) - // is not lost on replay. Every emitted event's `lsn` is this redo record's - // WAL LSN — the Event-Plane watermark keys on it to dedup against the - // forward-path event (both share this LSN in the same space). - RecordType::TransactionRedo => decompose_redo_to_events(record, sequence), + RecordType::Put => Some(RowRecord::Put), + RecordType::Delete => Some(RowRecord::Delete), + // A Calvin cross-shard or single-shard commit is durable as a + // `TransactionRedo` whose sub-ops carry each engine's own per-op + // payload. Decompose it into the same WriteEvents the forward path + // emitted, so the effect (triggers/CDC) is not lost on replay. Every + // emitted event's `lsn` is this redo record's WAL LSN — the Event-Plane + // watermark keys on it to dedup against the forward-path event. + RecordType::TransactionRedo => Some(RowRecord::Redo), // Graph node-label mutations carry no natural collection (they are // tenant-wide), so they surface on the nameable `__graph_node_labels__` // CDC stream. The forward-path emit (Data Plane `SetNodeLabels` / // `RemoveNodeLabels`) produces the same `(collection, row_id, op, value)` // shape, so replayed events dedup against forward events on LSN. - RecordType::GraphNodeLabelSet => parse_graph_node_label_record( - &record.payload, - true, - database_id, - tenant_id, - vshard_id, - lsn, - sequence, - ) - .into_iter() - .collect(), - RecordType::GraphNodeLabelRemove => parse_graph_node_label_record( - &record.payload, - false, - database_id, - tenant_id, - vshard_id, - lsn, - sequence, - ) - .into_iter() - .collect(), + RecordType::GraphNodeLabelSet => Some(RowRecord::LabelSet), + RecordType::GraphNodeLabelRemove => Some(RowRecord::LabelRemove), // `CalvinApplied` is a payload-free applied-marker: it records that a // sequencer `(epoch, position)` was applied, but carries no writes. Its // base writes, if any, ride a separate `TransactionRedo`; a pure-read or // CRDT-only commit has no base WriteEvents at all (CRDT effects ride // `CrdtDelta` records). Nothing to emit. - RecordType::CalvinApplied => Vec::new(), + RecordType::CalvinApplied => None, // The records below carry NO forward-path Data-Plane WriteEvent, so // there is nothing for replay to reconstruct. `record_to_events` // reconstructs exactly the forward WriteEvent stream the Data Plane @@ -283,31 +255,103 @@ fn record_to_events(record: &WalRecord, sequence: &mut u64) -> Vec { // `WalManager::replay_from`). The marker itself is not a row write. | RecordType::WriteAborted // ProposalApplied marks a Raft proposal as applied; it writes no row. - | RecordType::ProposalApplied => Vec::new(), + | RecordType::ProposalApplied => None, + } +} + +/// A record type that carries row writes. +#[derive(Debug, Clone, Copy)] +enum RowRecord { + Put, + Delete, + Redo, + LabelSet, + LabelRemove, +} + +/// The events of one row-write record. +/// +/// Every row write stores the source its write ran with. A record without one +/// cannot say whether its triggers may fire, so it rebuilds no event. +fn row_record_events(record: &WalRecord, kind: RowRecord, sequence: &mut u64) -> Vec { + let Some(source) = EventSource::from_wal_code(record.event_source()) else { + error!( + lsn = record.header.lsn, + record_type = record.logical_record_type(), + code = record.event_source(), + "WAL replay: a row-write record carries no event source; no event rebuilt" + ); + return Vec::new(); + }; + let sources = match kind { + RowRecord::Redo => RowSources::committed_redo(source), + RowRecord::Put | RowRecord::Delete | RowRecord::LabelSet | RowRecord::LabelRemove => { + RowSources::uniform(source) + } + }; + let scope = ReplayScope { + tenant_id: TenantId::new(record.header.tenant_id), + // `database_id` is part of the WAL envelope. Every WAL writer stamps + // the request database. + database_id: DatabaseId::new(record.header.database_id), + vshard_id: VShardId::new(record.header.vshard_id), + lsn: Lsn::new(record.header.lsn), + sources, + }; + match kind { + RowRecord::Redo => decompose_redo_to_events(&record.payload, &scope, sequence), + RowRecord::Put | RowRecord::Delete | RowRecord::LabelSet | RowRecord::LabelRemove => { + single_row_events(kind, &record.payload, &scope, sequence) + } } } +/// The event of one single-row payload of `kind`. A redo payload carries no +/// single row. +fn single_row_events( + kind: RowRecord, + payload: &[u8], + scope: &ReplayScope, + sequence: &mut u64, +) -> Vec { + let event = match kind { + RowRecord::Put => parse_put_record(payload, scope, sequence), + RowRecord::Delete => parse_delete_record(payload, scope, sequence), + RowRecord::LabelSet => parse_graph_node_label_record(payload, true, scope, sequence), + RowRecord::LabelRemove => parse_graph_node_label_record(payload, false, scope, sequence), + RowRecord::Redo => { + warn!( + lsn = scope.lsn.as_u64(), + "WAL replay: a redo sub-record is itself a redo; skipped" + ); + None + } + }; + event.into_iter().collect() +} + /// Decompose a `TransactionRedo` record into per-sub-op WriteEvents. /// /// Each `RedoSubRecord` carries its engine's own `record_type` and a payload in /// that engine's exact per-op WAL shape (the same encoders the autocommit path -/// uses). We reconstitute each sub-op as a standalone `WalRecord` — stamped with -/// the enclosing redo record's header identity, crucially its LSN — and feed it -/// back through [`record_to_events`]. That reuses the raw Put/Delete parsers -/// verbatim and inherits every current and future event mapping: a sub-op type -/// with no Event-Plane mapping (VectorPut, SpatialPut, …) yields no event and -/// does not touch `sequence`, exactly as its raw counterpart does. Because the -/// reconstituted record carries `record.header.lsn`, every emitted event sets -/// `lsn = record.header.lsn`, satisfying the watermark-dedup requirement. +/// uses). Each sub-op goes through the same single-row parsers as its raw +/// counterpart, under the enclosing record's `scope`: its LSN (the +/// watermark-dedup key), its tenant, vShard and database, and the row sources +/// of a committed redo. A sub-op type with no Event-Plane mapping (VectorPut, +/// SpatialPut, …) yields no event and does not touch `sequence`. /// /// A malformed redo payload is logged and skipped (never a panic), mirroring the /// decode-failure handling in the Data-Plane redo replay path. -fn decompose_redo_to_events(record: &WalRecord, sequence: &mut u64) -> Vec { - let redo = match crate::wal::RedoRecord::from_bytes(&record.payload) { +fn decompose_redo_to_events( + payload: &[u8], + scope: &ReplayScope, + sequence: &mut u64, +) -> Vec { + let redo = match crate::wal::RedoRecord::from_bytes(payload) { Ok(redo) => redo, Err(e) => { warn!( - lsn = record.header.lsn, + lsn = scope.lsn.as_u64(), error = %e, "WAL replay: skipping malformed TransactionRedo payload" ); @@ -317,32 +361,12 @@ fn decompose_redo_to_events(record: &WalRecord, sequence: &mut u64) -> Vec wr, - Err(e) => { - warn!( - lsn = record.header.lsn, - sub_record_type = sub.record_type, - error = %e, - "WAL replay: skipping redo sub-record that failed to reconstitute" - ); - continue; - } + let Some(record_type) = RecordType::from_raw(sub.record_type) else { + continue; }; - events.extend(record_to_events(&sub_record, sequence)); + if let Some(kind) = row_kind(record_type) { + events.extend(single_row_events(kind, &sub.payload, scope, sequence)); + } } events } @@ -875,6 +899,7 @@ mod tests { assert_eq!(seq, 0, "index sub-op consumes no sequence on decompose"); } + /// A record a client write appended. fn make_record( rt: RecordType, payload: &[u8], @@ -882,16 +907,122 @@ mod tests { vshard_id: u32, lsn: u64, ) -> WalRecord { - WalRecord::new(nodedb_wal::WalRecordArgs { - record_type: rt as u32, - lsn, - tenant_id, - vshard_id, - database_id: 0, - payload: payload.to_vec(), - encryption_key: None, - preamble_bytes: None, - }) + stamped_record( + rt, + payload, + (tenant_id, vshard_id, lsn), + Some(EventSource::User), + ) + } + + /// A tenant-1 record at `lsn`, appended with `source`. + fn make_sourced_record( + rt: RecordType, + payload: &[u8], + lsn: u64, + source: Option, + ) -> WalRecord { + stamped_record(rt, payload, (1, 0, lsn), source) + } + + /// A record at `(tenant, vshard, lsn)` appended with `source`. `None` + /// stores no event source. + fn stamped_record( + rt: RecordType, + payload: &[u8], + (tenant_id, vshard_id, lsn): (u64, u32, u64), + source: Option, + ) -> WalRecord { + WalRecord::new_stamped( + nodedb_wal::WalRecordArgs { + record_type: rt as u32, + lsn, + tenant_id, + vshard_id, + database_id: 0, + payload: payload.to_vec(), + encryption_key: None, + preamble_bytes: None, + }, + nodedb_wal::RecordStamp { + apply_key: 0, + event_source: source.map_or(nodedb_wal::NO_EVENT_SOURCE, EventSource::wal_code), + }, + ) .unwrap() } + + #[test] + fn a_replayed_row_carries_the_source_its_record_stored() { + let payload = zerompk::to_msgpack_vec(&("orders", "order-1", b"value")).unwrap(); + for source in [ + EventSource::Restore, + EventSource::Trigger, + EventSource::CrdtSync, + EventSource::Deferred, + EventSource::User, + ] { + let record = make_sourced_record(RecordType::Put, &payload, 300, Some(source)); + let mut seq = 0u64; + assert_eq!(one_event(&record, &mut seq).source, source); + } + } + + #[test] + fn a_row_record_without_a_source_rebuilds_no_event() { + let payload = zerompk::to_msgpack_vec(&("orders", "order-1", b"value")).unwrap(); + let record = make_sourced_record(RecordType::Put, &payload, 301, None); + let mut seq = 0u64; + assert!(record_to_events(&record, &mut seq).is_empty()); + assert_eq!(seq, 0); + } + + /// A committed redo's document rows follow `committed_row_source`, and its + /// KV rows keep the record's source, as the live apply emits them. + #[test] + fn a_redo_replays_rows_with_the_live_apply_sources() { + use crate::wal::{RedoRecord, RedoSubRecord}; + + let doc = zerompk::to_msgpack_vec(&("orders", "order-9", b"doc")).unwrap(); + let kv = zerompk::to_msgpack_vec(&("kv_put", "cache", b"k9", b"v9", 0u64)).unwrap(); + let redo = RedoRecord { + version: 1, + ops: vec![ + RedoSubRecord { + record_type: RecordType::Put as u32, + payload: doc, + }, + RedoSubRecord { + record_type: RecordType::Put as u32, + payload: kv, + }, + ], + calvin_stamp: None, + }; + let bytes = redo.to_bytes().unwrap(); + for (source, document, other) in [ + (EventSource::User, EventSource::Deferred, EventSource::User), + ( + EventSource::Restore, + EventSource::Restore, + EventSource::Restore, + ), + ( + EventSource::Trigger, + EventSource::Trigger, + EventSource::Trigger, + ), + ] { + let record = + make_sourced_record(RecordType::TransactionRedo, &bytes, 400, Some(source)); + let mut seq = 0u64; + let events = record_to_events(&record, &mut seq); + assert_eq!(events.len(), 2); + assert_eq!( + events[0].source, document, + "document row of a {source} redo" + ); + assert_eq!(events[1].source, other, "KV row of a {source} redo"); + } + } } diff --git a/nodedb/src/event/wal_replay_parse.rs b/nodedb/src/event/wal_replay_parse.rs index a6848870a..2621b0641 100644 --- a/nodedb/src/event/wal_replay_parse.rs +++ b/nodedb/src/event/wal_replay_parse.rs @@ -8,10 +8,9 @@ //! (document with/without surrogate/provenance, KV point / batch); the parser //! tries them in most-specific-first order and returns the first that decodes. //! -//! A `TransactionRedo` sub-op payload is byte-identical to the corresponding raw -//! per-op WAL record payload, so `wal_replay` reconstitutes each sub-op as a -//! standalone `WalRecord` and routes it back through the same dispatch — reusing -//! these parsers verbatim, with no redo-specific decode path. +//! A `TransactionRedo` sub-op payload is byte-identical to its raw per-op +//! record, so `wal_replay` parses each sub-op with these same parsers. Every +//! event takes its identity and event source from the [`ReplayScope`]. use std::sync::Arc; @@ -19,8 +18,8 @@ use nodedb_types::RowIdentity; use nodedb_types::sync::wire::SyncProvenance; use tracing::warn; -use crate::event::types::{EventSource, RecordPosition, RowId, WriteEvent, WriteOp}; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::event::types::{RecordPosition, RowId, WriteEvent, WriteOp}; +use crate::event::wal_replay_scope::ReplayScope; /// `(op, new_value, old_value)` for a node-label CDC event — the op tag plus /// the label-delta payload placed on whichever side its `WriteOp` implies. @@ -85,12 +84,16 @@ fn decode_kv_batch_put_event_fields(payload: &[u8]) -> Option<(String, Vec<(Vec< /// graph edge put — distinguished by the MessagePack structure. pub(super) fn parse_put_record( payload: &[u8], - database_id: DatabaseId, - tenant_id: TenantId, - vshard_id: VShardId, - lsn: Lsn, + scope: &ReplayScope, sequence: &mut u64, ) -> Option { + let ReplayScope { + database_id, + tenant_id, + vshard_id, + lsn, + sources, + } = *scope; // Try KV put first. Three arities decode: the current // `("kv_put", collection, key, value, ttl_ms, expire_at_ms, surrogate)` and // the two that predate the carried surrogate. The event stream keys on the @@ -115,7 +118,7 @@ pub(super) fn parse_put_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.other, new_value: Some(Arc::from(row.as_slice())), old_value: None, system_time_ms, @@ -141,7 +144,7 @@ pub(super) fn parse_put_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.other, new_value: None, old_value: None, system_time_ms: None, @@ -172,7 +175,7 @@ pub(super) fn parse_put_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.document, new_value: Some(Arc::from(value.as_slice())), old_value: None, system_time_ms, @@ -199,7 +202,7 @@ pub(super) fn parse_put_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.document, new_value: Some(Arc::from(value.as_slice())), old_value: None, system_time_ms, @@ -229,7 +232,7 @@ pub(super) fn parse_put_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.document, new_value: Some(Arc::from(value.as_slice())), old_value: None, system_time_ms, @@ -262,7 +265,7 @@ pub(super) fn parse_put_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.other, new_value: Some(Arc::from(properties.as_slice())), old_value: None, system_time_ms, @@ -294,12 +297,16 @@ pub(super) fn parse_put_record( pub(super) fn parse_graph_node_label_record( payload: &[u8], is_set: bool, - database_id: DatabaseId, - tenant_id: TenantId, - vshard_id: VShardId, - lsn: Lsn, + scope: &ReplayScope, sequence: &mut u64, ) -> Option { + let ReplayScope { + database_id, + tenant_id, + vshard_id, + lsn, + sources, + } = *scope; let (node_id, labels) = match zerompk::from_msgpack::<(String, Vec)>(payload) { Ok(decoded) => decoded, Err(_) => { @@ -328,7 +335,7 @@ pub(super) fn parse_graph_node_label_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.other, new_value, old_value, system_time_ms: None, @@ -341,12 +348,16 @@ pub(super) fn parse_graph_node_label_record( /// Parse a `RecordType::Delete` payload. May be a document delete or KV delete. pub(super) fn parse_delete_record( payload: &[u8], - database_id: DatabaseId, - tenant_id: TenantId, - vshard_id: VShardId, - lsn: Lsn, + scope: &ReplayScope, sequence: &mut u64, ) -> Option { + let ReplayScope { + database_id, + tenant_id, + vshard_id, + lsn, + sources, + } = *scope; // Try KV delete: ("kv_delete", collection, keys) if let Ok((disc, collection, keys)) = zerompk::from_msgpack::<(&str, String, Vec>)>(payload) @@ -365,7 +376,7 @@ pub(super) fn parse_delete_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.other, new_value: None, old_value: None, system_time_ms: None, @@ -392,7 +403,7 @@ pub(super) fn parse_delete_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.document, new_value: None, old_value: None, system_time_ms: None, @@ -417,7 +428,7 @@ pub(super) fn parse_delete_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.document, new_value: None, old_value: None, system_time_ms: None, @@ -440,7 +451,7 @@ pub(super) fn parse_delete_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.document, new_value: None, old_value: None, system_time_ms: None, @@ -469,7 +480,7 @@ pub(super) fn parse_delete_record( database_id, tenant_id, vshard_id, - source: EventSource::User, + source: sources.other, new_value: None, old_value: None, system_time_ms: None, diff --git a/nodedb/src/event/wal_replay_scope.rs b/nodedb/src/event/wal_replay_scope.rs new file mode 100644 index 000000000..e40580017 --- /dev/null +++ b/nodedb/src/event/wal_replay_scope.rs @@ -0,0 +1,69 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Where a replayed WAL record sits, and which event source its rows carry. +//! +//! A write record's header stores the event source its write ran with. A +//! replayed event carries the source its live event carried: +//! - a raw Put or Delete record: every row carries the record's source; +//! - a `TransactionRedo` record: a document row carries +//! [`EventSource::committed_row_source`] of the record's source, and a KV, +//! graph edge or node-label row carries the record's source. The live +//! committed-redo apply emits the same sources. + +use crate::event::types::EventSource; +use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + +/// The event source each kind of replayed row carries. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RowSources { + /// The source of a document row. + pub document: EventSource, + /// The source of a KV, graph edge or node-label row. + pub other: EventSource, +} + +impl RowSources { + /// A raw write record: every row carries `source`. + pub const fn uniform(source: EventSource) -> Self { + Self { + document: source, + other: source, + } + } + + /// A committed transaction's redo record whose writes ran with `source`. + pub const fn committed_redo(source: EventSource) -> Self { + Self { + document: source.committed_row_source(), + other: source, + } + } +} + +/// The header identity every event rebuilt from one record carries. +#[derive(Debug, Clone, Copy)] +pub struct ReplayScope { + pub database_id: DatabaseId, + pub tenant_id: TenantId, + pub vshard_id: VShardId, + pub lsn: Lsn, + pub sources: RowSources, +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_client_transaction_redo_fires_deferred_document_rows_only() { + let sources = RowSources::committed_redo(EventSource::User); + assert_eq!(sources.document, EventSource::Deferred); + assert_eq!(sources.other, EventSource::User); + } + + #[test] + fn a_restore_redo_keeps_restore_for_every_row() { + let sources = RowSources::committed_redo(EventSource::Restore); + assert_eq!(sources, RowSources::uniform(EventSource::Restore)); + } +} diff --git a/nodedb/src/wal/manager/append.rs b/nodedb/src/wal/manager/append.rs index 4877d35f2..043310841 100644 --- a/nodedb/src/wal/manager/append.rs +++ b/nodedb/src/wal/manager/append.rs @@ -13,7 +13,7 @@ impl WalAppender<'_> { db: DatabaseId, p: &[u8], ) -> crate::Result { - self.append_record(RecordType::Put, tid, vs, db, p) + self.append_row_record(RecordType::Put, tid, vs, db, p) } pub fn append_delete( @@ -23,7 +23,7 @@ impl WalAppender<'_> { db: DatabaseId, p: &[u8], ) -> crate::Result { - self.append_record(RecordType::Delete, tid, vs, db, p) + self.append_row_record(RecordType::Delete, tid, vs, db, p) } } @@ -72,7 +72,9 @@ mod tests { let v = VShardId::new(0); let db = DatabaseId::DEFAULT; - let appender = wal.appender(NO_APPLY_KEY); + let appender = wal + .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User); let lsn1 = appender.append_put(t, v, db, b"key1=value1").unwrap(); let lsn2 = appender.append_put(t, v, db, b"key2=value2").unwrap(); let lsn3 = appender.append_delete(t, v, db, b"key1").unwrap(); @@ -111,7 +113,9 @@ mod tests { calvin_stamp: None, }; - let appender = wal.appender(NO_APPLY_KEY); + let appender = wal + .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User); let lsn1 = appender.append_transaction_redo(t, v, db, &record).unwrap(); let lsn2 = appender.append_transaction_redo(t, v, db, &record).unwrap(); let lsn3 = appender.append_transaction_redo(t, v, db, &record).unwrap(); diff --git a/nodedb/src/wal/manager/append_index.rs b/nodedb/src/wal/manager/append_index.rs index d67a8af33..4772ca9da 100644 --- a/nodedb/src/wal/manager/append_index.rs +++ b/nodedb/src/wal/manager/append_index.rs @@ -66,7 +66,7 @@ impl WalAppender<'_> { db: DatabaseId, p: &[u8], ) -> crate::Result { - self.append_record(RecordType::GraphNodeLabelSet, tid, vs, db, p) + self.append_row_record(RecordType::GraphNodeLabelSet, tid, vs, db, p) } /// Append a `GraphNodeLabelRemove` record. Payload is produced by @@ -78,7 +78,7 @@ impl WalAppender<'_> { db: DatabaseId, p: &[u8], ) -> crate::Result { - self.append_record(RecordType::GraphNodeLabelRemove, tid, vs, db, p) + self.append_row_record(RecordType::GraphNodeLabelRemove, tid, vs, db, p) } /// Append a `WriteAborted` record naming `aborted_lsn`, the LSN of a diff --git a/nodedb/src/wal/manager/append_transaction.rs b/nodedb/src/wal/manager/append_transaction.rs index 0be102734..c8cdf98fe 100644 --- a/nodedb/src/wal/manager/append_transaction.rs +++ b/nodedb/src/wal/manager/append_transaction.rs @@ -32,7 +32,7 @@ impl WalAppender<'_> { record: &crate::wal::RedoRecord, ) -> crate::Result { let payload = record.to_bytes()?; - self.append_record(RecordType::TransactionRedo, tid, vs, db, &payload) + self.append_row_record(RecordType::TransactionRedo, tid, vs, db, &payload) } /// Append a `TransactionRedo` record whose payload is an already-encoded @@ -45,7 +45,7 @@ impl WalAppender<'_> { db: DatabaseId, payload: &[u8], ) -> crate::Result { - self.append_record(RecordType::TransactionRedo, tid, vs, db, payload) + self.append_row_record(RecordType::TransactionRedo, tid, vs, db, payload) } pub fn append_crdt_delta( diff --git a/nodedb/src/wal/manager/appender.rs b/nodedb/src/wal/manager/appender.rs index f2d7e54ed..39335f258 100644 --- a/nodedb/src/wal/manager/appender.rs +++ b/nodedb/src/wal/manager/appender.rs @@ -8,6 +8,12 @@ //! these keys. The key is part of the append call: a caller names it when it //! takes an appender, so no append can read a key another caller set and no //! caller has a key to clear. +//! +//! A row-write record (Put, Delete, TransactionRedo, graph node-label) also +//! carries the event source its write ran with, so WAL replay rebuilds the +//! event the live write emitted. The caller names the source with +//! [`WalAppender::with_event_source`]. A row-write append from an appender +//! with no source is refused: no row write is stored without one. use std::sync::Mutex; @@ -43,11 +49,15 @@ impl AppendSink for Mutex> { } } -/// Appends WAL records that all carry one apply key. +/// Appends WAL records that all carry one apply key, and row-write records +/// that carry one event source. #[derive(Clone, Copy)] pub struct WalAppender<'a> { wal: &'a WalManager, apply_key: u64, + /// The event source code row-write records carry. + /// `nodedb_wal::NO_EVENT_SOURCE` until the caller names one. + event_source: u8, /// Collects every record this appender writes, when set. sink: Option<&'a dyn AppendSink>, } @@ -60,6 +70,7 @@ impl WalManager { WalAppender { wal: self, apply_key, + event_source: nodedb_wal::NO_EVENT_SOURCE, sink: None, } } @@ -74,6 +85,7 @@ impl WalManager { WalAppender { wal: self, apply_key, + event_source: nodedb_wal::NO_EVENT_SOURCE, sink: Some(sink), } } @@ -85,7 +97,15 @@ impl WalAppender<'_> { self.apply_key } - /// Append one record of `record_type`. + /// This appender, with row-write records carrying `source`. + pub fn with_event_source(self, source: crate::event::EventSource) -> Self { + Self { + event_source: source.wal_code(), + ..self + } + } + + /// Append one record of `record_type` that carries no row write. pub(super) fn append_record( &self, record_type: RecordType, @@ -93,19 +113,56 @@ impl WalAppender<'_> { vshard_id: VShardId, database_id: DatabaseId, payload: &[u8], + ) -> crate::Result { + let target = RecordTarget { + record_type: record_type as u32, + tenant_id: tenant_id.as_u64(), + vshard_id: vshard_id.as_u32(), + database_id: database_id.as_u64(), + event_source: nodedb_wal::NO_EVENT_SOURCE, + }; + self.append_target(target, tenant_id, vshard_id, database_id, payload) + } + + /// Append one row-write record of `record_type`, carrying this + /// appender's event source. Refused when the caller named no source. + pub(super) fn append_row_record( + &self, + record_type: RecordType, + tenant_id: TenantId, + vshard_id: VShardId, + database_id: DatabaseId, + payload: &[u8], + ) -> crate::Result { + if self.event_source == nodedb_wal::NO_EVENT_SOURCE { + return Err(crate::Error::Internal { + detail: format!( + "WAL append: a {record_type:?} row-write record needs the event source \ + of its write; the appender names none" + ), + }); + } + let target = RecordTarget { + record_type: record_type as u32, + tenant_id: tenant_id.as_u64(), + vshard_id: vshard_id.as_u32(), + database_id: database_id.as_u64(), + event_source: self.event_source, + }; + self.append_target(target, tenant_id, vshard_id, database_id, payload) + } + + fn append_target( + &self, + target: RecordTarget, + tenant_id: TenantId, + vshard_id: VShardId, + database_id: DatabaseId, + payload: &[u8], ) -> crate::Result { let mut wal = self.wal.wal.lock().unwrap_or_else(|p| p.into_inner()); let lsn = wal - .append_keyed( - RecordTarget { - record_type: record_type as u32, - tenant_id: tenant_id.as_u64(), - vshard_id: vshard_id.as_u32(), - database_id: database_id.as_u64(), - }, - payload, - self.apply_key, - ) + .append_keyed(target, payload, self.apply_key) .map_err(crate::Error::Wal)?; let lsn = Lsn::new(lsn); if let Some(sink) = self.sink { @@ -131,9 +188,12 @@ mod tests { let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); let (t, v, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); - let keyed = wal.appender(0xAB); + let keyed = wal + .appender(0xAB) + .with_event_source(crate::event::EventSource::User); keyed.append_put(t, v, db, b"keyed").expect("keyed append"); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put(t, v, db, b"unkeyed") .expect("unkeyed append"); keyed @@ -157,9 +217,12 @@ mod tests { let (t, v, db) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); let sink = Mutex::new(Vec::new()); - let recording = wal.recording_appender(NO_APPLY_KEY, &sink); + let recording = wal + .recording_appender(NO_APPLY_KEY, &sink) + .with_event_source(crate::event::EventSource::User); let first = recording.append_put(t, v, db, b"a").expect("append"); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put(t, v, db, b"unrecorded") .expect("append"); let second = recording.append_put(t, v, db, b"b").expect("append"); diff --git a/nodedb/src/wal/manager/durable_commit.rs b/nodedb/src/wal/manager/durable_commit.rs index 6f84c08ce..feb92eb38 100644 --- a/nodedb/src/wal/manager/durable_commit.rs +++ b/nodedb/src/wal/manager/durable_commit.rs @@ -124,6 +124,7 @@ mod tests { let wal = open_wal(dir.path()); let lsn = wal .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -141,6 +142,7 @@ mod tests { let wal = open_wal(dir.path()); let lsn = wal .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -161,6 +163,7 @@ mod tests { for _ in 0..16 { lsns.push( wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), diff --git a/nodedb/src/wal/manager/encryption.rs b/nodedb/src/wal/manager/encryption.rs index 62bf9440b..16ca328f3 100644 --- a/nodedb/src/wal/manager/encryption.rs +++ b/nodedb/src/wal/manager/encryption.rs @@ -289,6 +289,7 @@ mod tests { let mut wal = WalManager::open_encrypted(&wal_dir, false, &key_a).unwrap(); let stable_root = wal.crdt_signing_root().unwrap().unwrap(); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -302,6 +303,7 @@ mod tests { let mut wal = WalManager::open_encrypted_rotating(&wal_dir, false, &key_b, &key_a).unwrap(); assert_eq!(wal.crdt_signing_root().unwrap(), Some(stable_root)); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -344,6 +346,7 @@ mod tests { .unwrap(); let root = wal.crdt_signing_root().unwrap(); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -358,6 +361,7 @@ mod tests { .unwrap(); assert_eq!(wal.crdt_signing_root().unwrap(), root); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), diff --git a/nodedb/src/wal/manager/ops.rs b/nodedb/src/wal/manager/ops.rs index f07c1557c..93365d435 100644 --- a/nodedb/src/wal/manager/ops.rs +++ b/nodedb/src/wal/manager/ops.rs @@ -71,6 +71,7 @@ mod tests { { let wal = WalManager::open_for_testing(&path).unwrap(); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -79,6 +80,7 @@ mod tests { ) .unwrap(); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -94,6 +96,7 @@ mod tests { let lsn = wal .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), @@ -117,6 +120,7 @@ mod tests { for i in 0..10u32 { wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put(t, v, db, format!("val-{i}").as_bytes()) .unwrap(); } @@ -136,6 +140,7 @@ mod tests { let wal = WalManager::open_for_testing(&path).unwrap(); wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) .append_put( TenantId::new(1), VShardId::new(0), diff --git a/nodedb/tests/inproc/cases/event_wal_replay_source.rs b/nodedb/tests/inproc/cases/event_wal_replay_source.rs new file mode 100644 index 000000000..89ffebc5f --- /dev/null +++ b/nodedb/tests/inproc/cases/event_wal_replay_source.rs @@ -0,0 +1,62 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Events rebuilt from the WAL carry the source their write ran with. +//! +//! The Event Plane rebuilds events from the WAL after its ring overflows or +//! after a crash. Each row-write record stores its event source, so a +//! restored, trigger, CRDT-sync or deferred row replays with that source. A +//! restored row therefore fires no AFTER trigger on replay, as on its live +//! write. + +use nodedb::event::EventSource; +use nodedb::event::wal_replay::replay_wal_to_events; +use nodedb::types::{DatabaseId, Lsn, TenantId, VShardId}; +use nodedb::wal::manager::{NO_APPLY_KEY, WalManager}; + +fn document_put(id: &str) -> Vec { + zerompk::to_msgpack_vec(&("orders", id, b"row".to_vec())).expect("encode put") +} + +#[test] +fn a_replayed_row_carries_the_source_of_its_write() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); + let sources = [ + EventSource::Restore, + EventSource::Trigger, + EventSource::CrdtSync, + EventSource::Deferred, + ]; + for (i, source) in sources.iter().enumerate() { + wal.appender(NO_APPLY_KEY) + .with_event_source(*source) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + &document_put(&format!("row-{i}")), + ) + .expect("append"); + } + wal.sync().expect("sync"); + + let events = replay_wal_to_events(&wal, Lsn::new(0), 0, 1, 0).expect("replay"); + let replayed: Vec = events.iter().map(|event| event.source).collect(); + assert_eq!(replayed, sources.to_vec()); +} + +#[test] +fn a_row_write_without_a_source_is_refused() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = WalManager::open_for_testing(&dir.path().join("wal")).expect("open wal"); + let appended = wal.appender(NO_APPLY_KEY).append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + &document_put("row"), + ); + assert!( + appended.is_err(), + "a row-write record must name the source of its write" + ); +} diff --git a/nodedb/tests/inproc/cases/mod.rs b/nodedb/tests/inproc/cases/mod.rs index c100a849c..70e3a5861 100644 --- a/nodedb/tests/inproc/cases/mod.rs +++ b/nodedb/tests/inproc/cases/mod.rs @@ -90,6 +90,7 @@ mod event_streaming_mv; mod event_topics; mod event_trigger; mod event_trigger_descriptor_fence; +mod event_wal_replay_source; mod fts_compaction_budget; mod gateway_local_descriptor_fence; mod graph_collection_isolation; From eabd4905ff861dd56e18ccb0e43d5410a2704cfb Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 15:05:10 +0800 Subject: [PATCH 40/64] feat(query): decode a KV row back into insert-shaped body fields kv_row_to_body_fields inverts kv_row_msgpack: it strips the injected primary key column when it matches the entry key, keeps a key column that differs, and picks Raw vs Map shape the same way SQL lowering does for a fresh insert. --- nodedb-query/src/msgpack_scan/kv_body.rs | 85 +++++++++++++++++++++++- nodedb-query/src/msgpack_scan/mod.rs | 4 +- 2 files changed, 87 insertions(+), 2 deletions(-) diff --git a/nodedb-query/src/msgpack_scan/kv_body.rs b/nodedb-query/src/msgpack_scan/kv_body.rs index 0e746142c..a5b22fd53 100644 --- a/nodedb-query/src/msgpack_scan/kv_body.rs +++ b/nodedb-query/src/msgpack_scan/kv_body.rs @@ -9,6 +9,8 @@ //! Decoding raw bytes as msgpack instead either fails (multi-byte value) or //! reads the first byte as a fixint and discards the rest. +use std::collections::HashMap; + use nodedb_types::{MsgpackError, NotScalar, Value, scalar_to_raw_bytes}; use crate::msgpack_scan::{map_header, write_map_header, write_str}; @@ -124,10 +126,91 @@ pub fn row_to_kv_body(row: &Value, shape: KvBodyShape) -> Result, KvBody } } +/// The fields a KV body stores for a msgpack-encoded row, and the body shape. +/// +/// This is the inverse of [`super::kv_row_msgpack`]. `key` is the entry's key. +/// A `key` field equal to it is the injected primary key and is dropped. A +/// `key` field that differs is a stored column and is kept. The remaining +/// fields pick the shape the SQL lowering picks for a fresh insert: +/// - `value` alone: [`KvBodyShape::Raw`]; +/// - any other field set: [`KvBodyShape::Map`]. +/// +/// The caller encodes the fields in its own body format. +/// [`row_to_kv_body`] is the Origin format. +pub fn kv_row_to_body_fields( + key: &str, + row: &[u8], +) -> Result<(HashMap, KvBodyShape), KvBodyError> { + let row = nodedb_types::value_from_msgpack(row)?; + let Value::Object(mut fields) = row else { + return Err(KvBodyError::RowNotObject { + kind: row.type_name(), + }); + }; + if matches!(fields.get("key"), Some(Value::String(stored)) if stored == key) { + fields.remove("key"); + } + let shape = if fields.len() == 1 && fields.contains_key("value") { + KvBodyShape::Raw + } else { + KvBodyShape::Map + }; + Ok((fields, shape)) +} + #[cfg(test)] mod tests { use super::*; - use std::collections::HashMap; + + fn body_from_row(key: &str, row: &[u8]) -> Vec { + let (fields, shape) = kv_row_to_body_fields(key, row).unwrap(); + row_to_kv_body(&Value::Object(fields), shape).unwrap() + } + + #[test] + fn a_raw_body_round_trips_through_its_row() { + for body in [b"v1".to_vec(), b"1".to_vec(), Vec::new()] { + let row = crate::msgpack_scan::kv_row_msgpack("k1", &body); + assert_eq!(body_from_row("k1", &row), body); + } + } + + #[test] + fn a_map_body_round_trips_through_its_row() { + let mut fields = HashMap::new(); + fields.insert("n".to_string(), Value::Integer(7)); + fields.insert("s".to_string(), Value::String("x".into())); + let body = row_to_kv_body(&Value::Object(fields), KvBodyShape::Map).unwrap(); + let row = crate::msgpack_scan::kv_row_msgpack("k2", &body); + assert_eq!(body_from_row("k2", &row), body); + } + + #[test] + fn a_key_column_that_differs_from_the_entry_key_is_kept() { + let mut fields = HashMap::new(); + fields.insert("key".to_string(), Value::String("stored".into())); + fields.insert("n".to_string(), Value::Integer(1)); + let body = row_to_kv_body(&Value::Object(fields), KvBodyShape::Map).unwrap(); + let row = crate::msgpack_scan::kv_row_msgpack("slot", &body); + assert_eq!(body_from_row("slot", &row), body); + } + + #[test] + fn a_row_that_is_not_a_map_is_an_error() { + let not_a_map = nodedb_types::value_to_msgpack(&Value::Integer(118)).unwrap(); + assert!(matches!( + kv_row_to_body_fields("k", ¬_a_map), + Err(KvBodyError::RowNotObject { kind: "int" }) + )); + } + + #[test] + fn bytes_that_are_not_msgpack_are_a_decode_error() { + assert!(matches!( + kv_row_to_body_fields("k", &[0x81]), + Err(KvBodyError::Decode(_)) + )); + } fn value_of(row: &Value) -> &Value { row.get("value").expect("row carries `value`") diff --git a/nodedb-query/src/msgpack_scan/mod.rs b/nodedb-query/src/msgpack_scan/mod.rs index af657f92b..939c67a8d 100644 --- a/nodedb-query/src/msgpack_scan/mod.rs +++ b/nodedb-query/src/msgpack_scan/mod.rs @@ -24,7 +24,9 @@ pub use compare::{compare_field_bytes, hash_field_bytes}; pub use field::{extract_field, extract_path}; pub use group_key::build_group_key; pub use index::FieldIndex; -pub use kv_body::{KvBodyError, KvBodyShape, kv_body_shape, kv_body_to_row, row_to_kv_body}; +pub use kv_body::{ + KvBodyError, KvBodyShape, kv_body_shape, kv_body_to_row, kv_row_to_body_fields, row_to_kv_body, +}; pub use kv_row::kv_row_msgpack; pub use reader::{ array_header, map_header, read_bin_advance, read_bool, read_f64, read_i64, read_null, read_str, From 8971b41d6958fa7bbe0d928fa8df5a5599ada6db Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 16:55:32 +0800 Subject: [PATCH 41/64] feat(sync): push KV writes from Lite through the idempotency gate Add KvPush/KvPushAck wire messages so Lite forwards KV writes to Origin, and RowPushReject so Lite can refuse an Origin row push it cannot apply. Thread SyncProvenance through KvOp::Put/Delete so a pushed write clears the sync idempotency gate instead of bypassing it, and introduce SyncHold (Duplicate/Fenced/Gap) to distinguish a held-back frame from an outright rejection across the bridge and cluster RPC codec. --- .../cases/cluster_execute_request.rs | 2 + .../common_suite/cases/gateway_execute.rs | 1 + .../cases/http_gateway_migration.rs | 2 + .../cases/listeners_gateway_smoke.rs | 5 + .../cases/listeners_typed_not_leader.rs | 4 + .../cases/native_gateway_migration.rs | 2 + .../cases/pgwire_gateway_migration.rs | 1 + .../cases/resp_gateway_migration.rs | 2 + .../src/rpc_codec/data_plane_error.rs | 14 + nodedb-cluster/src/rpc_codec/mod.rs | 2 +- nodedb-physical/src/physical_plan/kv/op.rs | 7 + nodedb-test-support/src/tx_batch_helpers.rs | 1 + nodedb-types/src/namespace.rs | 9 +- nodedb-types/src/sync/wire/ack_status.rs | 1 + nodedb-types/src/sync/wire/frame.rs | 14 + nodedb-types/src/sync/wire/kv.rs | 177 ++++++++ nodedb-types/src/sync/wire/mod.rs | 6 + nodedb-types/src/sync/wire/stream_id.rs | 1 + nodedb/src/bridge/admission_chokepoint.rs | 2 + nodedb/src/bridge/envelope/error_code.rs | 6 + nodedb/src/bridge/envelope/mod.rs | 2 + nodedb/src/bridge/envelope/sync_hold.rs | 42 ++ .../src/control/backup/restore/kv_reissue.rs | 1 + nodedb/src/control/clone/copyup.rs | 1 + .../control/cluster/data_plane_error_wire.rs | 48 +- nodedb/src/control/gateway/router.rs | 1 + .../maintenance/clone_materializer/kv.rs | 1 + .../src/control/planner/rls_injection/kv.rs | 4 + .../planner/sql_plan_convert/dml/kv_insert.rs | 1 + .../dml/update_delete/delete.rs | 1 + .../control/server/dispatch_utils/dispatch.rs | 1 + .../dispatch_utils/durability_barrier.rs | 1 + .../server/dispatch_utils/durable_write.rs | 57 +++ .../src/control/server/dispatch_utils/mod.rs | 4 +- .../server/dispatch_utils/write_abort.rs | 6 +- .../native/dispatch/plan_builder/document.rs | 2 + .../src/control/server/pgwire/handler/plan.rs | 1 + .../control/server/resp/handler_kv/strings.rs | 2 + .../src/control/server/resp/handler_sorted.rs | 2 + .../server/response_shape/calvin_fold.rs | 2 + .../types/plan_kind/describe.rs | 2 + .../control/server/shared/clone_write/kv.rs | 5 + .../server/shared/ddl/neutral/rate_gate.rs | 1 + .../shared/ddl/neutral/weighted_pick.rs | 1 + .../src/control/server/shared/ddl/sqlstate.rs | 7 + .../server/shared/session/commit/metering.rs | 1 + .../server/shared/sql/staging_predicates.rs | 1 + .../shared/write_admission/lock_keys.rs | 1 + .../predicate/plan_is_write.rs | 1 + .../predicate/txn_buffering/classify.rs | 2 + .../predicate/user_data_write.rs | 1 + nodedb/src/control/server/sync/kv_handler.rs | 267 +++++++++++ nodedb/src/control/server/sync/kv_session.rs | 424 ++++++++++++++++++ nodedb/src/control/server/sync/mod.rs | 2 + .../control/server/sync/session/dispatch.rs | 9 + nodedb/src/control/server/sync/session/mod.rs | 2 + .../server/sync/session/row_push_reject.rs | 127 ++++++ .../sync/session_handler/engine_dispatch.rs | 25 +- nodedb/src/control/server/sync/wire.rs | 13 +- .../control/server/wal_dispatch_kv/append.rs | 40 +- .../wal_replication/decode/entry_kv.rs | 24 +- .../src/control/wal_replication/decode/kv.rs | 76 +++- .../control/wal_replication/encode/entry.rs | 1 + .../wal_replication/encode/entry_kv.rs | 18 +- .../src/control/wal_replication/encode/kv.rs | 29 +- .../wal_replication/types/replicated_write.rs | 6 + .../handlers/control/calvin_reply/stage.rs | 3 + .../src/data/executor/handlers/kv/crud/mod.rs | 1 + .../executor/handlers/kv/crud/sync_write.rs | 205 +++++++++ .../src/data/executor/handlers/kv/dispatch.rs | 28 +- .../executor/handlers/kv/resolve/dispatch.rs | 4 + .../handlers/transaction/resolve/entry.rs | 2 + .../transaction/stage_write/stage_kv.rs | 3 + .../src/data/executor/wal_replay_kv_atomic.rs | 4 + .../src/data/executor/wal_replay_kv_expiry.rs | 3 + .../src/data/executor/wal_replay_kv_field.rs | 2 + .../src/data/executor/wal_replay_kv_incr.rs | 5 + .../src/data/executor/wal_replay_kv_index.rs | 1 + .../executor/wal_replay_kv_insert_conflict.rs | 3 + .../executor/wal_replay_kv_sorted_index.rs | 1 + .../data/executor/wal_replay_kv_transfer.rs | 3 + nodedb/src/data/executor/wal_replay_kv_ttl.rs | 3 + nodedb/src/error_from_data_plane.rs | 7 + .../cases/calvin_determinism_contract.rs | 2 + .../inproc/cases/calvin_executor_apply.rs | 1 + .../cases/calvin_executor_panic_recovery.rs | 1 + .../inproc/cases/calvin_two_phase_apply.rs | 1 + .../test_cross_type_join/basic_scans.rs | 2 + .../test_cross_type_join/join_budget.rs | 3 + .../test_cross_type_join/multi_core_joins.rs | 4 + .../test_cross_type_join/single_core_joins.rs | 2 + .../test_graph_savepoint_overlay.rs | 1 + .../inproc/cases/executor_tests/test_kv.rs | 12 + .../cases/executor_tests/test_kv_advanced.rs | 12 + .../executor_tests/test_kv_ttl_overlay.rs | 3 + .../test_tenant_isolation_kv.rs | 1 + .../test_tenant_isolation_kv_negative.rs | 4 + .../cases/executor_tests/test_tenant_purge.rs | 1 + .../test_transaction_matrix_kv.rs | 2 + .../inproc/cases/surrogate_round_trip.rs | 1 + .../inproc/cases/write_admission_fence.rs | 1 + 101 files changed, 1803 insertions(+), 54 deletions(-) create mode 100644 nodedb-types/src/sync/wire/kv.rs create mode 100644 nodedb/src/bridge/envelope/sync_hold.rs create mode 100644 nodedb/src/control/server/sync/kv_handler.rs create mode 100644 nodedb/src/control/server/sync/kv_session.rs create mode 100644 nodedb/src/control/server/sync/session/row_push_reject.rs create mode 100644 nodedb/src/data/executor/handlers/kv/crud/sync_write.rs diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs index 3cd22a251..73c924b54 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs @@ -42,6 +42,7 @@ fn make_kv_put_request( surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let plan_bytes = plan_wire::encode(&plan).expect("encode plan"); @@ -286,6 +287,7 @@ async fn execute_request_cross_node_dispatch() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); plan_wire::encode(&plan).expect("encode plan") }, diff --git a/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs b/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs index c9ee4a8ab..91c32d7c0 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs @@ -76,6 +76,7 @@ async fn gateway_execute_kv_put_get_single_node() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; let put_result = gateway.execute(&ctx, put_checked).await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs index ef889768d..286c119a4 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs @@ -77,6 +77,7 @@ async fn http_gateway_migration_single_node_query() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; let put_result = gateway.execute(&ctx, put_checked).await; @@ -156,6 +157,7 @@ async fn http_gateway_migration_cross_node_query() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&follower.shared, &ctx, put_plan).await; let put_result = gateway.execute(&ctx, put_checked).await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs b/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs index 05c97b55c..74c3a0104 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs @@ -78,6 +78,7 @@ async fn pgwire_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); @@ -148,6 +149,7 @@ async fn http_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); @@ -212,6 +214,7 @@ async fn resp_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); @@ -279,6 +282,7 @@ async fn ilp_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); @@ -343,6 +347,7 @@ async fn native_gateway_smoke_cache_hit() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway.execute(&ctx, checked).await.expect("gateway Put"); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs b/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs index 54bb5ae6c..08dd33abc 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs @@ -179,6 +179,7 @@ async fn pgwire_not_leader_retry_uses_shared_gateway() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let ctx = test_ctx(); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; @@ -244,6 +245,7 @@ async fn http_not_leader_gateway_error_mapping() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let ctx = test_ctx(); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; @@ -315,6 +317,7 @@ async fn resp_not_leader_gateway_error_mapping() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let ctx = test_ctx(); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; @@ -449,6 +452,7 @@ async fn native_not_leader_gateway_error_mapping() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let ctx = test_ctx(); let checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs index b9c9a820c..fc46e4228 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs @@ -77,6 +77,7 @@ async fn native_gateway_migration_single_node_select() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway @@ -147,6 +148,7 @@ async fn native_gateway_migration_cross_node_select() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&cluster.nodes[0].shared, &ctx, put_plan).await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs index d02c841b5..8a3b46cdb 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs @@ -191,6 +191,7 @@ async fn pgwire_gateway_migration_plan_cache_hits() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; gateway diff --git a/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs index 68bfecdae..2a77900cc 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs @@ -74,6 +74,7 @@ async fn resp_gateway_migration_single_node_set_get() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&node.shared, &ctx, put_plan).await; let put_result = gateway.execute(&ctx, put_checked).await; @@ -146,6 +147,7 @@ async fn resp_gateway_migration_cross_node_get() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_checked = common::authorize_gateway_plan(&cluster.nodes[0].shared, &ctx, put_plan).await; diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index 8d8953135..fc884563b 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -141,6 +141,20 @@ pub enum DataPlaneErrorCode { stream_id: u64, seq: u64, }, + /// A sync frame the idempotency gate held back. Nothing applied, and + /// the stream's mark did not move. + SyncNotApplied { + hold: DataPlaneSyncHold, + applied_seq: u64, + }, +} + +/// Wire mirror of `nodedb::bridge::envelope::SyncHold`. +#[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum DataPlaneSyncHold { + Duplicate, + Fenced, + Gap { expected: u64 }, } /// Wire mirror of `nodedb_physical::kv_atomic::CounterFault`. diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index 4b10a22cd..da881d23d 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -43,7 +43,7 @@ pub use cluster_mgmt::{ JoinGroupInfo, JoinNodeInfo, JoinRequest, JoinResponse, LEADER_REDIRECT_PREFIX, PingRequest, PongResponse, TopologyAck, TopologyUpdate, }; -pub use data_plane_error::{DataPlaneCounterFault, DataPlaneErrorCode}; +pub use data_plane_error::{DataPlaneCounterFault, DataPlaneErrorCode, DataPlaneSyncHold}; pub use data_propose::{ DataProposeRequest, DataProposeResponse, ForwardedProposeRefusal, ProposeTarget, }; diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index c65772208..9c392caef 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -61,6 +61,10 @@ pub enum KvOp { /// principal would show. #[serde(default)] rls_filters: Vec, + /// Sync provenance of a Lite KV push. `Some` puts the write behind + /// the sync idempotency gate. `None` for every other write. + #[serde(default)] + provenance: Option, }, /// SQL `INSERT` semantics: write only if the key does not already exist. @@ -141,6 +145,9 @@ pub enum KvOp { /// See `Put::rls_filters`. #[serde(default)] rls_filters: Vec, + /// See `Put::provenance`. + #[serde(default)] + provenance: Option, }, /// Cursor-based scan with optional filter predicate. diff --git a/nodedb-test-support/src/tx_batch_helpers.rs b/nodedb-test-support/src/tx_batch_helpers.rs index e53437dd1..4ab576dca 100644 --- a/nodedb-test-support/src/tx_batch_helpers.rs +++ b/nodedb-test-support/src/tx_batch_helpers.rs @@ -180,6 +180,7 @@ pub fn kv_put(key: &[u8], value: &[u8]) -> PhysicalPlan { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb-types/src/namespace.rs b/nodedb-types/src/namespace.rs index 15f034c90..8ac6e727f 100644 --- a/nodedb-types/src/namespace.rs +++ b/nodedb-types/src/namespace.rs @@ -114,6 +114,10 @@ pub enum Namespace { /// transport to Origin. Keys are big-endian monotonic u64 IDs; values are /// zerompk-encoded `PendingSpatialDelete` payloads. SpatialDeletePending = 24, + /// Durable FIFO queue of outbound KV writes waiting for transport to + /// Origin. Keys are big-endian monotonic u64 IDs; values are + /// zerompk-encoded `PendingKvWrite` payloads. + KvPushPending = 25, } impl Namespace { @@ -145,6 +149,7 @@ impl Namespace { 22 => Some(Self::FtsDeletePending), 23 => Some(Self::SpatialInsertPending), 24 => Some(Self::SpatialDeletePending), + 25 => Some(Self::KvPushPending), _ => None, } } @@ -156,10 +161,10 @@ mod tests { #[test] fn namespace_roundtrip() { - for v in 0u8..=24 { + for v in 0u8..=25 { let ns = Namespace::from_u8(v).unwrap(); assert_eq!(ns as u8, v); } - assert!(Namespace::from_u8(25).is_none()); + assert!(Namespace::from_u8(26).is_none()); } } diff --git a/nodedb-types/src/sync/wire/ack_status.rs b/nodedb-types/src/sync/wire/ack_status.rs index 74469ecc7..7720ecc82 100644 --- a/nodedb-types/src/sync/wire/ack_status.rs +++ b/nodedb-types/src/sync/wire/ack_status.rs @@ -13,6 +13,7 @@ use serde::{Deserialize, Serialize}; Clone, Default, PartialEq, + Eq, Serialize, Deserialize, zerompk::ToMessagePack, diff --git a/nodedb-types/src/sync/wire/frame.rs b/nodedb-types/src/sync/wire/frame.rs index 702a71752..dbb9617f8 100644 --- a/nodedb-types/src/sync/wire/frame.rs +++ b/nodedb-types/src/sync/wire/frame.rs @@ -15,6 +15,9 @@ //! - `0xAB` SpatialInsertAck (server → client) //! - `0xAC` SpatialDelete (client → server) //! - `0xAD` SpatialDeleteAck (server → client) +//! - `0xAE` KvPush (client → server) +//! - `0xAF` KvPushAck (server → client) +//! - `0x16` RowPushReject (client → server) /// Sync message type identifiers. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -44,6 +47,10 @@ pub enum SyncMessageType { /// originated on the server (SQL DML, DDL-managed system rows), where no /// client-authored CRDT operation exists to replicate. RowPush = 0x15, + /// Row push refusal (client → server, 0x16). + /// + /// Lite sends this for a [`Self::RowPush`] it could not apply. + RowPushReject = 0x16, ShapeSubscribe = 0x20, ShapeSnapshot = 0x21, ShapeDelta = 0x22, @@ -116,6 +123,10 @@ pub enum SyncMessageType { SpatialDelete = 0xAC, /// Spatial delete acknowledgment (server → client, 0xAD). SpatialDeleteAck = 0xAD, + /// KV write push (client → server, 0xAE). + KvPush = 0xAE, + /// KV push acknowledgment (server → client, 0xAF). + KvPushAck = 0xAF, PingPong = 0xFF, } @@ -130,6 +141,7 @@ impl SyncMessageType { 0x13 => Some(Self::CollectionSchema), 0x14 => Some(Self::CollectionPurged), 0x15 => Some(Self::RowPush), + 0x16 => Some(Self::RowPushReject), 0x20 => Some(Self::ShapeSubscribe), 0x21 => Some(Self::ShapeSnapshot), 0x22 => Some(Self::ShapeDelta), @@ -167,6 +179,8 @@ impl SyncMessageType { 0xAB => Some(Self::SpatialInsertAck), 0xAC => Some(Self::SpatialDelete), 0xAD => Some(Self::SpatialDeleteAck), + 0xAE => Some(Self::KvPush), + 0xAF => Some(Self::KvPushAck), 0xFF => Some(Self::PingPong), _ => None, } diff --git a/nodedb-types/src/sync/wire/kv.rs b/nodedb-types/src/sync/wire/kv.rs new file mode 100644 index 000000000..13f0e22f6 --- /dev/null +++ b/nodedb-types/src/sync/wire/kv.rs @@ -0,0 +1,177 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! KV row push messages, and the refusal of a row push. +//! +//! `KvPushMsg` carries one KV write from a Lite client to Origin. A put +//! carries the `{key, value…}` row every KV read returns, the same shape +//! Origin sends in a `RowPushMsg`. Origin answers each push with a +//! `KvPushAckMsg`. A terminal refusal travels in that ack as +//! `AckStatus::Rejected`, the way every engine push ack carries one. +//! +//! `RowPushRejectMsg` is Lite's refusal of an Origin `RowPushMsg` it could +//! not apply. +//! +//! Wire opcodes: +//! - `0x16` — `RowPushReject` (Lite → Origin) +//! - `0xAE` — `KvPush` (Lite → Origin) +//! - `0xAF` — `KvPushAck` (Origin → Lite) + +use serde::{Deserialize, Serialize}; + +use crate::sync::wire::ack_status::AckStatus; + +/// The write a `KvPushMsg` carries. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum KvPushOp { + /// Store `row` at the entry's key. + Put { + /// The `{key, value…}` row as standard MessagePack. + row: Vec, + /// Absolute expiry in milliseconds since the Unix epoch. `0` means + /// the entry never expires. + expire_at_ms: u64, + }, + /// Remove the entry's key. + Delete, +} + +/// KV write push (Lite → Origin, 0xAE). +#[derive( + Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, +)] +pub struct KvPushMsg { + /// Lite instance ID. + pub lite_id: String, + /// Target KV collection. + pub collection: String, + /// The entry's key bytes. + pub key: Vec, + /// The write. + pub op: KvPushOp, + /// Lite-assigned ID for ACK correlation. + pub batch_id: u64, + /// Stable identity of the originating producer. + pub producer_id: u64, + /// Producer epoch. + pub epoch: u64, + /// Per-stream monotonic sequence number within the epoch. + pub seq: u64, +} + +/// KV write push acknowledgment (Origin → Lite, 0xAF). +#[derive( + Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, +)] +pub struct KvPushAckMsg { + /// Collection acknowledged. + pub collection: String, + /// Key from the originating `KvPushMsg`. + pub key: Vec, + /// Batch ID from the originating `KvPushMsg`. + pub batch_id: u64, + /// `true` unless `status` is `AckStatus::Rejected`. + pub accepted: bool, + /// Refusal detail when `status` is `AckStatus::Rejected`. + pub reject_reason: Option, + /// Highest sequence from this producer's stream that Origin applied. + pub applied_seq: u64, + /// Idempotency outcome of the push. + pub status: AckStatus, +} + +/// Why Lite refused an Origin row push. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum RowPushRefusal { + /// The payload is not the row shape the collection's engine expects. + Malformed { detail: String }, + /// The payload decoded, and the local write failed. + ApplyFailed { detail: String }, +} + +impl std::fmt::Display for RowPushRefusal { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Malformed { detail } => write!(f, "malformed row push: {detail}"), + Self::ApplyFailed { detail } => write!(f, "row push apply failed: {detail}"), + } + } +} + +/// Lite's refusal of an Origin `RowPushMsg` (Lite → Origin, 0x16). +#[derive( + Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, +)] +pub struct RowPushRejectMsg { + /// Collection of the refused row. + pub collection: String, + /// Document ID of the refused row. + pub document_id: String, + /// `sequence` of the refused `RowPushMsg`. + pub sequence: u64, + /// `peer_id` of the refused `RowPushMsg`. + pub peer_id: u64, + /// Why Lite refused the row. + pub refusal: RowPushRefusal, +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_put_push_round_trips() { + let msg = KvPushMsg { + lite_id: "lite-1".into(), + collection: "cfg".into(), + key: b"k1".to_vec(), + op: KvPushOp::Put { + row: vec![0x81, 0xa1, b'k', 0x01], + expire_at_ms: 42, + }, + batch_id: 7, + producer_id: 3, + epoch: 1, + seq: 9, + }; + let bytes = zerompk::to_msgpack_vec(&msg).expect("encode"); + let back: KvPushMsg = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(back.op, msg.op); + assert_eq!(back.key, msg.key); + assert_eq!(back.seq, 9); + } + + #[test] + fn a_row_push_reject_round_trips() { + let msg = RowPushRejectMsg { + collection: "cfg".into(), + document_id: "k1".into(), + sequence: 4, + peer_id: 2, + refusal: RowPushRefusal::Malformed { + detail: "not a row map".into(), + }, + }; + let bytes = zerompk::to_msgpack_vec(&msg).expect("encode"); + let back: RowPushRejectMsg = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(back.refusal, msg.refusal); + assert_eq!(back.sequence, 4); + } +} diff --git a/nodedb-types/src/sync/wire/mod.rs b/nodedb-types/src/sync/wire/mod.rs index ffdd23dd6..a6f56e679 100644 --- a/nodedb-types/src/sync/wire/mod.rs +++ b/nodedb-types/src/sync/wire/mod.rs @@ -12,6 +12,8 @@ //! - `0x12` DeltaReject (server → client) //! - `0x13` CollectionSchema (bidirectional) //! - `0x14` CollectionPurged (server → client) +//! - `0x15` RowPush (server → client) +//! - `0x16` RowPushReject (client → server) //! - `0x20` ShapeSubscribe (client → server) //! - `0x21` ShapeSnapshot (server → client) //! - `0x22` ShapeDelta (server → client) @@ -49,6 +51,8 @@ //! - `0xAB` SpatialInsertAck (server → client) //! - `0xAC` SpatialDelete (client → server) //! - `0xAD` SpatialDeleteAck (server → client) +//! - `0xAE` KvPush (client → server) +//! - `0xAF` KvPushAck (server → client) //! - `0xFF` Ping/Pong (bidirectional) pub mod ack_result; @@ -59,6 +63,7 @@ pub mod columnar; pub mod delta; pub mod frame; pub mod fts; +pub mod kv; pub mod presence; pub mod provenance; pub mod resync; @@ -82,6 +87,7 @@ pub use delta::{ }; pub use frame::{SyncFrame, SyncMessageType}; pub use fts::{FtsDeleteAckMsg, FtsDeleteMsg, FtsIndexAckMsg, FtsIndexMsg}; +pub use kv::{KvPushAckMsg, KvPushMsg, KvPushOp, RowPushRefusal, RowPushRejectMsg}; pub use presence::{PeerPresence, PresenceBroadcastMsg, PresenceLeaveMsg, PresenceUpdateMsg}; pub use provenance::SyncProvenance; pub use resync::{ResyncReason, ResyncRequestMsg, ThrottleMsg}; diff --git a/nodedb-types/src/sync/wire/stream_id.rs b/nodedb-types/src/sync/wire/stream_id.rs index 9214c200f..c6fa8cb90 100644 --- a/nodedb-types/src/sync/wire/stream_id.rs +++ b/nodedb-types/src/sync/wire/stream_id.rs @@ -22,6 +22,7 @@ pub enum EngineKind { Fts, Spatial, Array, + Kv, } /// Derive a stable, deterministic `stream_id` for a `(engine, collection)` pair. diff --git a/nodedb/src/bridge/admission_chokepoint.rs b/nodedb/src/bridge/admission_chokepoint.rs index ab130da81..4927c17b9 100644 --- a/nodedb/src/bridge/admission_chokepoint.rs +++ b/nodedb/src/bridge/admission_chokepoint.rs @@ -108,6 +108,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -216,6 +217,7 @@ mod tests { rls_write_check: check, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index 150b6ce0a..2d3d8f7cf 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -33,6 +33,12 @@ pub enum ErrorCode { applied_seq: u64, provenance: nodedb_types::sync::wire::SyncProvenance, }, + /// A sync frame the idempotency gate held back. Nothing applied, and + /// the stream's mark did not move. `applied_seq` is that mark. + SyncNotApplied { + hold: super::SyncHold, + applied_seq: u64, + }, /// Document/collection not found. NotFound, /// Authorization failure. diff --git a/nodedb/src/bridge/envelope/mod.rs b/nodedb/src/bridge/envelope/mod.rs index 0a92b0870..f09bb0a47 100644 --- a/nodedb/src/bridge/envelope/mod.rs +++ b/nodedb/src/bridge/envelope/mod.rs @@ -7,6 +7,7 @@ pub mod payload; pub mod request; pub mod response; pub mod status; +pub mod sync_hold; pub use error_code::ErrorCode; pub use nodedb_physical::kv_atomic::CounterFault; @@ -15,3 +16,4 @@ pub use payload::Payload; pub use request::{Admission, ExemptReason, Request}; pub use response::{Response, WriteSetEntry}; pub use status::{Priority, Status}; +pub use sync_hold::SyncHold; diff --git a/nodedb/src/bridge/envelope/sync_hold.rs b/nodedb/src/bridge/envelope/sync_hold.rs new file mode 100644 index 000000000..7673f6007 --- /dev/null +++ b/nodedb/src/bridge/envelope/sync_hold.rs @@ -0,0 +1,42 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Why the sync idempotency gate held a frame back without applying it. + +use nodedb_types::sync::wire::AckStatus; + +/// A gate verdict that applies nothing and is not a refusal. +/// +/// The sender acts on each one differently, so each crosses the bridge by +/// name. None of them asks the sender to compensate. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SyncHold { + /// The frame's sequence is at or below the stream's mark: it applied + /// under an earlier delivery. + Duplicate, + /// The frame's producer epoch is below the producer's floor. + Fenced, + /// The frame skipped sequences. `expected` is the next one the stream + /// admits. + Gap { expected: u64 }, +} + +impl SyncHold { + /// The ack status a sender receives for this hold. + pub fn ack_status(self) -> AckStatus { + match self { + Self::Duplicate => AckStatus::Duplicate, + Self::Fenced => AckStatus::Fenced, + Self::Gap { expected } => AckStatus::Gap { expected }, + } + } +} + +impl std::fmt::Display for SyncHold { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Duplicate => write!(f, "duplicate"), + Self::Fenced => write!(f, "fenced producer epoch"), + Self::Gap { expected } => write!(f, "sequence gap, expected {expected}"), + } + } +} diff --git a/nodedb/src/control/backup/restore/kv_reissue.rs b/nodedb/src/control/backup/restore/kv_reissue.rs index 7b98a67c1..55f71afb9 100644 --- a/nodedb/src/control/backup/restore/kv_reissue.rs +++ b/nodedb/src/control/backup/restore/kv_reissue.rs @@ -84,6 +84,7 @@ pub(in crate::control::backup::restore) async fn reissue_kv_tables( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); super::durable::reissue_plan_durably(state, tenant, database_id, collection, plan) .await?; diff --git a/nodedb/src/control/clone/copyup.rs b/nodedb/src/control/clone/copyup.rs index 953f18b03..2508c6614 100644 --- a/nodedb/src/control/clone/copyup.rs +++ b/nodedb/src/control/clone/copyup.rs @@ -69,6 +69,7 @@ pub async fn perform_kv_clone_copyup(params: KvCopyUpParams<'_>) -> crate::Resul surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); let vshard_id = VShardId::from_collection_in_database(target_db_id, &target_coll_qualified); diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 80d849642..b99150b1d 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -8,9 +8,11 @@ //! fails to compile here until it is mirrored on the wire instead of silently //! degrading to `Internal` and losing its SQLSTATE at the coordinator. -use nodedb_cluster::rpc_codec::{DataPlaneCounterFault, DataPlaneErrorCode, TypedClusterError}; +use nodedb_cluster::rpc_codec::{ + DataPlaneCounterFault, DataPlaneErrorCode, DataPlaneSyncHold, TypedClusterError, +}; -use crate::bridge::envelope::{CounterFault, ErrorCode}; +use crate::bridge::envelope::{CounterFault, ErrorCode, SyncHold}; /// Map a local-execution [`crate::Error`] to the wire error a remote caller /// receives. @@ -89,6 +91,10 @@ impl From for DataPlaneErrorCode { stream_id: provenance.stream_id, seq: provenance.seq, }, + ErrorCode::SyncNotApplied { hold, applied_seq } => Self::SyncNotApplied { + hold: sync_hold_to_wire(hold), + applied_seq, + }, ErrorCode::NotFound => Self::NotFound, ErrorCode::RejectedAuthz { resource } => Self::RejectedAuthz { resource }, ErrorCode::ConflictRetry => Self::ConflictRetry, @@ -204,6 +210,10 @@ impl From for ErrorCode { seq, }, }, + DataPlaneErrorCode::SyncNotApplied { hold, applied_seq } => Self::SyncNotApplied { + hold: sync_hold_from_wire(hold), + applied_seq, + }, DataPlaneErrorCode::NotFound => Self::NotFound, DataPlaneErrorCode::RejectedAuthz { resource } => Self::RejectedAuthz { resource }, DataPlaneErrorCode::ConflictRetry => Self::ConflictRetry, @@ -299,6 +309,24 @@ impl From for ErrorCode { } } +/// The wire form of a sync hold. +fn sync_hold_to_wire(hold: SyncHold) -> DataPlaneSyncHold { + match hold { + SyncHold::Duplicate => DataPlaneSyncHold::Duplicate, + SyncHold::Fenced => DataPlaneSyncHold::Fenced, + SyncHold::Gap { expected } => DataPlaneSyncHold::Gap { expected }, + } +} + +/// The sync hold a wire form names. +fn sync_hold_from_wire(hold: DataPlaneSyncHold) -> SyncHold { + match hold { + DataPlaneSyncHold::Duplicate => SyncHold::Duplicate, + DataPlaneSyncHold::Fenced => SyncHold::Fenced, + DataPlaneSyncHold::Gap { expected } => SyncHold::Gap { expected }, + } +} + /// The wire form of a counter fault. Both types live in other crates, so the /// mapping is a function, not a `From` impl. fn counter_fault_to_wire(fault: CounterFault) -> DataPlaneCounterFault { @@ -382,6 +410,22 @@ mod tests { assert_eq!(ErrorCode::from(wire), original); } + #[test] + fn sync_not_applied_roundtrips_verbatim() { + for hold in [ + SyncHold::Duplicate, + SyncHold::Fenced, + SyncHold::Gap { expected: 7 }, + ] { + let original = ErrorCode::SyncNotApplied { + hold, + applied_seq: 6, + }; + let wire = DataPlaneErrorCode::from(original.clone()); + assert_eq!(ErrorCode::from(wire), original); + } + } + #[test] fn expired_before_execution_roundtrips_verbatim() { let wire = DataPlaneErrorCode::from(ErrorCode::ExpiredBeforeExecution); diff --git a/nodedb/src/control/gateway/router.rs b/nodedb/src/control/gateway/router.rs index 6cfb56bf7..e946b6581 100644 --- a/nodedb/src/control/gateway/router.rs +++ b/nodedb/src/control/gateway/router.rs @@ -354,6 +354,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let routes = route_plan( plan, diff --git a/nodedb/src/control/maintenance/clone_materializer/kv.rs b/nodedb/src/control/maintenance/clone_materializer/kv.rs index abe438050..63dd27502 100644 --- a/nodedb/src/control/maintenance/clone_materializer/kv.rs +++ b/nodedb/src/control/maintenance/clone_materializer/kv.rs @@ -105,6 +105,7 @@ pub(super) async fn materialize_kv_collection( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); let resp = dispatch_local(state, tenant_id, db_id, &target_qualified, plan, None).await?; diff --git a/nodedb/src/control/planner/rls_injection/kv.rs b/nodedb/src/control/planner/rls_injection/kv.rs index 36c074b78..43c642e35 100644 --- a/nodedb/src/control/planner/rls_injection/kv.rs +++ b/nodedb/src/control/planner/rls_injection/kv.rs @@ -281,6 +281,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -294,6 +295,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -353,6 +355,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); assert!(matches!( inject(&mut plan, &store), @@ -466,6 +469,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), + provenance: None, }, KvOp::FieldSet { collection: collection(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs index 7fb7e237d..48c6a126c 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs @@ -112,6 +112,7 @@ pub(in super::super) fn convert_kv_insert( // RLS injection pass. returning: None, rls_filters: Vec::new(), + provenance: None, }, }; tasks.push(PhysicalTask { diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs index 623db1e52..2805683d9 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs @@ -71,6 +71,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( // Attached by `inject_returning_spec` after plan conversion. returning: None, rls_filters: Vec::new(), + provenance: None, }), post_set_op: PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index 29b7f4f31..de5d7eba9 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -895,6 +895,7 @@ mod tests { surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let responder = tokio::spawn(respond_once_with( Arc::clone(&state), diff --git a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs index 4c6b021da..fc6d66b3e 100644 --- a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs +++ b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs @@ -149,6 +149,7 @@ mod tests { surrogate: Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/src/control/server/dispatch_utils/durable_write.rs b/nodedb/src/control/server/dispatch_utils/durable_write.rs index 6797151fd..4b4390d0a 100644 --- a/nodedb/src/control/server/dispatch_utils/durable_write.rs +++ b/nodedb/src/control/server/dispatch_utils/durable_write.rs @@ -51,6 +51,7 @@ pub(crate) async fn dispatch_durable_autocommit_write( vshard_id: write.vshard_id, }, &write.plan, + write.event_source, ) .await? { @@ -94,6 +95,7 @@ pub(crate) async fn dispatch_authorized_durable_write( vshard_id: checked.vshard_id(), }, checked.plan(), + crate::event::EventSource::User, ) .await? { @@ -102,6 +104,59 @@ pub(crate) async fn dispatch_authorized_durable_write( dispatch_authorized_autocommit_write(shared, checked, trace_id).await } +/// Dispatch an authorized autocommit write on the durable route, tagged with +/// `event_source`. +/// +/// Same routing as [`dispatch_authorized_durable_write`], for a write that +/// runs under a source other than a client's, such as a Lite sync push. Every +/// replica stamps the source on the write's events, so a synced write does +/// not re-fire AFTER triggers. +/// +/// A clustered write whose RLS write policy must resolve against current rows +/// before it is proposed is refused: the resolved write carries none of the +/// plan's sync provenance, so the idempotency gate could not run. +pub(crate) async fn dispatch_authorized_durable_write_with_source( + shared: &SharedState, + checked: CloneCheckedTask, + trace_id: TraceId, + event_source: crate::event::EventSource, +) -> crate::Result { + if checked.txn_id().is_none() + && shared.async_raft_proposer().is_some() + && crate::control::write_resolve::resolver_for_plan(checked.plan()).is_some() + { + return Err(crate::Error::PlanError { + detail: format!( + "a {event_source:?} write to '{}' needs its row-level-security policy resolved \ + before it is proposed, and the resolved write cannot carry its sync provenance", + checked.plan().collection().unwrap_or("") + ), + }); + } + if checked.txn_id().is_none() + && let Some(response) = propose_if_replicable( + shared, + WriteTarget { + tenant_id: checked.tenant_id(), + database_id: checked.database_id(), + vshard_id: checked.vshard_id(), + }, + checked.plan(), + event_source, + ) + .await? + { + return Ok(response); + } + super::dispatch::dispatch_authorized_autocommit_write_with_source( + shared, + checked, + trace_id, + event_source, + ) + .await +} + /// Dispatch one authorized task by its class: a write on the durable route, /// anything else on the read route. /// @@ -135,6 +190,7 @@ async fn propose_if_replicable( shared: &SharedState, target: WriteTarget, plan: &PhysicalPlan, + event_source: crate::event::EventSource, ) -> crate::Result> { let Some(proposer) = shared.async_raft_proposer() else { return Ok(None); @@ -148,6 +204,7 @@ async fn propose_if_replicable( let Some(entry) = to_replicated_entry(tenant_id, database_id, vshard_id, &replicable)? else { return Ok(None); }; + let entry = entry.with_event_source(event_source); let (payload, write_version) = propose_replicated_entry(shared, proposer, entry).await?; // Replicas apply with `ChangeFeedOwner::Unowned`. The proposing node // handled the write once, so it publishes the change event. diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index 40c0cf6be..3250b043f 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -29,8 +29,8 @@ pub(crate) use dispatch::{ }; pub use durability_barrier::writes_acked_without_durability; pub(crate) use durable_write::{ - dispatch_authorized_durable_write, dispatch_authorized_task_by_class, - dispatch_durable_autocommit_write, + dispatch_authorized_durable_write, dispatch_authorized_durable_write_with_source, + dispatch_authorized_task_by_class, dispatch_durable_autocommit_write, }; pub(crate) use error_status::reject_data_plane_error; pub(crate) use local_read::{LocalRead, dispatch_local_read}; diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index f96c73818..e580e2323 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -35,7 +35,8 @@ use crate::bridge::envelope::ErrorCode; /// precondition is not final: another replica can apply the same entry, and /// a redelivery here can too. That covers admission and capacity verdicts, /// a task that expired before it started, concurrency retries, the staging -/// byte budget, and `RetryableRefusal`, +/// byte budget, a sync hold, which depends on the core's own stream mark, +/// and `RetryableRefusal`, /// which a committed-redo apply answers with after it rolled a failed /// install back. pub(crate) fn refusal_is_final(code: &ErrorCode) -> bool { @@ -43,6 +44,7 @@ pub(crate) fn refusal_is_final(code: &ErrorCode) -> bool { && !matches!( code, ErrorCode::RetryableRefusal { .. } + | ErrorCode::SyncNotApplied { .. } | ErrorCode::RateExceeded { .. } | ErrorCode::CollectionDraining { .. } | ErrorCode::DispatchCapacity { .. } @@ -71,6 +73,8 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { | ErrorCode::RejectedPrevalidation { .. } // The sync gate refused the frame before its delta installed. | ErrorCode::SyncRejected { .. } + // The sync gate held the frame back before anything installed. + | ErrorCode::SyncNotApplied { .. } | ErrorCode::RejectedAuthz { .. } | ErrorCode::RejectedDanglingEdge { .. } | ErrorCode::AppendOnlyViolation { .. } diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index 5c0c0431e..eb664220a 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -83,6 +83,7 @@ pub(crate) fn build_point_put( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, })) } Some(CollectionType::Columnar(ColumnarProfile::Timeseries { .. })) => { @@ -151,6 +152,7 @@ pub(crate) fn build_point_delete( // The native point-delete carries no RETURNING clause. returning: None, rls_filters: Vec::new(), + provenance: None, })), Some(CollectionType::Columnar(ColumnarProfile::Timeseries { .. })) => { Err(crate::Error::BadRequest { diff --git a/nodedb/src/control/server/pgwire/handler/plan.rs b/nodedb/src/control/server/pgwire/handler/plan.rs index a0275f0cf..db3a0e90d 100644 --- a/nodedb/src/control/server/pgwire/handler/plan.rs +++ b/nodedb/src/control/server/pgwire/handler/plan.rs @@ -97,6 +97,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); let outcome = calvin_tag_for_plan(&plan).expect("an upsert folds without a round-trip"); let tag: pgwire::messages::response::CommandComplete = render(outcome).into(); diff --git a/nodedb/src/control/server/resp/handler_kv/strings.rs b/nodedb/src/control/server/resp/handler_kv/strings.rs index bfa7aa09b..44f9290d9 100644 --- a/nodedb/src/control/server/resp/handler_kv/strings.rs +++ b/nodedb/src/control/server/resp/handler_kv/strings.rs @@ -140,6 +140,7 @@ pub(in crate::control::server::resp) async fn handle_set( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); // A rejected write surfaces as the error it is, never as `OK`. @@ -170,6 +171,7 @@ pub(in crate::control::server::resp) async fn handle_del( // RESP has no RETURNING clause. returning: None, rls_filters: Vec::new(), + provenance: None, }); // A rejected delete surfaces as the error it is, never as `0` deleted. diff --git a/nodedb/src/control/server/resp/handler_sorted.rs b/nodedb/src/control/server/resp/handler_sorted.rs index db4f1d748..b82cda2bc 100644 --- a/nodedb/src/control/server/resp/handler_sorted.rs +++ b/nodedb/src/control/server/resp/handler_sorted.rs @@ -81,6 +81,7 @@ pub(super) async fn handle_zadd( surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); match dispatch_kv_write(state, session, plan).await { @@ -116,6 +117,7 @@ pub(super) async fn handle_zrem( // RESP has no RETURNING clause. returning: None, rls_filters: Vec::new(), + provenance: None, }); match dispatch_kv_write(state, session, plan).await { diff --git a/nodedb/src/control/server/response_shape/calvin_fold.rs b/nodedb/src/control/server/response_shape/calvin_fold.rs index acbff10fc..9cbfd05de 100644 --- a/nodedb/src/control/server/response_shape/calvin_fold.rs +++ b/nodedb/src/control/server/response_shape/calvin_fold.rs @@ -243,6 +243,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); assert_eq!( calvin_tag_for_plan(&plan), @@ -264,6 +265,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), + provenance: None, }); assert!(calvin_tag_for_plan(&delete).is_none()); diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs b/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs index 7c51c000e..cf9b6fad7 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs @@ -205,6 +205,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: spec(), rls_filters: Vec::new(), + provenance: None, }), PhysicalPlan::Kv(KvOp::BatchPut { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), @@ -242,6 +243,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/src/control/server/shared/clone_write/kv.rs b/nodedb/src/control/server/shared/clone_write/kv.rs index d84093828..d534e9221 100644 --- a/nodedb/src/control/server/shared/clone_write/kv.rs +++ b/nodedb/src/control/server/shared/clone_write/kv.rs @@ -38,6 +38,10 @@ pub(super) async fn intercept_kv_clone_write( rls_write_check, returning, rls_filters, + // The tombstone path answers a count, not a sync ack. A Lite KV + // push that lands here moves its stream mark on its own, after + // this returns `Handled`. + provenance: _, }) => { // Delete may have multiple keys; handle each. We serialize here // (one tombstone per key) and return Handled with synthetic OK. @@ -167,6 +171,7 @@ pub(super) async fn intercept_kv_clone_write( // Same statement, same projection and read gate. returning: returning.clone(), rls_filters: rls_filters.clone(), + provenance: None, }); let vshard_id = VShardId::from_collection_in_database(db_id, collection_qualified); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, delete_plan) diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index eaf8962fd..c6bd82749 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -254,6 +254,7 @@ pub async fn rate_reset( rls_write_check: nodedb_types::RlsWriteCheck::system_internal_collection(), returning: None, rls_filters: Vec::new(), + provenance: None, }); match dispatch_counter_write(state, tenant_id, vshard, plan).await { diff --git a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs index 5afe571cf..27a10e040 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs @@ -182,6 +182,7 @@ pub async fn weighted_pick( surrogate: audit_surrogate, returning: None, rls_filters: Vec::new(), + provenance: None, }); // The caller asked for an audited pick, so the pick is answered only // once its audit record is durable: Raft in cluster mode, else the diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index db5736066..630e2a889 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -40,6 +40,13 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, sqlstate::CHECK_VIOLATION, format!("sync frame rejected: {violation}"), ), + // Nothing applied, and the sender re-sends or retires the frame by + // the hold, so it takes the class drivers already retry on. + ErrorCode::SyncNotApplied { hold, .. } => ( + "ERROR", + sqlstate::SERIALIZATION_FAILURE, + format!("sync frame not applied: {hold}"), + ), // Nothing applied and the identical statement is expected to succeed // later, so drivers get the same class they already retry on rather // than a check violation they would surface as permanent. diff --git a/nodedb/src/control/server/shared/session/commit/metering.rs b/nodedb/src/control/server/shared/session/commit/metering.rs index 909d63f92..63827869e 100644 --- a/nodedb/src/control/server/shared/session/commit/metering.rs +++ b/nodedb/src/control/server/shared/session/commit/metering.rs @@ -114,6 +114,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, })) } diff --git a/nodedb/src/control/server/shared/sql/staging_predicates.rs b/nodedb/src/control/server/shared/sql/staging_predicates.rs index abd0a0a42..550fd2378 100644 --- a/nodedb/src/control/server/shared/sql/staging_predicates.rs +++ b/nodedb/src/control/server/shared/sql/staging_predicates.rs @@ -535,6 +535,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }; assert_eq!( kv_write_shape(&op).expect("stageable").tag_kind(&payload), diff --git a/nodedb/src/control/server/shared/write_admission/lock_keys.rs b/nodedb/src/control/server/shared/write_admission/lock_keys.rs index bf5b6f3d2..60981ece3 100644 --- a/nodedb/src/control/server/shared/write_admission/lock_keys.rs +++ b/nodedb/src/control/server/shared/write_admission/lock_keys.rs @@ -379,6 +379,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), + provenance: None, }), None ); diff --git a/nodedb/src/control/server/shared/write_admission/predicate/plan_is_write.rs b/nodedb/src/control/server/shared/write_admission/predicate/plan_is_write.rs index f72250e49..5cf191d85 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/plan_is_write.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/plan_is_write.rs @@ -77,6 +77,7 @@ mod tests { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); assert!(plan_is_write(&plan)); } diff --git a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs index ef8416f70..9fd8b2854 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/txn_buffering/classify.rs @@ -1196,6 +1196,7 @@ mod tests { surrogate: Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), PhysicalPlan::Kv(KvOp::Insert { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), @@ -1232,6 +1233,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), + provenance: None, }), PhysicalPlan::Kv(KvOp::PredicateUpdate { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), diff --git a/nodedb/src/control/server/shared/write_admission/predicate/user_data_write.rs b/nodedb/src/control/server/shared/write_admission/predicate/user_data_write.rs index 19e98dc53..854023b4b 100644 --- a/nodedb/src/control/server/shared/write_admission/predicate/user_data_write.rs +++ b/nodedb/src/control/server/shared/write_admission/predicate/user_data_write.rs @@ -69,6 +69,7 @@ mod tests { surrogate: Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }); assert!(plan_writes_user_data(&plan)); } diff --git a/nodedb/src/control/server/sync/kv_handler.rs b/nodedb/src/control/server/sync/kv_handler.rs new file mode 100644 index 000000000..accee6289 --- /dev/null +++ b/nodedb/src/control/server/sync/kv_handler.rs @@ -0,0 +1,267 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! KV push dispatch for sync sessions. +//! +//! A Lite `KvPushMsg` becomes a `KvOp::Put` or `KvOp::Delete` carrying the +//! frame's sync provenance. It takes the durable route every client KV write +//! takes, tagged `EventSource::CrdtSync`: +//! - the funnel appends its redo record and a `SyncSeqAdvance` record under +//! one outcome-floor window; +//! - a clustered write is proposed through Raft, and every replica gates it +//! on the same stream mark; +//! - the Data Plane's sync gate answers the frame's ack. +//! +//! The session handler lives in `kv_session.rs`. + +use async_trait::async_trait; + +use nodedb_physical::physical_plan::KvOp; +use nodedb_types::sync::wire::{AckStatus, SyncAckResult, SyncProvenance}; +use nodedb_types::{QualifiedCollection, RlsWriteCheck}; + +use crate::bridge::envelope::{ErrorCode, PhysicalPlan, SyncHold}; +use crate::control::server::dispatch_utils::RecordOwner; +use crate::control::server::shared::clone_write::{ + CloneCheckedOutcome, InterceptAndAuthorizeParams, intercept_and_authorize, +}; +use crate::event::EventSource; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; + +/// The write a KV push applies. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum KvPushWriteOp { + /// Store `body` at the key. `ttl_ms` is `0` for an entry that never + /// expires. + Put { body: Vec, ttl_ms: u64 }, + /// Remove the key. + Delete, +} + +/// One KV push, ready to dispatch. +#[derive(Debug, Clone)] +pub struct KvPushWrite { + pub collection: String, + pub key: Vec, + pub op: KvPushWriteOp, + pub provenance: SyncProvenance, +} + +/// Encapsulates the dispatch of a KV push. +#[async_trait] +pub trait KvPushDispatcher: Send + Sync { + /// Apply `write` and return the Data Plane's `SyncAckResult` payload. + async fn apply(&self, tenant_id: TenantId, write: KvPushWrite) -> crate::Result>; + + /// Move the stream mark past a frame Origin refused before it reached + /// the Data Plane, so the producer's next frame is not a gap. + async fn skip( + &self, + tenant_id: TenantId, + collection: &str, + provenance: SyncProvenance, + ) -> crate::Result<()>; +} + +// ── SharedState adapter ────────────────────────────────────────────────────── + +/// Production dispatcher: runs admission, authorization and row-level +/// security, then the durable write route. +pub struct SharedStateKvDispatcher<'a> { + pub shared: &'a crate::control::state::SharedState, + pub(crate) identity: Option<&'a crate::control::security::identity::AuthenticatedIdentity>, + pub(crate) database_id: DatabaseId, + /// The session's remote address, for the blacklist and risk checks. + pub(crate) peer_addr: &'a str, +} + +#[async_trait] +impl KvPushDispatcher for SharedStateKvDispatcher<'_> { + async fn apply(&self, tenant_id: TenantId, write: KvPushWrite) -> crate::Result> { + let identity = self.identity.ok_or_else(|| crate::Error::RejectedAuthz { + tenant_id, + resource: "authenticated sync identity required".into(), + })?; + let database_id = self.database_id; + let request = crate::control::security::request_scope::ClientRequestScope::for_database( + identity, + self.shared.auth_stores(), + database_id, + self.peer_addr, + ); + crate::control::server::session_auth::check_blacklist_and_status(self.shared, &request)?; + self.shared.check_tenant_quota(tenant_id)?; + + let KvPushWrite { + collection, + key, + op, + provenance, + } = write; + let qualified = QualifiedCollection::new(database_id, &collection); + let mut plan = match op { + KvPushWriteOp::Put { body, ttl_ms } => { + let surrogate = self.shared.surrogate_assigner.assign( + database_id, + tenant_id, + &collection, + &key, + )?; + PhysicalPlan::Kv(KvOp::Put { + collection: qualified, + key: key.clone(), + value: body, + ttl_ms, + surrogate, + returning: None, + rls_filters: Vec::new(), + provenance: Some(provenance.clone()), + }) + } + KvPushWriteOp::Delete => PhysicalPlan::Kv(KvOp::Delete { + collection: qualified, + keys: vec![key.clone()], + rls_write_check: RlsWriteCheck::pending_injection(), + returning: None, + rls_filters: Vec::new(), + provenance: Some(provenance.clone()), + }), + }; + crate::control::planner::rls_injection::inject_rls_for_single_plan( + tenant_id.as_u64(), + database_id, + &mut plan, + &self.shared.rls, + request.scope().auth(), + )?; + + let task = nodedb_physical::physical_task::PhysicalTask { + tenant_id, + vshard_id: VShardId::from_collection_in_database(database_id, &collection), + database_id, + plan, + post_set_op: nodedb_physical::physical_task::PostSetOp::None, + txn_id: None, + }; + let emitter = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( + &self.shared.audit, + )); + let checked = match intercept_and_authorize(InterceptAndAuthorizeParams { + state: self.shared, + task, + identity, + tenant_id, + permissions: &self.shared.permissions, + roles: &self.shared.roles, + emitter: &emitter, + }) + .await? + { + CloneCheckedOutcome::Proceed(checked) => checked, + // The clone copy-up path applied the write itself, outside the + // gate. The frame still takes its sequence. + CloneCheckedOutcome::Handled(response) => { + crate::control::server::shared::response_payload::payload_or_typed_error(response)?; + self.skip(tenant_id, &collection, provenance.clone()) + .await?; + return encode_applied(provenance.seq); + } + }; + let response = + crate::control::server::dispatch_utils::dispatch_authorized_durable_write_with_source( + self.shared, + checked, + TraceId::ZERO, + EventSource::CrdtSync, + ) + .await?; + crate::control::server::shared::response_payload::payload_or_typed_error(response) + } + + async fn skip( + &self, + tenant_id: TenantId, + collection: &str, + provenance: SyncProvenance, + ) -> crate::Result<()> { + use crate::control::server::wal_dispatch::{WalAppendRequest, wal_append}; + + let database_id = self.database_id; + let owner = RecordOwner { + tenant_id, + database_id, + vshard_id: VShardId::from_collection_in_database(database_id, collection), + }; + // A delete of no keys applies nothing. Its provenance moves the mark, + // and its records make the move durable. + let plan = PhysicalPlan::Kv(KvOp::Delete { + collection: QualifiedCollection::new(database_id, collection), + keys: Vec::new(), + rls_write_check: RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + provenance: Some(provenance), + }); + let (minted, _) = super::raft_dispatch::append_under_window(self.shared, owner, |wal| { + wal_append(WalAppendRequest { + wal, + event_source: EventSource::CrdtSync, + tenant_id, + vshard_id: owner.vshard_id, + database_id, + plan: &plan, + credentials: None, + now_override: None, + }) + .map(|outcome| outcome.lsn) + }) + .await?; + let response = super::raft_dispatch::dispatch_trusted_internal_minted_sync_response( + self.shared, + owner, + plan, + EventSource::CrdtSync, + minted, + ) + .await?; + match crate::control::server::shared::response_payload::payload_or_typed_error(response) { + Ok(_) => Ok(()), + // An earlier delivery already moved the mark past this frame. + Err(crate::Error::DataPlane(ErrorCode::SyncNotApplied { + hold: SyncHold::Duplicate, + .. + })) => Ok(()), + Err(error) => Err(error), + } + } +} + +/// The `SyncAckResult` payload of an applied frame. +fn encode_applied(seq: u64) -> crate::Result> { + zerompk::to_msgpack_vec(&SyncAckResult::acked(AckStatus::Applied, seq)).map_err(|e| { + crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("kv push ack: {e}"), + } + }) +} + +// ── NoOp dispatcher (loud failure) ────────────────────────────────────────── + +/// Dispatcher used when `SharedState` is unavailable. +pub struct NoOpKvDispatcher; + +#[async_trait] +impl KvPushDispatcher for NoOpKvDispatcher { + async fn apply(&self, _tenant_id: TenantId, _write: KvPushWrite) -> crate::Result> { + Err(super::raft_dispatch::noop_dispatch_error("kv push")) + } + + async fn skip( + &self, + _tenant_id: TenantId, + _collection: &str, + _provenance: SyncProvenance, + ) -> crate::Result<()> { + Err(super::raft_dispatch::noop_dispatch_error("kv push skip")) + } +} diff --git a/nodedb/src/control/server/sync/kv_session.rs b/nodedb/src/control/server/sync/kv_session.rs new file mode 100644 index 000000000..640cef809 --- /dev/null +++ b/nodedb/src/control/server/sync/kv_session.rs @@ -0,0 +1,424 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Session-level KV push handler: `SyncSession::handle_kv_push`. +//! +//! A put's `{key, value…}` row decodes to the stored body through +//! `kv_row_to_body_fields` and `row_to_kv_body`, the inverse of the row every +//! KV read returns. A put whose absolute expiry already passed applies as a +//! delete: the entry is gone on the Lite that wrote it. +//! +//! Every outcome is one `KvPushAckMsg`: +//! - a gate verdict (`Applied`, `Duplicate`, `Fenced`, `Gap`) passes +//! through; +//! - a dispatch that never got a verdict is a retryable `Gap` at the frame's +//! own sequence; +//! - a terminal refusal is `Rejected`. A refusal decided before the Data +//! Plane still moves the stream mark past the frame, so the producer's +//! next frame is not a gap. + +use tracing::{debug, error}; + +use nodedb_query::msgpack_scan::{kv_row_to_body_fields, row_to_kv_body}; +use nodedb_types::Value; +use nodedb_types::sync::wire::{ + AckStatus, EngineKind, KvPushAckMsg, KvPushMsg, KvPushOp, SyncFrame, SyncMessageType, + SyncProvenance, stream_id_for, +}; + +use super::kv_handler::{KvPushDispatcher, KvPushWrite, KvPushWriteOp}; +use super::session::SyncSession; +use crate::bridge::envelope::ErrorCode; +use crate::types::TenantId; + +/// What the session tells the sender about one push. +enum KvPushOutcome { + /// A status at a stream mark. + Status { status: AckStatus, applied_seq: u64 }, + /// A terminal refusal. `mark_moved` says whether the Data Plane already + /// moved the stream mark past the frame. + Refused { reason: String, mark_moved: bool }, +} + +impl SyncSession { + /// Process a `KvPushMsg` and return its `KvPushAckMsg` frame. + pub async fn handle_kv_push( + &mut self, + msg: &KvPushMsg, + dispatcher: &D, + ) -> Option { + self.last_activity = std::time::Instant::now(); + + if !self.authenticated { + return kv_ack( + msg, + KvPushOutcome::Refused { + reason: "unauthenticated".to_string(), + mark_moved: false, + }, + ); + } + + let tenant_id = self.tenant_id.unwrap_or(TenantId::new(0)); + let provenance = SyncProvenance { + producer_id: self.producer_id, + epoch: self.accepted_epoch, + stream_id: stream_id_for(EngineKind::Kv, &msg.collection), + seq: msg.seq, + }; + debug!( + session = %self.session_id, + collection = %msg.collection, + batch_id = msg.batch_id, + seq = msg.seq, + lite_id = %msg.lite_id, + "kv push: dispatching" + ); + + let outcome = match push_write(msg, provenance.clone()) { + Ok(write) => match dispatcher.apply(tenant_id, write).await { + Ok(payload) => { + let wire = super::ack_decode::decode_sync_ack( + &payload, + "kv push", + &self.session_id, + &msg.collection, + msg.seq, + ) + .into_wire(); + match wire.status { + AckStatus::Rejected { reason } => KvPushOutcome::Refused { + reason, + mark_moved: true, + }, + status => KvPushOutcome::Status { + status, + applied_seq: wire.applied_seq, + }, + } + } + Err(e) => dispatch_outcome(&e, msg.seq), + }, + Err(reason) => KvPushOutcome::Refused { + reason, + mark_moved: false, + }, + }; + + let outcome = match outcome { + KvPushOutcome::Refused { + reason, + mark_moved: false, + } => { + error!( + session = %self.session_id, + collection = %msg.collection, + batch_id = msg.batch_id, + seq = msg.seq, + %reason, + "kv push refused before the Data Plane" + ); + let mark_moved = match dispatcher + .skip(tenant_id, &msg.collection, provenance) + .await + { + Ok(()) => true, + Err(e) => { + error!( + session = %self.session_id, + collection = %msg.collection, + seq = msg.seq, + error = %e, + "kv push: the stream mark could not move past a refused frame; \ + the producer's next frame reports a gap" + ); + false + } + }; + KvPushOutcome::Refused { reason, mark_moved } + } + other => other, + }; + + match &outcome { + KvPushOutcome::Status { + status: AckStatus::Applied, + .. + } => self.mutations_applied += 1, + KvPushOutcome::Status { + status: AckStatus::Duplicate, + .. + } => self.mutations_deduplicated += 1, + KvPushOutcome::Status { .. } => self.mutations_not_applied += 1, + KvPushOutcome::Refused { .. } => self.mutations_rejected += 1, + } + self.mutations_processed += 1; + kv_ack(msg, outcome) + } +} + +/// The write a push applies, or why Origin refuses it on its content. +fn push_write(msg: &KvPushMsg, provenance: SyncProvenance) -> Result { + let op = match &msg.op { + KvPushOp::Delete => KvPushWriteOp::Delete, + KvPushOp::Put { row, expire_at_ms } => { + let now_ms = crate::engine::kv::current_ms(); + if *expire_at_ms != 0 && *expire_at_ms <= now_ms { + KvPushWriteOp::Delete + } else { + let key = String::from_utf8_lossy(&msg.key); + let (fields, shape) = kv_row_to_body_fields(&key, row) + .map_err(|e| format!("KV push row does not decode: {e}"))?; + let body = row_to_kv_body(&Value::Object(fields), shape) + .map_err(|e| format!("KV push row does not encode as a body: {e}"))?; + let ttl_ms = if *expire_at_ms == 0 { + 0 + } else { + expire_at_ms - now_ms + }; + KvPushWriteOp::Put { body, ttl_ms } + } + } + }; + Ok(KvPushWrite { + collection: msg.collection.clone(), + key: msg.key.clone(), + op, + provenance, + }) +} + +/// The outcome of a dispatch that returned an error. +fn dispatch_outcome(error: &crate::Error, seq: u64) -> KvPushOutcome { + match error { + crate::Error::DataPlane(ErrorCode::SyncNotApplied { hold, applied_seq }) => { + KvPushOutcome::Status { + status: hold.ack_status(), + applied_seq: *applied_seq, + } + } + crate::Error::DataPlane(ErrorCode::SyncRejected { violation, .. }) => { + KvPushOutcome::Refused { + reason: violation.to_string(), + mark_moved: true, + } + } + other => match super::refusal::ack_status_for_dispatch_error(other, seq) { + AckStatus::Rejected { reason } => KvPushOutcome::Refused { + reason, + mark_moved: false, + }, + status => KvPushOutcome::Status { + status, + applied_seq: seq.saturating_sub(1), + }, + }, + } +} + +/// Encode the ack frame for `outcome`. +fn kv_ack(msg: &KvPushMsg, outcome: KvPushOutcome) -> Option { + let (status, applied_seq) = match outcome { + KvPushOutcome::Status { + status, + applied_seq, + } => (status, applied_seq), + KvPushOutcome::Refused { reason, mark_moved } => ( + AckStatus::Rejected { reason }, + if mark_moved { + msg.seq + } else { + msg.seq.saturating_sub(1) + }, + ), + }; + let ack = KvPushAckMsg { + collection: msg.collection.clone(), + key: msg.key.clone(), + batch_id: msg.batch_id, + accepted: !matches!(status, AckStatus::Rejected { .. }), + reject_reason: super::refusal::reject_reason_for(&status), + applied_seq, + status, + }; + SyncFrame::try_encode(SyncMessageType::KvPushAck, &ack) +} + +#[cfg(test)] +mod tests { + use std::sync::{Arc, Mutex}; + + use async_trait::async_trait; + use nodedb_query::msgpack_scan::kv_row_msgpack; + + use super::*; + use crate::bridge::envelope::SyncHold; + + /// Records every dispatch and answers with a fixed result. + struct MockDispatcher { + applied: Arc>>, + skipped: Arc>>, + answer: fn(&KvPushWrite) -> crate::Result>, + } + + impl MockDispatcher { + fn answering(answer: fn(&KvPushWrite) -> crate::Result>) -> Self { + Self { + applied: Arc::new(Mutex::new(Vec::new())), + skipped: Arc::new(Mutex::new(Vec::new())), + answer, + } + } + } + + #[async_trait] + impl KvPushDispatcher for MockDispatcher { + async fn apply(&self, _tenant_id: TenantId, write: KvPushWrite) -> crate::Result> { + let answer = (self.answer)(&write); + self.applied.lock().expect("applied").push(write); + answer + } + + async fn skip( + &self, + _tenant_id: TenantId, + _collection: &str, + provenance: SyncProvenance, + ) -> crate::Result<()> { + self.skipped.lock().expect("skipped").push(provenance.seq); + Ok(()) + } + } + + fn applied_at(write: &KvPushWrite) -> crate::Result> { + Ok( + zerompk::to_msgpack_vec(&nodedb_types::sync::wire::SyncAckResult::acked( + AckStatus::Applied, + write.provenance.seq, + )) + .expect("encode ack"), + ) + } + + fn duplicate(_write: &KvPushWrite) -> crate::Result> { + Err(crate::Error::DataPlane(ErrorCode::SyncNotApplied { + hold: SyncHold::Duplicate, + applied_seq: 5, + })) + } + + fn authenticated() -> SyncSession { + let mut session = SyncSession::new("kv-push".to_string()); + session.authenticated = true; + session.producer_id = 11; + session.accepted_epoch = 1; + session + } + + fn put_msg(key: &str, row: Vec, seq: u64) -> KvPushMsg { + KvPushMsg { + lite_id: "lite".into(), + collection: "cfg".into(), + key: key.as_bytes().to_vec(), + op: KvPushOp::Put { + row, + expire_at_ms: 0, + }, + batch_id: seq, + producer_id: 11, + epoch: 1, + seq, + } + } + + fn ack_of(frame: Option) -> KvPushAckMsg { + frame + .expect("ack frame") + .decode_body() + .expect("decode KvPushAck") + } + + #[tokio::test] + async fn a_pushed_row_dispatches_its_stored_body_and_acks_applied() { + let mut session = authenticated(); + let mock = MockDispatcher::answering(applied_at); + let msg = put_msg("k1", kv_row_msgpack("k1", b"v1"), 5); + + let ack = ack_of(session.handle_kv_push(&msg, &mock).await); + + assert_eq!(ack.status, AckStatus::Applied); + assert!(ack.accepted); + assert_eq!(ack.applied_seq, 5); + let applied = mock.applied.lock().expect("applied"); + assert_eq!( + applied[0].op, + KvPushWriteOp::Put { + body: b"v1".to_vec(), + ttl_ms: 0 + } + ); + assert_eq!( + applied[0].provenance.stream_id, + stream_id_for(EngineKind::Kv, "cfg") + ); + } + + #[tokio::test] + async fn a_resent_push_acks_duplicate() { + let mut session = authenticated(); + let mock = MockDispatcher::answering(duplicate); + let msg = put_msg("k1", kv_row_msgpack("k1", b"v1"), 5); + + let ack = ack_of(session.handle_kv_push(&msg, &mock).await); + + assert_eq!(ack.status, AckStatus::Duplicate); + assert!(ack.accepted); + assert_eq!(session.mutations_deduplicated, 1); + } + + #[tokio::test] + async fn a_row_that_is_not_a_row_map_is_rejected_and_its_sequence_is_skipped() { + let mut session = authenticated(); + let mock = MockDispatcher::answering(applied_at); + let msg = put_msg("k1", b"v1".to_vec(), 6); + + let ack = ack_of(session.handle_kv_push(&msg, &mock).await); + + assert!(matches!(ack.status, AckStatus::Rejected { .. })); + assert!(!ack.accepted); + assert!(mock.applied.lock().expect("applied").is_empty()); + assert_eq!(*mock.skipped.lock().expect("skipped"), vec![6]); + assert_eq!(ack.applied_seq, 6); + } + + #[tokio::test] + async fn an_expired_put_applies_as_a_delete() { + let mut session = authenticated(); + let mock = MockDispatcher::answering(applied_at); + let mut msg = put_msg("k1", kv_row_msgpack("k1", b"v1"), 7); + msg.op = KvPushOp::Put { + row: kv_row_msgpack("k1", b"v1"), + expire_at_ms: 1, + }; + + let ack = ack_of(session.handle_kv_push(&msg, &mock).await); + + assert_eq!(ack.status, AckStatus::Applied); + assert_eq!( + mock.applied.lock().expect("applied")[0].op, + KvPushWriteOp::Delete + ); + } + + #[tokio::test] + async fn an_unauthenticated_push_is_rejected_without_dispatch() { + let mut session = SyncSession::new("kv-push".to_string()); + let mock = MockDispatcher::answering(applied_at); + let msg = put_msg("k1", kv_row_msgpack("k1", b"v1"), 1); + + let ack = ack_of(session.handle_kv_push(&msg, &mock).await); + + assert!(matches!(ack.status, AckStatus::Rejected { .. })); + assert!(mock.applied.lock().expect("applied").is_empty()); + assert!(mock.skipped.lock().expect("skipped").is_empty()); + } +} diff --git a/nodedb/src/control/server/sync/mod.rs b/nodedb/src/control/server/sync/mod.rs index 94a76131e..4604d8767 100644 --- a/nodedb/src/control/server/sync/mod.rs +++ b/nodedb/src/control/server/sync/mod.rs @@ -7,6 +7,8 @@ pub mod definition_fanout; pub mod dlq; pub mod fts_handler; mod fts_session; +pub mod kv_handler; +mod kv_session; pub mod listener; pub mod presence; pub mod raft_dispatch; diff --git a/nodedb/src/control/server/sync/session/dispatch.rs b/nodedb/src/control/server/sync/session/dispatch.rs index 5c7ef9d48..2369d9cc9 100644 --- a/nodedb/src/control/server/sync/session/dispatch.rs +++ b/nodedb/src/control/server/sync/session/dispatch.rs @@ -176,6 +176,15 @@ impl SyncSession { | SyncMessageType::SpatialDelete | SyncMessageType::SpatialInsertAck | SyncMessageType::SpatialDeleteAck => None, + // KvPush is intercepted in session_handler before reaching here. + // KvPushAck is server→client; receiving it here means a mis-wired + // client — ignore silently. + SyncMessageType::KvPush | SyncMessageType::KvPushAck => None, + SyncMessageType::RowPushReject => { + let msg: RowPushRejectMsg = frame.decode_body()?; + self.handle_row_push_reject(&msg, shared); + None + } SyncMessageType::ResyncRequest => { // In the production path (shared = Some), session_handler.rs // intercepts ResyncRequest before process_frame and dispatches diff --git a/nodedb/src/control/server/sync/session/mod.rs b/nodedb/src/control/server/sync/session/mod.rs index 7c3075f26..eb5d233b6 100644 --- a/nodedb/src/control/server/sync/session/mod.rs +++ b/nodedb/src/control/server/sync/session/mod.rs @@ -14,6 +14,7 @@ //! - `clock_ping.rs` — `handle_vector_clock_sync` + `handle_ping`. //! - `token.rs` — `handle_token_refresh`. //! - `dispatch.rs` — `process_frame` (match on `msg_type`, route). +//! - `row_push_reject.rs` — `handle_row_push_reject` (peer refused a row push). pub mod clock_ping; pub mod collection_schema; @@ -21,6 +22,7 @@ pub mod delta; pub mod dispatch; pub mod fencing; pub mod handshake; +pub mod row_push_reject; pub mod state; pub mod token; diff --git a/nodedb/src/control/server/sync/session/row_push_reject.rs b/nodedb/src/control/server/sync/session/row_push_reject.rs new file mode 100644 index 000000000..93a6f6bcc --- /dev/null +++ b/nodedb/src/control/server/sync/session/row_push_reject.rs @@ -0,0 +1,127 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `SyncSession::handle_row_push_reject`: a Lite peer refused a row push. +//! +//! Origin does not re-send a row push, so a refused row never reaches the +//! peer on its own. The refusal goes to the sync dead-letter queue with the +//! peer's reason, the way a refused delta does, so an operator can inspect +//! it and repair the peer. + +use std::sync::Arc; + +use tracing::{error, warn}; + +use crate::control::state::SharedState; + +use super::super::dlq::{DlqEnqueueParams, ViolationType}; +use super::super::wire::{RowPushRefusal, RowPushRejectMsg}; +use super::state::SyncSession; + +impl SyncSession { + /// Record a row push the peer refused. + /// + /// Without `SharedState` there is no dead-letter queue to hold it, so + /// the refusal is only logged. + pub fn handle_row_push_reject( + &mut self, + msg: &RowPushRejectMsg, + shared: Option<&Arc>, + ) { + // Origin sends row pushes only to authenticated sessions, so an + // unauthenticated refusal names no row this session received. + if !self.authenticated { + warn!( + session = %self.session_id, + collection = %msg.collection, + "row push refusal from an unauthenticated session ignored" + ); + return; + } + error!( + session = %self.session_id, + collection = %msg.collection, + document_id = %msg.document_id, + sequence = msg.sequence, + refusal = %msg.refusal, + "sync peer refused a row push; the row is not on that peer" + ); + let Some(shared) = shared else { + return; + }; + let entry = self.row_push_reject_entry(msg); + let mut dlq = shared.sync_dlq.lock().unwrap_or_else(|p| p.into_inner()); + dlq.enqueue(entry); + } + + /// The dead-letter entry for a refused row push. + fn row_push_reject_entry(&self, msg: &RowPushRejectMsg) -> DlqEnqueueParams { + let violation_type = match &msg.refusal { + RowPushRefusal::Malformed { detail } => ViolationType::MalformedDelta { + detail: detail.clone(), + }, + RowPushRefusal::ApplyFailed { detail } => ViolationType::ConstraintViolation { + detail: detail.clone(), + }, + }; + DlqEnqueueParams { + session_id: self.session_id.clone(), + tenant_id: self.tenant_id.map(|t| t.as_u64()).unwrap_or(0), + username: self.username.clone().unwrap_or_default(), + collection: msg.collection.clone(), + document_id: msg.document_id.clone(), + mutation_id: msg.sequence, + peer_id: msg.peer_id, + delta: Vec::new(), + violation_type, + compensation: None, + device_metadata: self.device_metadata.clone(), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn reject(refusal: RowPushRefusal) -> RowPushRejectMsg { + RowPushRejectMsg { + collection: "cfg".into(), + document_id: "k1".into(), + sequence: 3, + peer_id: 9, + refusal, + } + } + + #[test] + fn a_malformed_row_is_dead_lettered_as_a_malformed_delta() { + let session = SyncSession::new("row-push-reject".to_string()); + let entry = session.row_push_reject_entry(&reject(RowPushRefusal::Malformed { + detail: "not a row map".into(), + })); + assert_eq!(entry.collection, "cfg"); + assert_eq!(entry.document_id, "k1"); + assert_eq!(entry.mutation_id, 3); + assert_eq!(entry.peer_id, 9); + assert_eq!( + entry.violation_type, + ViolationType::MalformedDelta { + detail: "not a row map".into() + } + ); + } + + #[test] + fn a_failed_apply_is_dead_lettered_as_a_constraint_violation() { + let session = SyncSession::new("row-push-reject".to_string()); + let entry = session.row_push_reject_entry(&reject(RowPushRefusal::ApplyFailed { + detail: "storage full".into(), + })); + assert_eq!( + entry.violation_type, + ViolationType::ConstraintViolation { + detail: "storage full".into() + } + ); + } +} diff --git a/nodedb/src/control/server/sync/session_handler/engine_dispatch.rs b/nodedb/src/control/server/sync/session_handler/engine_dispatch.rs index c61bbcde4..df893f66c 100644 --- a/nodedb/src/control/server/sync/session_handler/engine_dispatch.rs +++ b/nodedb/src/control/server/sync/session_handler/engine_dispatch.rs @@ -2,8 +2,8 @@ //! Per-engine sync-message dispatch for the session loop. //! -//! Every engine sync message (timeseries / columnar / vector / FTS / spatial) -//! follows the same shape: decode the typed body, pick the production +//! Every engine sync message (timeseries / columnar / vector / FTS / spatial / +//! KV) follows the same shape: decode the typed body, pick the production //! (`SharedState`) or no-op dispatcher, invoke the session handler, and forward //! the ACK frame. [`dispatch_engine_frame`] factors that boilerplate into one //! place so adding an engine is a single match arm. @@ -17,7 +17,7 @@ use tokio_tungstenite::tungstenite::Message; use super::super::session::SyncSession; use super::super::wire::{ - ColumnarInsertMsg, FtsDeleteMsg, FtsIndexMsg, SpatialDeleteMsg, SpatialInsertMsg, + ColumnarInsertMsg, FtsDeleteMsg, FtsIndexMsg, KvPushMsg, SpatialDeleteMsg, SpatialInsertMsg, SyncMessageType, TimeseriesPushMsg, VectorDeleteMsg, VectorInsertMsg, }; use crate::control::state::SharedState; @@ -67,11 +67,13 @@ pub(super) async fn dispatch_engine_frame( shared: &Option>, ) -> EngineOutcome { use super::super::{ - columnar_handler, fts_handler, spatial_handler, timeseries_handler, vector_handler, + columnar_handler, fts_handler, kv_handler, spatial_handler, timeseries_handler, + vector_handler, }; let dispatcher_identity = session.identity.clone(); let dispatcher_database = session.database_id(); + let dispatcher_peer = session.device_metadata.remote_addr.clone(); match frame.msg_type { SyncMessageType::TimeseriesPush => dispatch!( @@ -188,6 +190,21 @@ pub(super) async fn dispatch_engine_frame( }, spatial_handler::NoOpSpatialDispatcher ), + SyncMessageType::KvPush => dispatch!( + ws, + session, + frame, + shared, + KvPushMsg, + handle_kv_push, + |s| kv_handler::SharedStateKvDispatcher { + shared: s, + identity: dispatcher_identity.as_ref(), + database_id: dispatcher_database, + peer_addr: &dispatcher_peer, + }, + kv_handler::NoOpKvDispatcher + ), _ => EngineOutcome::NotEngine, } } diff --git a/nodedb/src/control/server/sync/wire.rs b/nodedb/src/control/server/sync/wire.rs index 6e8fab4ca..c4462e725 100644 --- a/nodedb/src/control/server/sync/wire.rs +++ b/nodedb/src/control/server/sync/wire.rs @@ -11,12 +11,13 @@ pub use nodedb_types::sync::wire::{ AckStatus, CollectionDescriptor, CollectionSchemaSyncMsg, ColumnarInsertAckMsg, ColumnarInsertMsg, DefinitionSyncMsg, DeltaAckMsg, DeltaPushMsg, DeltaRejectMsg, FtsDeleteAckMsg, FtsDeleteMsg, FtsIndexAckMsg, FtsIndexMsg, HandshakeAckMsg, HandshakeMsg, - PeerPresence, PingPongMsg, PresenceBroadcastMsg, PresenceLeaveMsg, PresenceUpdateMsg, - ResyncReason, ResyncRequestMsg, ShapeDeltaMsg, ShapeSnapshotMsg, ShapeSubscribeMsg, - ShapeUnsubscribeMsg, SpatialDeleteAckMsg, SpatialDeleteMsg, SpatialInsertAckMsg, - SpatialInsertMsg, SyncFrame, SyncMessageType, SyncProvenance, ThrottleMsg, TimeseriesAckMsg, - TimeseriesPushMsg, TokenRefreshAckMsg, TokenRefreshMsg, VectorClockSyncMsg, VectorDeleteAckMsg, - VectorDeleteMsg, VectorInsertAckMsg, VectorInsertMsg, + KvPushAckMsg, KvPushMsg, KvPushOp, PeerPresence, PingPongMsg, PresenceBroadcastMsg, + PresenceLeaveMsg, PresenceUpdateMsg, ResyncReason, ResyncRequestMsg, RowPushRefusal, + RowPushRejectMsg, ShapeDeltaMsg, ShapeSnapshotMsg, ShapeSubscribeMsg, ShapeUnsubscribeMsg, + SpatialDeleteAckMsg, SpatialDeleteMsg, SpatialInsertAckMsg, SpatialInsertMsg, SyncFrame, + SyncMessageType, SyncProvenance, ThrottleMsg, TimeseriesAckMsg, TimeseriesPushMsg, + TokenRefreshAckMsg, TokenRefreshMsg, VectorClockSyncMsg, VectorDeleteAckMsg, VectorDeleteMsg, + VectorInsertAckMsg, VectorInsertMsg, }; // ── Re-export CompensationHint (used by dlq.rs and session.rs) ── diff --git a/nodedb/src/control/server/wal_dispatch_kv/append.rs b/nodedb/src/control/server/wal_dispatch_kv/append.rs index 491c7545d..a93b2782a 100644 --- a/nodedb/src/control/server/wal_dispatch_kv/append.rs +++ b/nodedb/src/control/server/wal_dispatch_kv/append.rs @@ -41,6 +41,20 @@ fn resolve_expiry(ttl_ms: u64, now_override: Option) -> (Option, Optio } } +/// Append the `SyncSeqAdvance` record of a Lite KV push, ahead of the +/// write's own record. +/// +/// Both records land in the write's outcome-floor window. A frame the gate +/// holds back cancels both, so the mark never moves for a write that did not +/// apply. The write's record is the higher LSN, so the durable-at-ack wait +/// on it covers the mark too. +fn append_sync_mark( + wal: WalAppender<'_>, + prov: &nodedb_types::sync::wire::SyncProvenance, +) -> crate::Result { + wal.append_sync_seq_advance(prov.producer_id, prov.epoch, prov.stream_id, prov.seq) +} + /// Serialize a KV operation and append to the WAL — see [`KvAppendOutcome`]. /// `now_override` pins `expire_at_ms` to an instant decided elsewhere (e.g. a /// Raft-committed entry), so every replica's redo installs it verbatim. @@ -60,9 +74,25 @@ pub fn wal_append_kv_op( value, ttl_ms, surrogate, + provenance, .. + } => { + if let Some(prov) = provenance { + append_sync_mark(wal, prov)?; + } + let (now_ms, expire_at_ms) = resolve_expiry(*ttl_ms, now_override); + resolved_now_ms = now_ms; + let entry = encode_kv_put( + collection.as_str(), + key, + value, + *ttl_ms, + expire_at_ms, + surrogate.as_u32(), + )?; + Some(wal.append_put(tenant_id, vshard_id, database_id, &entry)?) } - | KvOp::Insert { + KvOp::Insert { collection, key, value, @@ -116,8 +146,14 @@ pub fn wal_append_kv_op( Some(wal.append_put(tenant_id, vshard_id, database_id, &entry)?) } KvOp::Delete { - collection, keys, .. + collection, + keys, + provenance, + .. } => { + if let Some(prov) = provenance { + append_sync_mark(wal, prov)?; + } let entry = encode_kv_delete(collection.as_str(), keys)?; Some(wal.append_delete(tenant_id, vshard_id, database_id, &entry)?) } diff --git a/nodedb/src/control/wal_replication/decode/entry_kv.rs b/nodedb/src/control/wal_replication/decode/entry_kv.rs index c8896d86f..7631714ee 100644 --- a/nodedb/src/control/wal_replication/decode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/decode/entry_kv.rs @@ -5,10 +5,10 @@ //! `Kv*` variants stamp `resolved_now_ms` so every replica installs the same //! `expire_at_ms`. See `entry_document::decode_arm` for the trailing-arm contract. -use super::super::decode_sync_engines::decode_returning; +use super::super::decode_sync_engines::{decode_provenance, decode_returning}; use super::super::types::ReplicatedWrite; use super::kv; -use super::kv::ReturningFields; +use super::kv::{PutFields, ReturningFields}; use crate::bridge::envelope::PhysicalPlan; pub(super) fn decode_arm(write: &ReplicatedWrite) -> crate::Result<(PhysicalPlan, Option)> { @@ -27,18 +27,22 @@ pub(super) fn decode_arm(write: &ReplicatedWrite) -> crate::Result<(PhysicalPlan resolved_now_ms: rn, returning, rls_filters, + provenance, } => { resolved_now_ms = *rn; kv::put( - collection, - key, - value, - *ttl_ms, - *surrogate, + PutFields { + collection, + key, + value, + ttl_ms: *ttl_ms, + surrogate: *surrogate, + }, ReturningFields { returning: decode_returning(returning)?, rls_filters, }, + decode_provenance(provenance)?, )? } ReplicatedWrite::KvDelete { @@ -46,6 +50,7 @@ pub(super) fn decode_arm(write: &ReplicatedWrite) -> crate::Result<(PhysicalPlan keys, returning, rls_filters, + provenance, } => kv::delete( collection, keys, @@ -53,6 +58,7 @@ pub(super) fn decode_arm(write: &ReplicatedWrite) -> crate::Result<(PhysicalPlan returning: decode_returning(returning)?, rls_filters, }, + decode_provenance(provenance)?, ), ReplicatedWrite::KvInsert { collection, @@ -354,6 +360,7 @@ mod tests { resolved_now_ms: Some(resolved_now_ms), returning: None, rls_filters: Vec::new(), + provenance: None, }, ); let bytes = entry.to_bytes(); @@ -400,6 +407,7 @@ mod tests { resolved_now_ms: None, returning: None, rls_filters: Vec::new(), + provenance: None, }, ); let bytes = entry.to_bytes(); @@ -525,6 +533,7 @@ mod tests { surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let entry = to_replicated_entry(tenant, DatabaseId::DEFAULT, vshard, &plan) .expect("encode must not error") @@ -547,6 +556,7 @@ mod tests { surrogate: nodedb_types::Surrogate::new(2), returning: None, rls_filters: Vec::new(), + provenance: None, }); let entry_no_ttl = to_replicated_entry(tenant, DatabaseId::DEFAULT, vshard, &plan_no_ttl) .expect("encode must not error") diff --git a/nodedb/src/control/wal_replication/decode/kv.rs b/nodedb/src/control/wal_replication/decode/kv.rs index 291f9e8d8..99814a20e 100644 --- a/nodedb/src/control/wal_replication/decode/kv.rs +++ b/nodedb/src/control/wal_replication/decode/kv.rs @@ -8,6 +8,7 @@ use crate::bridge::envelope::PhysicalPlan; use nodedb_physical::physical_plan::{KvCounterShape, KvOp, ReturningSpec}; use nodedb_types::RlsWriteCheck; +use nodedb_types::sync::wire::SyncProvenance; /// A decoded RETURNING projection spec plus the read filters gating it — see /// `ReplicatedWrite::KvPut::returning`. Bundled — plain positional arguments @@ -17,14 +18,27 @@ pub(super) struct ReturningFields<'a> { pub rls_filters: &'a [u8], } +/// The row a replicated KV put writes, and the identity it is written under. +pub(super) struct PutFields<'a> { + pub collection: &'a str, + pub key: &'a [u8], + pub value: &'a [u8], + pub ttl_ms: u64, + pub surrogate: u32, +} + pub(super) fn put( - collection: &str, - key: &[u8], - value: &[u8], - ttl_ms: u64, - surrogate: u32, + put: PutFields<'_>, returning: ReturningFields<'_>, + provenance: Option, ) -> crate::Result { + let PutFields { + collection, + key, + value, + ttl_ms, + surrogate, + } = put; let surrogate = nodedb_types::Surrogate::new(surrogate); Ok(PhysicalPlan::Kv(KvOp::Put { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), @@ -35,6 +49,8 @@ pub(super) fn put( // Carried on the record — a replay re-executes for the originating request. returning: returning.returning, rls_filters: returning.rls_filters.to_vec(), + // Every replica gates the write on the same stream mark. + provenance, })) } @@ -45,6 +61,7 @@ pub(super) fn delete( collection: &str, keys: &[Vec], returning: ReturningFields<'_>, + provenance: Option, ) -> PhysicalPlan { PhysicalPlan::Kv(KvOp::Delete { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), @@ -52,6 +69,7 @@ pub(super) fn delete( rls_write_check: RlsWriteCheck::already_decided_elsewhere(), returning: returning.returning, rls_filters: returning.rls_filters.to_vec(), + provenance, }) } @@ -633,6 +651,53 @@ mod tests { } } + /// A Lite KV push carries its sync provenance to every replica, so each + /// one gates the write on the same stream mark. + #[test] + fn kv_put_and_delete_provenance_roundtrips() { + let tenant = TenantId::new(1); + let vshard = VShardId::new(0); + let prov = SyncProvenance { + producer_id: 7, + epoch: 2, + stream_id: 99, + seq: 5, + }; + let put = PhysicalPlan::Kv(KvOp::Put { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "kv"), + key: b"k1".to_vec(), + value: b"v1".to_vec(), + ttl_ms: 0, + surrogate: Surrogate::new(1), + returning: None, + rls_filters: Vec::new(), + provenance: Some(prov.clone()), + }); + let delete = PhysicalPlan::Kv(KvOp::Delete { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "kv"), + keys: vec![b"k1".to_vec()], + rls_write_check: RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + provenance: Some(prov.clone()), + }); + for plan in [put, delete] { + let entry = to_replicated_entry(tenant, DatabaseId::DEFAULT, vshard, &plan) + .expect("encode must not error") + .expect("a KV write produces a ReplicatedEntry"); + let (_, _, decoded, _) = decode::from_replicated_entry(&entry.to_bytes(), None) + .expect("from_replicated_entry error") + .expect("from_replicated_entry returned None"); + match decoded { + PhysicalPlan::Kv(KvOp::Put { provenance, .. }) + | PhysicalPlan::Kv(KvOp::Delete { provenance, .. }) => { + assert_eq!(provenance, Some(prov.clone())); + } + other => panic!("expected a KV write, got {other:?}"), + } + } + } + /// `entry_kv::kv_write` must not drop `KvOp::Put::returning` / `rls_filters` /// — the leader re-derives its plan from the committed entry, so a drop here /// loses `RETURNING` for the originating request too, not just followers. @@ -651,6 +716,7 @@ mod tests { surrogate: Surrogate::new(1), returning: Some(spec.clone()), rls_filters: b"rls-predicate".to_vec(), + provenance: None, }); let entry = to_replicated_entry(tenant, DatabaseId::DEFAULT, vshard, &plan) .expect("encode must not error") diff --git a/nodedb/src/control/wal_replication/encode/entry.rs b/nodedb/src/control/wal_replication/encode/entry.rs index 8b3af78ea..eb35d11f6 100644 --- a/nodedb/src/control/wal_replication/encode/entry.rs +++ b/nodedb/src/control/wal_replication/encode/entry.rs @@ -236,6 +236,7 @@ mod tests { surrogate: Surrogate::new(7), returning: None, rls_filters: Vec::new(), + provenance: None, }); assert!( to_replicated_entry(tenant, DatabaseId::DEFAULT, vshard, &kv_put) diff --git a/nodedb/src/control/wal_replication/encode/entry_kv.rs b/nodedb/src/control/wal_replication/encode/entry_kv.rs index 59bf3c677..28f42808e 100644 --- a/nodedb/src/control/wal_replication/encode/entry_kv.rs +++ b/nodedb/src/control/wal_replication/encode/entry_kv.rs @@ -9,7 +9,7 @@ use super::super::types::ReplicatedWrite; use super::kv; -use super::kv::WireReturning; +use super::kv::{WirePut, WireReturning}; use nodedb_physical::physical_plan::KvOp; /// Encode a `KvOp` write variant, `Ok(None)` when not a single-shard replicated @@ -25,16 +25,20 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { surrogate, returning, rls_filters, + provenance, } => kv::put( - collection.as_str(), - key, - value, - *ttl_ms, - surrogate.as_u32(), + WirePut { + collection: collection.as_str(), + key, + value, + ttl_ms: *ttl_ms, + surrogate: surrogate.as_u32(), + }, WireReturning { returning, rls_filters, }, + provenance, ), // The compiled RLS predicate is absent from the durable record, so a // replay re-applies the already-admitted write, not re-deciding it. @@ -45,6 +49,7 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { rls_write_check: _, returning, rls_filters, + provenance, } => kv::delete( collection.as_str(), keys, @@ -52,6 +57,7 @@ pub(super) fn kv_write(op: &KvOp) -> crate::Result> { returning, rls_filters, }, + provenance, ), KvOp::Insert { collection, diff --git a/nodedb/src/control/wal_replication/encode/kv.rs b/nodedb/src/control/wal_replication/encode/kv.rs index 20558e899..8b5652160 100644 --- a/nodedb/src/control/wal_replication/encode/kv.rs +++ b/nodedb/src/control/wal_replication/encode/kv.rs @@ -3,9 +3,10 @@ //! Encode `PhysicalPlan::Kv` variants into `ReplicatedWrite`. use super::super::types::ReplicatedWrite; -use super::entry::encode_returning; +use super::entry::{encode_provenance, encode_returning}; use nodedb_physical::physical_plan::{KvCounterShape, ReturningSpec, UpdateValue}; use nodedb_types::Surrogate; +use nodedb_types::sync::wire::SyncProvenance; /// Resolve the wall-clock instant for a TTL-bearing write once, at proposal /// time. `None` when `ttl_ms == 0`. Every replica computes `expire_at_ms` @@ -26,14 +27,27 @@ pub(super) struct WireReturning<'a> { pub rls_filters: &'a [u8], } +/// The row a KV put writes, and the identity it is written under. +pub(super) struct WirePut<'a> { + pub collection: &'a str, + pub key: &'a [u8], + pub value: &'a [u8], + pub ttl_ms: u64, + pub surrogate: u32, +} + pub(super) fn put( - collection: &str, - key: &[u8], - value: &[u8], - ttl_ms: u64, - surrogate: u32, + put: WirePut<'_>, returning: WireReturning<'_>, + provenance: &Option, ) -> ReplicatedWrite { + let WirePut { + collection, + key, + value, + ttl_ms, + surrogate, + } = put; ReplicatedWrite::KvPut { collection: collection.to_owned(), key: key.to_vec(), @@ -43,6 +57,7 @@ pub(super) fn put( resolved_now_ms: resolve_now_ms(ttl_ms), returning: encode_returning(returning.returning), rls_filters: returning.rls_filters.to_vec(), + provenance: encode_provenance(provenance), } } @@ -81,12 +96,14 @@ pub(super) fn delete( collection: &str, keys: &[Vec], returning: WireReturning<'_>, + provenance: &Option, ) -> ReplicatedWrite { ReplicatedWrite::KvDelete { collection: collection.to_owned(), keys: keys.to_vec(), returning: encode_returning(returning.returning), rls_filters: returning.rls_filters.to_vec(), + provenance: encode_provenance(provenance), } } diff --git a/nodedb/src/control/wal_replication/types/replicated_write.rs b/nodedb/src/control/wal_replication/types/replicated_write.rs index edceec980..f8c9c1e75 100644 --- a/nodedb/src/control/wal_replication/types/replicated_write.rs +++ b/nodedb/src/control/wal_replication/types/replicated_write.rs @@ -402,6 +402,9 @@ pub enum ReplicatedWrite { /// See `ReplicatedWrite::PointPut::rls_filters`. #[serde(default)] rls_filters: Vec, + /// Sync provenance of a Lite KV push, encoded as zerompk bytes. + #[serde(default)] + provenance: Option>, }, KvDelete { collection: String, @@ -412,6 +415,9 @@ pub enum ReplicatedWrite { /// See `ReplicatedWrite::PointPut::rls_filters`. #[serde(default)] rls_filters: Vec, + /// Sync provenance of a Lite KV push, encoded as zerompk bytes. + #[serde(default)] + provenance: Option>, }, KvInsert { collection: String, diff --git a/nodedb/src/data/executor/handlers/control/calvin_reply/stage.rs b/nodedb/src/data/executor/handlers/control/calvin_reply/stage.rs index 260053674..67f512512 100644 --- a/nodedb/src/data/executor/handlers/control/calvin_reply/stage.rs +++ b/nodedb/src/data/executor/handlers/control/calvin_reply/stage.rs @@ -614,6 +614,7 @@ mod tests { surrogate: Surrogate::new(5), returning: returning(&["v"]), rls_filters: Vec::new(), + provenance: None, }) } @@ -635,6 +636,7 @@ mod tests { rls_write_check: RlsWriteCheck::NoPolicyApplies, returning: returning(&["v"]), rls_filters: Vec::new(), + provenance: None, }); let response = commit(&[kv_put("put"), delete], |_| {}); assert_eq!(returned(&response), vec![text("put")]); @@ -652,6 +654,7 @@ mod tests { rls_write_check: RlsWriteCheck::NoPolicyApplies, returning: returning(&["v"]), rls_filters: Vec::new(), + provenance: None, }); let dir = tempfile::tempdir().expect("tempdir"); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); diff --git a/nodedb/src/data/executor/handlers/kv/crud/mod.rs b/nodedb/src/data/executor/handlers/kv/crud/mod.rs index 0d9799356..229b08812 100644 --- a/nodedb/src/data/executor/handlers/kv/crud/mod.rs +++ b/nodedb/src/data/executor/handlers/kv/crud/mod.rs @@ -4,6 +4,7 @@ mod delete; mod get; +mod sync_write; mod types; mod write_basic; mod write_upsert; diff --git a/nodedb/src/data/executor/handlers/kv/crud/sync_write.rs b/nodedb/src/data/executor/handlers/kv/crud/sync_write.rs new file mode 100644 index 000000000..073882600 --- /dev/null +++ b/nodedb/src/data/executor/handlers/kv/crud/sync_write.rs @@ -0,0 +1,205 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! KV writes from a Lite KV push, behind the sync idempotency gate. +//! +//! A pushed put or delete runs the same handler a SQL write runs. The gate +//! wraps it: +//! - a frame the gate holds back (duplicate, fenced epoch, sequence gap) +//! applies nothing and answers `SyncNotApplied`, so the funnel cancels the +//! frame's records and restart replay never applies them; +//! - an applied frame advances the stream's mark and answers the gate's ack +//! payload; +//! - a frame the row-level-security write policy refuses advances the mark +//! and answers `SyncRejected`, so the producer's next frame is not a gap. + +use nodedb_types::sync::violation::ViolationType; +use nodedb_types::sync::wire::{AckStatus, SyncProvenance}; + +use super::types::{KvDeleteParams, KvWriteParams}; +use crate::bridge::envelope::{ErrorCode, Response, Status, SyncHold}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::sync_gate::SyncAdmit; +use crate::data::executor::task::ExecutionTask; + +impl CoreLoop { + /// Run a pushed KV put behind the sync gate. + pub(in crate::data::executor) fn execute_kv_sync_put( + &mut self, + task: &ExecutionTask, + params: KvWriteParams<'_>, + prov: &SyncProvenance, + ) -> Response { + if let Some(held) = self.kv_sync_hold(task, prov) { + return held; + } + let response = self.execute_kv_put(task, params); + self.kv_sync_outcome(task, response, prov) + } + + /// Run a pushed KV delete behind the sync gate. + pub(in crate::data::executor) fn execute_kv_sync_delete( + &mut self, + task: &ExecutionTask, + params: KvDeleteParams<'_>, + prov: &SyncProvenance, + ) -> Response { + if let Some(held) = self.kv_sync_hold(task, prov) { + return held; + } + let response = self.execute_kv_delete(task, params); + self.kv_sync_outcome(task, response, prov) + } + + /// The `SyncNotApplied` response for a frame the gate holds back, or + /// `None` when the frame is admitted. + fn kv_sync_hold(&mut self, task: &ExecutionTask, prov: &SyncProvenance) -> Option { + let hold = match self.sync_admit(prov) { + SyncAdmit::Apply => return None, + SyncAdmit::Duplicate => SyncHold::Duplicate, + SyncAdmit::Fenced => SyncHold::Fenced, + SyncAdmit::Gap { expected } => SyncHold::Gap { expected }, + }; + let applied_seq = self.sync_hwm_value(prov.producer_id, prov.stream_id); + Some(self.response_error(task, ErrorCode::SyncNotApplied { hold, applied_seq })) + } + + /// Turn the handler's response into the sync outcome. + fn kv_sync_outcome( + &mut self, + task: &ExecutionTask, + response: Response, + prov: &SyncProvenance, + ) -> Response { + match response.status { + Status::Ok | Status::Partial => { + self.sync_commit(prov); + self.sync_ack_response(task, AckStatus::Applied, prov.seq) + } + Status::Error => { + let refusing_policy = match response.error_code.as_deref() { + Some(ErrorCode::RejectedAuthz { resource }) => Some(resource.clone()), + Some(_) | None => None, + }; + match refusing_policy { + Some(policy_name) => self.sync_reject_response( + task, + ViolationType::RlsPolicyViolation { policy_name }, + prov, + ), + None => response, + } + } + } + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::sync::wire::{SyncAckResult, SyncOutcome}; + use nodedb_types::{DatabaseId, RlsWriteCheck, Surrogate}; + + use super::*; + use crate::data::executor::core_loop::tests::{make_core_with_dir, make_default_task}; + + const TID: u64 = 1; + + fn prov(seq: u64) -> SyncProvenance { + SyncProvenance { + producer_id: 7, + epoch: 1, + stream_id: 42, + seq, + } + } + + fn put<'a>(value: &'a [u8]) -> KvWriteParams<'a> { + KvWriteParams { + did: DatabaseId::DEFAULT.as_u64(), + tid: TID, + collection: "cfg", + key: b"k1", + value, + ttl_ms: 0, + surrogate: Surrogate::ZERO, + returning: None, + rls_filters: &[], + } + } + + fn applied_seq(response: &Response) -> u64 { + assert_eq!(response.status, Status::Ok); + let ack: SyncAckResult = zerompk::from_msgpack(&response.payload).expect("sync ack"); + assert_eq!(ack.outcome, SyncOutcome::Ack(AckStatus::Applied)); + ack.applied_seq + } + + fn hold(response: &Response) -> SyncHold { + match response.error_code.as_deref() { + Some(ErrorCode::SyncNotApplied { hold, .. }) => *hold, + other => panic!("expected SyncNotApplied, got {other:?}"), + } + } + + fn stored(core: &CoreLoop) -> Option> { + core.kv_engine.get( + DatabaseId::DEFAULT.as_u64(), + TID, + "cfg", + b"k1", + crate::engine::kv::current_ms(), + ) + } + + #[test] + fn a_frame_applies_once_and_its_resend_is_a_duplicate() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let task = make_default_task(); + + let first = core.execute_kv_sync_put(&task, put(b"v1"), &prov(1)); + assert_eq!(applied_seq(&first), 1); + + let resent = core.execute_kv_sync_put(&task, put(b"v1"), &prov(1)); + assert_eq!(hold(&resent), SyncHold::Duplicate); + assert_eq!(stored(&core).as_deref(), Some(b"v1".as_slice())); + } + + #[test] + fn a_frame_past_a_gap_applies_nothing() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let task = make_default_task(); + + let skipped = core.execute_kv_sync_put(&task, put(b"v1"), &prov(3)); + assert_eq!(hold(&skipped), SyncHold::Gap { expected: 1 }); + assert_eq!(stored(&core), None); + } + + #[test] + fn a_pushed_delete_removes_the_key_behind_the_gate() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let task = make_default_task(); + assert_eq!( + applied_seq(&core.execute_kv_sync_put(&task, put(b"v1"), &prov(1))), + 1 + ); + + let keys = vec![b"k1".to_vec()]; + let deleted = core.execute_kv_sync_delete( + &task, + KvDeleteParams { + did: DatabaseId::DEFAULT.as_u64(), + tid: TID, + collection: "cfg", + keys: &keys, + rls_write_check: &RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: &[], + }, + &prov(2), + ); + assert_eq!(applied_seq(&deleted), 2); + assert_eq!(stored(&core), None); + } +} diff --git a/nodedb/src/data/executor/handlers/kv/dispatch.rs b/nodedb/src/data/executor/handlers/kv/dispatch.rs index 0825d9de5..f4775d5a5 100644 --- a/nodedb/src/data/executor/handlers/kv/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/dispatch.rs @@ -41,9 +41,9 @@ impl CoreLoop { surrogate, returning, rls_filters, - } => self.execute_kv_put( - task, - super::crud::KvWriteParams { + provenance, + } => { + let params = super::crud::KvWriteParams { did, tid, collection: collection.as_str(), @@ -53,8 +53,12 @@ impl CoreLoop { surrogate: *surrogate, returning: returning.as_ref(), rls_filters, - }, - ), + }; + match provenance { + Some(prov) => self.execute_kv_sync_put(task, params, prov), + None => self.execute_kv_put(task, params), + } + } KvOp::Insert { collection, key, @@ -131,9 +135,9 @@ impl CoreLoop { rls_write_check, returning, rls_filters, - } => self.execute_kv_delete( - task, - super::crud::KvDeleteParams { + provenance, + } => { + let params = super::crud::KvDeleteParams { did, tid, collection: collection.as_str(), @@ -141,8 +145,12 @@ impl CoreLoop { rls_write_check, returning: returning.as_ref(), rls_filters, - }, - ), + }; + match provenance { + Some(prov) => self.execute_kv_sync_delete(task, params, prov), + None => self.execute_kv_delete(task, params), + } + } KvOp::Scan { .. } => self.dispatch_kv_scan(task, did, tid, op), KvOp::Expire { collection, diff --git a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs index 233c5f57a..a3c3f27e2 100644 --- a/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs +++ b/nodedb/src/data/executor/handlers/kv/resolve/dispatch.rs @@ -60,12 +60,16 @@ impl CoreLoop { }, task, ), + // A sync-gated delete never takes the resolve route: the resolved + // write carries no provenance, so the gate would not run. It + // falls to the refusal below. KvOp::Delete { collection, keys, rls_write_check, returning, rls_filters, + provenance: None, } => self.resolve_kv_delete(KvDeleteParams { did, tid, diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index 24513cdeb..6dc7957c0 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -509,6 +509,7 @@ mod tests { surrogate: Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -674,6 +675,7 @@ mod tests { rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), + provenance: None, })], ); let redo = decode_redo(&resp); diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs index 84d136bfe..3d6f3a219 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_kv.rs @@ -108,6 +108,9 @@ impl CoreLoop { // before the write is staged, so no row image is projected here. returning: _, rls_filters: _, + // A Lite KV push dispatches with no transaction id, so a + // staged delete never carries sync provenance. + provenance: _, } => self.stage_kv_delete(task, tid, txn_id, collection.as_str(), keys, rls_write_check), // Predicate DML staged like Document `BulkUpdate`/`BulkDelete`: // the row set resolves against BASE ∪ OVERLAY. Same `RETURNING` diff --git a/nodedb/src/data/executor/wal_replay_kv_atomic.rs b/nodedb/src/data/executor/wal_replay_kv_atomic.rs index a261c96d6..c73d64fe5 100644 --- a/nodedb/src/data/executor/wal_replay_kv_atomic.rs +++ b/nodedb/src/data/executor/wal_replay_kv_atomic.rs @@ -413,6 +413,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let cas = PhysicalPlan::Kv(KvOp::Cas { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "state"), @@ -445,6 +446,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); // Live dispatch would have failed this compare (expected "idle" but // the seeded value is "fighting"); the WAL record still exists @@ -535,6 +537,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let incr = PhysicalPlan::Kv(KvOp::IncrFloat { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "scores"), @@ -567,6 +570,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let getset = PhysicalPlan::Kv(KvOp::GetSet { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "session"), diff --git a/nodedb/src/data/executor/wal_replay_kv_expiry.rs b/nodedb/src/data/executor/wal_replay_kv_expiry.rs index 434d183c7..8c569b995 100644 --- a/nodedb/src/data/executor/wal_replay_kv_expiry.rs +++ b/nodedb/src/data/executor/wal_replay_kv_expiry.rs @@ -272,6 +272,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let entry = crate::control::server::wal_dispatch_kv::encode::encode_kv_expire( "sessions", b"tok1", 5_000, 6_000, @@ -325,6 +326,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); wal_append_if_write( &wal, @@ -396,6 +398,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let persist_p = PhysicalPlan::Kv(KvOp::Persist { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "sessions"), diff --git a/nodedb/src/data/executor/wal_replay_kv_field.rs b/nodedb/src/data/executor/wal_replay_kv_field.rs index 71f743aac..085210376 100644 --- a/nodedb/src/data/executor/wal_replay_kv_field.rs +++ b/nodedb/src/data/executor/wal_replay_kv_field.rs @@ -189,6 +189,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let field_set = PhysicalPlan::Kv(KvOp::FieldSet { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "players"), @@ -272,6 +273,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let field_set = PhysicalPlan::Kv(KvOp::FieldSet { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "players"), diff --git a/nodedb/src/data/executor/wal_replay_kv_incr.rs b/nodedb/src/data/executor/wal_replay_kv_incr.rs index 35096d31a..108f6680a 100644 --- a/nodedb/src/data/executor/wal_replay_kv_incr.rs +++ b/nodedb/src/data/executor/wal_replay_kv_incr.rs @@ -278,6 +278,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let incr = PhysicalPlan::Kv(KvOp::Incr { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), @@ -311,6 +312,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let incr1 = PhysicalPlan::Kv(KvOp::Incr { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), @@ -353,6 +355,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let incr = PhysicalPlan::Kv(KvOp::Incr { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), @@ -391,6 +394,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let entry = encode_kv_incr(KvIncrRecord { collection: "counters", @@ -446,6 +450,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let incr = PhysicalPlan::Kv(KvOp::Incr { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), diff --git a/nodedb/src/data/executor/wal_replay_kv_index.rs b/nodedb/src/data/executor/wal_replay_kv_index.rs index 5f48bffad..2f2a4fcc0 100644 --- a/nodedb/src/data/executor/wal_replay_kv_index.rs +++ b/nodedb/src/data/executor/wal_replay_kv_index.rs @@ -204,6 +204,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs b/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs index 424cdfd92..be461c8a7 100644 --- a/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs +++ b/nodedb/src/data/executor/wal_replay_kv_insert_conflict.rs @@ -366,6 +366,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let updates = vec![( "mana".to_string(), @@ -459,6 +460,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let excluded = obj_bytes(&[("hp", 1)]); @@ -549,6 +551,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let upsert = PhysicalPlan::Kv(KvOp::InsertOnConflictUpdate { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "raw"), diff --git a/nodedb/src/data/executor/wal_replay_kv_sorted_index.rs b/nodedb/src/data/executor/wal_replay_kv_sorted_index.rs index 2b4dd3c6e..d5cf46ebf 100644 --- a/nodedb/src/data/executor/wal_replay_kv_sorted_index.rs +++ b/nodedb/src/data/executor/wal_replay_kv_sorted_index.rs @@ -219,6 +219,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/src/data/executor/wal_replay_kv_transfer.rs b/nodedb/src/data/executor/wal_replay_kv_transfer.rs index 76e6e80c9..bcf4488ca 100644 --- a/nodedb/src/data/executor/wal_replay_kv_transfer.rs +++ b/nodedb/src/data/executor/wal_replay_kv_transfer.rs @@ -410,6 +410,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let put_bob = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "accounts"), @@ -419,6 +420,7 @@ mod tests { surrogate: Surrogate::new(2), returning: None, rls_filters: Vec::new(), + provenance: None, }); let transfer = PhysicalPlan::Kv(KvOp::Transfer { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "accounts"), @@ -486,6 +488,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let transfer_item = PhysicalPlan::Kv(KvOp::TransferItem { source_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "inventory"), diff --git a/nodedb/src/data/executor/wal_replay_kv_ttl.rs b/nodedb/src/data/executor/wal_replay_kv_ttl.rs index e5a4e9611..b7d95c0d5 100644 --- a/nodedb/src/data/executor/wal_replay_kv_ttl.rs +++ b/nodedb/src/data/executor/wal_replay_kv_ttl.rs @@ -174,6 +174,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let outcome = wal_append_if_write( &wal, @@ -225,6 +226,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); let outcome = wal_append_if_write( &wal, @@ -289,6 +291,7 @@ mod tests { surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), + provenance: None, }); // 1_000 is vastly less than the real wall clock, so a live-apply path // that ignores `resolved_now_ms` and reads the wall clock instead diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index 4d421eaf3..0f6ec6b12 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -40,6 +40,13 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { ErrorCode::SyncRejected { violation, .. } => { NodeDbError::constraint_violation("", "sync", violation.to_string()) } + // The gate held the frame back without applying it. The sender + // re-sends or retires it by the hold, so it presents as the + // retriable class. + ErrorCode::SyncNotApplied { hold, .. } => NodeDbError::from_wire( + PublicCode::WRITE_CONFLICT, + format!("sync frame not applied: {hold}"), + ), // Nothing was applied and the identical frame is expected to succeed // once the transient precondition resolves, so it presents as the // retriable class rather than a permanent refusal. diff --git a/nodedb/tests/inproc/cases/calvin_determinism_contract.rs b/nodedb/tests/inproc/cases/calvin_determinism_contract.rs index 3476d65e4..663dd5322 100644 --- a/nodedb/tests/inproc/cases/calvin_determinism_contract.rs +++ b/nodedb/tests/inproc/cases/calvin_determinism_contract.rs @@ -177,6 +177,7 @@ fn kv_no_ttl_byte_identical() { surrogate: nodedb_types::Surrogate::new(i), returning: None, rls_filters: Vec::new(), + provenance: None, }) }) .collect(); @@ -361,6 +362,7 @@ fn kv_with_ttl_byte_identical() { surrogate: nodedb_types::Surrogate::new(i), returning: None, rls_filters: Vec::new(), + provenance: None, }) }) .collect(); diff --git a/nodedb/tests/inproc/cases/calvin_executor_apply.rs b/nodedb/tests/inproc/cases/calvin_executor_apply.rs index 35720daa6..0fd2379e4 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_apply.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_apply.rs @@ -102,6 +102,7 @@ fn kv_put(coll: &str, key: &[u8], value: &[u8]) -> PhysicalPlan { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs b/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs index 1dae5af34..103c2c1ef 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_panic_recovery.rs @@ -139,6 +139,7 @@ fn kv_put_in(coll: &str, key: &[u8], value: &[u8]) -> PhysicalPlan { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs index 2446966e3..3f82e35bb 100644 --- a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs +++ b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs @@ -192,6 +192,7 @@ fn kv_put(coll: &str, key: &[u8], value: &[u8]) -> PhysicalPlan { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/basic_scans.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/basic_scans.rs index d5a07d7f0..54dbbfab7 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/basic_scans.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/basic_scans.rs @@ -35,6 +35,7 @@ fn kv_put_scan_roundtrip() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -54,6 +55,7 @@ fn kv_put_scan_roundtrip() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/join_budget.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/join_budget.rs index e6370374f..213a7d6c3 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/join_budget.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/join_budget.rs @@ -92,6 +92,7 @@ fn hash_join_completeness_past_50k_cap() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -165,6 +166,7 @@ fn sort_merge_join_completeness_past_50k_cap() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -223,6 +225,7 @@ fn nested_loop_join_completeness_past_50k_cap() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs index 6a4115058..0899ab6bc 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/multi_core_joins.rs @@ -59,6 +59,7 @@ fn multi_core_broadcast_inner_join() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -218,6 +219,7 @@ fn multi_core_broadcast_left_join() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -373,6 +375,7 @@ fn multi_core_broadcast_merge_simulation() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -393,6 +396,7 @@ fn multi_core_broadcast_merge_simulation() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } diff --git a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/single_core_joins.rs b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/single_core_joins.rs index 247bcf397..9a5426c81 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/single_core_joins.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_cross_type_join/single_core_joins.rs @@ -58,6 +58,7 @@ fn single_core_cross_type_hash_join() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -177,6 +178,7 @@ fn single_core_left_join_with_nulls() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } diff --git a/nodedb/tests/inproc/cases/executor_tests/test_graph_savepoint_overlay.rs b/nodedb/tests/inproc/cases/executor_tests/test_graph_savepoint_overlay.rs index 0415b8656..efe73a5f2 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_graph_savepoint_overlay.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_graph_savepoint_overlay.rs @@ -297,6 +297,7 @@ fn one_savepoint_reverts_value_and_graph_overlays_together() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv.rs index 39577ca63..b8767e4b7 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv.rs @@ -31,6 +31,7 @@ fn kv_put_get_delete() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -65,6 +66,7 @@ fn kv_put_get_delete() { rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); let json = payload_value(&payload); @@ -107,6 +109,7 @@ fn kv_overwrite_returns_ok() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -126,6 +129,7 @@ fn kv_overwrite_returns_ok() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -217,6 +221,7 @@ fn kv_scan_returns_entries() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -267,6 +272,7 @@ fn kv_scan_with_match_pattern() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -322,6 +328,7 @@ fn kv_expire_and_persist() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -409,6 +416,7 @@ fn kv_register_index_and_lookup() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); send_ok( @@ -426,6 +434,7 @@ fn kv_register_index_and_lookup() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -485,6 +494,7 @@ fn kv_drop_index() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -527,6 +537,7 @@ fn kv_tenant_isolation() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, })) }; tx.try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored(req)) @@ -549,6 +560,7 @@ fn kv_tenant_isolation() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, })) }; tx.try_push(nodedb::bridge::dispatch::BridgeRequest::unfloored(req)) diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs index a7d8f227d..5de1e3753 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv_advanced.rs @@ -32,6 +32,7 @@ fn kv_protocol_command_sequence() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -68,6 +69,7 @@ fn kv_protocol_command_sequence() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -119,6 +121,7 @@ fn kv_protocol_command_sequence() { rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); let json: serde_json::Value = payload_value(&payload); @@ -197,6 +200,7 @@ fn kv_protocol_command_sequence() { rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); let json: serde_json::Value = payload_value(&payload); @@ -231,6 +235,7 @@ fn kv_and_vector_coexist() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -322,6 +327,7 @@ fn ttl_expiry_produces_expired_key_info() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -366,6 +372,7 @@ fn ttl_expiry_produces_expired_key_info() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -422,6 +429,7 @@ fn kv_field_get_and_set() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -512,6 +520,7 @@ fn kv_truncate_clears_all() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -600,6 +609,7 @@ fn kv_index_write_amp_ratio_matches() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -657,6 +667,7 @@ fn kv_mass_expiry_respects_reap_budget() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } @@ -677,6 +688,7 @@ fn kv_mass_expiry_respects_reap_budget() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs b/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs index f25942443..b44990267 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_kv_ttl_overlay.rs @@ -96,6 +96,7 @@ fn staged_expire_is_observed_by_in_tx_get_ttl_then_reverts_on_rollback() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -181,6 +182,7 @@ fn staged_persist_hides_base_ttl_then_reverts_on_rollback() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -277,6 +279,7 @@ fn staged_expire_with_zero_ttl_makes_key_appear_absent_to_in_tx_get() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv.rs b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv.rs index 24de9da2d..da75093cf 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv.rs @@ -28,6 +28,7 @@ fn kv_get_isolated() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv_negative.rs b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv_negative.rs index 8241d0954..93d9b1e8c 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv_negative.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_tenant_isolation_kv_negative.rs @@ -33,6 +33,7 @@ fn kv_cross_tenant_put_does_not_overwrite() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -54,6 +55,7 @@ fn kv_cross_tenant_put_does_not_overwrite() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -107,6 +109,7 @@ fn kv_cross_tenant_delete_does_not_affect_owner() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); @@ -127,6 +130,7 @@ fn kv_cross_tenant_delete_does_not_affect_owner() { rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); // Either Ok (deleted 0 rows from B's namespace) or NotFound — both correct. diff --git a/nodedb/tests/inproc/cases/executor_tests/test_tenant_purge.rs b/nodedb/tests/inproc/cases/executor_tests/test_tenant_purge.rs index 5253861e3..7a930bcbb 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_tenant_purge.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_tenant_purge.rs @@ -78,6 +78,7 @@ fn purge_removes_all_tenant_data() { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }), ); diff --git a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs index aec1998f8..21ef464c4 100644 --- a/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs +++ b/nodedb/tests/inproc/cases/executor_tests/test_transaction_matrix_kv.rs @@ -33,6 +33,7 @@ fn kv_put(key: &[u8], value: &[u8]) -> PhysicalPlan { surrogate: nodedb_types::Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } @@ -196,6 +197,7 @@ fn rollback_matrix_kv_delete_then_doc_fail() { rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, returning: None, rls_filters: Vec::new(), + provenance: None, }), doc_insert_conflict("docs"), ], diff --git a/nodedb/tests/inproc/cases/surrogate_round_trip.rs b/nodedb/tests/inproc/cases/surrogate_round_trip.rs index 9a9e9977d..2ac7ca62e 100644 --- a/nodedb/tests/inproc/cases/surrogate_round_trip.rs +++ b/nodedb/tests/inproc/cases/surrogate_round_trip.rs @@ -302,6 +302,7 @@ fn surrogate_round_trip_all_engines() { surrogate: Surrogate::new(s), returning: None, rls_filters: Vec::new(), + provenance: None, }), ); } diff --git a/nodedb/tests/inproc/cases/write_admission_fence.rs b/nodedb/tests/inproc/cases/write_admission_fence.rs index 0ee7cde5d..ba2627cd3 100644 --- a/nodedb/tests/inproc/cases/write_admission_fence.rs +++ b/nodedb/tests/inproc/cases/write_admission_fence.rs @@ -89,6 +89,7 @@ fn kv_put(collection: &str, key: &[u8]) -> PhysicalPlan { surrogate: Surrogate::ZERO, returning: None, rls_filters: Vec::new(), + provenance: None, }) } From ffec08bf63cd302750b42684215ba2ec7f382792 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 19:45:23 +0800 Subject: [PATCH 42/64] feat(startup): bind client sockets early, listen only once serving Boot now binds every protocol socket before it waits on cluster readiness, so a port conflict fails boot before anything is exposed, then opens the sockets for accept only after entering a new terminal Serving startup phase. A bound-but-not-listening socket refuses connections at once instead of leaving a client to wait out the rest of boot in the kernel's accept queue. The HTTP listener is the exception: it starts serving early so orchestrator probes can watch startup, gated by a new middleware that lets through only the health and metrics routes until Serving. Reaching authorization readiness during cluster-ready no longer waits on the bounded lease alone: a node that leads the metadata group as its only voter now qualifies through a pinned lease with no expiry, since a second voter would end that leadership before it could vote. Lease status reporting and Prometheus rendering account for this new SoleVoter state. --- .../node/lifecycle/spawn_full.rs | 13 +- nodedb/src/bootstrap/cluster_ready.rs | 11 +- nodedb/src/bootstrap/listeners.rs | 195 ++++++++++----- nodedb/src/bootstrap/state_wiring.rs | 14 +- .../src/control/security/auth_fence/view.rs | 12 +- .../control/security/auth_lease/barrier.rs | 4 +- .../src/control/security/auth_lease/holder.rs | 22 +- .../control/security/auth_lease/leadership.rs | 17 ++ nodedb/src/control/security/auth_lease/mod.rs | 2 +- .../control/security/auth_lease/service.rs | 39 ++- .../src/control/security/auth_lease/status.rs | 235 ++++++++++++++++-- .../src/control/security/auth_lease/table.rs | 128 +++++++++- nodedb/src/control/server/http/mod.rs | 1 + .../src/control/server/http/routes/health.rs | 32 ++- .../src/control/server/http/routes/metrics.rs | 5 +- nodedb/src/control/server/http/server.rs | 57 +---- .../src/control/server/http/startup_gate.rs | 184 ++++++++++++++ nodedb/src/control/server/ilp_listener.rs | 6 +- nodedb/src/control/server/listener.rs | 6 +- nodedb/src/control/server/mod.rs | 1 + nodedb/src/control/server/pgwire/listener.rs | 6 +- nodedb/src/control/server/reserved_socket.rs | 112 +++++++++ nodedb/src/control/server/resp/listener.rs | 5 + nodedb/src/control/server/sync/listener.rs | 26 +- nodedb/src/control/startup/gate.rs | 7 +- nodedb/src/control/startup/health.rs | 30 ++- nodedb/src/control/startup/phase.rs | 24 +- .../src/control/startup/startup_sequencer.rs | 11 +- nodedb/src/ctl/healthcheck.rs | 2 +- nodedb/src/main.rs | 37 ++- nodedb/src/main_boot/gates.rs | 4 + nodedb/src/main_boot/listeners.rs | 32 +-- nodedb/tests/crash_harness/pgwire.rs | 2 +- nodedb/tests/inproc/cases/http_health.rs | 5 +- .../tests/inproc/cases/startup_gate_http.rs | 44 ++-- .../tests/inproc/cases/startup_gate_native.rs | 19 +- nodedb/tests/inproc/cases/startup_ordering.rs | 26 +- 37 files changed, 1062 insertions(+), 314 deletions(-) create mode 100644 nodedb/src/control/server/http/startup_gate.rs create mode 100644 nodedb/src/control/server/reserved_socket.rs diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs index 33b163290..5c0498945 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs @@ -454,12 +454,13 @@ impl TestClusterNode { // authorization lease, as a production node opens its gateway only // once it holds one. if let Some(timing) = shared.authorization_fence.timing() { - shared - .authorization_fence - .holder() - .await_valid(Duration::from_secs(15), timing.renew_every) - .await - .map_err(|e| format!("node {node_id}: {e}"))?; + nodedb::control::security::auth_lease::await_planning_admitted( + &shared, + Duration::from_secs(15), + timing.renew_every, + ) + .await + .map_err(|e| format!("node {node_id}: {e}"))?; } Ok(Self { diff --git a/nodedb/src/bootstrap/cluster_ready.rs b/nodedb/src/bootstrap/cluster_ready.rs index 46b1fa1c4..b4e549d6a 100644 --- a/nodedb/src/bootstrap/cluster_ready.rs +++ b/nodedb/src/bootstrap/cluster_ready.rs @@ -211,11 +211,12 @@ pub async fn await_cluster_ready( // opens once the first lease is granted, so the first statements are // not refused. if let Some(timing) = shared.authorization_fence.timing() - && let Err(error) = shared - .authorization_fence - .holder() - .await_valid(RAFT_READY_STALL_TIMEOUT, timing.renew_every) - .await + && let Err(error) = crate::control::security::auth_lease::await_planning_admitted( + shared, + RAFT_READY_STALL_TIMEOUT, + timing.renew_every, + ) + .await { gateway_enable_gate.fail(format!("authorization lease not granted: {error}")); return Err(anyhow::anyhow!( diff --git a/nodedb/src/bootstrap/listeners.rs b/nodedb/src/bootstrap/listeners.rs index 49d3509b9..1200fc300 100644 --- a/nodedb/src/bootstrap/listeners.rs +++ b/nodedb/src/bootstrap/listeners.rs @@ -13,18 +13,19 @@ use crate::control::cluster::ClusterHandle; use crate::control::server::ilp_listener::IlpListener; use crate::control::server::listener::Listener; use crate::control::server::pgwire::listener::PgListener; +use crate::control::server::reserved_socket::ReservedSocket; use crate::control::server::resp::RespListener; use crate::control::shutdown::ShutdownBus; -use crate::control::startup::StartupGate; +use crate::control::startup::{ReadyGate, StartupGate}; use crate::control::state::SharedState; -/// The pre-bound protocol listeners passed to [`spawn_protocol_listeners`]. +/// The listening client-protocol sockets passed to +/// [`spawn_protocol_listeners`]. /// -/// Every socket here is already bound by [`bind_listeners`], so spawning -/// cannot fail on a port conflict. +/// Every socket here was bound by [`bind_listeners`] and opened by +/// [`open_listeners`], so spawning cannot fail on a port conflict. pub struct ProtocolListeners { pub pg_listener: PgListener, - pub http_listener: TcpListener, pub sync_listener: TcpListener, pub ilp_listener: Option, pub resp_listener: Option, @@ -37,14 +38,16 @@ pub struct ListenerInfra { pub shutdown_bus: ShutdownBus, } -/// Spawn all non-native protocol listeners as background tasks. +/// Spawn all non-native client-protocol listeners as background tasks. /// /// The native listener is not spawned here — it is run on the main task -/// by the caller after this returns. +/// by the caller after this returns. The HTTP server is spawned earlier by +/// [`spawn_http_listener`]. /// /// Infallible by construction: every socket was bound by [`bind_listeners`] -/// before this point, so a port conflict has already aborted boot while -/// nothing was exposed. Nothing here may silently swallow a bind failure. +/// and opened by [`open_listeners`] before this point, so a port conflict +/// has already aborted boot. Nothing here may silently swallow a bind +/// failure. pub async fn spawn_protocol_listeners( listeners: ProtocolListeners, shared: Arc, @@ -55,7 +58,6 @@ pub async fn spawn_protocol_listeners( ) { let ProtocolListeners { pg_listener, - http_listener, sync_listener, ilp_listener, resp_listener, @@ -70,7 +72,6 @@ pub async fn spawn_protocol_listeners( }; let tls_flags = config.server.tls.as_ref(); let pgwire_tls_enabled = tls_flags.is_some_and(|t| t.pgwire); - let http_tls_enabled = tls_flags.is_some_and(|t| t.http); let resp_tls_enabled = tls_flags.is_some_and(|t| t.resp); let ilp_tls_enabled = tls_flags.is_some_and(|t| t.ilp); @@ -97,29 +98,6 @@ pub async fn spawn_protocol_listeners( } }); - // HTTP API server (on the socket bound by `bind_listeners`). - let shared_http = Arc::clone(&shared); - let http_auth_mode = config.auth.mode.clone(); - let http_tls = if http_tls_enabled { - config.server.tls.clone() - } else { - None - }; - let bus_http = shutdown_bus.clone(); - tokio::spawn(async move { - if let Err(e) = crate::control::server::http::server::run( - http_listener, - shared_http, - http_auth_mode, - http_tls.as_ref(), - bus_http, - ) - .await - { - tracing::error!(error = %e, "HTTP API server failed"); - } - }); - // ILP TCP listener (if configured). if let Some(ilp) = ilp_listener { let shared_ilp = Arc::clone(&shared); @@ -194,11 +172,61 @@ pub async fn spawn_protocol_listeners( nodedb_cluster::readiness::notify_ready(); } -/// Every protocol socket, bound before any accept loop starts. +/// Spawn the HTTP API server on `http_listener`. +/// +/// Boot calls this before it waits for the node to become ready, so +/// orchestrator probes can watch startup. Until the `Serving` phase only the +/// probe and metrics routes answer (see +/// `control::server::http::startup_gate`). +pub fn spawn_http_listener( + http_listener: TcpListener, + shared: Arc, + config: &ServerConfig, + shutdown_bus: ShutdownBus, +) { + let http_auth_mode = config.auth.mode.clone(); + let http_tls = config.server.tls.as_ref().filter(|tls| tls.http).cloned(); + tokio::spawn(async move { + if let Err(e) = crate::control::server::http::server::run( + http_listener, + shared, + http_auth_mode, + http_tls.as_ref(), + shutdown_bus, + ) + .await + { + tracing::error!(error = %e, "HTTP API server failed"); + } + }); +} + +/// Every protocol socket, bound before the node waits to become ready. +/// +/// The HTTP socket listens from the start, so probes answer during boot. The +/// client-protocol sockets are bound but not listening. pub struct BoundListeners { + pub http: TcpListener, + pub clients: ClientSockets, +} + +/// The client-protocol sockets, bound but not yet listening. +/// +/// A bound socket that does not listen refuses each connection attempt at +/// once. Boot listens through [`open_listeners`] only once the node can +/// serve, so no client waits in a kernel accept queue through boot. +pub struct ClientSockets { + pub native: ReservedSocket, + pub pgwire: ReservedSocket, + pub sync: ReservedSocket, + pub ilp: Option, + pub resp: Option, +} + +/// Every client-protocol socket, listening. +pub struct OpenListeners { pub native: Listener, pub pgwire: PgListener, - pub http: TcpListener, pub sync: TcpListener, pub ilp: Option, pub resp: Option, @@ -209,35 +237,82 @@ pub struct BoundListeners { /// This is the single fail-fast point for listener setup: it runs before the /// node waits on cluster readiness and before any accept loop is spawned, so /// a port conflict on *any* protocol — including HTTP and sync, which serve -/// from detached tasks — aborts boot while nothing is exposed yet. Never -/// move a bind out of here into a spawned task; that is how a server ends up -/// running for days missing a listener behind one warning line. -pub async fn bind_listeners(config: &ServerConfig) -> anyhow::Result { - let native = crate::control::server::listener::Listener::bind(config.native_addr()).await?; - let pgwire = - crate::control::server::pgwire::listener::PgListener::bind(config.pgwire_addr()).await?; - let http = TcpListener::bind(config.http_addr()) - .await - .with_context(|| format!("bind HTTP API listener to {}", config.http_addr()))?; - let sync = crate::control::server::sync::listener::bind_sync_listener(config.sync_addr()) - .await - .context("sync listener failed to bind")?; - let ilp = if let Some(ilp_addr) = config.ilp_addr() { - Some(crate::control::server::ilp_listener::IlpListener::bind(ilp_addr).await?) - } else { - None - }; - let resp = if let Some(resp_addr) = config.resp_addr() { - Some(crate::control::server::resp::RespListener::bind(resp_addr).await?) - } else { - None +/// from detached tasks — aborts boot early. Never move a bind out of here +/// into a spawned task; that is how a server ends up running for days +/// missing a listener behind one warning line. +pub fn bind_listeners(config: &ServerConfig) -> anyhow::Result { + let reserve = |name: &str, addr: std::net::SocketAddr| { + ReservedSocket::bind(addr).with_context(|| format!("bind {name} listener to {addr}")) }; + let http = reserve("HTTP API", config.http_addr())? + .listen() + .context("listen on the HTTP API address")?; + let native = reserve("native protocol", config.native_addr())?; + let pgwire = reserve("pgwire", config.pgwire_addr())?; + let sync = crate::control::server::sync::listener::reserve_sync_listener(config.sync_addr()) + .context("sync listener failed to bind")?; + let ilp = config + .ilp_addr() + .map(|addr| reserve("ILP", addr)) + .transpose()?; + let resp = config + .resp_addr() + .map(|addr| reserve("RESP", addr)) + .transpose()?; Ok(BoundListeners { + http, + clients: ClientSockets { + native, + pgwire, + sync, + ilp, + resp, + }, + }) +} + +/// Start listening on every client-protocol socket, then fire +/// `serving_gate`. +/// +/// Boot calls this once the node can serve and before any accept loop is +/// spawned. The gate advances the startup sequencer to +/// [`StartupPhase::Serving`](crate::control::startup::StartupPhase::Serving), +/// which opens every HTTP route and lets `/healthz` report `ok`, at the same +/// point the client protocols start listening. A socket that cannot listen +/// fails the gate and aborts boot: another process started listening on its +/// address after the bind. +pub fn open_listeners( + clients: ClientSockets, + serving_gate: ReadyGate, +) -> anyhow::Result { + let ClientSockets { native, pgwire, - http, sync, ilp, resp, - }) + } = clients; + let open = || -> crate::Result { + Ok(OpenListeners { + native: Listener::from_listener(native.listen()?)?, + pgwire: PgListener::from_listener(pgwire.listen()?)?, + sync: sync.listen()?, + ilp: ilp + .map(|socket| IlpListener::from_listener(socket.listen()?)) + .transpose()?, + resp: resp + .map(|socket| RespListener::from_listener(socket.listen()?)) + .transpose()?, + }) + }; + match open() { + Ok(open) => { + serving_gate.fire(); + Ok(open) + } + Err(error) => { + serving_gate.fail(error.to_string()); + Err(error.into()) + } + } } diff --git a/nodedb/src/bootstrap/state_wiring.rs b/nodedb/src/bootstrap/state_wiring.rs index 155949d57..4c3bb1173 100644 --- a/nodedb/src/bootstrap/state_wiring.rs +++ b/nodedb/src/bootstrap/state_wiring.rs @@ -43,10 +43,16 @@ pub async fn wire_state( array_catalog, maintenance_budget, } = components; - // Install startup gate. - if let Some(state) = Arc::get_mut(shared) { - state.startup = Arc::clone(startup_gate); - } + // Install startup gate. `/healthz` and the HTTP startup gate read it, so + // a state left on the test helpers' pre-fired gate would report ready + // and open every route during boot. + Arc::get_mut(shared) + .ok_or_else(|| { + anyhow::anyhow!( + "startup gate: SharedState is already shared before the gate was installed" + ) + })? + .startup = Arc::clone(startup_gate); // Replay surrogate WAL records. // Note: wal_records are not passed here — caller must handle surrogate replay diff --git a/nodedb/src/control/security/auth_fence/view.rs b/nodedb/src/control/security/auth_fence/view.rs index 4b5c22ecf..1c46b8dc9 100644 --- a/nodedb/src/control/security/auth_fence/view.rs +++ b/nodedb/src/control/security/auth_fence/view.rs @@ -15,7 +15,9 @@ //! 4. In a cluster, the node must hold a valid authorization lease. A writer //! acknowledges an authorization change only after every lease holder //! covered it or its lease expired, so a valid lease means the state read -//! here holds every change acknowledged before this point. +//! here holds every change acknowledged before this point. A node that +//! leads the metadata group as its only voter holds a pinned lease, which +//! never expires: every barrier waits for its coverage instead. //! //! The lease is checked after the cache guard is taken. The guard fixes the //! cache for the whole plan, and a change acknowledged after the check was @@ -25,6 +27,7 @@ use std::time::Instant; use tokio::sync::RwLockReadGuard; +use crate::control::security::auth_lease::lease_status; use crate::control::security::permission_tree::{PermissionCache, reload}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, VShardId}; @@ -58,12 +61,7 @@ pub async fn permission_view( } } } - if state.authorization_fence.timing().is_some() - && !state - .authorization_fence - .holder() - .is_valid_at(Instant::now()) - { + if !lease_status(state, Instant::now()).admits_planning() { return Err(behind( "this node holds no valid authorization lease; it has not confirmed the latest \ authorization changes", diff --git a/nodedb/src/control/security/auth_lease/barrier.rs b/nodedb/src/control/security/auth_lease/barrier.rs index 50712374b..b96676cdb 100644 --- a/nodedb/src/control/security/auth_lease/barrier.rs +++ b/nodedb/src/control/security/auth_lease/barrier.rs @@ -34,7 +34,9 @@ use super::leadership::{metadata_leader, send_to_leader}; /// holder reports its coverage only with its next renewal. This covers every /// DDL that bears authorization, `CREATE COLLECTION` included. A holder that /// cannot renew adds up to one lease duration (the election timeout), until -/// its lease expires. On a single node the wait is the permission step's +/// its lease expires. A pinned holder, the leader that is the only voter of +/// the metadata group, has no expiry: the barrier waits for its next renewal +/// however late it runs. On a single node the wait is the permission step's /// lag only. pub async fn authorization_barrier( state: &SharedState, diff --git a/nodedb/src/control/security/auth_lease/holder.rs b/nodedb/src/control/security/auth_lease/holder.rs index b1f6c7f26..0d940b3e4 100644 --- a/nodedb/src/control/security/auth_lease/holder.rs +++ b/nodedb/src/control/security/auth_lease/holder.rs @@ -6,6 +6,9 @@ //! lease is valid. The lease ends on this node's clock before it ends on the //! leader's (see [`super::timing`]), so once the leader treats it as expired //! no statement here can still plan under it. +//! +//! A node that leads the metadata group as its only voter also plans under a +//! pinned lease, which has no expiry (see [`super::status`]). use std::sync::Mutex; use std::time::Instant; @@ -38,25 +41,6 @@ impl LeaseHolder { pub fn valid_until(&self) -> Option { *self.valid_until.lock().unwrap_or_else(|p| p.into_inner()) } - - /// Wait until a lease is valid, polling every `poll`, or refuse once - /// `timeout` passes. - pub async fn await_valid( - &self, - timeout: std::time::Duration, - poll: std::time::Duration, - ) -> crate::Result<()> { - let deadline = Instant::now() + timeout; - while !self.is_valid_at(Instant::now()) { - if Instant::now() >= deadline { - return Err(crate::Error::AuthorizationStateBehind { - detail: format!("no authorization lease was granted within {timeout:?}"), - }); - } - tokio::time::sleep(poll).await; - } - Ok(()) - } } #[cfg(test)] diff --git a/nodedb/src/control/security/auth_lease/leadership.rs b/nodedb/src/control/security/auth_lease/leadership.rs index c9189e281..d475ed5c3 100644 --- a/nodedb/src/control/security/auth_lease/leadership.rs +++ b/nodedb/src/control/security/auth_lease/leadership.rs @@ -27,6 +27,23 @@ pub(crate) fn leading_term(state: &SharedState) -> Option { .map(|(_, term)| term) } +/// The term this node leads the metadata group in, when it is also the +/// group's only voter. +/// +/// A single-voter group commits a configuration change inside the propose +/// call and applies it at once. A second voter therefore ends this before it +/// can vote. No other node can lead the group while this returns a term. +pub(crate) fn sole_voter_term(state: &SharedState) -> Option { + let status = state.raft_status_fn.get()?; + status() + .into_iter() + .find(|group| group.group_id == METADATA_GROUP_ID) + .filter(|group| { + group.role == "Leader" && group.leader_id == state.node_id && group.member_count == 1 + }) + .map(|group| group.term) +} + /// The leader hint to send back with a refusal. pub(crate) fn leader_hint(state: &SharedState) -> Option { metadata_leader(state) diff --git a/nodedb/src/control/security/auth_lease/mod.rs b/nodedb/src/control/security/auth_lease/mod.rs index a6bb95512..907385bad 100644 --- a/nodedb/src/control/security/auth_lease/mod.rs +++ b/nodedb/src/control/security/auth_lease/mod.rs @@ -18,5 +18,5 @@ pub use barrier::{ pub use calvin_acks::CalvinAckCoverage; pub use holder::LeaseHolder; pub use service::LeaderLeaseService; -pub use status::{LeaseStatus, lease_status}; +pub use status::{LeaseStatus, await_planning_admitted, lease_status}; pub use timing::LeaseTiming; diff --git a/nodedb/src/control/security/auth_lease/service.rs b/nodedb/src/control/security/auth_lease/service.rs index 270169f3c..98f8abad4 100644 --- a/nodedb/src/control/security/auth_lease/service.rs +++ b/nodedb/src/control/security/auth_lease/service.rs @@ -17,6 +17,11 @@ //! leadership against a quorum, taken after the decision. A leader deposed //! meanwhile answers `NotLeader`, so no lease it grants and no barrier it //! releases outlives its term unseen. +//! +//! A leader that is the only voter of the metadata group pins its own lease +//! (see [`super::table`]). No other node can lead the group then, so every +//! barrier releases here, and each one waits for this node's coverage. +//! Planning on this node then needs no lease that expires on the clock. use std::collections::HashSet; use std::sync::{Mutex, Weak}; @@ -36,7 +41,7 @@ use crate::control::security::auth_fence::cluster::{ use crate::control::security::auth_fence::view::apply_committed_tree_defs; use crate::control::state::SharedState; -use super::leadership::{leader_hint, leading_term}; +use super::leadership::{leader_hint, leading_term, sole_voter_term}; use super::table::{BarrierState, LeaseTable, RenewDecision}; use super::timing::LeaseTiming; use super::withheld_warn::WithheldWarnings; @@ -80,6 +85,24 @@ impl LeaderLeaseService { self.table.lock().unwrap_or_else(|p| p.into_inner()) } + /// Whether the table of `term` pins the lease of `node_id`. + /// + /// The caller passes the term in which it read that this node is the + /// only voter of the metadata group. + pub fn holds_pinned_lease(&self, term: u64, node_id: u64) -> bool { + self.table() + .as_ref() + .is_some_and(|table| table.term() == term && table.is_pinned(node_id)) + } + + /// Run `edit` on the current table, creating one for `term` first. + #[cfg(test)] + pub(crate) fn edit_table(&self, term: u64, now: Instant, edit: impl FnOnce(&mut LeaseTable)) { + let mut table = self.table(); + let table = table.get_or_insert_with(|| LeaseTable::new(term, now)); + edit(table); + } + /// Make the table of `term` current and load its floors. async fn table_ready(&self, state: &SharedState, term: u64) -> crate::Result<()> { { @@ -184,16 +207,24 @@ impl LeaderLeaseService { outcome: AuthLeaseRenewOutcome::Withheld, }; } + let sole_voter = sole_voter_term(&state) == Some(term); let (decision, shortfall) = { let mut table = self.table(); match table.as_mut().filter(|t| t.term() == term) { Some(table) => { + table.observe_sole_voter(sole_voter); let decision = table.renew( req.node_id, &req.coverage, Instant::now(), self.timing.lease, ); + if decision == RenewDecision::Granted + && sole_voter + && req.node_id == state.node_id + { + table.pin(req.node_id); + } let shortfall = match decision { RenewDecision::Withheld => table.shortfall(&req.coverage), RenewDecision::Granted => Vec::new(), @@ -257,6 +288,7 @@ impl LeaderLeaseService { tokio::time::sleep(self.timing.renew_every).await; continue; } + let sole_voter = sole_voter_term(&state) == Some(term); let notified = self.changed.notified(); tokio::pin!(notified); notified.as_mut().enable(); @@ -264,6 +296,7 @@ impl LeaderLeaseService { let mut table = self.table(); match table.as_mut().filter(|t| t.term() == term) { Some(table) => { + table.observe_sole_voter(sole_voter); table.raise_floors(&req.targets); table.barrier(&req.targets, Instant::now(), self.timing.lease) } @@ -282,6 +315,10 @@ impl LeaderLeaseService { } BarrierState::NotReady => tokio::time::sleep(self.timing.renew_every).await, BarrierState::Waiting { until } => { + // A pinned holder releases the barrier by renewing, which + // wakes `notified`. A check every renewal interval also + // sees the pin end when a second voter joins. + let until = until.unwrap_or_else(|| Instant::now() + self.timing.renew_every); let wake = tokio::time::Instant::from_std(until.min(deadline)); drop(state); tokio::select! { diff --git a/nodedb/src/control/security/auth_lease/status.rs b/nodedb/src/control/security/auth_lease/status.rs index f4a182ce0..bec4b3a68 100644 --- a/nodedb/src/control/security/auth_lease/status.rs +++ b/nodedb/src/control/security/auth_lease/status.rs @@ -11,6 +11,9 @@ use std::fmt::Write as _; use std::time::{Duration, Instant}; use crate::control::security::auth_fence::AuthorizationFence; +use crate::control::state::SharedState; + +use super::leadership::sole_voter_term; /// Whether this node can plan permission-checked statements now. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -19,37 +22,98 @@ pub enum LeaseStatus { NotRequired, /// The lease is valid for `remaining`. Valid { remaining: Duration }, + /// This node leads the metadata group as its only voter and holds a + /// pinned lease, which has no expiry (see [`super::table`]). + SoleVoter, /// No valid lease. `expired_for` is how long ago the last one ended, or /// `None` when no lease was ever granted. Invalid { expired_for: Option }, } -/// The lease state at `now`. -pub fn lease_status(fence: &AuthorizationFence, now: Instant) -> LeaseStatus { +impl LeaseStatus { + /// Whether this node can plan permission-checked statements. + pub fn admits_planning(&self) -> bool { + !matches!(self, Self::Invalid { .. }) + } +} + +/// The lease state of this node at `now`. +pub fn lease_status(state: &SharedState, now: Instant) -> LeaseStatus { + lease_status_of( + &state.authorization_fence, + || sole_voter_term(state), + state.node_id, + now, + ) +} + +/// The lease state of node `node_id` at `now`. +/// +/// `sole_voter_term` returns the term in which this node leads the metadata +/// group as its only voter. It runs only when the bounded lease is not +/// valid, so the planning path reads the Raft status only after a lapse. +pub(crate) fn lease_status_of( + fence: &AuthorizationFence, + sole_voter_term: impl FnOnce() -> Option, + node_id: u64, + now: Instant, +) -> LeaseStatus { if fence.timing().is_none() { return LeaseStatus::NotRequired; } - match fence.holder().valid_until() { - Some(until) if now < until => LeaseStatus::Valid { + let valid_until = fence.holder().valid_until(); + if let Some(until) = valid_until.filter(|until| now < *until) { + return LeaseStatus::Valid { remaining: until - now, - }, - Some(until) => LeaseStatus::Invalid { - expired_for: Some(now.saturating_duration_since(until)), - }, - None => LeaseStatus::Invalid { expired_for: None }, + }; + } + let pinned = sole_voter_term().is_some_and(|term| { + fence + .leader() + .is_some_and(|service| service.holds_pinned_lease(term, node_id)) + }); + if pinned { + return LeaseStatus::SoleVoter; + } + LeaseStatus::Invalid { + expired_for: valid_until.map(|until| now.saturating_duration_since(until)), + } +} + +/// Wait until this node can plan permission-checked statements, polling +/// every `poll`, or refuse once `timeout` passes. +pub async fn await_planning_admitted( + state: &SharedState, + timeout: Duration, + poll: Duration, +) -> crate::Result<()> { + let deadline = Instant::now() + timeout; + while !lease_status(state, Instant::now()).admits_planning() { + if Instant::now() >= deadline { + return Err(crate::Error::AuthorizationStateBehind { + detail: format!("no authorization lease was granted within {timeout:?}"), + }); + } + tokio::time::sleep(poll).await; } + Ok(()) } /// Append the lease gauges. A single node holds no lease and emits none. /// /// - `nodedb_authorization_lease_valid`: 1 while the lease is valid, else 0. /// - `nodedb_authorization_lease_remaining_seconds`: time left on the lease, -/// 0 without one. -pub fn render_prometheus(fence: &AuthorizationFence, out: &mut String) { - let (valid, remaining) = match lease_status(fence, Instant::now()) { +/// 0 without one, `+Inf` for a pinned lease. +pub fn render_prometheus(state: &SharedState, out: &mut String) { + render_status(lease_status(state, Instant::now()), out); +} + +fn render_status(status: LeaseStatus, out: &mut String) { + let (valid, remaining) = match status { LeaseStatus::NotRequired => return, - LeaseStatus::Valid { remaining } => (1, remaining), - LeaseStatus::Invalid { .. } => (0, Duration::ZERO), + LeaseStatus::Valid { remaining } => (1, remaining.as_secs_f64()), + LeaseStatus::SoleVoter => (1, f64::INFINITY), + LeaseStatus::Invalid { .. } => (0, 0.0), }; let _ = writeln!( out, @@ -61,60 +125,177 @@ pub fn render_prometheus(fence: &AuthorizationFence, out: &mut String) { node's authorization lease\n\ # TYPE nodedb_authorization_lease_remaining_seconds gauge\n\ nodedb_authorization_lease_remaining_seconds {}", - remaining.as_secs_f64() + prometheus_float(remaining) ); } +/// A gauge value in the Prometheus text format, which spells infinity `+Inf`. +fn prometheus_float(value: f64) -> String { + if value.is_infinite() { + "+Inf".to_string() + } else { + value.to_string() + } +} + #[cfg(test)] mod tests { + use std::sync::{Arc, Weak}; + + use nodedb_cluster::GroupCoverage; + use super::*; - use crate::control::security::auth_lease::LeaseTiming; + use crate::control::security::auth_lease::table::RenewDecision; + use crate::control::security::auth_lease::{LeaderLeaseService, LeaseTiming}; use crate::control::security::permission_tree::SourceIndex; + const NODE: u64 = 1; + const TERM: u64 = 3; + fn timing() -> LeaseTiming { LeaseTiming::from_raft(Duration::from_millis(1000), Duration::from_millis(100)) .expect("timing") } + fn fence() -> AuthorizationFence { + AuthorizationFence::new(Arc::new(SourceIndex::default())) + } + + /// A fence whose leader service granted this node's lease at `granted_at` + /// in `TERM`, pinned when `sole_voter` holds, as one renewal round does. + fn fence_granted_at(granted_at: Instant, sole_voter: bool) -> AuthorizationFence { + let fence = fence(); + let timing = timing(); + assert!(fence.install_timing(timing)); + let service = Arc::new(LeaderLeaseService::new(Weak::new(), timing)); + let coverage = [GroupCoverage { + group_id: 0, + through: 10, + }]; + service.edit_table(TERM, granted_at, |table| { + table.load_floors(&coverage); + table.observe_sole_voter(sole_voter); + assert_eq!( + table.renew(NODE, &coverage, granted_at, timing.lease), + RenewDecision::Granted + ); + if sole_voter { + table.pin(NODE); + } + }); + assert!(fence.install_leader(service)); + fence + .holder() + .install(timing.holder_expiry(granted_at, timing.lease)); + fence + } + #[test] fn a_node_without_timing_needs_no_lease() { - let fence = AuthorizationFence::new(std::sync::Arc::new(SourceIndex::default())); - assert_eq!( - lease_status(&fence, Instant::now()), - LeaseStatus::NotRequired - ); + let fence = fence(); + let status = lease_status_of(&fence, || None, NODE, Instant::now()); + assert_eq!(status, LeaseStatus::NotRequired); let mut out = String::new(); - render_prometheus(&fence, &mut out); + render_status(status, &mut out); assert!(out.is_empty()); } #[test] fn a_lapsed_lease_is_invalid_and_reports_zero() { - let fence = AuthorizationFence::new(std::sync::Arc::new(SourceIndex::default())); + let fence = fence(); assert!(fence.install_timing(timing())); let now = Instant::now(); assert_eq!( - lease_status(&fence, now), + lease_status_of(&fence, || None, NODE, now), LeaseStatus::Invalid { expired_for: None } ); fence.holder().install(now + Duration::from_secs(2)); + let status = lease_status_of(&fence, || None, NODE, now); assert_eq!( - lease_status(&fence, now), + status, LeaseStatus::Valid { remaining: Duration::from_secs(2) } ); let mut out = String::new(); - render_prometheus(&fence, &mut out); + render_status(status, &mut out); assert!(out.contains("nodedb_authorization_lease_valid 1")); let later = now + Duration::from_secs(3); + let status = lease_status_of(&fence, || None, NODE, later); assert_eq!( - lease_status(&fence, later), + status, LeaseStatus::Invalid { expired_for: Some(Duration::from_secs(1)) } ); + assert!(!status.admits_planning()); + } + + /// The renewal task of a sole-voter node is starved for a whole lease + /// duration. Its bounded lease lapses, yet the node still plans: its + /// lease is pinned, and every barrier waits for its coverage. + #[test] + fn a_starved_renewal_on_a_sole_voter_keeps_planning() { + let granted_at = Instant::now(); + let fence = fence_granted_at(granted_at, true); + let starved = granted_at + timing().lease; + assert!( + !fence.holder().is_valid_at(starved), + "the bounded lease lapsed" + ); + + let status = lease_status_of(&fence, || Some(TERM), NODE, starved); + assert_eq!(status, LeaseStatus::SoleVoter); + assert!(status.admits_planning()); + let mut out = String::new(); + render_status(status, &mut out); + assert!(out.contains("nodedb_authorization_lease_valid 1")); + assert!(out.contains("nodedb_authorization_lease_remaining_seconds +Inf")); + } + + /// A cluster node whose renewal lapses cannot confirm it holds the latest + /// authorization state. It refuses, whatever its table once recorded. + #[test] + fn a_node_with_a_peer_voter_refuses_once_its_lease_lapses() { + let granted_at = Instant::now(); + let starved = granted_at + timing().lease; + + // A lease granted while a peer voter existed was never pinned. + let fence = fence_granted_at(granted_at, false); + let status = lease_status_of(&fence, || None, NODE, starved); + assert!(matches!(status, LeaseStatus::Invalid { .. })); + assert!(!status.admits_planning()); + + // A pinned lease stops counting the moment a peer voter joins. + let fence = fence_granted_at(granted_at, true); + let status = lease_status_of(&fence, || None, NODE, starved); + assert!(!status.admits_planning()); + } + + /// A barrier that runs once a peer voter joined removes the pin. The node + /// then refuses even after it is the only voter again, until a new grant. + #[test] + fn a_pin_removed_by_a_barrier_stays_removed() { + let granted_at = Instant::now(); + let fence = fence_granted_at(granted_at, true); + let starved = granted_at + timing().lease; + fence + .leader() + .expect("leader service") + .edit_table(TERM, starved, |table| table.observe_sole_voter(false)); + let status = lease_status_of(&fence, || Some(TERM), NODE, starved); + assert!(!status.admits_planning()); + } + + /// A pin belongs to the leadership term that granted it. + #[test] + fn a_pin_from_an_earlier_term_does_not_count() { + let granted_at = Instant::now(); + let fence = fence_granted_at(granted_at, true); + let starved = granted_at + timing().lease; + let status = lease_status_of(&fence, || Some(TERM + 1), NODE, starved); + assert!(!status.admits_planning()); } } diff --git a/nodedb/src/control/security/auth_lease/table.rs b/nodedb/src/control/security/auth_lease/table.rs index 3b2a53fc1..161f6152a 100644 --- a/nodedb/src/control/security/auth_lease/table.rs +++ b/nodedb/src/control/security/auth_lease/table.rs @@ -13,6 +13,12 @@ //! in the table. They end within one lease duration of this leader taking //! over, so a barrier also waits until then. //! +//! A leader that is the only voter of the metadata group can pin its own +//! lease. A pinned lease has no expiry: every barrier waits for the pinned +//! node's coverage, however late its renewal runs. The caller reports on +//! every renewal and barrier whether the leader is still the only voter. The +//! first report that it is not removes the pin. +//! //! The table is pure: callers pass the clock, so every rule is testable. use std::collections::HashMap; @@ -50,8 +56,10 @@ pub enum BarrierState { /// No node can plan against state older than the targets. Released, /// Waiting for a report or an expiry. Nothing changes on its own before - /// the instant named, except a renewal. - Waiting { until: Instant }, + /// `until`, except a renewal. `until` is `None` while the pinned holder + /// has not covered the targets: only its renewal, or the end of the pin, + /// can release the barrier then. + Waiting { until: Option }, /// The floors of this term are not loaded yet. NotReady, } @@ -64,6 +72,9 @@ pub struct LeaseTable { floors_ready: bool, floors: HashMap, holders: HashMap, + /// The holder whose lease has no expiry while the leader stays the only + /// voter of the metadata group. + pinned: Option, } impl LeaseTable { @@ -75,6 +86,7 @@ impl LeaseTable { floors_ready: false, floors: HashMap::new(), holders: HashMap::new(), + pinned: None, } } @@ -125,6 +137,25 @@ impl LeaseTable { RenewDecision::Granted } + /// Record whether the leader is still the only voter of the metadata + /// group. A leader that is not removes the pin. + pub fn observe_sole_voter(&mut self, sole_voter: bool) { + if !sole_voter { + self.pinned = None; + } + } + + /// Pin the lease of `node_id`, which a renewal just granted while the + /// leader was the only voter. + pub fn pin(&mut self, node_id: u64) { + self.pinned = Some(node_id); + } + + /// Whether the lease of `node_id` is pinned. + pub fn is_pinned(&self, node_id: u64) -> bool { + self.pinned == Some(node_id) + } + /// Every group whose floor `coverage` does not reach, as /// `(group_id, floor, reported)`. `reported` is `None` for a group the /// report omits. @@ -159,23 +190,33 @@ impl LeaseTable { let mut wait_for = |instant: Instant| { until = Some(until.map_or(instant, |current: Instant| current.min(instant))); }; + let mut wait_for_pinned = false; let earlier_leases_end = self.leader_since + lease; if now < earlier_leases_end { wait_for(earlier_leases_end); } - for record in self.holders.values() { - let Some(expires_at) = record.expires_at.filter(|end| *end > now) else { + for (node_id, record) in &self.holders { + let pinned = self.pinned == Some(*node_id); + let live_until = record.expires_at.filter(|end| *end > now); + if !pinned && live_until.is_none() { continue; - }; + } let covered = targets .iter() .all(|target| record.covers(target.group_id, target.through)); - if !covered { - wait_for(expires_at); + if covered { + continue; } + match live_until { + Some(expires_at) if !pinned => wait_for(expires_at), + _ => wait_for_pinned = true, + } + } + if wait_for_pinned { + return BarrierState::Waiting { until: None }; } match until { - Some(until) => BarrierState::Waiting { until }, + Some(until) => BarrierState::Waiting { until: Some(until) }, None => BarrierState::Released, } } @@ -248,7 +289,7 @@ mod tests { assert_eq!( table.barrier(&[cover(0, 5)], start, LEASE), BarrierState::Waiting { - until: start + LEASE + until: Some(start + LEASE) } ); assert_eq!( @@ -271,7 +312,9 @@ mod tests { // Node 2 holds a lease and has not covered index 12. assert_eq!( table.barrier(&target, now, LEASE), - BarrierState::Waiting { until: now + LEASE } + BarrierState::Waiting { + until: Some(now + LEASE) + } ); // Its renewal below the new floor is withheld, and its lease is not // extended. @@ -291,7 +334,9 @@ mod tests { table.raise_floors(&later); assert_eq!( table.barrier(&later, now, LEASE), - BarrierState::Waiting { until: now + LEASE } + BarrierState::Waiting { + until: Some(now + LEASE) + } ); assert_eq!( table.barrier(&later, now + LEASE, LEASE), @@ -303,4 +348,65 @@ mod tests { RenewDecision::Withheld ); } + + /// A pinned lease holds a barrier past its bounded expiry. Its renewal + /// can run arbitrarily late without any barrier releasing behind it. + #[test] + fn a_pinned_lease_holds_a_barrier_past_its_expiry() { + let start = Instant::now(); + let mut table = settled_table(start); + let now = start + LEASE; + table.observe_sole_voter(true); + assert_eq!( + table.renew(1, &[cover(0, 10)], now, LEASE), + RenewDecision::Granted + ); + table.pin(1); + let target = [cover(0, 12)]; + table.raise_floors(&target); + + // Ten lease durations pass with no renewal. The bounded lease ended + // long ago, but the barrier still waits for node 1. + let starved = now + LEASE * 10; + table.observe_sole_voter(true); + assert_eq!( + table.barrier(&target, starved, LEASE), + BarrierState::Waiting { until: None } + ); + + // A late renewal that covers the target releases it. + assert_eq!( + table.renew(1, &[cover(0, 12)], starved, LEASE), + RenewDecision::Granted + ); + assert_eq!( + table.barrier(&target, starved, LEASE), + BarrierState::Released + ); + } + + /// Once the leader is not the only voter, the pin ends and the lease + /// expires on the clock again. + #[test] + fn a_second_voter_ends_the_pin() { + let start = Instant::now(); + let mut table = settled_table(start); + let now = start + LEASE; + assert_eq!( + table.renew(1, &[cover(0, 10)], now, LEASE), + RenewDecision::Granted + ); + table.pin(1); + assert!(table.is_pinned(1)); + let target = [cover(0, 12)]; + table.raise_floors(&target); + + let starved = now + LEASE * 10; + table.observe_sole_voter(false); + assert!(!table.is_pinned(1)); + assert_eq!( + table.barrier(&target, starved, LEASE), + BarrierState::Released + ); + } } diff --git a/nodedb/src/control/server/http/mod.rs b/nodedb/src/control/server/http/mod.rs index e533b4cb8..3cfe7f079 100644 --- a/nodedb/src/control/server/http/mod.rs +++ b/nodedb/src/control/server/http/mod.rs @@ -6,6 +6,7 @@ pub mod peer; pub(crate) mod rate_limit_headers; pub mod routes; pub mod server; +pub mod startup_gate; pub(crate) mod tls_accept; pub mod transport; pub mod types; diff --git a/nodedb/src/control/server/http/routes/health.rs b/nodedb/src/control/server/http/routes/health.rs index abafab8b5..a348bbaa7 100644 --- a/nodedb/src/control/server/http/routes/health.rs +++ b/nodedb/src/control/server/http/routes/health.rs @@ -6,7 +6,7 @@ //! |-------------------|--------|-----------------------------|---------------| //! | `/healthz` | GET | Ready to serve traffic | readiness | //! | `/health/live` | GET | Process alive (always 200) | liveness | -//! | `/health/ready` | GET | WAL recovered | readiness alt | +//! | `/health/ready` | GET | WAL recovered, serving | readiness alt | //! | `/health/drain` | POST | Trigger graceful drain | preStop hook | use std::sync::atomic::Ordering; @@ -34,13 +34,13 @@ pub async fn live() -> impl IntoResponse { /// GET /healthz — k8s-style readiness probe. /// -/// Returns `200 OK` when the node has reached `GatewayEnable`, is +/// Returns `200 OK` when the node has reached `Serving`, is /// serving traffic, is NOT draining/decommissioned, and — on a node that /// runs a Calvin sequencer — can actually sequence a cross-shard write. /// Returns `503 Service Unavailable` otherwise. /// /// Every condition is evaluated live on each call, not latched: sequencer -/// leadership and the epoch seed can both be lost long after `GatewayEnable`. +/// leadership and the epoch seed can both be lost long after `Serving`. pub async fn healthz(State(state): State) -> impl IntoResponse { // The coordinator signals this canonical watch before progressing drain // phases, so readiness must fail immediately even before lifecycle state @@ -192,13 +192,14 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { (status, axum::Json(body)) } -/// The lease state `/healthz` reports: `valid`, `invalid`, or `not_required` -/// on a single node without a cluster. +/// The lease state `/healthz` reports: `valid`, `invalid`, `sole_voter` for +/// a pinned lease, or `not_required` on a single node without a cluster. fn lease_label(state: &AppState) -> &'static str { use crate::control::security::auth_lease::{LeaseStatus, lease_status}; - match lease_status(&state.shared.authorization_fence, std::time::Instant::now()) { + match lease_status(&state.shared, std::time::Instant::now()) { LeaseStatus::NotRequired => "not_required", LeaseStatus::Valid { .. } => "valid", + LeaseStatus::SoleVoter => "sole_voter", LeaseStatus::Invalid { .. } => "invalid", } } @@ -207,8 +208,8 @@ fn lease_label(state: &AppState) -> &'static str { /// `None` when it holds one or needs none. fn lease_invalid_body(state: &AppState) -> Option { use crate::control::security::auth_lease::{LeaseStatus, lease_status}; - match lease_status(&state.shared.authorization_fence, std::time::Instant::now()) { - LeaseStatus::NotRequired | LeaseStatus::Valid { .. } => None, + match lease_status(&state.shared, std::time::Instant::now()) { + LeaseStatus::NotRequired | LeaseStatus::Valid { .. } | LeaseStatus::SoleVoter => None, LeaseStatus::Invalid { expired_for } => Some(json!({ "status": "degraded", "reason": "authorization_lease_invalid", @@ -302,16 +303,23 @@ fn sequencer_not_servable( None } -/// GET /health/ready — readiness check (WAL recovered, cores initialized). +/// GET /health/ready — readiness check: WAL recovered and the `Serving` phase reached. +/// +/// Boot recovers the WAL long before it listens on the client protocols, so +/// the WAL alone never makes the node ready. pub async fn ready(State(state): State) -> impl IntoResponse { - let wal_ready = state.shared.wal.next_lsn().as_u64() > 0; - let status = if wal_ready { + let serving = matches!( + crate::control::startup::health::observe(&state.shared.startup), + crate::control::startup::health::HealthState::Ok + ); + let ready = serving && state.shared.wal.next_lsn().as_u64() > 0; + let status = if ready { StatusCode::OK } else { StatusCode::SERVICE_UNAVAILABLE }; let body = json!({ - "status": if wal_ready { "ready" } else { "not_ready" }, + "status": if ready { "ready" } else { "not_ready" }, "wal_lsn": state.shared.wal.next_lsn().as_u64(), "node_id": state.shared.node_id, }); diff --git a/nodedb/src/control/server/http/routes/metrics.rs b/nodedb/src/control/server/http/routes/metrics.rs index b80157944..3afb56ad1 100644 --- a/nodedb/src/control/server/http/routes/metrics.rs +++ b/nodedb/src/control/server/http/routes/metrics.rs @@ -276,10 +276,7 @@ pub async fn metrics( // Authorization lease validity. A node without a valid lease refuses // permission-checked statements. - crate::control::security::auth_lease::status::render_prometheus( - &state.shared.authorization_fence, - &mut output, - ); + crate::control::security::auth_lease::status::render_prometheus(&state.shared, &mut output); // Metering capacity: dropped-entry counters, so a refused (i.e. never // billed) usage record is observable without reading server logs. diff --git a/nodedb/src/control/server/http/server.rs b/nodedb/src/control/server/http/server.rs index fcabd76cc..7e8981a2e 100644 --- a/nodedb/src/control/server/http/server.rs +++ b/nodedb/src/control/server/http/server.rs @@ -3,11 +3,14 @@ //! HTTP API server using axum + axum-server (for TLS). //! //! Probe routes (unversioned, always reachable): -//! - GET /healthz — k8s readiness/liveness (always reachable; 503 until GatewayEnable) +//! - GET /healthz — k8s readiness (always reachable; 503 until the `Serving` phase) //! - GET /health/live — unconditional liveness probe -//! - GET /health/ready — readiness (WAL recovered) +//! - GET /health/ready — readiness (WAL recovered and the `Serving` phase reached) //! - POST /health/drain — trigger graceful drain -//! - GET /metrics — Prometheus-format metrics (requires monitor role) +//! - GET /metrics — Prometheus-format metrics (requires monitor role; always reachable) +//! +//! Every other route returns 503 until the `Serving` startup phase (see +//! [`super::startup_gate`]). //! //! All other routes are versioned under `/v1/`. //! @@ -40,9 +43,8 @@ use std::sync::Arc; use axum::Router; -use axum::extract::{DefaultBodyLimit, State}; -use axum::middleware::{self, Next}; -use axum::response::Response; +use axum::extract::DefaultBodyLimit; +use axum::middleware; use axum::routing::{get, post, put}; use tracing::info; @@ -64,7 +66,7 @@ use super::routes; /// SSE and WebSocket routes are kept on a separate sub-router that does NOT /// carry the `map_response` layer — those handlers set their own /// `Content-Type` (text/event-stream, or the WS upgrade response). -fn build_router(state: AppState) -> Router { +pub(super) fn build_router(state: AppState) -> Router { // ── Streaming / non-JSON routes (no Content-Type stamp) ────────────────── let streaming_routes = Router::new() // WebSocket RPC — upgrade response, not JSON. @@ -172,50 +174,11 @@ fn build_router(state: AppState) -> Router { .merge(streaming_routes) .layer(middleware::from_fn_with_state( state.clone(), - startup_gate_middleware, + super::startup_gate::startup_gate_middleware, )) .with_state(state) } -/// Axum middleware that gates non-health routes on [`StartupPhase::GatewayEnable`]. -/// -/// All `/health*` paths (liveness, readiness, drain) are always let through so -/// k8s probes can observe startup progress. All other routes receive a -/// `503 Service Unavailable` until the node reaches `GatewayEnable`. -async fn startup_gate_middleware( - State(app_state): State, - req: axum::http::Request, - next: Next, -) -> Response { - use axum::http::StatusCode; - use axum::response::IntoResponse; - - let path = req.uri().path(); - // Health-probe paths bypass the gate — these must be reachable during startup. - let is_health_path = path == "/healthz" || path.starts_with("/health/"); - - if !is_health_path { - let gate = &app_state.shared.startup; - let snap = gate.current_phase(); - if let Some(err) = gate.is_failed() { - let body = serde_json::json!({ - "status": "failed", - "error": err.to_string(), - }); - return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)).into_response(); - } - if snap < crate::control::startup::StartupPhase::GatewayEnable { - let body = serde_json::json!({ - "status": "starting", - "phase": snap.name(), - }); - return (StatusCode::SERVICE_UNAVAILABLE, axum::Json(body)).into_response(); - } - } - - next.run(req).await -} - /// Start the HTTP API server from an already-bound [`tokio::net::TcpListener`]. /// /// Alias of [`run`], kept for call sites (tests, harnesses) that bind an diff --git a/nodedb/src/control/server/http/startup_gate.rs b/nodedb/src/control/server/http/startup_gate.rs new file mode 100644 index 000000000..c8fc31382 --- /dev/null +++ b/nodedb/src/control/server/http/startup_gate.rs @@ -0,0 +1,184 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The HTTP gate that holds every non-probe route until the node serves. +//! +//! The HTTP listener serves from early in boot so orchestrator probes can +//! watch startup. Until the startup gate reports [`HealthState::Ok`], only +//! these routes answer: +//! +//! - `/healthz`, which reports `starting` with `503`; +//! - `/health/*` (liveness, readiness, drain); +//! - `/metrics`. +//! +//! Every other route, an unmatched path included, returns `503` with a +//! `starting` body. The one readiness signal is the startup phase, read +//! through [`observe`]: the node is ready at [`StartupPhase::Serving`], which +//! boot enters where it opens the client protocols. +//! +//! [`StartupPhase::Serving`]: crate::control::startup::StartupPhase::Serving + +use axum::extract::State; +use axum::middleware::Next; +use axum::response::{IntoResponse, Response}; +use serde_json::json; + +use crate::control::startup::health::{HealthState, observe, to_http_response}; + +use super::auth::AppState; + +/// The error text of a route refused while the node boots. +pub const NODE_STARTING: &str = "node is starting"; + +/// Whether `path` is a probe or metrics route, which answers during boot. +fn answers_during_boot(path: &str) -> bool { + path == "/healthz" || path.starts_with("/health/") || path == "/metrics" +} + +/// Refuse every non-probe route until the node reaches the final phase. +pub(super) async fn startup_gate_middleware( + State(app_state): State, + req: axum::http::Request, + next: Next, +) -> Response { + if answers_during_boot(req.uri().path()) { + return next.run(req).await; + } + let health = observe(&app_state.shared.startup); + if matches!(health, HealthState::Ok) { + return next.run(req).await; + } + let (status, mut body) = to_http_response(&health); + if matches!(health, HealthState::Starting { .. }) { + body["error"] = json!(NODE_STARTING); + } + (status, axum::Json(body)).into_response() +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use axum::body::Body; + use axum::http::{Method, Request, StatusCode}; + use tower::ServiceExt; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::config::auth::AuthMode; + use crate::control::startup::{ReadyGate, StartupPhase, StartupSequencer}; + use crate::control::state::SharedState; + use crate::wal::WalManager; + + /// A router over a state whose boot reached `GatewayEnable` but not + /// `Serving`, as between `await_cluster_ready` and the opening of the + /// client protocols. Firing the returned gate enters `Serving`. + fn booting_router(dir: &tempfile::TempDir) -> (axum::Router, StartupSequencer, ReadyGate) { + let wal = + Arc::new(WalManager::open_for_testing(&dir.path().join("gate.wal")).expect("open WAL")); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let mut shared = SharedState::new(dispatcher, wal).expect("shared state"); + let (sequencer, gate) = StartupSequencer::new(); + let serving = sequencer.register_gate(StartupPhase::Serving, "test-serving"); + sequencer + .register_gate(StartupPhase::GatewayEnable, "test-gateway") + .fire(); + assert_eq!(gate.current_phase(), StartupPhase::GatewayEnable); + Arc::get_mut(&mut shared) + .expect("state is uniquely owned here") + .startup = Arc::clone(&gate); + let state = AppState { + shutdown_bus: crate::control::shutdown::ShutdownBus::new(Arc::clone(&shared.shutdown)) + .0, + query_ctx: Arc::new(crate::control::planner::context::QueryContext::for_state( + &shared, + )), + shared, + auth_mode: AuthMode::Trust, + }; + let router = super::super::server::build_router(state).layer(axum::Extension( + crate::control::security::tls_policy::TransportSecurity::Cleartext, + )); + (router, sequencer, serving) + } + + async fn get(router: &axum::Router, path: &str) -> (StatusCode, String) { + let req = Request::builder() + .method(Method::GET) + .uri(path) + .body(Body::empty()) + .expect("request"); + let response = router.clone().oneshot(req).await.expect("response"); + let status = response.status(); + let bytes = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .expect("read body"); + (status, String::from_utf8_lossy(&bytes).into_owned()) + } + + #[tokio::test] + async fn healthz_reports_starting_until_serving() { + let dir = tempfile::tempdir().expect("tempdir"); + let (router, _sequencer, serving) = booting_router(&dir); + + let (status, body) = get(&router, "/healthz").await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE); + let body: serde_json::Value = sonic_rs::from_str(&body).expect("healthz body is JSON"); + assert_eq!(body["status"], "starting"); + + let (status, _) = get(&router, "/health/live").await; + assert_eq!(status, StatusCode::OK); + + serving.fire(); + let (status, body) = get(&router, "/healthz").await; + assert_eq!(status, StatusCode::OK, "healthz after admission: {body}"); + let body: serde_json::Value = sonic_rs::from_str(&body).expect("healthz body is JSON"); + assert_eq!(body["status"], "ok"); + } + + #[tokio::test] + async fn a_non_probe_route_is_refused_until_serving() { + let dir = tempfile::tempdir().expect("tempdir"); + let (router, _sequencer, serving) = booting_router(&dir); + + let (status, body) = get(&router, "/v1/status").await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE); + let parsed: serde_json::Value = sonic_rs::from_str(&body).expect("refusal body is JSON"); + assert_eq!(parsed["status"], "starting"); + assert_eq!(parsed["error"], NODE_STARTING); + + // Metrics answer during boot. + let (_, body) = get(&router, "/metrics").await; + assert!(!body.contains(NODE_STARTING)); + + serving.fire(); + let (_, body) = get(&router, "/v1/status").await; + assert!(!body.contains(NODE_STARTING)); + } + + /// The gate layer also wraps the fallback: an unmatched path returns 503 + /// during boot and 404 once the node serves. + #[tokio::test] + async fn an_unmatched_path_is_refused_until_serving_then_not_found() { + let dir = tempfile::tempdir().expect("tempdir"); + let (router, _sequencer, serving) = booting_router(&dir); + + let (status, body) = get(&router, "/health").await; + assert_eq!(status, StatusCode::SERVICE_UNAVAILABLE); + assert!(body.contains(NODE_STARTING)); + + serving.fire(); + let (status, _) = get(&router, "/health").await; + assert_eq!(status, StatusCode::NOT_FOUND); + } + + #[test] + fn probe_and_metrics_paths_answer_during_boot() { + assert!(answers_during_boot("/healthz")); + assert!(answers_during_boot("/health/live")); + assert!(answers_during_boot("/health/ready")); + assert!(answers_during_boot("/metrics")); + assert!(!answers_during_boot("/v1/query")); + assert!(!answers_during_boot("/healthzz")); + assert!(!answers_during_boot("/metrics/extra")); + } +} diff --git a/nodedb/src/control/server/ilp_listener.rs b/nodedb/src/control/server/ilp_listener.rs index 32e0852cd..79eb1ad59 100644 --- a/nodedb/src/control/server/ilp_listener.rs +++ b/nodedb/src/control/server/ilp_listener.rs @@ -51,7 +51,11 @@ pub struct IlpListener { impl IlpListener { /// Bind to the given address. pub async fn bind(addr: SocketAddr) -> crate::Result { - let tcp = TcpListener::bind(addr).await.map_err(crate::Error::Io)?; + Self::from_listener(TcpListener::bind(addr).await.map_err(crate::Error::Io)?) + } + + /// Serve on a socket that already listens. + pub fn from_listener(tcp: TcpListener) -> crate::Result { let local_addr = tcp.local_addr().map_err(crate::Error::Io)?; info!(%local_addr, "ILP TCP listener bound"); Ok(Self { diff --git a/nodedb/src/control/server/listener.rs b/nodedb/src/control/server/listener.rs index c3d87df20..5bc131d9b 100644 --- a/nodedb/src/control/server/listener.rs +++ b/nodedb/src/control/server/listener.rs @@ -48,7 +48,11 @@ pub struct ListenerRunParams { impl Listener { /// Bind to the given address. pub async fn bind(addr: SocketAddr) -> crate::Result { - let tcp = TcpListener::bind(addr).await?; + Self::from_listener(TcpListener::bind(addr).await?) + } + + /// Serve on a socket that already listens. + pub fn from_listener(tcp: TcpListener) -> crate::Result { let local_addr = tcp.local_addr()?; info!(%local_addr, "control plane listener bound"); Ok(Self { diff --git a/nodedb/src/control/server/mod.rs b/nodedb/src/control/server/mod.rs index b2653b9db..1e49f7491 100644 --- a/nodedb/src/control/server/mod.rs +++ b/nodedb/src/control/server/mod.rs @@ -16,6 +16,7 @@ pub mod payload_merge; pub mod pgwire; pub mod post_aggregate; pub mod reservation; +pub mod reserved_socket; pub mod resp; pub mod response_shape; pub mod response_translate; diff --git a/nodedb/src/control/server/pgwire/listener.rs b/nodedb/src/control/server/pgwire/listener.rs index 0228c684f..5dc8393e7 100644 --- a/nodedb/src/control/server/pgwire/listener.rs +++ b/nodedb/src/control/server/pgwire/listener.rs @@ -39,7 +39,11 @@ fn forced_drain_cleanup_ids( impl PgListener { pub async fn bind(addr: SocketAddr) -> crate::Result { - let tcp = TcpListener::bind(addr).await?; + Self::from_listener(TcpListener::bind(addr).await?) + } + + /// Serve on a socket that already listens. + pub fn from_listener(tcp: TcpListener) -> crate::Result { let local_addr = tcp.local_addr()?; info!(%local_addr, "pgwire listener bound"); Ok(Self { diff --git a/nodedb/src/control/server/reserved_socket.rs b/nodedb/src/control/server/reserved_socket.rs new file mode 100644 index 000000000..e11d061bb --- /dev/null +++ b/nodedb/src/control/server/reserved_socket.rs @@ -0,0 +1,112 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A TCP address bound early and opened for connections late. +//! +//! Boot binds every protocol address before it waits for the node to become +//! ready, so a port conflict fails boot while nothing is exposed. A bound +//! socket that is not listening refuses each connection attempt at once. A +//! listening socket completes the TCP handshake in the kernel before any +//! accept runs, so a client would wait out the whole boot. Boot therefore +//! calls [`ReservedSocket::listen`] only once the node can serve. + +use std::net::SocketAddr; + +use tokio::net::{TcpListener, TcpSocket}; + +/// Pending-connection queue length, the same value +/// `tokio::net::TcpListener::bind` uses. +const LISTEN_BACKLOG: u32 = 1024; + +/// A TCP socket bound to its address but not yet listening. +#[derive(Debug)] +pub struct ReservedSocket { + socket: TcpSocket, + addr: SocketAddr, +} + +impl ReservedSocket { + /// Bind `addr` without listening. + /// + /// The bind fails when another socket listens on `addr`. `SO_REUSEADDR` + /// is set on Unix, as `TcpListener::bind` does, so a restart can rebind + /// while old connections sit in `TIME_WAIT`. + pub fn bind(addr: SocketAddr) -> crate::Result { + let bind_error = |e: std::io::Error| crate::Error::Config { + detail: format!("bind {addr}: {e}"), + }; + let socket = if addr.is_ipv4() { + TcpSocket::new_v4() + } else { + TcpSocket::new_v6() + } + .map_err(bind_error)?; + #[cfg(not(windows))] + socket.set_reuseaddr(true).map_err(bind_error)?; + socket.bind(addr).map_err(bind_error)?; + let addr = socket.local_addr().map_err(bind_error)?; + Ok(Self { socket, addr }) + } + + /// The bound address. For port `0` it holds the port the OS assigned. + pub fn local_addr(&self) -> SocketAddr { + self.addr + } + + /// Start listening. Connection attempts are accepted from here on. + /// + /// Fails when another socket started listening on the same address after + /// this one was bound. + pub fn listen(self) -> crate::Result { + let addr = self.addr; + self.socket + .listen(LISTEN_BACKLOG) + .map_err(|e| crate::Error::Config { + detail: format!("listen on {addr}: {e}"), + }) + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::*; + + fn loopback_any_port() -> SocketAddr { + SocketAddr::from(([127, 0, 0, 1], 0)) + } + + /// A reserved socket refuses connections at once, so a client connecting + /// during boot never waits in the kernel's accept queue. + #[tokio::test] + async fn a_reserved_socket_refuses_connections_until_it_listens() { + let reserved = ReservedSocket::bind(loopback_any_port()).expect("bind"); + let addr = reserved.local_addr(); + assert_ne!(addr.port(), 0); + + let refused = + tokio::time::timeout(Duration::from_secs(5), tokio::net::TcpStream::connect(addr)) + .await + .expect("a refusal is immediate"); + assert!( + refused.is_err(), + "a reserved socket must refuse connections" + ); + + let listener = reserved.listen().expect("listen"); + let (connected, accepted) = + tokio::join!(tokio::net::TcpStream::connect(addr), listener.accept()); + connected.expect("connect after listen"); + accepted.expect("accept after listen"); + } + + /// An address another socket listens on cannot be reserved. + #[tokio::test] + async fn an_address_in_use_cannot_be_reserved() { + let occupied = TcpListener::bind(loopback_any_port()) + .await + .expect("occupy a port"); + let addr = occupied.local_addr().expect("occupied addr"); + assert!(ReservedSocket::bind(addr).is_err()); + } +} diff --git a/nodedb/src/control/server/resp/listener.rs b/nodedb/src/control/server/resp/listener.rs index 9bc3327f4..502681f94 100644 --- a/nodedb/src/control/server/resp/listener.rs +++ b/nodedb/src/control/server/resp/listener.rs @@ -39,6 +39,11 @@ impl RespListener { .map_err(|e| crate::Error::Config { detail: format!("failed to bind RESP listener on {addr}: {e}"), })?; + Self::from_listener(tcp) + } + + /// Serve on a socket that already listens. + pub fn from_listener(tcp: TcpListener) -> crate::Result { let local_addr = tcp.local_addr().map_err(|e| crate::Error::Config { detail: format!("failed to get RESP local address: {e}"), })?; diff --git a/nodedb/src/control/server/sync/listener.rs b/nodedb/src/control/server/sync/listener.rs index e5c1a5c14..bbae3e044 100644 --- a/nodedb/src/control/server/sync/listener.rs +++ b/nodedb/src/control/server/sync/listener.rs @@ -14,6 +14,7 @@ use tokio::net::TcpListener; use tokio::task::{JoinHandle, JoinSet}; use tracing::{info, warn}; +use crate::control::server::reserved_socket::ReservedSocket; use crate::control::server::shared::{ConnectionFutureOutcome, isolate_connection_future}; use crate::control::shutdown::{ShutdownBus, ShutdownPhase}; use crate::control::state::SharedState; @@ -250,13 +251,13 @@ async fn run_accepted_session( isolate_connection_future(future).await } -/// Bind the sync WebSocket listener socket. +/// Bind the sync listener address without listening. /// -/// Separate from [`serve_sync_listener`] so boot can bind every protocol -/// socket up front — before any accept loop is spawned and before the node -/// joins the cluster — and fail loudly on a port conflict while nothing is -/// yet exposed. See `bootstrap::listeners::bind_listeners`. -pub async fn bind_sync_listener(addr: SocketAddr) -> crate::Result { +/// Boot reserves every protocol socket up front, before the node joins the +/// cluster, and fails loudly on a port conflict while nothing is exposed. +/// It listens only once the node can serve. See +/// `bootstrap::listeners::bind_listeners`. +pub fn reserve_sync_listener(addr: SocketAddr) -> crate::Result { // Plaintext `ws://` sync must terminate TLS at a loopback proxy: reject any // public bind here so both the fail-fast boot path (`bind_listeners`) and // the convenience `start_sync_listener` path are covered by one guard. @@ -277,16 +278,17 @@ pub async fn bind_sync_listener(addr: SocketAddr) -> crate::Result ), }); } - TcpListener::bind(&addr) - .await - .map_err(|e| crate::Error::Config { - detail: format!("bind sync listener to {addr}: {e}"), - }) + ReservedSocket::bind(addr) +} + +/// Bind the sync listener address and listen at once. +pub async fn bind_sync_listener(addr: SocketAddr) -> crate::Result { + reserve_sync_listener(addr)?.listen() } /// Start the sync WebSocket listener with full security context. /// -/// Binds and serves in one step. Boot uses [`bind_sync_listener`] + +/// Binds and serves in one step. Boot uses [`reserve_sync_listener`] + /// [`serve_sync_listener`] instead so the bind is fail-fast; this is the /// convenience path for callers that own the whole lifecycle (tests, tools) /// and can provide that lifecycle's canonical shutdown bus. diff --git a/nodedb/src/control/startup/gate.rs b/nodedb/src/control/startup/gate.rs index 757ed2660..d66f95913 100644 --- a/nodedb/src/control/startup/gate.rs +++ b/nodedb/src/control/startup/gate.rs @@ -74,7 +74,8 @@ impl StartupGate { Self { rx } } - /// Create a gate that is pre-fired at [`StartupPhase::GatewayEnable`]. + /// Create a gate that is pre-fired at [`StartupPhase::Serving`], the + /// final phase. /// /// Used by test helpers that construct a [`SharedState`] without a real /// [`StartupSequencer`]. Any call to [`await_phase`] on this gate returns @@ -83,14 +84,14 @@ impl StartupGate { /// [`await_phase`]: StartupGate::await_phase pub fn pre_fired() -> Arc { let (tx, rx) = watch::channel(SequencerSnapshot { - phase: StartupPhase::GatewayEnable, + phase: StartupPhase::Serving, failed: None, }); // Keep the sender alive inside the gate so the receiver never sees // the channel as closed and returns `AlreadyTerminated`. let gate = Arc::new(Self { rx }); // The sender is dropped intentionally: no further phase changes will - // occur. The already-received value (GatewayEnable) is what all + // occur. The already-received value (Serving) is what all // `await_phase` callers will see. drop(tx); gate diff --git a/nodedb/src/control/startup/health.rs b/nodedb/src/control/startup/health.rs index a753ae242..8bb3ffdae 100644 --- a/nodedb/src/control/startup/health.rs +++ b/nodedb/src/control/startup/health.rs @@ -21,7 +21,8 @@ use super::phase::StartupPhase; pub enum HealthState { /// Still advancing through startup phases. Starting { phase: StartupPhase }, - /// Node has reached [`StartupPhase::GatewayEnable`] and is serving. + /// Node has reached [`StartupPhase::Serving`], the final phase: its + /// client protocols listen. Ok, /// Startup failed; includes the original error. Failed { error: Arc }, @@ -33,7 +34,7 @@ pub fn observe(gate: &StartupGate) -> HealthState { return HealthState::Failed { error: err }; } let phase = gate.current_phase(); - if phase >= StartupPhase::GatewayEnable { + if phase >= StartupPhase::Serving { HealthState::Ok } else { HealthState::Starting { phase } @@ -55,7 +56,7 @@ pub fn to_http_response(state: &HealthState) -> (axum::http::StatusCode, serde_j StatusCode::OK, serde_json::json!({ "status": "ok", - "phase": StartupPhase::GatewayEnable.name(), + "phase": StartupPhase::Serving.name(), }), ), HealthState::Starting { phase } => ( @@ -112,7 +113,7 @@ mod tests { use crate::control::startup::StartupSequencer; #[test] - fn observe_starting_before_gateway_enable() { + fn observe_starting_before_serving() { // A pre-fired gate (used by test helpers) reports Ok immediately. let gate = StartupGate::pre_fired(); let state = observe(&gate); @@ -161,4 +162,25 @@ mod tests { assert_eq!(NativeStatus::Starting.to_string(), "Starting"); assert_eq!(NativeStatus::Failed.to_string(), "Failed"); } + + /// Boot reaches `GatewayEnable` before it listens on the client + /// protocols. The node reports starting until the final phase. + #[test] + fn observe_starting_until_serving() { + let (seq, gate) = StartupSequencer::new(); + let gateway = seq.register_gate(StartupPhase::GatewayEnable, "gateway"); + let serving = seq.register_gate(StartupPhase::Serving, "serving"); + gateway.fire(); + assert_eq!(gate.current_phase(), StartupPhase::GatewayEnable); + assert!(matches!( + observe(&gate), + HealthState::Starting { + phase: StartupPhase::GatewayEnable + } + )); + + serving.fire(); + assert_eq!(gate.current_phase(), StartupPhase::Serving); + assert!(matches!(observe(&gate), HealthState::Ok)); + } } diff --git a/nodedb/src/control/startup/phase.rs b/nodedb/src/control/startup/phase.rs index 6570b5579..321ee5018 100644 --- a/nodedb/src/control/startup/phase.rs +++ b/nodedb/src/control/startup/phase.rs @@ -15,7 +15,7 @@ use std::fmt; /// Total number of phases. Kept in sync with the enum below by /// the `phase_order_matches_u8` unit test. -pub const PHASE_COUNT: usize = 12; +pub const PHASE_COUNT: usize = 13; /// Startup phase. Ordered — use `Ord` / `PartialOrd` to compare. #[derive(Copy, Clone, Debug, Eq, PartialEq, Ord, PartialOrd, Hash)] @@ -52,9 +52,13 @@ pub enum StartupPhase { WarmPeers = 8, /// Health monitor running. HealthLoopStart = 9, - /// Listeners may now process accepted requests. - /// `StartupGate::await_phase(GatewayEnable)` resolves. + /// Cluster readiness completed: the node holds its first authorization + /// lease and its gateway is installed. Listener accept loops may process + /// requests once they exist. The client protocols do not listen yet. GatewayEnable = 10, + /// The client protocols listen and every HTTP route answers. The final + /// phase: the node is ready, and `/healthz` reports `ok`. + Serving = 11, /// Terminal state — entered via [`StartupSequencer::fail`] or /// when a [`ReadyGate`] is dropped without firing. All /// [`StartupGate::await_phase`] waiters wake with an error. @@ -62,7 +66,7 @@ pub enum StartupPhase { /// [`StartupSequencer::fail`]: super::startup_sequencer::StartupSequencer::fail /// [`ReadyGate`]: super::gate::ReadyGate /// [`StartupGate::await_phase`]: super::gate::StartupGate::await_phase - Failed = 11, + Failed = 12, } impl StartupPhase { @@ -81,6 +85,7 @@ impl StartupPhase { Self::WarmPeers => "warm_peers", Self::HealthLoopStart => "health_loop_start", Self::GatewayEnable => "gateway_enable", + Self::Serving => "serving", Self::Failed => "failed", } } @@ -98,7 +103,8 @@ impl StartupPhase { Self::TransportBind => Some(Self::WarmPeers), Self::WarmPeers => Some(Self::HealthLoopStart), Self::HealthLoopStart => Some(Self::GatewayEnable), - Self::GatewayEnable => None, + Self::GatewayEnable => Some(Self::Serving), + Self::Serving => None, Self::Failed => None, } } @@ -118,7 +124,8 @@ impl StartupPhase { 8 => Some(Self::WarmPeers), 9 => Some(Self::HealthLoopStart), 10 => Some(Self::GatewayEnable), - 11 => Some(Self::Failed), + 11 => Some(Self::Serving), + 12 => Some(Self::Failed), _ => None, } } @@ -155,6 +162,7 @@ mod tests { StartupPhase::WarmPeers, StartupPhase::HealthLoopStart, StartupPhase::GatewayEnable, + StartupPhase::Serving, StartupPhase::Failed, ]; assert_eq!(ordered.len(), PHASE_COUNT); @@ -169,7 +177,7 @@ mod tests { } #[test] - fn next_chain_terminates_at_gateway() { + fn next_chain_terminates_at_serving() { let mut cur = StartupPhase::Boot; let mut count = 1; while let Some(n) = cur.next() { @@ -179,7 +187,7 @@ mod tests { panic!("phase chain failed to terminate"); } } - assert_eq!(cur, StartupPhase::GatewayEnable); + assert_eq!(cur, StartupPhase::Serving); } #[test] diff --git a/nodedb/src/control/startup/startup_sequencer.rs b/nodedb/src/control/startup/startup_sequencer.rs index 971db12af..e358aa01b 100644 --- a/nodedb/src/control/startup/startup_sequencer.rs +++ b/nodedb/src/control/startup/startup_sequencer.rs @@ -147,7 +147,7 @@ impl SequencerState { if self.failed.is_some() { return; } - if self.current == StartupPhase::GatewayEnable { + if self.current == StartupPhase::Serving { return; } let Some(next) = self.current.next() else { @@ -330,7 +330,7 @@ mod tests { /// that `current_phase()` advances in lock-step. /// /// Without the sentinel gate the sequencer would advance all the way to - /// `GatewayEnable` after the last registered gate fires, because no + /// `Serving` after the last registered gate fires, because no /// pending gates block the remaining phases. The sentinel makes the /// stopping point explicit and deterministic. #[tokio::test] @@ -502,7 +502,7 @@ mod tests { // ── 6. Matchstick: StartupPhase::next() is exhaustive ─────────────────── /// Every non-terminal phase must return `Some(_)` from `next()`, and - /// the chain must terminate exactly at `GatewayEnable`. If a new + /// the chain must terminate exactly at `Serving`. If a new /// variant is added without a branch in `next()`, the compiler rejects /// the match — catching the omission at compile time. #[test] @@ -523,8 +523,8 @@ mod tests { } assert_eq!( cur, - StartupPhase::GatewayEnable, - "chain must terminate at GatewayEnable" + StartupPhase::Serving, + "chain must terminate at Serving" ); // Exhaustive match — compile error if a variant is added without @@ -541,6 +541,7 @@ mod tests { StartupPhase::WarmPeers => StartupPhase::WarmPeers.next(), StartupPhase::HealthLoopStart => StartupPhase::HealthLoopStart.next(), StartupPhase::GatewayEnable => StartupPhase::GatewayEnable.next(), + StartupPhase::Serving => StartupPhase::Serving.next(), StartupPhase::Failed => StartupPhase::Failed.next(), }; } diff --git a/nodedb/src/ctl/healthcheck.rs b/nodedb/src/ctl/healthcheck.rs index 773c5ae83..b3d7e544a 100644 --- a/nodedb/src/ctl/healthcheck.rs +++ b/nodedb/src/ctl/healthcheck.rs @@ -8,7 +8,7 @@ //! exits 0 on a 2xx response, non-zero otherwise. //! //! `/healthz` is the readiness probe: it reports 503 while the node is -//! draining or has not reached `GatewayEnable`, so a container runtime stops +//! draining or has not yet admitted clients, so a container runtime stops //! routing to a node that cannot serve. //! //! Kept dependency-free (`std::net` only) so it's cheap to invoke from diff --git a/nodedb/src/main.rs b/nodedb/src/main.rs index eefc2c946..bca0a3268 100644 --- a/nodedb/src/main.rs +++ b/nodedb/src/main.rs @@ -163,6 +163,7 @@ async fn server_main() -> anyhow::Result<()> { warm_peers_gate, health_loop_gate, gateway_enable_gate, + serving_gate, } = gates::register_startup_gates(&startup_seq); let data_plane::DataPlaneBootstrap { @@ -237,12 +238,7 @@ async fn server_main() -> anyhow::Result<()> { let listeners::ListenerSetup { conn_semaphore, admission_registry, - listener, - pg_listener, - http_listener, - sync_listener, - ilp_listener, - resp_listener, + bound, base_acceptor, native_tls_enabled, } = listeners::setup( @@ -254,6 +250,22 @@ async fn server_main() -> anyhow::Result<()> { ) .await?; + let nodedb::bootstrap::listeners::BoundListeners { + http: http_listener, + clients: client_sockets, + } = bound; + + // Serve HTTP from here on, so orchestrator probes can watch the rest of + // boot. Only `/healthz`, `/health/*` and `/metrics` answer until the + // `Serving` phase below. Every other route returns 503, so no request can + // bypass a quota before `replay_quotas` runs. + nodedb::bootstrap::listeners::spawn_http_listener( + http_listener, + Arc::clone(&shared), + &config, + shutdown_bus.clone(), + ); + // Per-protocol TLS: returns the acceptor only if the protocol flag is true. let tls_for = |enabled: bool| -> Option { if enabled { base_acceptor.clone() } else { None } @@ -281,11 +293,22 @@ async fn server_main() -> anyhow::Result<()> { // listener has accepted a connection that could bypass a cap. nodedb::bootstrap::quota_replay::replay_quotas(&shared); + // Listen on the client protocols only now, and enter `Serving`. Until here + // every client socket was bound but not listening, so a client + // connecting during boot was refused at once instead of waiting in the + // kernel's accept queue for the node to become ready. + let nodedb::bootstrap::listeners::OpenListeners { + native: listener, + pgwire: pg_listener, + sync: sync_listener, + ilp: ilp_listener, + resp: resp_listener, + } = nodedb::bootstrap::listeners::open_listeners(client_sockets, serving_gate)?; + // Spawn all non-native protocol listeners. nodedb::bootstrap::listeners::spawn_protocol_listeners( nodedb::bootstrap::listeners::ProtocolListeners { pg_listener, - http_listener, sync_listener, ilp_listener, resp_listener, diff --git a/nodedb/src/main_boot/gates.rs b/nodedb/src/main_boot/gates.rs index 8f2be9810..c5820cb86 100644 --- a/nodedb/src/main_boot/gates.rs +++ b/nodedb/src/main_boot/gates.rs @@ -19,6 +19,8 @@ pub(crate) struct StartupGates { pub(crate) warm_peers_gate: ReadyGate, pub(crate) health_loop_gate: ReadyGate, pub(crate) gateway_enable_gate: ReadyGate, + /// Fired where the client protocols start listening. + pub(crate) serving_gate: ReadyGate, } /// Register all gates up-front so the sequencer knows every phase has @@ -40,6 +42,7 @@ pub(crate) fn register_startup_gates(startup_seq: &StartupSequencer) -> StartupG let health_loop_gate = startup_seq.register_gate(StartupPhase::HealthLoopStart, "health-loop"); let gateway_enable_gate = startup_seq.register_gate(StartupPhase::GatewayEnable, "gateway-enable"); + let serving_gate = startup_seq.register_gate(StartupPhase::Serving, "client-listeners"); StartupGates { wal_gate, @@ -52,5 +55,6 @@ pub(crate) fn register_startup_gates(startup_seq: &StartupSequencer) -> StartupG warm_peers_gate, health_loop_gate, gateway_enable_gate, + serving_gate, } } diff --git a/nodedb/src/main_boot/listeners.rs b/nodedb/src/main_boot/listeners.rs index e80ec28a4..425b4863c 100644 --- a/nodedb/src/main_boot/listeners.rs +++ b/nodedb/src/main_boot/listeners.rs @@ -10,10 +10,6 @@ use nodedb::ServerConfig; use nodedb::bootstrap; use nodedb::bootstrap::tls::build_tls_acceptor; use nodedb::control::cluster::ClusterHandle; -use nodedb::control::server::ilp_listener::IlpListener; -use nodedb::control::server::listener::Listener; -use nodedb::control::server::pgwire::listener::PgListener; -use nodedb::control::server::resp::RespListener; use nodedb::control::shutdown::ShutdownBus; use nodedb::control::state::SharedState; @@ -23,12 +19,9 @@ use nodedb::control::state::SharedState; pub(crate) struct ListenerSetup { pub(crate) conn_semaphore: Arc, pub(crate) admission_registry: Arc, - pub(crate) listener: Listener, - pub(crate) pg_listener: PgListener, - pub(crate) http_listener: tokio::net::TcpListener, - pub(crate) sync_listener: tokio::net::TcpListener, - pub(crate) ilp_listener: Option, - pub(crate) resp_listener: Option, + /// Every protocol socket. HTTP listens at once. The client protocols + /// listen only once the node is ready. + pub(crate) bound: bootstrap::listeners::BoundListeners, pub(crate) base_acceptor: Option, pub(crate) native_tls_enabled: bool, } @@ -57,15 +50,9 @@ pub(crate) async fn setup( // Bind all listeners — every protocol, including HTTP and sync — before // starting any accept loop, so a port conflict fails boot here rather - // than after the node is already serving other protocols. - let bootstrap::listeners::BoundListeners { - native: listener, - pgwire: pg_listener, - http: http_listener, - sync: sync_listener, - ilp: ilp_listener, - resp: resp_listener, - } = bootstrap::listeners::bind_listeners(config).await?; + // than after the node is already serving other protocols. Only HTTP + // listens now. The client protocols listen once the node is ready. + let bound = bootstrap::listeners::bind_listeners(config)?; // Startup banner (and trust-mode warning if applicable). bootstrap::credentials::print_startup_banner(config, cluster_mode_str); @@ -106,12 +93,7 @@ pub(crate) async fn setup( Ok(ListenerSetup { conn_semaphore, admission_registry, - listener, - pg_listener, - http_listener, - sync_listener, - ilp_listener, - resp_listener, + bound, base_acceptor, native_tls_enabled, }) diff --git a/nodedb/tests/crash_harness/pgwire.rs b/nodedb/tests/crash_harness/pgwire.rs index d7bc08621..271a90a03 100644 --- a/nodedb/tests/crash_harness/pgwire.rs +++ b/nodedb/tests/crash_harness/pgwire.rs @@ -152,7 +152,7 @@ impl CrashHarness { /// reports ready. /// /// `/healthz` is a one-shot boot-phase latch (`control/startup/health.rs`) - /// that flips to OK at `GatewayEnable`; the Calvin sequencer is + /// that flips to OK at `Serving`; the Calvin sequencer is /// deliberately not a data group in that readiness gate, so a write can /// still hit `calvin-submit: no sequencer leader elected yet; cannot /// submit cross-shard transaction` after `/healthz` is already green. diff --git a/nodedb/tests/inproc/cases/http_health.rs b/nodedb/tests/inproc/cases/http_health.rs index dfbee9ca2..7f0473aeb 100644 --- a/nodedb/tests/inproc/cases/http_health.rs +++ b/nodedb/tests/inproc/cases/http_health.rs @@ -88,9 +88,8 @@ async fn start_http(auth_mode: AuthMode) -> TestServer { .ok(); }); - // Wait for the gate (startup phase must reach GatewayEnable). - // For testing purposes the Trust-mode server starts in Trust mode which - // also fires the gate because startup is bypassed in test builds. + // The test state's pre-fired startup gate starts at `Serving`, the final + // phase, so every route answers at once. Give the server time to start. tokio::time::sleep(Duration::from_millis(40)).await; TestServer { diff --git a/nodedb/tests/inproc/cases/startup_gate_http.rs b/nodedb/tests/inproc/cases/startup_gate_http.rs index b31afead1..45cba172a 100644 --- a/nodedb/tests/inproc/cases/startup_gate_http.rs +++ b/nodedb/tests/inproc/cases/startup_gate_http.rs @@ -1,13 +1,14 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Integration test: HTTP middleware gates non-health routes on GatewayEnable. +//! Integration test: HTTP middleware gates non-probe routes on `Serving`, the +//! final startup phase. //! //! The test: -//! 1. Builds a minimal node with a real StartupSequencer (gate held). +//! 1. Builds a minimal node with a real StartupSequencer (serving gate held). //! 2. Binds and spawns the HTTP server. //! 3. Verifies that GET /healthz returns 503 with `{"status":"starting",...}`. //! 4. Verifies that POST /query returns 503 during startup. -//! 5. Fires the gate. +//! 5. Fires the serving gate. //! 6. Verifies that GET /healthz now returns 200. use std::sync::Arc; @@ -31,18 +32,18 @@ fn make_gated_state() -> ( let mut shared = SharedState::new(dispatcher, wal).unwrap(); let (seq, gate) = StartupSequencer::new(); - let gw_gate = seq.register_gate(StartupPhase::GatewayEnable, "gateway-enable-http-test"); + let serving_gate = seq.register_gate(StartupPhase::Serving, "serving-http-test"); Arc::get_mut(&mut shared) .expect("SharedState not yet cloned") .startup = Arc::clone(&gate); - (shared, seq, gw_gate, dir) + (shared, seq, serving_gate, dir) } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn http_healthz_returns_503_before_gateway_enable() { - let (shared, _seq, _gw_gate, _dir) = make_gated_state(); +async fn http_healthz_returns_503_before_serving() { + let (shared, _seq, _serving_gate, _dir) = make_gated_state(); // Bind the HTTP server on an ephemeral port. let listen: std::net::SocketAddr = "127.0.0.1:0".parse().unwrap(); @@ -55,7 +56,7 @@ async fn http_healthz_returns_503_before_gateway_enable() { let bus_http = shutdown_bus.clone(); tokio::spawn(async move { // Run the HTTP server. It binds immediately and serves /healthz from - // the start, but non-health routes get 503 until GatewayEnable. + // the start, but non-probe routes get 503 until Serving. nodedb::control::server::http::server::run_with_listener( listener, shared_http, @@ -82,7 +83,7 @@ async fn http_healthz_returns_503_before_gateway_enable() { assert_eq!( resp.status(), reqwest::StatusCode::SERVICE_UNAVAILABLE, - "/healthz should return 503 before GatewayEnable" + "/healthz should return 503 before Serving" ); let body: serde_json::Value = resp.json().await.unwrap(); assert_eq!( @@ -101,13 +102,13 @@ async fn http_healthz_returns_503_before_gateway_enable() { assert_eq!( resp.status(), reqwest::StatusCode::SERVICE_UNAVAILABLE, - "/query should return 503 before GatewayEnable" + "/query should return 503 before Serving" ); } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn http_healthz_returns_200_after_gateway_enable() { - let (shared, _seq, gw_gate, _dir) = make_gated_state(); +async fn http_healthz_returns_200_once_serving() { + let (shared, _seq, serving_gate, _dir) = make_gated_state(); let listen: std::net::SocketAddr = "127.0.0.1:0".parse().unwrap(); let listener = tokio::net::TcpListener::bind(listen).await.unwrap(); @@ -129,8 +130,8 @@ async fn http_healthz_returns_200_after_gateway_enable() { .ok(); }); - // Fire the gate, then check /healthz returns 200. - gw_gate.fire(); + // Enter Serving, then check /healthz returns 200. + serving_gate.fire(); tokio::time::sleep(Duration::from_millis(20)).await; @@ -145,7 +146,7 @@ async fn http_healthz_returns_200_after_gateway_enable() { assert_eq!( resp.status(), reqwest::StatusCode::OK, - "/healthz should return 200 after GatewayEnable" + "/healthz should return 200 once Serving" ); let body: serde_json::Value = resp.json().await.unwrap(); assert_eq!(body["status"], "ok", "body.status should be 'ok'"); @@ -155,7 +156,7 @@ async fn http_healthz_returns_200_after_gateway_enable() { async fn http_health_bare_returns_404() { // The bare /health route was removed in favour of /healthz (k8s convention). // Requests to /health must fall through to axum's default 404 handler. - let (shared, _seq, gw_gate, _dir) = make_gated_state(); + let (shared, _seq, serving_gate, _dir) = make_gated_state(); let listen: std::net::SocketAddr = "127.0.0.1:0".parse().unwrap(); let listener = tokio::net::TcpListener::bind(listen).await.unwrap(); @@ -177,8 +178,9 @@ async fn http_health_bare_returns_404() { .ok(); }); - // Fire the gate so the startup middleware doesn't interfere. - gw_gate.fire(); + // Enter Serving so the startup middleware lets unmatched paths reach the + // router's 404 fallback. Before Serving they return 503. + serving_gate.fire(); tokio::time::sleep(Duration::from_millis(20)).await; let base = format!("http://{local_addr}"); @@ -205,7 +207,7 @@ async fn http_health_bare_returns_404() { /// serving perfectly well. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn the_healthcheck_subcommand_probes_a_route_the_server_serves() { - let (shared, _seq, gw_gate, _dir) = make_gated_state(); + let (shared, _seq, serving_gate, _dir) = make_gated_state(); let listener = tokio::net::TcpListener::bind("127.0.0.1:0".parse::().unwrap()) @@ -237,10 +239,10 @@ async fn the_healthcheck_subcommand_probes_a_route_the_server_serves() { .unwrap(); assert_eq!( starting, 1, - "the probe must report unhealthy before GatewayEnable" + "the probe must report unhealthy before Serving" ); - gw_gate.fire(); + serving_gate.fire(); tokio::time::sleep(Duration::from_millis(20)).await; let ready = tokio::task::spawn_blocking(move || nodedb::ctl::healthcheck::run(port)) diff --git a/nodedb/tests/inproc/cases/startup_gate_native.rs b/nodedb/tests/inproc/cases/startup_gate_native.rs index cf2d04ff1..3ae903672 100644 --- a/nodedb/tests/inproc/cases/startup_gate_native.rs +++ b/nodedb/tests/inproc/cases/startup_gate_native.rs @@ -1,7 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Integration test: native protocol STATUS command returns "OK" after -//! GatewayEnable fires and returns "Starting" before it fires. +//! Integration test: native protocol STATUS command returns "OK" once the +//! sequencer reaches `Serving`, the final phase. //! //! The native protocol is a simple framing format: //! [4-byte big-endian payload_len][payload] @@ -35,13 +35,16 @@ fn make_gated_state() -> ( let mut shared = SharedState::new(dispatcher, wal).unwrap(); let (seq, gate) = StartupSequencer::new(); - let gw_gate = seq.register_gate(StartupPhase::GatewayEnable, "gateway-enable-native-test"); + // No gate holds the phases before Serving. Firing this gate passes + // GatewayEnable, which opens the accept loop, and enters Serving, where + // STATUS reports ready. + let serving_gate = seq.register_gate(StartupPhase::Serving, "serving-native-test"); Arc::get_mut(&mut shared) .expect("SharedState not yet cloned") .startup = Arc::clone(&gate); - (shared, seq, gw_gate, dir) + (shared, seq, serving_gate, dir) } /// Encode a JSON payload as a native protocol frame (4-byte length prefix). @@ -70,8 +73,8 @@ async fn read_json_frame(stream: &mut TcpStream) -> Vec { } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn native_status_returns_ok_after_gateway_enable() { - let (shared, _seq, gw_gate, _dir) = make_gated_state(); +async fn native_status_returns_ok_once_serving() { + let (shared, _seq, serving_gate, _dir) = make_gated_state(); let startup_gate = Arc::clone(&shared.startup); // Bind the native protocol listener on an ephemeral port. @@ -100,8 +103,8 @@ async fn native_status_returns_ok_after_gateway_enable() { .await; }); - // Fire the gate so the listener starts accepting. - gw_gate.fire(); + // Enter Serving so STATUS reports ready. + serving_gate.fire(); // Give the listener time to reach the accept loop. tokio::time::sleep(Duration::from_millis(30)).await; diff --git a/nodedb/tests/inproc/cases/startup_ordering.rs b/nodedb/tests/inproc/cases/startup_ordering.rs index 20f226c8d..3a55ad4b4 100644 --- a/nodedb/tests/inproc/cases/startup_ordering.rs +++ b/nodedb/tests/inproc/cases/startup_ordering.rs @@ -8,7 +8,8 @@ //! is determined by the `StartupPhase` passed to `register_gate`. //! - Firing a later-phase gate before an earlier-phase gate does not advance //! past the earlier phase until all earlier gates also fire. -//! - `GatewayEnable` is only reached after all prior phases complete. +//! - `Serving`, the final phase, is only reached after all prior phases +//! complete. use std::sync::Arc; use std::time::Duration; @@ -46,6 +47,7 @@ async fn phases_advance_in_order_when_gates_fire() { let peers_gate = seq.register_gate(StartupPhase::WarmPeers, "peers"); let health_gate = seq.register_gate(StartupPhase::HealthLoopStart, "health"); let gw_gate = seq.register_gate(StartupPhase::GatewayEnable, "gateway"); + let serving_gate = seq.register_gate(StartupPhase::Serving, "serving"); // Initial phase is Boot. assert_eq!(gate.current_phase(), StartupPhase::Boot); @@ -80,6 +82,14 @@ async fn phases_advance_in_order_when_gates_fire() { gw_gate.fire(); assert_phase_reaches(&gate, StartupPhase::GatewayEnable).await; + assert_eq!( + gate.current_phase(), + StartupPhase::GatewayEnable, + "the pending serving gate holds the sequencer at GatewayEnable" + ); + + serving_gate.fire(); + assert_phase_reaches(&gate, StartupPhase::Serving).await; } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] @@ -87,10 +97,10 @@ async fn later_phase_gate_fires_first_does_not_advance_past_earlier_phase() { let (seq, gate) = StartupSequencer::new(); let wal_gate = seq.register_gate(StartupPhase::WalRecovery, "wal"); - let gw_gate = seq.register_gate(StartupPhase::GatewayEnable, "gateway"); + let serving_gate = seq.register_gate(StartupPhase::Serving, "serving"); - // Fire GatewayEnable first — phase must not advance past Boot until WalRecovery fires. - gw_gate.fire(); + // Fire Serving first — phase must not advance past Boot until WalRecovery fires. + serving_gate.fire(); // Wait a bit and confirm we're still at Boot. tokio::time::sleep(Duration::from_millis(20)).await; @@ -100,10 +110,10 @@ async fn later_phase_gate_fires_first_does_not_advance_past_earlier_phase() { "phase advanced past Boot even though WalRecovery gate has not fired" ); - // Now fire WalRecovery — phase should advance all the way to GatewayEnable - // since the GatewayEnable gate already fired. + // Now fire WalRecovery — phase should advance all the way to Serving + // since the Serving gate already fired. wal_gate.fire(); - assert_phase_reaches(&gate, StartupPhase::GatewayEnable).await; + assert_phase_reaches(&gate, StartupPhase::Serving).await; } #[tokio::test(flavor = "multi_thread", worker_threads = 2)] @@ -141,6 +151,6 @@ async fn gate_fire_is_idempotent() { // Firing three times must succeed and advance the phase at least to WalRecovery. // With no later gates registered, the sequencer may advance all the way to - // GatewayEnable — that is expected and correct. + // Serving — that is expected and correct. assert_phase_reaches(&gate, StartupPhase::WalRecovery).await; } From 9f09e0a66bb49f3d187e892190e35f1b0fb67c3a Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sat, 26 Sep 2026 23:38:29 +0800 Subject: [PATCH 43/64] feat(query): fail scalar function faults instead of folding to NULL Give EvalError three new variants alongside DivisionByZero: UnknownFunction, VectorDimensionMismatch, ArgumentType, and InvalidJsonPath. A call to an unregistered name, a vector distance over mismatched dimensions or a non-vector operand, and a document function given a malformed JSONPath now surface as typed errors instead of silently evaluating to NULL. The new ErrorCode::DataException (SQLSTATE 22000) and NodeDbError::data_exception carry these through the bridge, cluster RPC codec, plan error map, and pgwire/native/HTTP error rendering, alongside the constant-fold path and every enforcement/executor site that used to hardcode DivisionByZero on any evaluator error. Add vector_distance/vector_cosine_distance/vector_neg_inner_product, doc_get/doc_exists/doc_array_contains/nav document accessors, and ndb_chunk_text as real per-row evaluators, plus make_array for non-literal ARRAY[...] elements. Rewrite sql_like_match as a tokenizing matcher shared by the scan-filter and scalar like/ilike paths, adding escape-character support. Add planner::search_scope::refuse_row_scoped_search_functions so an index-owned search function (bm25_score, text_match, rrf_score, sparse_score, ...) used outside a search plan is refused at plan time with SqlError::SearchFunctionOutsideSearch instead of resolving to a NULL score column. --- .../src/raft_loop/handle_rpc/plan_dispatch.rs | 5 +- .../src/rpc_codec/data_plane_error.rs | 9 + nodedb-crdt/src/validator/types.rs | 2 +- nodedb-query/src/expr/eval.rs | 47 +- nodedb-query/src/functions/array.rs | 4 + nodedb-query/src/functions/eval.rs | 181 +++++- nodedb-query/src/functions/json/dispatch.rs | 23 +- nodedb-query/src/functions/json/doc.rs | 271 ++++++++ nodedb-query/src/functions/json/mod.rs | 1 + nodedb-query/src/functions/json/pg_ops.rs | 2 +- nodedb-query/src/functions/mod.rs | 2 + nodedb-query/src/functions/string.rs | 46 ++ nodedb-query/src/functions/text_chunk.rs | 90 +++ nodedb-query/src/functions/vector.rs | 253 ++++++++ nodedb-query/src/scan_filter/like.rs | 153 ++++- nodedb-query/src/window/value_eval.rs | 2 +- nodedb-sql/src/error.rs | 20 + nodedb-sql/src/lib.rs | 3 + nodedb-sql/src/planner/const_fold.rs | 31 + nodedb-sql/src/planner/mod.rs | 1 + nodedb-sql/src/planner/search_scope.rs | 608 ++++++++++++++++++ .../src/planner/select/order_by/projection.rs | 4 +- nodedb-types/src/error/code.rs | 3 + nodedb-types/src/error/code_table.rs | 1 + .../src/error/ctors/read_query_auth.rs | 13 + nodedb-types/src/error/details.rs | 4 + nodedb-types/src/error/msgpack/constants.rs | 2 + .../error/msgpack/decode/from_messagepack.rs | 12 + nodedb-types/src/error/msgpack/encode.rs | 1 + nodedb-types/src/error/sqlstate.rs | 6 + nodedb-types/src/error/types.rs | 1 + nodedb/src/bridge/envelope/error_code.rs | 28 + .../control/cluster/data_plane_error_wire.rs | 4 + .../src/control/gateway/error_map/pgwire.rs | 3 + nodedb/src/control/planner/plan_error_map.rs | 9 +- .../sql_plan_convert/expr/bridge_expr.rs | 108 +++- .../server/dispatch_utils/write_abort.rs | 2 + .../server/http/routes/result_shape.rs | 1 + .../control/server/native/sqlstate_code.rs | 1 + .../control/server/pgwire/types/error_map.rs | 5 + .../src/control/server/shared/ddl/result.rs | 1 + .../src/control/server/shared/ddl/sqlstate.rs | 6 + .../executor/enforcement/transition_check.rs | 13 +- .../data/executor/enforcement/typeguard.rs | 22 +- .../data/executor/handlers/aggregate/exec.rs | 22 +- .../handlers/aggregate/streaming/finalize.rs | 14 +- .../handlers/columnar_read/scan/execute.rs | 8 +- .../executor/handlers/columnar_resolve.rs | 4 +- .../executor/handlers/document/index_fetch.rs | 4 +- .../executor/handlers/document/read/scan.rs | 13 +- .../handlers/document/sort/in_memory.rs | 50 ++ .../executor/handlers/document/sort/mod.rs | 2 +- .../src/data/executor/handlers/generated.rs | 5 +- .../executor/handlers/grouping_sets_exec.rs | 41 +- .../executor/handlers/join/grace_probe.rs | 6 +- .../executor/handlers/join/hash_handlers.rs | 6 +- .../executor/handlers/join/nested_loop.rs | 4 +- .../executor/handlers/kv/predicate/matches.rs | 2 +- nodedb/src/data/executor/handlers/kv/scan.rs | 4 +- .../handlers/point/update/post_image.rs | 4 +- .../data/executor/handlers/provider_scan.rs | 18 +- .../src/data/executor/handlers/recursive.rs | 12 +- .../executor/handlers/spatial/full_scan.rs | 8 +- .../executor/handlers/spatial/rtree_scan.rs | 8 +- .../handlers/timeseries/raw_scan/scan.rs | 4 +- .../stage_write/stage_bulk_delete.rs | 4 +- .../stage_write/stage_columnar_dml.rs | 14 +- .../handlers/vector_direct_targets.rs | 2 +- nodedb/src/error/types.rs | 7 + nodedb/src/error_classify.rs | 1 + nodedb/src/error_from.rs | 13 +- nodedb/src/error_from_data_plane.rs | 2 + nodedb/src/query/materialized_sum_delta.rs | 5 +- nodedb/src/storage/cold_filter.rs | 16 +- .../wire/cases/expr_predicate_like_array.rs | 88 +++ nodedb/tests/wire/cases/mod.rs | 2 + .../cases/row_functions_and_window_order.rs | 226 +++++++ .../cases/sql_search_subquery_composition.rs | 22 + 78 files changed, 2425 insertions(+), 220 deletions(-) create mode 100644 nodedb-query/src/functions/json/doc.rs create mode 100644 nodedb-query/src/functions/text_chunk.rs create mode 100644 nodedb-query/src/functions/vector.rs create mode 100644 nodedb-sql/src/planner/search_scope.rs create mode 100644 nodedb/tests/wire/cases/expr_predicate_like_array.rs create mode 100644 nodedb/tests/wire/cases/row_functions_and_window_order.rs diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs index e0115c725..a9d825a23 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs @@ -149,6 +149,9 @@ mod tests { let resp = propose_forwarded(&mut mr, sequencer_request()); assert!(!resp.success); - assert_eq!(resp.error_message, "not leader"); + assert_eq!( + resp.refusal, + Some(crate::rpc_codec::ForwardedProposeRefusal::NotLeader) + ); } } diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index fc884563b..b33c57ed3 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -113,6 +113,15 @@ pub enum DataPlaneErrorCode { limit: u64, }, DivisionByZero, + /// Expression evaluation called a function no evaluator implements. + UndefinedFunction { + name: String, + }, + /// A function received an argument it cannot compute on (SQLSTATE + /// `22000`). `detail` names the function and the value. + DataException { + detail: String, + }, /// A period-lock reference row exists but does not carry the /// configured `status_column` — a misconfigured column name, not a /// locked period. diff --git a/nodedb-crdt/src/validator/types.rs b/nodedb-crdt/src/validator/types.rs index ae06f1870..ec0ef0e45 100644 --- a/nodedb-crdt/src/validator/types.rs +++ b/nodedb-crdt/src/validator/types.rs @@ -23,7 +23,7 @@ pub enum ValidationOutcome { EvalError { /// The CHECK constraint whose predicate failed to evaluate. constraint_name: String, - /// The underlying evaluation error (currently only `DivisionByZero`). + /// The underlying evaluation error. error: nodedb_query::EvalError, }, } diff --git a/nodedb-query/src/expr/eval.rs b/nodedb-query/src/expr/eval.rs index 04f60e2d6..e73651660 100644 --- a/nodedb-query/src/expr/eval.rs +++ b/nodedb-query/src/expr/eval.rs @@ -18,15 +18,50 @@ use super::types::SqlExpr; /// /// Mirrors the [`WindowError`](crate::window::WindowError) idiom: a small, /// `thiserror`-derived enum living next to the evaluator it describes. -/// Everything that historically folded to `Value::Null` (bad casts, wrong -/// arg counts, unknown-value coercions) keeps doing so — this type exists -/// solely for division/modulo by a zero divisor, which must surface as -/// SQLSTATE `22012` (`division_by_zero`) instead of silently evaluating to -/// `NULL`. -#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] +/// Bad casts, wrong argument counts and unknown-value coercions fold to +/// `Value::Null`. These faults fail the statement instead: +/// - division or modulo by a zero divisor, SQLSTATE `22012`; +/// - a call to a function no evaluator implements. It never evaluates to a +/// silent `NULL`; +/// - a function argument it cannot compute on: vectors of different +/// dimensions, an argument of the wrong type, a malformed JSONPath. +/// SQLSTATE `22000`. A `NULL` argument stays `NULL` instead. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] pub enum EvalError { #[error("division by zero")] DivisionByZero, + #[error("function {name}() has no evaluator")] + UnknownFunction { + /// The function name as called. + name: String, + }, + /// Two vector operands have different dimensions. Same message shape + /// as the vector engine's dimension error. + #[error("{function}(): vector dimension mismatch: expected {expected}, got {got}")] + VectorDimensionMismatch { + function: &'static str, + /// Dimension of the first operand. + expected: usize, + /// Dimension of the second operand. + got: usize, + }, + /// An argument holds a value of a type the function cannot compute on. + #[error("{function}(): argument {position} must be {expected}, got {got}")] + ArgumentType { + function: &'static str, + /// 1-based argument position. + position: usize, + expected: &'static str, + /// `Value::type_name` of the argument received. + got: &'static str, + }, + /// A path argument is not a supported JSONPath. + #[error("{function}(): invalid JSONPath {path:?}: {reason}")] + InvalidJsonPath { + function: &'static str, + path: String, + reason: String, + }, } /// Row scope for `SqlExpr::eval_scope`: how `Column(..)` and `OldColumn(..)` diff --git a/nodedb-query/src/functions/array.rs b/nodedb-query/src/functions/array.rs index 45298022a..9b44f625c 100644 --- a/nodedb-query/src/functions/array.rs +++ b/nodedb-query/src/functions/array.rs @@ -88,6 +88,10 @@ pub(super) fn try_eval(name: &str, args: &[Value]) -> Option { reversed.reverse(); Value::Array(reversed) } + // `ARRAY[a, b, ...]` with a non-literal element lowers to this call, + // so the array is built per row from the evaluated elements. A NULL + // element stays a NULL element. + "make_array" => Value::Array(args.to_vec()), _ => return None, }; Some(v) diff --git a/nodedb-query/src/functions/eval.rs b/nodedb-query/src/functions/eval.rs index 636c70aac..cd6e0a920 100644 --- a/nodedb-query/src/functions/eval.rs +++ b/nodedb-query/src/functions/eval.rs @@ -6,19 +6,25 @@ use nodedb_types::Value; use crate::expr::EvalError; -use super::{array, conditional, datetime, fts, id, json, math, string, system, types}; +use super::{ + array, conditional, datetime, fts, id, json, math, string, system, text_chunk, types, vector, +}; /// Evaluate a scalar function call. /// -/// Every function returns `Ok(Value::Null)` on invalid/missing arguments -/// (SQL NULL propagation semantics) — the sole exception is `mod`'s -/// zero-modulus arm (`math::try_eval`), which returns -/// `Err(EvalError::DivisionByZero)`. Every other -/// sibling module here stays `Option`-shaped internally; only the -/// `math` arm is threaded as `Option>` and the -/// rest are wrapped in `Ok` at this dispatch boundary, so a single -/// fallible arm doesn't force every scalar-function module to carry a -/// `Result` it can never actually produce. +/// A `NULL` argument gives `Ok(Value::Null)` (SQL NULL propagation). These +/// calls fail instead: +/// - `mod`'s zero-modulus arm (`math::try_eval`) returns +/// `Err(EvalError::DivisionByZero)`; +/// - a vector distance over operands of different dimensions or over a +/// non-vector operand (`vector::try_eval`); +/// - a document function given a malformed JSONPath (`json::try_eval`); +/// - a name no family module implements returns +/// `Err(EvalError::UnknownFunction)`, never a silent `NULL`. +/// +/// The fallible families (`math`, `vector`, `json`) return +/// `Option>`. The rest stay `Option`-shaped +/// and are wrapped in `Ok` at this dispatch boundary. pub fn eval_function(name: &str, args: &[Value]) -> Result { if let Some(v) = string::try_eval(name, args) { return Ok(v); @@ -35,8 +41,8 @@ pub fn eval_function(name: &str, args: &[Value]) -> Result { if let Some(v) = datetime::try_eval(name, args) { return Ok(v); } - if let Some(v) = json::try_eval(name, args) { - return Ok(v); + if let Some(r) = json::try_eval(name, args) { + return r; } if let Some(v) = types::try_eval(name, args) { return Ok(v); @@ -50,8 +56,16 @@ pub fn eval_function(name: &str, args: &[Value]) -> Result { if let Some(v) = system::try_eval(name, args) { return Ok(v); } + if let Some(r) = vector::try_eval(name, args) { + return r; + } + if let Some(v) = text_chunk::try_eval(name, args) { + return Ok(v); + } // Geo / Spatial functions — delegated to geo_functions module. - Ok(crate::geo_functions::eval_geo_function(name, args).unwrap_or(Value::Null)) + crate::geo_functions::eval_geo_function(name, args).ok_or_else(|| EvalError::UnknownFunction { + name: name.to_owned(), + }) } #[cfg(test)] @@ -69,6 +83,147 @@ mod tests { assert_eq!(err, EvalError::DivisionByZero); } + #[test] + fn an_unknown_function_is_an_error_not_null() { + let err = eval_function("no_such_function", &[Value::Integer(1)]).unwrap_err(); + assert_eq!( + err, + EvalError::UnknownFunction { + name: "no_such_function".into() + } + ); + } + + /// Registered SQL scalars with a per-row meaning dispatch to a real + /// evaluator, never to the unknown-function error. + #[test] + fn registered_row_scalars_have_evaluators() { + let doc = Value::Object( + [( + "tags".to_string(), + Value::Array(vec![Value::String("a".into())]), + )] + .into_iter() + .collect(), + ); + let path = Value::String("$.tags[0]".into()); + let vector = Value::Array(vec![Value::Float(1.0), Value::Float(0.0)]); + let calls: [(&str, Vec); 8] = [ + ("doc_get", vec![doc.clone(), path.clone()]), + ("doc_exists", vec![doc.clone(), path.clone()]), + ( + "doc_array_contains", + vec![ + doc.clone(), + Value::String("$.tags".into()), + Value::String("a".into()), + ], + ), + ("nav", vec![doc, path]), + ("vector_distance", vec![vector.clone(), vector.clone()]), + ( + "vector_cosine_distance", + vec![vector.clone(), vector.clone()], + ), + ("vector_neg_inner_product", vec![vector.clone(), vector]), + ( + "ndb_chunk_text", + vec![Value::String("abc".into()), Value::Integer(2)], + ), + ]; + for (name, args) in calls { + let value = eval_function(name, &args) + .unwrap_or_else(|e| panic!("{name}() must evaluate, got {e}")); + assert_ne!(value, Value::Null, "{name}() gave NULL"); + } + } + + fn text(s: &str) -> Value { + Value::String(s.into()) + } + + #[test] + fn like_matches_percent_and_underscore() { + assert_eq!( + eval_fn("like", vec![text("alice"), text("a%")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("alice"), text("a_ice")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("alice"), text("b%")]), + Value::Bool(false) + ); + assert_eq!( + eval_fn("like", vec![text("Alice"), text("a%")]), + Value::Bool(false) + ); + } + + #[test] + fn like_honours_the_escape_character() { + assert_eq!( + eval_fn("like", vec![text("50%"), text("50\\%")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("500"), text("50\\%")]), + Value::Bool(false) + ); + assert_eq!( + eval_fn("like", vec![text("50%"), text("50!%"), text("!")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("a\\b"), text("a\\b"), text("")]), + Value::Bool(true) + ); + } + + #[test] + fn like_with_a_null_operand_is_null() { + assert_eq!(eval_fn("like", vec![Value::Null, text("a%")]), Value::Null); + assert_eq!(eval_fn("ilike", vec![text("a"), Value::Null]), Value::Null); + } + + #[test] + fn ilike_folds_unicode_case() { + assert_eq!( + eval_fn("ilike", vec![text("ÉCOLE"), text("éc%")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("ilike", vec![text("Straße"), text("STRA%")]), + Value::Bool(true) + ); + assert_eq!( + eval_fn("like", vec![text("ÉCOLE"), text("éc%")]), + Value::Bool(false) + ); + } + + #[test] + fn not_like_negates_the_call() { + let expr = SqlExpr::Negate(Box::new(SqlExpr::Function { + name: "like".into(), + args: vec![SqlExpr::Literal(text("bob")), SqlExpr::Literal(text("a%"))], + })); + assert_eq!(expr.eval(&Value::Null).unwrap(), Value::Bool(true)); + } + + #[test] + fn make_array_keeps_every_evaluated_element() { + assert_eq!( + eval_fn( + "make_array", + vec![Value::Integer(5), Value::Null, text("x")] + ), + Value::Array(vec![Value::Integer(5), Value::Null, text("x")]) + ); + } + #[test] fn upper() { assert_eq!( diff --git a/nodedb-query/src/functions/json/dispatch.rs b/nodedb-query/src/functions/json/dispatch.rs index d5270733f..2e2dd3fc1 100644 --- a/nodedb-query/src/functions/json/dispatch.rs +++ b/nodedb-query/src/functions/json/dispatch.rs @@ -4,7 +4,28 @@ use nodedb_types::Value; -pub(in crate::functions) fn try_eval(name: &str, args: &[Value]) -> Option { +use crate::expr::EvalError; + +/// `None` when no JSON-family function has `name`. The document functions +/// can fail with a typed error; every other JSON function returns a value. +pub(in crate::functions) fn try_eval( + name: &str, + args: &[Value], +) -> Option> { + let doc_result = match name { + "doc_get" => Some(super::doc::doc_get(args)), + "doc_exists" => Some(super::doc::doc_exists(args)), + "doc_array_contains" => Some(super::doc::doc_array_contains(args)), + "nav" => Some(super::doc::nav(args)), + _ => None, + }; + if doc_result.is_some() { + return doc_result; + } + try_eval_value(name, args).map(Ok) +} + +fn try_eval_value(name: &str, args: &[Value]) -> Option { // PostgreSQL JSON operator functions (lowered from AST BinaryOp). let pg_result = match name { "pg_json_get" => { diff --git a/nodedb-query/src/functions/json/doc.rs b/nodedb-query/src/functions/json/doc.rs new file mode 100644 index 000000000..204284a2c --- /dev/null +++ b/nodedb-query/src/functions/json/doc.rs @@ -0,0 +1,271 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Document navigation functions: `doc_get`, `doc_exists`, +//! `doc_array_contains`, and `nav`. +//! +//! Each takes a document value and a path. The path is JSONPath (`$.a.b`, +//! `$.arr[0]`). A path without the leading `$` is read as `$.` plus the +//! path, so `'user.name'` and `'$.user.name'` name the same field. A document +//! held as JSON text is parsed before the walk. +//! +//! A malformed path, or a path argument that is not text, fails the +//! statement with a typed [`EvalError`]. A `NULL` path gives `NULL`. A path +//! that resolves to nothing gives the default, `NULL`, or `false`. + +use nodedb_types::Value; + +use super::path::{PathStep, parse_jsonpath, walk_path}; +use super::pg_ops::coerce_json_string; +use crate::expr::EvalError; +use crate::value_ops::coerced_eq; + +/// Parse the path argument at 1-based `position`. `Ok(None)` for a `NULL` +/// path. +fn path_steps( + function: &'static str, + position: usize, + path: &Value, +) -> Result>, EvalError> { + let text = match path { + Value::Null => return Ok(None), + Value::String(text) => text, + other => { + return Err(EvalError::ArgumentType { + function, + position, + expected: "a JSONPath text", + got: other.type_name(), + }); + } + }; + let parsed = if text.starts_with('$') { + parse_jsonpath(text) + } else { + parse_jsonpath(&format!("$.{text}")) + }; + parsed.map(Some).map_err(|e| EvalError::InvalidJsonPath { + function, + path: text.clone(), + reason: e.to_string(), + }) +} + +/// The non-null value at the path in `args[1]` inside the document in +/// `args[0]`, cloned. `Ok(None)` when the path is `NULL` or resolves to +/// nothing. +fn resolve(function: &'static str, args: &[Value]) -> Result, EvalError> { + let doc = args.first().unwrap_or(&Value::Null); + let path = args.get(1).unwrap_or(&Value::Null); + let Some(steps) = path_steps(function, 2, path)? else { + return Ok(None); + }; + let doc = coerce_json_string(doc); + Ok(match walk_path(&doc, &steps) { + Some(Value::Null) | None => None, + Some(found) => Some(found.clone()), + }) +} + +/// `doc_get(doc, path [, default])`: the value at `path`, or `default` when +/// the path is missing or null. `default` is `NULL` when omitted. +pub(super) fn doc_get(args: &[Value]) -> Result { + Ok(resolve("doc_get", args)?.unwrap_or_else(|| args.get(2).cloned().unwrap_or(Value::Null))) +} + +/// `nav(doc, path)`: the value at `path`, or `NULL`. +pub(super) fn nav(args: &[Value]) -> Result { + Ok(resolve("nav", args)?.unwrap_or(Value::Null)) +} + +/// `doc_exists(doc, path)`: whether `path` holds a non-null value. +pub(super) fn doc_exists(args: &[Value]) -> Result { + Ok(Value::Bool(resolve("doc_exists", args)?.is_some())) +} + +/// `doc_array_contains(doc, path, value)`: whether the array at `path` holds +/// an element equal to `value`. Equality coerces a numeric string to a +/// number and an ISO-8601 string to an instant. A missing path or a +/// non-array value at the path gives `false`. +pub(super) fn doc_array_contains(args: &[Value]) -> Result { + let needle = args.get(2).unwrap_or(&Value::Null); + let contains = match resolve("doc_array_contains", args)? { + Some(Value::Array(items) | Value::Set(items)) => { + items.iter().any(|item| coerced_eq(item, needle)) + } + Some(_) | None => false, + }; + Ok(Value::Bool(contains)) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::HashMap; + + fn obj(pairs: &[(&str, Value)]) -> Value { + Value::Object( + pairs + .iter() + .map(|(k, v)| (k.to_string(), v.clone())) + .collect::>(), + ) + } + + fn text(s: &str) -> Value { + Value::String(s.into()) + } + + fn event() -> Value { + obj(&[ + ( + "user", + obj(&[("name", text("ada")), ("email", Value::Null)]), + ), + ( + "tags", + Value::Array(vec![text("important"), text("ops"), Value::Integer(7)]), + ), + ]) + } + + #[test] + fn doc_get_reads_a_nested_field() { + assert_eq!(doc_get(&[event(), text("$.user.name")]), Ok(text("ada"))); + } + + #[test] + fn doc_get_reads_a_path_without_the_dollar() { + assert_eq!(doc_get(&[event(), text("user.name")]), Ok(text("ada"))); + } + + #[test] + fn doc_get_reads_an_array_element() { + assert_eq!(doc_get(&[event(), text("$.tags[1]")]), Ok(text("ops"))); + } + + #[test] + fn doc_get_missing_path_gives_the_default() { + assert_eq!( + doc_get(&[event(), text("$.user.age"), Value::Integer(0)]), + Ok(Value::Integer(0)) + ); + assert_eq!(doc_get(&[event(), text("$.user.age")]), Ok(Value::Null)); + } + + #[test] + fn doc_get_null_field_gives_the_default() { + assert_eq!( + doc_get(&[event(), text("$.user.email"), text("none")]), + Ok(text("none")) + ); + } + + #[test] + fn doc_get_parses_a_json_text_document() { + let doc = text(r#"{"user":{"name":"ada"}}"#); + assert_eq!(doc_get(&[doc, text("$.user.name")]), Ok(text("ada"))); + } + + #[test] + fn doc_get_malformed_path_is_an_error() { + let err = doc_get(&[event(), text("$..name"), text("d")]).unwrap_err(); + assert!( + matches!( + &err, + EvalError::InvalidJsonPath { function: "doc_get", path, .. } if path == "$..name" + ), + "{err:?}" + ); + let err = doc_get(&[event(), text("$.tags[x]")]).unwrap_err(); + assert!(matches!(err, EvalError::InvalidJsonPath { .. }), "{err:?}"); + } + + #[test] + fn doc_array_contains_malformed_path_is_an_error() { + let err = doc_array_contains(&[event(), text("$.tags["), text("ops")]).unwrap_err(); + assert!( + matches!( + err, + EvalError::InvalidJsonPath { + function: "doc_array_contains", + .. + } + ), + "{err:?}" + ); + } + + #[test] + fn a_non_text_path_is_an_argument_error() { + let err = doc_exists(&[event(), Value::Integer(3)]).unwrap_err(); + assert_eq!( + err, + EvalError::ArgumentType { + function: "doc_exists", + position: 2, + expected: "a JSONPath text", + got: "int", + } + ); + } + + #[test] + fn a_null_path_resolves_to_nothing() { + assert_eq!(doc_get(&[event(), Value::Null, text("d")]), Ok(text("d"))); + assert_eq!(doc_exists(&[event(), Value::Null]), Ok(Value::Bool(false))); + } + + #[test] + fn nav_matches_doc_get_without_a_default() { + assert_eq!(nav(&[event(), text("$.user.name")]), Ok(text("ada"))); + assert_eq!(nav(&[event(), text("$.missing")]), Ok(Value::Null)); + } + + #[test] + fn doc_exists_is_true_only_for_a_non_null_value() { + assert_eq!( + doc_exists(&[event(), text("$.user.name")]), + Ok(Value::Bool(true)) + ); + assert_eq!( + doc_exists(&[event(), text("$.user.email")]), + Ok(Value::Bool(false)) + ); + assert_eq!( + doc_exists(&[event(), text("$.nope")]), + Ok(Value::Bool(false)) + ); + assert_eq!( + doc_exists(&[Value::Null, text("$.a")]), + Ok(Value::Bool(false)) + ); + } + + #[test] + fn doc_array_contains_finds_an_element() { + assert_eq!( + doc_array_contains(&[event(), text("$.tags"), text("important")]), + Ok(Value::Bool(true)) + ); + assert_eq!( + doc_array_contains(&[event(), text("$.tags"), text("absent")]), + Ok(Value::Bool(false)) + ); + } + + #[test] + fn doc_array_contains_coerces_a_numeric_string() { + assert_eq!( + doc_array_contains(&[event(), text("$.tags"), text("7")]), + Ok(Value::Bool(true)) + ); + } + + #[test] + fn doc_array_contains_on_a_non_array_is_false() { + assert_eq!( + doc_array_contains(&[event(), text("$.user.name"), text("ada")]), + Ok(Value::Bool(false)) + ); + } +} diff --git a/nodedb-query/src/functions/json/mod.rs b/nodedb-query/src/functions/json/mod.rs index 8b5652181..3677ec16c 100644 --- a/nodedb-query/src/functions/json/mod.rs +++ b/nodedb-query/src/functions/json/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: Apache-2.0 mod dispatch; +mod doc; pub(super) mod legacy; pub(crate) mod path; pub(super) mod pg_ops; diff --git a/nodedb-query/src/functions/json/pg_ops.rs b/nodedb-query/src/functions/json/pg_ops.rs index 49757a0f3..900094c3b 100644 --- a/nodedb-query/src/functions/json/pg_ops.rs +++ b/nodedb-query/src/functions/json/pg_ops.rs @@ -17,7 +17,7 @@ use nodedb_types::Value; /// /// This is a cheap path: non-string values pass through a single `matches!` /// check; strings that are not valid JSON also return quickly from the parser. -fn coerce_json_string(v: &Value) -> std::borrow::Cow<'_, Value> { +pub(super) fn coerce_json_string(v: &Value) -> std::borrow::Cow<'_, Value> { if let Value::String(s) = v && let Ok(parsed) = sonic_rs::from_str::(s) { diff --git a/nodedb-query/src/functions/mod.rs b/nodedb-query/src/functions/mod.rs index 04dcfdbb8..a7b26ec68 100644 --- a/nodedb-query/src/functions/mod.rs +++ b/nodedb-query/src/functions/mod.rs @@ -16,6 +16,8 @@ mod math; pub(crate) mod shared; mod string; mod system; +mod text_chunk; mod types; +mod vector; pub use eval::eval_function; diff --git a/nodedb-query/src/functions/string.rs b/nodedb-query/src/functions/string.rs index 84f4b80a3..bd68ef50f 100644 --- a/nodedb-query/src/functions/string.rs +++ b/nodedb-query/src/functions/string.rs @@ -3,6 +3,7 @@ //! String scalar functions. use super::shared::{num_arg, str_arg}; +use crate::scan_filter::like::{DEFAULT_LIKE_ESCAPE, sql_like_match_escaped}; use crate::value_ops::value_to_display_string; use nodedb_types::Value; @@ -48,7 +49,52 @@ pub(super) fn try_eval(name: &str, args: &[Value]) -> Option { "reverse" => { str_arg(args, 0).map_or(Value::Null, |s| Value::String(s.chars().rev().collect())) } + "like" => like(args, false), + "ilike" => like(args, true), _ => return None, }; Some(v) } + +/// `like(input, pattern[, escape])` / `ilike(...)`: SQL `LIKE` / `ILIKE`. +/// +/// - A NULL input or pattern gives NULL. +/// - A non-text scalar operand matches as its display text. +/// - `escape` defaults to `\`. An empty escape disables escaping. An escape +/// longer than one character is invalid and gives NULL. +/// +/// `NOT LIKE` is the negation of this call, so NULL stays NULL. +fn like(args: &[Value], case_insensitive: bool) -> Value { + let (Some(input), Some(pattern)) = (like_text(args.first()), like_text(args.get(1))) else { + return Value::Null; + }; + let escape = match args.get(2) { + None => Some(DEFAULT_LIKE_ESCAPE), + Some(v) => { + let Some(text) = like_text(Some(v)) else { + return Value::Null; + }; + let mut chars = text.chars(); + match (chars.next(), chars.next()) { + (None, _) => None, + (Some(c), None) => Some(c), + (Some(_), Some(_)) => return Value::Null, + } + } + }; + Value::Bool(sql_like_match_escaped( + &input, + &pattern, + case_insensitive, + escape, + )) +} + +/// The text a LIKE operand matches as. `None` for NULL or a missing operand. +fn like_text(v: Option<&Value>) -> Option { + match v? { + Value::Null => None, + Value::String(s) => Some(s.clone()), + other => Some(value_to_display_string(other)), + } +} diff --git a/nodedb-query/src/functions/text_chunk.rs b/nodedb-query/src/functions/text_chunk.rs new file mode 100644 index 000000000..35456c8af --- /dev/null +++ b/nodedb-query/src/functions/text_chunk.rs @@ -0,0 +1,90 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! `ndb_chunk_text(text, chunk_size [, overlap])` as a per-row scalar. +//! +//! The scalar form returns the chunk texts as an array, split on character +//! boundaries, the default strategy of the `SELECT * FROM NDB_CHUNK_TEXT(...)` +//! table function. `overlap` is 0 when omitted. A `NULL` or non-text `text`, +//! a `chunk_size` of 0 or less, or an `overlap` not below `chunk_size` gives +//! `NULL`. + +use nodedb_types::Value; + +use crate::chunk_text::{ChunkStrategy, chunk_text}; + +pub(super) fn try_eval(name: &str, args: &[Value]) -> Option { + if name != "ndb_chunk_text" { + return None; + } + Some(eval_chunk_text(args).unwrap_or(Value::Null)) +} + +fn eval_chunk_text(args: &[Value]) -> Option { + let text = args.first()?.as_str()?; + let chunk_size = count_arg(args.get(1)?)?; + let overlap = match args.get(2) { + Some(value) => count_arg(value)?, + None => 0, + }; + let chunks = chunk_text(text, chunk_size, overlap, ChunkStrategy::Character).ok()?; + Some(Value::Array( + chunks + .into_iter() + .map(|chunk| Value::String(chunk.text)) + .collect(), + )) +} + +fn count_arg(value: &Value) -> Option { + match value { + Value::Integer(n) => usize::try_from(*n).ok(), + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn text(s: &str) -> Value { + Value::String(s.into()) + } + + #[test] + fn splits_into_character_chunks() { + let out = try_eval("ndb_chunk_text", &[text("abcdef"), Value::Integer(4)]); + assert_eq!(out, Some(Value::Array(vec![text("abcd"), text("ef")]))); + } + + #[test] + fn overlap_repeats_the_tail() { + let out = try_eval( + "ndb_chunk_text", + &[text("abcdef"), Value::Integer(4), Value::Integer(2)], + ); + assert_eq!(out, Some(Value::Array(vec![text("abcd"), text("cdef")]))); + } + + #[test] + fn invalid_sizes_are_null() { + assert_eq!( + try_eval("ndb_chunk_text", &[text("abc"), Value::Integer(0)]), + Some(Value::Null) + ); + assert_eq!( + try_eval( + "ndb_chunk_text", + &[text("abc"), Value::Integer(2), Value::Integer(2)] + ), + Some(Value::Null) + ); + } + + #[test] + fn null_text_is_null() { + assert_eq!( + try_eval("ndb_chunk_text", &[Value::Null, Value::Integer(2)]), + Some(Value::Null) + ); + } +} diff --git a/nodedb-query/src/functions/vector.rs b/nodedb-query/src/functions/vector.rs new file mode 100644 index 000000000..b598cd2a1 --- /dev/null +++ b/nodedb-query/src/functions/vector.rs @@ -0,0 +1,253 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Per-row vector distance functions. +//! +//! `vector_distance` (the `<->` operator), `vector_cosine_distance` (`<=>`) +//! and `vector_neg_inner_product` (`<#>`) give the same numbers a vector +//! search reports for the metric each name selects: +//! - `vector_distance`: squared Euclidean (L2) distance; +//! - `vector_cosine_distance`: `1 - cosine similarity`; +//! - `vector_neg_inner_product`: the negated dot product. +//! +//! A vector search plan serves these calls in `ORDER BY`. Every other +//! position evaluates them here, once per row. A `NULL` operand gives `NULL`. +//! Operands of different dimensions, or an operand that is not a numeric +//! vector, fail the statement with a typed [`EvalError`]. + +use nodedb_types::Value; +use nodedb_types::vector_distance::{cosine_distance, l2_squared, neg_inner_product}; + +use crate::expr::EvalError; + +/// The expected-type text an argument error names. +const NUMERIC_VECTOR: &str = "a numeric vector"; + +/// A distance between two vectors of equal dimension. +type Metric = fn(&[f32], &[f32]) -> f32; + +pub(super) fn try_eval(name: &str, args: &[Value]) -> Option> { + let (function, metric): (&'static str, Metric) = match name { + "vector_distance" => ("vector_distance", l2_squared), + "vector_cosine_distance" => ("vector_cosine_distance", cosine_distance), + "vector_neg_inner_product" => ("vector_neg_inner_product", neg_inner_product), + _ => return None, + }; + Some(eval_distance(function, metric, args)) +} + +fn eval_distance( + function: &'static str, + metric: Metric, + args: &[Value], +) -> Result { + let left = args.first().unwrap_or(&Value::Null); + let right = args.get(1).unwrap_or(&Value::Null); + if matches!(left, Value::Null) || matches!(right, Value::Null) { + return Ok(Value::Null); + } + let a = as_vector(function, 1, left)?; + let b = as_vector(function, 2, right)?; + if a.len() != b.len() { + return Err(EvalError::VectorDimensionMismatch { + function, + expected: a.len(), + got: b.len(), + }); + } + Ok(Value::Float(f64::from(metric(&a, &b)))) +} + +/// Read a vector argument: a vector value, an array of numbers, or JSON text +/// holding an array of numbers. `position` is 1-based. +fn as_vector( + function: &'static str, + position: usize, + value: &Value, +) -> Result, EvalError> { + let wrong_type = |got: &'static str| EvalError::ArgumentType { + function, + position, + expected: NUMERIC_VECTOR, + got, + }; + match value { + Value::Vector(floats) => Ok(floats.to_vec()), + Value::Array(items) => items + .iter() + .map(|item| number(item).ok_or_else(|| wrong_type(item.type_name()))) + .collect(), + Value::String(text) => sonic_rs::from_str::>(text) + .map(|parsed| parsed.into_iter().map(|x| x as f32).collect()) + .map_err(|_| wrong_type(value.type_name())), + other => Err(wrong_type(other.type_name())), + } +} + +fn number(value: &Value) -> Option { + match value { + Value::Float(f) => Some(*f as f32), + Value::Integer(i) => Some(*i as f32), + Value::Decimal(d) => { + use rust_decimal::prelude::ToPrimitive; + d.to_f32() + } + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn vector(items: &[f64]) -> Value { + Value::Array(items.iter().map(|x| Value::Float(*x)).collect()) + } + + fn eval(name: &str, a: Value, b: Value) -> Result { + try_eval(name, &[a, b]).expect("vector function") + } + + #[test] + fn l2_is_squared_euclidean() { + assert_eq!( + eval("vector_distance", vector(&[0.0, 0.0]), vector(&[3.0, 4.0])), + Ok(Value::Float(25.0)) + ); + } + + #[test] + fn decimal_literal_elements_are_numeric() { + let decimals = Value::Array(vec![ + Value::Decimal(rust_decimal::Decimal::new(30, 1)), + Value::Decimal(rust_decimal::Decimal::new(40, 1)), + ]); + assert_eq!( + eval("vector_distance", vector(&[0.0, 0.0]), decimals), + Ok(Value::Float(25.0)) + ); + } + + #[test] + fn cosine_of_orthogonal_vectors_is_one() { + let Ok(Value::Float(d)) = eval( + "vector_cosine_distance", + vector(&[1.0, 0.0]), + vector(&[0.0, 1.0]), + ) else { + panic!("expected a float"); + }; + assert!((d - 1.0).abs() < 1e-6); + } + + #[test] + fn neg_inner_product_negates_the_dot_product() { + assert_eq!( + eval( + "vector_neg_inner_product", + vector(&[1.0, 2.0]), + vector(&[3.0, 4.0]) + ), + Ok(Value::Float(-11.0)) + ); + } + + #[test] + fn vector_values_integer_elements_and_json_text_are_vectors() { + let ints = Value::Array(vec![Value::Integer(0), Value::Integer(0)]); + assert_eq!( + eval("vector_distance", ints, Value::String("[3, 4]".into())), + Ok(Value::Float(25.0)) + ); + let packed = Value::Vector(vec![3.0_f32, 4.0].into()); + assert_eq!( + eval("vector_distance", packed, vector(&[0.0, 0.0])), + Ok(Value::Float(25.0)) + ); + } + + #[test] + fn dimension_mismatch_names_both_dimensions() { + let err = eval("vector_distance", vector(&[1.0]), vector(&[1.0, 2.0])).unwrap_err(); + assert_eq!( + err, + EvalError::VectorDimensionMismatch { + function: "vector_distance", + expected: 1, + got: 2, + } + ); + assert_eq!( + err.to_string(), + "vector_distance(): vector dimension mismatch: expected 1, got 2" + ); + } + + #[test] + fn non_vector_argument_names_its_position_and_type() { + let err = eval("vector_cosine_distance", vector(&[1.0]), Value::Integer(1)).unwrap_err(); + assert_eq!( + err, + EvalError::ArgumentType { + function: "vector_cosine_distance", + position: 2, + expected: NUMERIC_VECTOR, + got: "int", + } + ); + } + + #[test] + fn non_numeric_element_names_the_element_type() { + let mixed = Value::Array(vec![Value::Float(1.0), Value::String("x".into())]); + let err = eval("vector_distance", mixed, vector(&[1.0, 2.0])).unwrap_err(); + assert!( + matches!( + err, + EvalError::ArgumentType { + position: 1, + got: "string", + .. + } + ), + "{err:?}" + ); + } + + #[test] + fn text_that_is_not_a_vector_is_an_argument_error() { + let err = eval( + "vector_distance", + Value::String("not a vector".into()), + vector(&[1.0]), + ) + .unwrap_err(); + assert!( + matches!( + err, + EvalError::ArgumentType { + position: 1, + got: "string", + .. + } + ), + "{err:?}" + ); + } + + #[test] + fn null_operand_is_null() { + assert_eq!( + eval("vector_distance", Value::Null, vector(&[1.0])), + Ok(Value::Null) + ); + assert_eq!( + eval("vector_distance", vector(&[1.0]), Value::Null), + Ok(Value::Null) + ); + } + + #[test] + fn other_names_are_not_claimed() { + assert!(try_eval("bm25_score", &[]).is_none()); + } +} diff --git a/nodedb-query/src/scan_filter/like.rs b/nodedb-query/src/scan_filter/like.rs index 7604715f9..90033bd22 100644 --- a/nodedb-query/src/scan_filter/like.rs +++ b/nodedb-query/src/scan_filter/like.rs @@ -1,60 +1,155 @@ // SPDX-License-Identifier: Apache-2.0 -/// SQL LIKE pattern matching. -/// -/// Supports `%` (zero or more characters) and `_` (exactly one character). -/// When `case_insensitive` is true, both input and pattern are lowercased (ILIKE). +//! SQL `LIKE` / `ILIKE` pattern matching: the one matcher every LIKE path +//! uses, the scan filters and the `like` / `ilike` scalar functions alike. +//! +//! - `%` matches zero or more characters. +//! - `_` matches exactly one character (a Unicode scalar value, not a byte). +//! - The escape character makes the next pattern character literal. The +//! default escape is `\`, as in PostgreSQL. A trailing escape character +//! matches itself. +//! - `ILIKE` lowercases input and pattern by Unicode rules before matching. + +/// The escape character a pattern uses when none is given. +pub const DEFAULT_LIKE_ESCAPE: char = '\\'; + +/// Match `input` against the SQL LIKE `pattern`, with `\` as the escape. pub fn sql_like_match(input: &str, pattern: &str, case_insensitive: bool) -> bool { + sql_like_match_escaped(input, pattern, case_insensitive, Some(DEFAULT_LIKE_ESCAPE)) +} + +/// Match `input` against the SQL LIKE `pattern` with the escape character +/// `escape`. `None` means the pattern has no escape character. +pub fn sql_like_match_escaped( + input: &str, + pattern: &str, + case_insensitive: bool, + escape: Option, +) -> bool { let (input, pattern) = if case_insensitive { (input.to_lowercase(), pattern.to_lowercase()) } else { - (input.to_string(), pattern.to_string()) + (input.to_owned(), pattern.to_owned()) }; + let input: Vec = input.chars().collect(); + let tokens = tokenize(&pattern, escape); + matches(&input, &tokens) +} - let input = input.as_bytes(); - let pattern = pattern.as_bytes(); - - let (mut i, mut j) = (0usize, 0usize); - let (mut star_j, mut star_i) = (usize::MAX, 0usize); +/// One element of a compiled pattern. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Token { + /// A character that must match itself. + Char(char), + /// `_`: any one character. + AnyOne, + /// `%`: any run of characters, the empty run included. + AnyRun, +} - while i < input.len() { - if j < pattern.len() && (pattern[j] == b'_' || pattern[j] == input[i]) { - i += 1; - j += 1; - } else if j < pattern.len() && pattern[j] == b'%' { - star_j = j; - star_i = i; - j += 1; - } else if star_j != usize::MAX { - star_i += 1; - i = star_i; - j = star_j + 1; +fn tokenize(pattern: &str, escape: Option) -> Vec { + let mut tokens = Vec::with_capacity(pattern.len()); + let mut chars = pattern.chars(); + while let Some(c) = chars.next() { + let token = if Some(c) == escape { + // The escaped character is literal. A trailing escape is itself. + Token::Char(chars.next().unwrap_or(c)) } else { - return false; - } + match c { + '%' => Token::AnyRun, + '_' => Token::AnyOne, + other => Token::Char(other), + } + }; + tokens.push(token); } + tokens +} - while j < pattern.len() && pattern[j] == b'%' { - j += 1; +/// Greedy match with single-point backtracking to the last `%`. +fn matches(input: &[char], tokens: &[Token]) -> bool { + let (mut i, mut t) = (0usize, 0usize); + let mut backtrack: Option<(usize, usize)> = None; + while i < input.len() { + match tokens.get(t) { + Some(Token::AnyRun) => { + backtrack = Some((t, i)); + t += 1; + continue; + } + Some(Token::AnyOne) => { + i += 1; + t += 1; + continue; + } + Some(Token::Char(c)) if *c == input[i] => { + i += 1; + t += 1; + continue; + } + Some(Token::Char(_)) | None => {} + } + match backtrack { + Some((run_t, run_i)) => { + backtrack = Some((run_t, run_i + 1)); + i = run_i + 1; + t = run_t + 1; + } + None => return false, + } } - - j == pattern.len() + tokens[t..].iter().all(|token| *token == Token::AnyRun) } #[cfg(test)] mod tests { - use super::sql_like_match; + use super::*; #[test] fn like_basic() { assert!(sql_like_match("hello world", "%world", false)); assert!(sql_like_match("hello world", "hello%", false)); assert!(!sql_like_match("hello world", "xyz%", false)); + assert!(sql_like_match("", "%", false)); + assert!(!sql_like_match("", "_", false)); + } + + #[test] + fn underscore_matches_one_character_not_one_byte() { + assert!(sql_like_match("é", "_", false)); + assert!(sql_like_match("naïve", "na_ve", false)); + assert!(!sql_like_match("naïve", "na__ve", false)); + } + + #[test] + fn escape_makes_wildcards_literal() { + assert!(sql_like_match("100%", "100\\%", false)); + assert!(!sql_like_match("1000", "100\\%", false)); + assert!(sql_like_match("a_b", "a\\_b", false)); + assert!(!sql_like_match("axb", "a\\_b", false)); + assert!(sql_like_match("a\\", "a\\", false)); + assert!(sql_like_match_escaped("50!%", "50!!!%", false, Some('!'))); + assert!(sql_like_match_escaped("a\\b", "a\\b", false, None)); } #[test] fn ilike_case_insensitive() { assert!(sql_like_match("Hello", "hello", true)); assert!(sql_like_match("WORLD", "%world%", true)); + assert!(!sql_like_match("WORLD", "%world%", false)); + } + + #[test] + fn ilike_folds_unicode_case() { + assert!(sql_like_match("ÉCOLE", "école", true)); + assert!(sql_like_match("ΣΟΦΙΑ", "σοφια", true)); + assert!(!sql_like_match("ÉCOLE", "école", false)); + } + + #[test] + fn backtracking_finds_a_later_match() { + assert!(sql_like_match("abcabd", "%abd", false)); + assert!(sql_like_match("aaa", "%a%a", false)); + assert!(!sql_like_match("abc", "%d%", false)); } } diff --git a/nodedb-query/src/window/value_eval.rs b/nodedb-query/src/window/value_eval.rs index 1ba2da874..b4d1cbabb 100644 --- a/nodedb-query/src/window/value_eval.rs +++ b/nodedb-query/src/window/value_eval.rs @@ -28,7 +28,7 @@ pub enum WindowError { #[error("window frame error: {detail}")] BadFrame { detail: String }, - #[error("division by zero in window expression")] + #[error("window expression: {0}")] Eval(#[from] crate::expr::EvalError), } diff --git a/nodedb-sql/src/error.rs b/nodedb-sql/src/error.rs index eaa5733aa..b02210889 100644 --- a/nodedb-sql/src/error.rs +++ b/nodedb-sql/src/error.rs @@ -37,6 +37,19 @@ pub enum SqlError { )] SequencePerRowUnsupported { name: String }, + /// An index-owned search function (`bm25_score`, `text_match`, + /// `rrf_score`, `sparse_score`, ...) sits in a position the row evaluator + /// runs: the planner could not lower it into its search plan. The + /// function reads a search index and has no per-row value. + /// + /// Rendered as SQLSTATE `0A000` (feature_not_supported). + #[error( + "{name}(...) reads a search index and has no per-row value here; \ + call it in ORDER BY, in WHERE as the search predicate, or in the \ + SELECT list of a search, with a literal query" + )] + SearchFunctionOutsideSearch { name: String }, + /// A statement names a database object that does not exist — a sequence, /// most commonly. Distinct from [`SqlError::UndefinedFunction`]: the /// function exists, the object it names does not. PostgreSQL rejects the @@ -71,6 +84,13 @@ pub enum SqlError { #[error("division by zero")] DivisionByZero, + /// A constant function call received an argument it cannot compute on: + /// vectors of different dimensions, an argument of the wrong type, a + /// malformed JSONPath. Same distinction as [`SqlError::DivisionByZero`]. + /// Rendered as SQLSTATE `22000` (`data_exception`). + #[error("{detail}")] + DataException { detail: String }, + /// A LIMIT, OFFSET, or FETCH FIRST clause resolved to a value outside /// `[0, usize::MAX]`, or to an expression the planner cannot read as a /// literal. PostgreSQL rejects the same input with SQLSTATE `2201W`. diff --git a/nodedb-sql/src/lib.rs b/nodedb-sql/src/lib.rs index 62fbf1692..0f2ee0ad0 100644 --- a/nodedb-sql/src/lib.rs +++ b/nodedb-sql/src/lib.rs @@ -182,6 +182,9 @@ fn plan_statements( } } + for plan in &plans { + planner::search_scope::refuse_row_scoped_search_functions(plan, &functions)?; + } Ok(plans) } diff --git a/nodedb-sql/src/planner/const_fold.rs b/nodedb-sql/src/planner/const_fold.rs index c06a9fbbb..ec8e8f00e 100644 --- a/nodedb-sql/src/planner/const_fold.rs +++ b/nodedb-sql/src/planner/const_fold.rs @@ -389,6 +389,17 @@ pub fn fold_function_call_scoped( match nodedb_query::functions::eval_function(&name_lower, &folded_args) { Ok(result) => Ok(Some(ndb_to_sql_value(result))), Err(nodedb_query::EvalError::DivisionByZero) => Err(SqlError::DivisionByZero), + Err( + e @ (nodedb_query::EvalError::VectorDimensionMismatch { .. } + | nodedb_query::EvalError::ArgumentType { .. } + | nodedb_query::EvalError::InvalidJsonPath { .. }), + ) => Err(SqlError::DataException { + detail: e.to_string(), + }), + // A registered name some other evaluator owns (a search score): + // not foldable here. Its search plan serves it, or the plan-time + // search-scope pass refuses it. + Err(nodedb_query::EvalError::UnknownFunction { .. }) => Ok(None), } } @@ -473,6 +484,26 @@ mod tests { } } + /// A constant call whose argument the function cannot compute on fails + /// the statement at plan time instead of folding to NULL. + #[test] + fn fold_of_a_malformed_json_path_is_a_data_exception() { + let registry = FunctionRegistry::new(); + let expr = SqlExpr::Function { + name: "doc_get".into(), + args: vec![ + SqlExpr::Literal(SqlValue::String("{\"a\":1}".into())), + SqlExpr::Literal(SqlValue::String("$..a".into())), + ], + distinct: false, + }; + let err = fold_constant(&expr, ®istry).expect_err("malformed path must fail"); + assert!( + matches!(&err, SqlError::DataException { detail } if detail.contains("invalid JSONPath")), + "{err:?}" + ); + } + #[test] fn fold_current_timestamp_produces_timestamptz_once_only() { let registry = FunctionRegistry::new(); diff --git a/nodedb-sql/src/planner/mod.rs b/nodedb-sql/src/planner/mod.rs index ae80ed820..3a1a72da3 100644 --- a/nodedb-sql/src/planner/mod.rs +++ b/nodedb-sql/src/planner/mod.rs @@ -31,6 +31,7 @@ pub mod lateral; pub mod merge; pub mod predicate_coerce; pub mod returning; +pub mod search_scope; pub mod select; pub use returning::resolve_returning_items; diff --git a/nodedb-sql/src/planner/search_scope.rs b/nodedb-sql/src/planner/search_scope.rs new file mode 100644 index 000000000..76a595276 --- /dev/null +++ b/nodedb-sql/src/planner/search_scope.rs @@ -0,0 +1,608 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Plan-time refusal of index-owned search functions in row-evaluated +//! positions. +//! +//! `bm25_score`, `search_score`, `text_match`, `search`, `rrf_score`, +//! `sparse_score`, `graph_score`, `multi_vector_score` and +//! `multi_vector_search` read a search index. The planner lowers each call it +//! recognises into its search plan, which serves the score as a column. A +//! call the planner could not lower stays in a filter, projection, sort key, +//! assignment or aggregate argument. The row evaluator has no index and no +//! value for it, so this pass refuses the statement at plan time. The +//! refusal does not depend on whether the collection holds rows. +//! +//! A wrapper plan (a subquery tail, aggregate, join or lateral join) over a +//! search plan is not checked for its own expressions: its projection names +//! the score column the search plan serves. + +use crate::error::{Result, SqlError}; +use crate::functions::registry::{FunctionRegistry, SearchTrigger}; +use crate::types::query::{AggregateExpr, Projection, SortKey, WindowSpec}; +use crate::types::{Filter, FilterExpr, MergePlanAction, SqlPlan}; +use crate::types_expr::SqlExpr; + +/// Refuse `plan` when an index-owned search function sits where the row +/// evaluator runs it. +pub fn refuse_row_scoped_search_functions( + plan: &SqlPlan, + functions: &FunctionRegistry, +) -> Result<()> { + Scope { functions }.plan(plan) +} + +struct Scope<'a> { + functions: &'a FunctionRegistry, +} + +impl Scope<'_> { + fn plan(&self, plan: &SqlPlan) -> Result<()> { + match plan { + SqlPlan::Scan { + filters, + projection, + sort_keys, + window_functions, + .. + } + | SqlPlan::DocumentIndexLookup { + filters, + projection, + sort_keys, + window_functions, + .. + } => { + self.filters(filters)?; + self.projection(projection)?; + self.sort_keys(sort_keys)?; + self.windows(window_functions) + } + SqlPlan::PointGet { projection, .. } | SqlPlan::RangeScan { projection, .. } => { + self.projection(projection) + } + SqlPlan::KvInsert { + on_conflict_updates, + .. + } + | SqlPlan::Upsert { + on_conflict_updates, + .. + } + | SqlPlan::VectorPrimaryInsert { + on_conflict_updates, + .. + } => self.assignments(on_conflict_updates), + SqlPlan::InsertSelect { + source, column_map, .. + } => { + self.plan(source)?; + self.assignments(column_map) + } + SqlPlan::Update { + assignments, + filters, + .. + } + | SqlPlan::VectorPrimaryUpdate { + assignments, + filters, + .. + } => { + self.assignments(assignments)?; + self.filters(filters) + } + SqlPlan::UpdateFrom { + source, + assignments, + target_filters, + .. + } => { + self.plan(source)?; + self.assignments(assignments)?; + self.filters(target_filters) + } + SqlPlan::Delete { filters, .. } | SqlPlan::VectorPrimaryDelete { filters, .. } => { + self.filters(filters) + } + SqlPlan::Join { + left, + right, + condition, + projection, + filters, + .. + } => { + self.plan(left)?; + self.plan(right)?; + if has_search_plan(left) || has_search_plan(right) { + return Ok(()); + } + if let Some(condition) = condition { + self.expr(condition)?; + } + self.projection(projection)?; + self.filters(filters) + } + SqlPlan::Aggregate { + input, + group_by, + aggregates, + having, + sort_keys, + .. + } => { + self.plan(input)?; + if has_search_plan(input) { + return Ok(()); + } + self.exprs(group_by)?; + self.aggregates(aggregates)?; + self.filters(having)?; + self.sort_keys(sort_keys) + } + SqlPlan::TimeseriesScan { + aggregates, + filters, + projection, + sort_keys, + .. + } => { + self.aggregates(aggregates)?; + self.filters(filters)?; + self.projection(projection)?; + self.sort_keys(sort_keys) + } + // A search plan serves its own score call as a column; only its + // residual filters run on the row evaluator. + SqlPlan::VectorSearch { filters, .. } | SqlPlan::TextSearch { filters, .. } => { + self.filters(filters) + } + SqlPlan::SpatialScan { + attribute_filters, .. + } => self.filters(attribute_filters), + SqlPlan::RecursiveScan { + base_filters, + recursive_filters, + .. + } => { + self.filters(base_filters)?; + self.filters(recursive_filters) + } + SqlPlan::Union { inputs, .. } => inputs.iter().try_for_each(|input| self.plan(input)), + SqlPlan::Intersect { left, right, .. } | SqlPlan::Except { left, right, .. } => { + self.plan(left)?; + self.plan(right) + } + SqlPlan::Cte { definitions, outer } => { + for (_, definition) in definitions { + self.plan(definition)?; + } + self.plan(outer) + } + SqlPlan::Subquery { + input, + filters, + projection, + window_functions, + sort_keys, + .. + } => { + self.plan(input)?; + if has_search_plan(input) { + return Ok(()); + } + self.filters(filters)?; + self.projection(projection)?; + self.windows(window_functions)?; + self.sort_keys(sort_keys) + } + SqlPlan::Merge { + source, clauses, .. + } => { + self.plan(source)?; + for clause in clauses { + self.filters(&clause.extra_predicate)?; + match &clause.action { + MergePlanAction::Update { assignments } => self.assignments(assignments)?, + MergePlanAction::Insert { values, .. } => self.exprs(values)?, + MergePlanAction::Delete | MergePlanAction::DoNothing => {} + } + } + Ok(()) + } + SqlPlan::LateralTopK { + outer, + inner_filters, + inner_order_by, + projection, + .. + } => { + self.plan(outer)?; + self.filters(inner_filters)?; + self.sort_keys(inner_order_by)?; + if has_search_plan(outer) { + return Ok(()); + } + self.projection(projection) + } + SqlPlan::LateralLoop { + outer, + inner, + projection, + .. + } => { + self.plan(outer)?; + self.plan(inner)?; + if has_search_plan(outer) || has_search_plan(inner) { + return Ok(()); + } + self.projection(projection) + } + // No row-evaluated expression: constants, literal-row writes, + // search plans that carry no residual filter, array statements, + // and DDL. + SqlPlan::ConstantResult { .. } + | SqlPlan::Insert { .. } + | SqlPlan::Truncate { .. } + | SqlPlan::TimeseriesIngest { .. } + | SqlPlan::MultiVectorSearch { .. } + | SqlPlan::SparseSearch { .. } + | SqlPlan::HybridSearch { .. } + | SqlPlan::HybridSearchTriple { .. } + | SqlPlan::RecursiveValue { .. } + | SqlPlan::CreateArray { .. } + | SqlPlan::DropArray { .. } + | SqlPlan::AlterArray { .. } + | SqlPlan::InsertArray { .. } + | SqlPlan::DeleteArray { .. } + | SqlPlan::ArraySlice { .. } + | SqlPlan::ArrayProject { .. } + | SqlPlan::ArrayAgg { .. } + | SqlPlan::ArrayElementwise { .. } + | SqlPlan::ArrayFlush { .. } + | SqlPlan::ArrayCompact { .. } + | SqlPlan::VectorPrimaryTruncate { .. } + | SqlPlan::CreateIndex { .. } + | SqlPlan::DropIndex { .. } => Ok(()), + } + } + + fn filters(&self, filters: &[Filter]) -> Result<()> { + filters + .iter() + .try_for_each(|filter| self.filter(&filter.expr)) + } + + fn filter(&self, expr: &FilterExpr) -> Result<()> { + match expr { + FilterExpr::Expr(expr) => self.expr(expr), + FilterExpr::And(children) | FilterExpr::Or(children) => self.filters(children), + FilterExpr::Not(child) => self.filter(&child.expr), + FilterExpr::Comparison { .. } + | FilterExpr::InList { .. } + | FilterExpr::Between { .. } + | FilterExpr::IsNull { .. } + | FilterExpr::IsNotNull { .. } => Ok(()), + } + } + + fn projection(&self, projection: &[Projection]) -> Result<()> { + for item in projection { + match item { + Projection::Computed { expr, .. } | Projection::CpComputed { expr, .. } => { + self.expr(expr)? + } + Projection::Column(_) | Projection::Star | Projection::QualifiedStar(_) => {} + } + } + Ok(()) + } + + fn sort_keys(&self, sort_keys: &[SortKey]) -> Result<()> { + sort_keys.iter().try_for_each(|key| self.expr(&key.expr)) + } + + fn windows(&self, windows: &[WindowSpec]) -> Result<()> { + for window in windows { + self.exprs(&window.args)?; + self.exprs(&window.partition_by)?; + self.sort_keys(&window.order_by)?; + } + Ok(()) + } + + fn aggregates(&self, aggregates: &[AggregateExpr]) -> Result<()> { + aggregates + .iter() + .try_for_each(|aggregate| self.exprs(&aggregate.args)) + } + + fn assignments(&self, assignments: &[(String, SqlExpr)]) -> Result<()> { + assignments.iter().try_for_each(|(_, expr)| self.expr(expr)) + } + + fn exprs(&self, exprs: &[SqlExpr]) -> Result<()> { + exprs.iter().try_for_each(|expr| self.expr(expr)) + } + + fn expr(&self, expr: &SqlExpr) -> Result<()> { + match first_search_function(expr, self.functions) { + Some(name) => Err(SqlError::SearchFunctionOutsideSearch { + name: name.to_owned(), + }), + None => Ok(()), + } + } +} + +/// Whether the search trigger `trigger` names a function that reads an +/// index and has no per-row value. +fn is_index_owned(trigger: SearchTrigger) -> bool { + match trigger { + SearchTrigger::MultiVectorSearch + | SearchTrigger::SparseSearch + | SearchTrigger::TextSearch + | SearchTrigger::HybridSearch + | SearchTrigger::TextMatch + | SearchTrigger::GraphSearch => true, + // The vector distances and the spatial predicates evaluate per row; + // the time bucket is a scalar; the array functions are table-valued + // and planned from FROM. + SearchTrigger::None + | SearchTrigger::VectorSearch + | SearchTrigger::SpatialDWithin + | SearchTrigger::SpatialContains + | SearchTrigger::SpatialIntersects + | SearchTrigger::SpatialWithin + | SearchTrigger::TimeBucket + | SearchTrigger::ArraySlice + | SearchTrigger::ArrayProject + | SearchTrigger::ArrayAgg + | SearchTrigger::ArrayElementwise + | SearchTrigger::ArrayFlush + | SearchTrigger::ArrayCompact => false, + } +} + +/// The first index-owned search function `expr` calls outside a subquery. +fn first_search_function<'e>(expr: &'e SqlExpr, functions: &FunctionRegistry) -> Option<&'e str> { + let find = |e: &'e SqlExpr| first_search_function(e, functions); + match expr { + SqlExpr::Function { name, args, .. } => { + if is_index_owned(functions.search_trigger(name)) { + return Some(name.as_str()); + } + args.iter().find_map(find) + } + SqlExpr::BinaryOp { left, right, .. } => find(left).or_else(|| find(right)), + SqlExpr::UnaryOp { expr, .. } + | SqlExpr::Cast { expr, .. } + | SqlExpr::IsNull { expr, .. } => find(expr), + SqlExpr::Case { + operand, + when_then, + else_expr, + } => operand + .as_deref() + .and_then(find) + .or_else(|| { + when_then + .iter() + .find_map(|(when, then)| find(when).or_else(|| find(then))) + }) + .or_else(|| else_expr.as_deref().and_then(find)), + SqlExpr::InList { expr, list, .. } => find(expr).or_else(|| list.iter().find_map(find)), + SqlExpr::Between { + expr, low, high, .. + } => find(expr).or_else(|| find(low)).or_else(|| find(high)), + SqlExpr::Like { expr, pattern, .. } => find(expr).or_else(|| find(pattern)), + SqlExpr::ArrayLiteral(items) => items.iter().find_map(find), + SqlExpr::Column { .. } | SqlExpr::Literal(_) | SqlExpr::Subquery(_) | SqlExpr::Wildcard => { + None + } + } +} + +/// Whether `plan` is, or wraps, a search plan that serves a score column. +fn has_search_plan(plan: &SqlPlan) -> bool { + match plan { + SqlPlan::VectorSearch { .. } + | SqlPlan::MultiVectorSearch { .. } + | SqlPlan::SparseSearch { .. } + | SqlPlan::TextSearch { .. } + | SqlPlan::HybridSearch { .. } + | SqlPlan::HybridSearchTriple { .. } => true, + SqlPlan::Subquery { input, .. } | SqlPlan::Aggregate { input, .. } => { + has_search_plan(input) + } + SqlPlan::Join { left, right, .. } => has_search_plan(left) || has_search_plan(right), + SqlPlan::LateralTopK { outer, .. } => has_search_plan(outer), + SqlPlan::LateralLoop { outer, inner, .. } => { + has_search_plan(outer) || has_search_plan(inner) + } + SqlPlan::Union { inputs, .. } => inputs.iter().any(has_search_plan), + SqlPlan::Intersect { left, right, .. } | SqlPlan::Except { left, right, .. } => { + has_search_plan(left) || has_search_plan(right) + } + SqlPlan::Cte { outer, .. } => has_search_plan(outer), + SqlPlan::ConstantResult { .. } + | SqlPlan::Scan { .. } + | SqlPlan::PointGet { .. } + | SqlPlan::DocumentIndexLookup { .. } + | SqlPlan::RangeScan { .. } + | SqlPlan::Insert { .. } + | SqlPlan::KvInsert { .. } + | SqlPlan::Upsert { .. } + | SqlPlan::InsertSelect { .. } + | SqlPlan::Update { .. } + | SqlPlan::UpdateFrom { .. } + | SqlPlan::Delete { .. } + | SqlPlan::Truncate { .. } + | SqlPlan::TimeseriesScan { .. } + | SqlPlan::TimeseriesIngest { .. } + | SqlPlan::SpatialScan { .. } + | SqlPlan::RecursiveScan { .. } + | SqlPlan::RecursiveValue { .. } + | SqlPlan::CreateArray { .. } + | SqlPlan::DropArray { .. } + | SqlPlan::AlterArray { .. } + | SqlPlan::InsertArray { .. } + | SqlPlan::DeleteArray { .. } + | SqlPlan::ArraySlice { .. } + | SqlPlan::ArrayProject { .. } + | SqlPlan::ArrayAgg { .. } + | SqlPlan::ArrayElementwise { .. } + | SqlPlan::ArrayFlush { .. } + | SqlPlan::ArrayCompact { .. } + | SqlPlan::Merge { .. } + | SqlPlan::VectorPrimaryInsert { .. } + | SqlPlan::VectorPrimaryDelete { .. } + | SqlPlan::VectorPrimaryTruncate { .. } + | SqlPlan::VectorPrimaryUpdate { .. } + | SqlPlan::CreateIndex { .. } + | SqlPlan::DropIndex { .. } => false, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::types::query::EngineType; + use crate::types_expr::SqlValue; + + fn call(name: &str) -> SqlExpr { + SqlExpr::Function { + name: name.into(), + args: vec![ + SqlExpr::Column { + table: None, + name: "body".into(), + }, + SqlExpr::Literal(SqlValue::String("rust".into())), + ], + distinct: false, + } + } + + fn scan(projection: Vec, filters: Vec) -> SqlPlan { + SqlPlan::Scan { + collection: "docs".into(), + alias: None, + engine: EngineType::DocumentSchemaless, + filters, + projection, + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + window_functions: Vec::new(), + temporal: Default::default(), + } + } + + fn check(plan: &SqlPlan) -> Result<()> { + refuse_row_scoped_search_functions(plan, &FunctionRegistry::new()) + } + + #[test] + fn a_score_in_a_scan_projection_is_refused() { + let plan = scan( + vec![Projection::Computed { + expr: call("bm25_score"), + alias: "s".into(), + }], + Vec::new(), + ); + assert_eq!( + check(&plan), + Err(SqlError::SearchFunctionOutsideSearch { + name: "bm25_score".into() + }) + ); + } + + #[test] + fn a_match_nested_in_a_scan_filter_is_refused() { + let nested = SqlExpr::BinaryOp { + left: Box::new(call("text_match")), + op: crate::types_expr::BinaryOp::Or, + right: Box::new(SqlExpr::Literal(SqlValue::Bool(false))), + }; + let plan = scan( + Vec::new(), + vec![Filter { + expr: FilterExpr::Expr(nested), + }], + ); + assert!(matches!( + check(&plan), + Err(SqlError::SearchFunctionOutsideSearch { .. }) + )); + } + + #[test] + fn row_scalars_in_a_scan_pass() { + let plan = scan( + vec![ + Projection::Computed { + expr: call("vector_distance"), + alias: "d".into(), + }, + Projection::Computed { + expr: call("doc_get"), + alias: "g".into(), + }, + ], + Vec::new(), + ); + assert_eq!(check(&plan), Ok(())); + } + + #[test] + fn a_search_plan_projection_serves_its_score() { + let plan = SqlPlan::TextSearch { + collection: "docs".into(), + query: crate::fts_types::FtsQuery::Plain { + text: "rust".into(), + fuzzy: true, + }, + top_k: 10, + filters: Vec::new(), + score_alias: Some("s".into()), + projection: vec![Projection::Computed { + expr: call("bm25_score"), + alias: "s".into(), + }], + }; + assert_eq!(check(&plan), Ok(())); + } + + #[test] + fn a_subquery_tail_over_a_search_plan_is_not_checked() { + let search = SqlPlan::TextSearch { + collection: "docs".into(), + query: crate::fts_types::FtsQuery::Plain { + text: "rust".into(), + fuzzy: true, + }, + top_k: 10, + filters: Vec::new(), + score_alias: Some("s".into()), + projection: Vec::new(), + }; + let plan = SqlPlan::Subquery { + input: Box::new(search), + filters: Vec::new(), + projection: vec![Projection::Computed { + expr: call("bm25_score"), + alias: "s".into(), + }], + window_functions: Vec::new(), + sort_keys: Vec::new(), + offset: 0, + distinct: false, + limit: None, + }; + assert_eq!(check(&plan), Ok(())); + } +} diff --git a/nodedb-sql/src/planner/select/order_by/projection.rs b/nodedb-sql/src/planner/select/order_by/projection.rs index 9ed17389b..9b99f3fb3 100644 --- a/nodedb-sql/src/planner/select/order_by/projection.rs +++ b/nodedb-sql/src/planner/select/order_by/projection.rs @@ -7,8 +7,8 @@ //! `bm25_score(...)` call may still appear directly in the SELECT projection. //! The canonical shape `SELECT id, rrf_score(...) AS score FROM c WHERE ... LIMIT N` //! requires this entry path because there is no ORDER BY clause to inspect. -//! Without it the score column resolves to NULL via scalar evaluation that -//! has no implementation. +//! A score call no search plan serves is refused at plan time +//! (`planner::search_scope`): it has no per-row value. //! //! Text-search shape: `SELECT id, bm25_score(field, term) FROM c ORDER BY id`. //! The plan stays a Scan after ORDER BY (non-search sort key). This pass diff --git a/nodedb-types/src/error/code.rs b/nodedb-types/src/error/code.rs index 2ff0461b7..6a462952a 100644 --- a/nodedb-types/src/error/code.rs +++ b/nodedb-types/src/error/code.rs @@ -67,6 +67,9 @@ impl ErrorCode { pub const UNDEFINED_COLUMN: Self = Self(1206); /// A bare column name resolves against more than one relation in scope. pub const AMBIGUOUS_COLUMN: Self = Self(1207); + /// A function received a value it cannot compute on: a vector of the + /// wrong dimension, an argument of the wrong shape, a malformed path. + pub const DATA_EXCEPTION: Self = Self(1208); // Engine ops (1300–1399) pub const ARRAY: Self = Self(1300); diff --git a/nodedb-types/src/error/code_table.rs b/nodedb-types/src/error/code_table.rs index f9fdfdc14..11524d5de 100644 --- a/nodedb-types/src/error/code_table.rs +++ b/nodedb-types/src/error/code_table.rs @@ -95,6 +95,7 @@ error_code_table! { UNDEFINED_COLUMN => UndefinedColumn { column: String::new() }, AMBIGUOUS_COLUMN => AmbiguousColumn { column: String::new() }, DIVISION_BY_ZERO => DivisionByZero, + DATA_EXCEPTION => DataException { detail: message.to_owned() }, INVALID_LIMIT_VALUE => InvalidLimitValue { clause: "remote".into(), value: message.to_owned() }, // Auth / tenant quota. diff --git a/nodedb-types/src/error/ctors/read_query_auth.rs b/nodedb-types/src/error/ctors/read_query_auth.rs index 9ac2d70a5..9bf562b46 100644 --- a/nodedb-types/src/error/ctors/read_query_auth.rs +++ b/nodedb-types/src/error/ctors/read_query_auth.rs @@ -199,6 +199,19 @@ impl NodeDbError { } } + /// A function received a value it cannot compute on: a vector of the + /// wrong dimension, an argument of the wrong shape, a malformed path. + /// SQLSTATE `22000` (`data_exception`). `detail` is the full message. + pub fn data_exception(detail: impl Into) -> Self { + let detail = detail.into(); + Self { + code: ErrorCode::DATA_EXCEPTION, + message: detail.clone(), + details: ErrorDetails::DataException { detail }, + cause: None, + } + } + /// A LIMIT/OFFSET/FETCH bound did not resolve to `[0, usize::MAX]`. /// Distinct from `plan_error` so clients match the code, SQLSTATE /// `2201W`, instead of parsing the message. diff --git a/nodedb-types/src/error/details.rs b/nodedb-types/src/error/details.rs index fbf0e4fc4..3286eee9d 100644 --- a/nodedb-types/src/error/details.rs +++ b/nodedb-types/src/error/details.rs @@ -123,6 +123,10 @@ pub enum ErrorDetails { /// Expression evaluation divided or took a modulus by zero. #[serde(rename = "division_by_zero")] DivisionByZero, + /// A function received a value it cannot compute on. `detail` names + /// the function and the offending value. + #[serde(rename = "data_exception")] + DataException { detail: String }, /// A LIMIT/OFFSET/FETCH bound resolved outside `[0, usize::MAX]`. #[serde(rename = "invalid_limit_value")] InvalidLimitValue { clause: String, value: String }, diff --git a/nodedb-types/src/error/msgpack/constants.rs b/nodedb-types/src/error/msgpack/constants.rs index 9627de4b7..e6625c049 100644 --- a/nodedb-types/src/error/msgpack/constants.rs +++ b/nodedb-types/src/error/msgpack/constants.rs @@ -85,6 +85,7 @@ // | 79 | UndefinedColumn | // | 80 | AmbiguousColumn | // | 81 | PeriodLockMisconfigured | +// | 82 | DataException | pub(super) const TAG_CONSTRAINT_VIOLATION: u16 = 1; pub(super) const TAG_WRITE_CONFLICT: u16 = 2; @@ -167,3 +168,4 @@ pub(super) const TAG_INVALID_LIMIT_VALUE: u16 = 78; pub(super) const TAG_UNDEFINED_COLUMN: u16 = 79; pub(super) const TAG_AMBIGUOUS_COLUMN: u16 = 80; pub(super) const TAG_PERIOD_LOCK_MISCONFIGURED: u16 = 81; +pub(super) const TAG_DATA_EXCEPTION: u16 = 82; diff --git a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs index 5666fcd4d..0dc316edd 100644 --- a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs +++ b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs @@ -155,6 +155,10 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { skip_fields(reader, field_count)?; Ok(ErrorDetails::DivisionByZero) } + TAG_DATA_EXCEPTION => { + let (detail,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::DataException { detail }) + } TAG_INVALID_LIMIT_VALUE => { let (clause, value) = read2_str(reader, field_count)?; Ok(ErrorDetails::InvalidLimitValue { clause, value }) @@ -500,6 +504,14 @@ mod tests { assert_eq!(roundtrip(&v), v); } + #[test] + fn data_exception_roundtrip() { + let v = ErrorDetails::DataException { + detail: "vector_distance(): vector dimension mismatch: expected 3, got 2".into(), + }; + assert_eq!(roundtrip(&v), v); + } + #[test] fn bridge_enriched_roundtrip() { let v = ErrorDetails::Bridge { diff --git a/nodedb-types/src/error/msgpack/encode.rs b/nodedb-types/src/error/msgpack/encode.rs index dfb983f48..7c36827b1 100644 --- a/nodedb-types/src/error/msgpack/encode.rs +++ b/nodedb-types/src/error/msgpack/encode.rs @@ -178,6 +178,7 @@ impl ToMessagePack for ErrorDetails { write1(writer, TAG_AMBIGUOUS_COLUMN, column) } ErrorDetails::DivisionByZero => write_unit(writer, TAG_DIVISION_BY_ZERO), + ErrorDetails::DataException { detail } => write1(writer, TAG_DATA_EXCEPTION, detail), ErrorDetails::InvalidLimitValue { clause, value } => { write2(writer, TAG_INVALID_LIMIT_VALUE, clause, value) } diff --git a/nodedb-types/src/error/sqlstate.rs b/nodedb-types/src/error/sqlstate.rs index 778b1c432..422884d27 100644 --- a/nodedb-types/src/error/sqlstate.rs +++ b/nodedb-types/src/error/sqlstate.rs @@ -49,6 +49,11 @@ pub const CANNOT_DROP_DEFAULT_DATABASE: AmbiguousSqlstate = AmbiguousSqlstate("0 // ── Class 22 — Data Exception ──────────────────────────────────────────────── +/// `22000` — `data_exception` (a function received a value it cannot compute +/// on: a vector of the wrong dimension, an argument of the wrong shape, a +/// malformed JSONPath) +pub const DATA_EXCEPTION: &str = "22000"; + /// `22003` — `numeric_value_out_of_range` pub const NUMERIC_VALUE_OUT_OF_RANGE: &str = "22003"; @@ -351,6 +356,7 @@ mod tests { WARNING, NO_DATA, FEATURE_NOT_SUPPORTED, + DATA_EXCEPTION, NUMERIC_VALUE_OUT_OF_RANGE, DIVISION_BY_ZERO, INVALID_LIMIT_VALUE, diff --git a/nodedb-types/src/error/types.rs b/nodedb-types/src/error/types.rs index be809a036..b967ae3df 100644 --- a/nodedb-types/src/error/types.rs +++ b/nodedb-types/src/error/types.rs @@ -102,6 +102,7 @@ impl NodeDbError { | ErrorDetails::UndefinedColumn { .. } | ErrorDetails::AmbiguousColumn { .. } | ErrorDetails::DivisionByZero + | ErrorDetails::DataException { .. } | ErrorDetails::InvalidLimitValue { .. } | ErrorDetails::BackupTenantMismatch { .. } | ErrorDetails::BackupKeyMismatch diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index 2d3d8f7cf..f70d38443 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -150,6 +150,14 @@ pub enum ErrorCode { /// special-cases `NotFound`) and reaches the client as SQLSTATE `22012` /// rather than the generic `XX000` every `Internal` maps to. DivisionByZero, + /// Expression evaluation called a function no evaluator implements. + /// Surfaces as SQLSTATE `42883` (`undefined_function`), never as a + /// silent `NULL`. + UndefinedFunction { name: String }, + /// A function received an argument it cannot compute on: vectors of + /// different dimensions, an argument of the wrong type, a malformed + /// JSONPath. Surfaces as SQLSTATE `22000` (`data_exception`). + DataException { detail: String }, /// The bridge dispatcher refused the request at a capacity limit, so /// nothing was enqueued or applied. Transient: the same request succeeds /// once capacity frees. `reason` names the limit and its counts. @@ -161,6 +169,24 @@ pub enum ErrorCode { ExpiredBeforeExecution, } +/// An expression evaluation failure, as the Data Plane reports it. +/// +/// Exhaustive, so a new evaluator error picks its own code rather than +/// defaulting to one. +impl From for ErrorCode { + fn from(e: nodedb_query::EvalError) -> Self { + match e { + nodedb_query::EvalError::DivisionByZero => Self::DivisionByZero, + nodedb_query::EvalError::UnknownFunction { name } => Self::UndefinedFunction { name }, + e @ (nodedb_query::EvalError::VectorDimensionMismatch { .. } + | nodedb_query::EvalError::ArgumentType { .. } + | nodedb_query::EvalError::InvalidJsonPath { .. }) => Self::DataException { + detail: e.to_string(), + }, + } + } +} + impl From for ErrorCode { fn from(e: crate::Error) -> Self { match e { @@ -253,6 +279,8 @@ impl From for ErrorCode { Self::TxnOverlayMemoryExceeded { limit } } crate::Error::DivisionByZero => Self::DivisionByZero, + crate::Error::UndefinedFunction { name } => Self::UndefinedFunction { name }, + crate::Error::DataException { detail } => Self::DataException { detail }, crate::Error::UndefinedColumn { column } => Self::UndefinedColumn { column }, // Same condition an undefined column reports at plan time, raised // here by the strict encoder for a transport the planner never diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index b99150b1d..662d7c879 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -176,6 +176,8 @@ impl From for DataPlaneErrorCode { limit: to_wire_count(limit), }, ErrorCode::DivisionByZero => Self::DivisionByZero, + ErrorCode::UndefinedFunction { name } => Self::UndefinedFunction { name }, + ErrorCode::DataException { detail } => Self::DataException { detail }, ErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, ErrorCode::ExpiredBeforeExecution => Self::ExpiredBeforeExecution, } @@ -303,6 +305,8 @@ impl From for ErrorCode { } } DataPlaneErrorCode::DivisionByZero => Self::DivisionByZero, + DataPlaneErrorCode::UndefinedFunction { name } => Self::UndefinedFunction { name }, + DataPlaneErrorCode::DataException { detail } => Self::DataException { detail }, DataPlaneErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, DataPlaneErrorCode::ExpiredBeforeExecution => Self::ExpiredBeforeExecution, } diff --git a/nodedb/src/control/gateway/error_map/pgwire.rs b/nodedb/src/control/gateway/error_map/pgwire.rs index cbae262d0..f643056b7 100644 --- a/nodedb/src/control/gateway/error_map/pgwire.rs +++ b/nodedb/src/control/gateway/error_map/pgwire.rs @@ -106,6 +106,9 @@ mod tests { column: "x".into(), }, Error::DivisionByZero, + Error::DataException { + detail: "vector_distance(): vector dimension mismatch: expected 3, got 2".into(), + }, Error::InvalidLimitValue { clause: "LIMIT", value: "-1".into(), diff --git a/nodedb/src/control/planner/plan_error_map.rs b/nodedb/src/control/planner/plan_error_map.rs index 1e9dd8dbc..796b2fc3d 100644 --- a/nodedb/src/control/planner/plan_error_map.rs +++ b/nodedb/src/control/planner/plan_error_map.rs @@ -35,9 +35,11 @@ pub(crate) fn map_plan_error( nodedb_sql::SqlError::UndefinedFunction { name } => { crate::Error::UndefinedFunction { name } } - // A per-row sequence accessor is a refusal, not a syntax error, so it - // keeps SQLSTATE `0A000` rather than the `42601` the fallback gives. - nodedb_sql::SqlError::SequencePerRowUnsupported { .. } => { + // A per-row sequence accessor and a search function outside its + // search plan are refusals, not syntax errors, so they keep SQLSTATE + // `0A000` rather than the `42601` the fallback gives. + nodedb_sql::SqlError::SequencePerRowUnsupported { .. } + | nodedb_sql::SqlError::SearchFunctionOutsideSearch { .. } => { crate::Error::FeatureNotSupported { detail: error.to_string(), } @@ -51,6 +53,7 @@ pub(crate) fn map_plan_error( // A constant expression that divides by zero is the same condition the // row-scope evaluator raises, so it carries the same code. nodedb_sql::SqlError::DivisionByZero => crate::Error::DivisionByZero, + nodedb_sql::SqlError::DataException { detail } => crate::Error::DataException { detail }, nodedb_sql::SqlError::InvalidLimitValue { clause, value } => { crate::Error::InvalidLimitValue { clause, value } } diff --git a/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs b/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs index 4924eba8b..f90ab6949 100644 --- a/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs +++ b/nodedb/src/control/planner/sql_plan_convert/expr/bridge_expr.rs @@ -235,31 +235,99 @@ fn convert_expr_inner(expr: &SqlExpr, qualify: bool) -> crate::bridge::expr_eval } } - // `ARRAY['a', 'b', ...]` — lower each element and, when all resolve to - // `BExpr::Literal`, fold into a single `Value::Array` literal so that - // functions like `pg_json_has_any_key` / `pg_json_has_all_keys` receive - // a proper `Value::Array` argument rather than `Value::Null`. + // `ARRAY[a, b, ...]`: an array of literals folds to one `Value::Array` + // literal, so functions like `pg_json_has_any_key` receive a real + // array argument. An array with any other element lowers to a + // `make_array` call, which builds the array per row from the + // evaluated elements. SqlExpr::ArrayLiteral(elems) => { - let mut values = Vec::with_capacity(elems.len()); - let mut all_literal = true; - for elem in elems { - match convert_expr_inner(elem, qualify) { - BExpr::Literal(v) => values.push(v), - other => { - all_literal = false; - // Non-literal element: fall back to Null for that slot. - let _ = other; - values.push(nodedb_types::Value::Null); + let lowered: Vec = elems + .iter() + .map(|elem| convert_expr_inner(elem, qualify)) + .collect(); + let literals: Option> = lowered + .iter() + .map(|elem| { + if let BExpr::Literal(v) = elem { + Some(v.clone()) + } else { + None } - } - } - if all_literal { - BExpr::Literal(nodedb_types::Value::Array(values)) - } else { - BExpr::Literal(nodedb_types::Value::Null) + }) + .collect(); + match literals { + Some(values) => BExpr::Literal(nodedb_types::Value::Array(values)), + None => BExpr::Function { + name: "make_array".into(), + args: lowered, + }, } } _ => BExpr::Literal(nodedb_types::Value::Null), } } + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use nodedb_sql::types::SqlValue; + use nodedb_types::Value; + + use super::*; + + fn col(name: &str) -> SqlExpr { + SqlExpr::Column { + table: None, + name: name.into(), + } + } + + fn row() -> Value { + Value::Object(HashMap::from([ + ("n".to_string(), Value::Integer(7)), + ("name".to_string(), Value::String("Alice".into())), + ])) + } + + #[test] + fn an_array_with_a_column_element_is_built_per_row() { + let expr = SqlExpr::ArrayLiteral(vec![col("n"), SqlExpr::Literal(SqlValue::Int(1))]); + let lowered = sql_expr_to_bridge_expr(&expr); + assert_eq!( + lowered.eval(&row()).expect("eval"), + Value::Array(vec![Value::Integer(7), Value::Integer(1)]) + ); + } + + #[test] + fn an_array_of_literals_folds_to_one_literal() { + let expr = SqlExpr::ArrayLiteral(vec![ + SqlExpr::Literal(SqlValue::Int(1)), + SqlExpr::Literal(SqlValue::Int(2)), + ]); + assert_eq!( + sql_expr_to_bridge_expr(&expr), + crate::bridge::expr_eval::SqlExpr::Literal(Value::Array(vec![ + Value::Integer(1), + Value::Integer(2) + ])) + ); + } + + #[test] + fn like_and_not_like_evaluate_through_the_like_function() { + let like = |negated, case_insensitive, pattern: &str| SqlExpr::Like { + expr: Box::new(col("name")), + pattern: Box::new(SqlExpr::Literal(SqlValue::String(pattern.into()))), + negated, + case_insensitive, + }; + let eval = |e: SqlExpr| sql_expr_to_bridge_expr(&e).eval(&row()).expect("eval"); + assert_eq!(eval(like(false, false, "Al%")), Value::Bool(true)); + assert_eq!(eval(like(true, false, "Al%")), Value::Bool(false)); + assert_eq!(eval(like(false, true, "al%")), Value::Bool(true)); + assert_eq!(eval(like(false, false, "al%")), Value::Bool(false)); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index e580e2323..1bf261f09 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -110,6 +110,8 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { | ErrorCode::TxnOverlayMemoryExceeded { .. } // Expression evaluation failed before producing a value to write. | ErrorCode::DivisionByZero + | ErrorCode::UndefinedFunction { .. } + | ErrorCode::DataException { .. } | ErrorCode::UndefinedColumn { .. } => true, // NOT established — every one of these can be reported by a request diff --git a/nodedb/src/control/server/http/routes/result_shape.rs b/nodedb/src/control/server/http/routes/result_shape.rs index a3022bd33..50fcbf793 100644 --- a/nodedb/src/control/server/http/routes/result_shape.rs +++ b/nodedb/src/control/server/http/routes/result_shape.rs @@ -33,6 +33,7 @@ pub(super) fn shape_error_to_api(e: NodeDbError) -> ApiError { | ErrorCode::OBJECT_NOT_READY | ErrorCode::PLAN_ERROR | ErrorCode::DIVISION_BY_ZERO + | ErrorCode::DATA_EXCEPTION | ErrorCode::BAD_REQUEST ) { StatusCode::BAD_REQUEST diff --git a/nodedb/src/control/server/native/sqlstate_code.rs b/nodedb/src/control/server/native/sqlstate_code.rs index 7c6c9d721..6f9f23212 100644 --- a/nodedb/src/control/server/native/sqlstate_code.rs +++ b/nodedb/src/control/server/native/sqlstate_code.rs @@ -68,6 +68,7 @@ pub(crate) fn ndb_code_for_sqlstate(sqlstate_str: &str) -> u16 { sqlstate::UNDEFINED_FUNCTION => ErrorCode::UNDEFINED_FUNCTION, sqlstate::UNDEFINED_COLUMN => ErrorCode::UNDEFINED_COLUMN, sqlstate::AMBIGUOUS_COLUMN => ErrorCode::AMBIGUOUS_COLUMN, + sqlstate::DATA_EXCEPTION => ErrorCode::DATA_EXCEPTION, // Both a malformed request and a plan that cannot be built render as // `42601`, so this cannot say which. It does not have to: the two // differ in which side wrote the bad statement, not in how a client diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 553ed1633..ea60089bf 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -105,6 +105,9 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str ("ERROR", sqlstate::UNDEFINED_COLUMN, err.to_string()) } crate::Error::DivisionByZero => ("ERROR", sqlstate::DIVISION_BY_ZERO, err.to_string()), + crate::Error::DataException { detail } => { + ("ERROR", sqlstate::DATA_EXCEPTION, detail.clone()) + } crate::Error::InvalidLimitValue { .. } => { ("ERROR", sqlstate::INVALID_LIMIT_VALUE, err.to_string()) } @@ -279,6 +282,8 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> Ec::AMBIGUOUS_COLUMN => sqlstate::AMBIGUOUS_COLUMN, // Mirrors the `DivisionByZero` arm. Ec::DIVISION_BY_ZERO => sqlstate::DIVISION_BY_ZERO, + // Mirrors the `DataException` arm. + Ec::DATA_EXCEPTION => sqlstate::DATA_EXCEPTION, // Mirrors the `InvalidLimitValue` arm. Ec::INVALID_LIMIT_VALUE => sqlstate::INVALID_LIMIT_VALUE, // Mirrors the `FanOutExceeded` arm. diff --git a/nodedb/src/control/server/shared/ddl/result.rs b/nodedb/src/control/server/shared/ddl/result.rs index 460a9512a..4e3102882 100644 --- a/nodedb/src/control/server/shared/ddl/result.rs +++ b/nodedb/src/control/server/shared/ddl/result.rs @@ -191,6 +191,7 @@ pub fn code_for_sqlstate(sqlstate_str: &str) -> ErrorCode { sqlstate::UNDEFINED_FUNCTION => ErrorCode::UNDEFINED_FUNCTION, sqlstate::UNDEFINED_COLUMN => ErrorCode::UNDEFINED_COLUMN, sqlstate::AMBIGUOUS_COLUMN => ErrorCode::AMBIGUOUS_COLUMN, + sqlstate::DATA_EXCEPTION => ErrorCode::DATA_EXCEPTION, // A malformed request and a plan that cannot be built both render as // `42601`; both are non-retriable client errors, so one code covers // both without losing anything a client acts on. diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index 630e2a889..234031df9 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -199,6 +199,12 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, sqlstate::DIVISION_BY_ZERO, "division by zero".into(), ), + ErrorCode::UndefinedFunction { name } => ( + "ERROR", + sqlstate::UNDEFINED_FUNCTION, + format!("function {name}() does not exist"), + ), + ErrorCode::DataException { detail } => ("ERROR", sqlstate::DATA_EXCEPTION, detail.clone()), // Transient: the client retries after a backoff. ErrorCode::DispatchCapacity { reason } => { ("ERROR", sqlstate::SERVER_OVERLOAD, reason.clone()) diff --git a/nodedb/src/data/executor/enforcement/transition_check.rs b/nodedb/src/data/executor/enforcement/transition_check.rs index 1925bd058..b7f22d273 100644 --- a/nodedb/src/data/executor/enforcement/transition_check.rs +++ b/nodedb/src/data/executor/enforcement/transition_check.rs @@ -25,16 +25,15 @@ pub fn check_transition_predicates( let old_val = nodedb_types::Value::from(old_doc.clone()); let new_val = nodedb_types::Value::from(new_doc.clone()); for check in checks { - // A division/modulo-by-zero inside the predicate is neither a PASS - // nor an ordinary FAIL — it's an evaluation error. `eval_with_old` - // can only fail with `EvalError::DivisionByZero`, so it surfaces as - // SQLSTATE 22012 (`ErrorCode::DivisionByZero`), matching generated - // columns / materialized-sum enforcement and Postgres, rather than - // being reported under this check's own 23xxx violation code. + // An evaluation error inside the predicate is neither a PASS nor an + // ordinary FAIL. It surfaces under its own SQLSTATE (22012 for a + // division by zero, 42883 for an unknown function), matching + // generated columns / materialized-sum enforcement and Postgres, + // rather than under this check's own 23xxx violation code. let result = check .predicate .eval_with_old(&new_val, &old_val) - .map_err(|_e| ErrorCode::DivisionByZero)?; + .map_err(ErrorCode::from)?; let passed = match result { nodedb_types::Value::Bool(b) => b, nodedb_types::Value::Null => false, // NULL treated as FALSE for constraint purposes. diff --git a/nodedb/src/data/executor/enforcement/typeguard.rs b/nodedb/src/data/executor/enforcement/typeguard.rs index 2f519a153..d81bfa795 100644 --- a/nodedb/src/data/executor/enforcement/typeguard.rs +++ b/nodedb/src/data/executor/enforcement/typeguard.rs @@ -39,10 +39,10 @@ pub fn inject_defaults( } })?; let doc = nodedb_types::Value::Object(fields.clone()); - // A division/modulo-by-zero — the only error `eval` can produce — - // surfaces as SQLSTATE 22012, not this guard's own violation - // code, matching generated-column enforcement. - let val = expr.eval(&doc).map_err(|_e| ErrorCode::DivisionByZero)?; + // An evaluation error (division by zero, an unknown function) + // surfaces as its own SQLSTATE, not this guard's violation code, + // matching generated-column enforcement. + let val = expr.eval(&doc).map_err(ErrorCode::from)?; fields.insert(guard.field.clone(), val); } // DEFAULT: inject only if absent or null. @@ -60,9 +60,9 @@ pub fn inject_defaults( ), })?; let doc = nodedb_types::Value::Object(fields.clone()); - // Div-by-zero → SQLSTATE 22012, not this guard's violation - // code (see the VALUE arm above). - let val = expr.eval(&doc).map_err(|_e| ErrorCode::DivisionByZero)?; + // An evaluation error keeps its own SQLSTATE, not this + // guard's violation code (see the VALUE arm above). + let val = expr.eval(&doc).map_err(ErrorCode::from)?; fields.insert(guard.field.clone(), val); } } @@ -181,11 +181,9 @@ pub fn check_type_guards( guard.field, check_str ), })?; - // Div-by-zero → SQLSTATE 22012, not this guard's violation code, - // matching CHECK-constraint enforcement elsewhere. - let result = check_expr - .eval(doc) - .map_err(|_e| ErrorCode::DivisionByZero)?; + // An evaluation error keeps its own SQLSTATE, not this guard's + // violation code, matching CHECK-constraint enforcement elsewhere. + let result = check_expr.eval(doc).map_err(ErrorCode::from)?; match result { nodedb_types::Value::Bool(true) => {} // CHECK passed nodedb_types::Value::Null => {} // NULL passes CHECK (SQL semantics) diff --git a/nodedb/src/data/executor/handlers/aggregate/exec.rs b/nodedb/src/data/executor/handlers/aggregate/exec.rs index 2fa0fa50a..71169442c 100644 --- a/nodedb/src/data/executor/handlers/aggregate/exec.rs +++ b/nodedb/src/data/executor/handlers/aggregate/exec.rs @@ -286,28 +286,28 @@ impl CoreLoop { } }; if !having_predicates.is_empty() { - // `Vec::retain`'s closure must return `bool`, so a - // division/modulo-by-zero in a HAVING predicate is - // captured via this `Cell` side-channel and checked - // once the retain finishes — HAVING is WHERE-shaped, - // so it gets the full error treatment. - let predicate_err: std::cell::Cell> = - std::cell::Cell::new(None); + // `Vec::retain`'s closure must return `bool`, so an + // evaluation error in a HAVING predicate is captured + // via this side-channel and checked once the retain + // finishes. HAVING is WHERE-shaped, so it gets the + // full error treatment. + let predicate_err: std::cell::RefCell> = + std::cell::RefCell::new(None); agg_result.rows.retain(|row| { - if predicate_err.get().is_some() { + if predicate_err.borrow().is_some() { return true; } let mp = nodedb_types::json_to_msgpack_or_empty(row); match ScanFilter::all_match_binary(&having_predicates, &mp) { Ok(keep) => keep, Err(e) => { - predicate_err.set(Some(e)); + predicate_err.replace(Some(e)); true } } }); - if predicate_err.take().is_some() { - return self.response_error(task, ErrorCode::DivisionByZero); + if let Some(e) = predicate_err.take() { + return self.response_error(task, ErrorCode::from(e)); } } } diff --git a/nodedb/src/data/executor/handlers/aggregate/streaming/finalize.rs b/nodedb/src/data/executor/handlers/aggregate/streaming/finalize.rs index d608f8422..ec52a2730 100644 --- a/nodedb/src/data/executor/handlers/aggregate/streaming/finalize.rs +++ b/nodedb/src/data/executor/handlers/aggregate/streaming/finalize.rs @@ -153,20 +153,20 @@ impl CoreLoop { } }; if !having_predicates.is_empty() { - // `Vec::retain`'s closure must return `bool`, so a division/ - // modulo-by-zero in a HAVING predicate is captured via this - // `Cell` side-channel and checked once the retain finishes. - let predicate_err: std::cell::Cell> = - std::cell::Cell::new(None); + // `Vec::retain`'s closure must return `bool`, so an evaluation + // error in a HAVING predicate is captured via this side-channel + // and checked once the retain finishes. + let predicate_err: std::cell::RefCell> = + std::cell::RefCell::new(None); results.retain(|row| { - if predicate_err.get().is_some() { + if predicate_err.borrow().is_some() { return true; } let mp = nodedb_types::json_to_msgpack_or_empty(row); match ScanFilter::all_match_binary(&having_predicates, &mp) { Ok(keep) => keep, Err(e) => { - predicate_err.set(Some(e)); + predicate_err.replace(Some(e)); true } } diff --git a/nodedb/src/data/executor/handlers/columnar_read/scan/execute.rs b/nodedb/src/data/executor/handlers/columnar_read/scan/execute.rs index 9787d637f..3f3cc11d2 100644 --- a/nodedb/src/data/executor/handlers/columnar_read/scan/execute.rs +++ b/nodedb/src/data/executor/handlers/columnar_read/scan/execute.rs @@ -270,12 +270,8 @@ impl CoreLoop { ) { Ok(true) => {} Ok(false) => continue, - // `EvalError` has exactly one variant and no direct - // `Into` — mirrors `stage_columnar_dml.rs`'s - // identical call site, which hardcodes the same typed - // code rather than collapsing to `Internal`/`XX000`. - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } let obj = match row_to_projected_value( diff --git a/nodedb/src/data/executor/handlers/columnar_resolve.rs b/nodedb/src/data/executor/handlers/columnar_resolve.rs index 8bc5f2130..7ee552d42 100644 --- a/nodedb/src/data/executor/handlers/columnar_resolve.rs +++ b/nodedb/src/data/executor/handlers/columnar_resolve.rs @@ -77,7 +77,7 @@ pub(in crate::data::executor) fn resolve_update_rows( match row_matches_filters(&row, schema, filter_predicates) { Ok(true) => {} Ok(false) => continue, - Err(_) => return Err(crate::Error::DivisionByZero), + Err(e) => return Err(crate::Error::from(e)), } } @@ -121,7 +121,7 @@ pub(in crate::data::executor) fn resolve_delete_rows( match row_matches_filters(&row, schema, filter_predicates) { Ok(true) => {} Ok(false) => continue, - Err(_) => return Err(crate::Error::DivisionByZero), + Err(e) => return Err(crate::Error::from(e)), } } admit_columnar_row(rls_write_check, &row, schema, tid, collection)?; diff --git a/nodedb/src/data/executor/handlers/document/index_fetch.rs b/nodedb/src/data/executor/handlers/document/index_fetch.rs index b8a370439..e9af74832 100644 --- a/nodedb/src/data/executor/handlers/document/index_fetch.rs +++ b/nodedb/src/data/executor/handlers/document/index_fetch.rs @@ -299,8 +299,8 @@ impl CoreLoop { } { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } let payload = if let Some(ref schema) = strict_schema { diff --git a/nodedb/src/data/executor/handlers/document/read/scan.rs b/nodedb/src/data/executor/handlers/document/read/scan.rs index 74e92e88f..bfbbbc17a 100644 --- a/nodedb/src/data/executor/handlers/document/read/scan.rs +++ b/nodedb/src/data/executor/handlers/document/read/scan.rs @@ -226,8 +226,8 @@ impl CoreLoop { } else { self.merge_overlay_into_scan(txn_id, &coll_key, &mut filtered, &matches); } - if predicate_err.take().is_some() { - return self.response_error(task, ErrorCode::DivisionByZero); + if let Some(e) = predicate_err.take() { + return self.response_error(task, ErrorCode::from(e)); } } @@ -284,7 +284,10 @@ impl CoreLoop { .collect() }; - let sorted = if sort_keys.is_empty() { + // With window functions the sort runs after the window pass + // (below), so ORDER BY can name a window alias, as the + // provider scan orders it. + let sorted = if sort_keys.is_empty() || !window_specs.is_empty() { filtered } else if filtered.len() <= self.query_tuning.sort_run_size { let mut v = filtered; @@ -366,6 +369,10 @@ impl CoreLoop { ) { return self.response_error(task, crate::Error::from(e)); } + let decoded_rows = match sort::sort_decoded_rows(decoded_rows, sort_keys) { + Ok(rows) => rows, + Err(e) => return self.response_error(task, e), + }; // Project first, then dedupe on the projected JSON value // so `SELECT DISTINCT col` honours SQL semantics. diff --git a/nodedb/src/data/executor/handlers/document/sort/in_memory.rs b/nodedb/src/data/executor/handlers/document/sort/in_memory.rs index 4e78a1347..1f5cd680f 100644 --- a/nodedb/src/data/executor/handlers/document/sort/in_memory.rs +++ b/nodedb/src/data/executor/handlers/document/sort/in_memory.rs @@ -27,6 +27,41 @@ pub(in crate::data::executor) fn sort_rows( sort_rows_by_expression(rows, sort_keys) } +/// Sort decoded document rows by the ORDER BY terms, with the same ordering +/// rules [`sort_rows`] applies to stored rows. +/// +/// A scan with window functions sorts after the window pass, so ORDER BY can +/// name a window alias; its rows are decoded by then. Each row is encoded +/// once under its position, sorted by [`sort_rows`], and taken back in the +/// sorted order. +pub(in crate::data::executor) fn sort_decoded_rows( + rows: Vec<(String, serde_json::Value)>, + sort_keys: &[SortKeySpec], +) -> crate::Result> { + if sort_keys.is_empty() { + return Ok(rows); + } + let mut keyed: Vec<(String, Vec)> = rows + .iter() + .enumerate() + .map(|(i, (_, doc))| (i.to_string(), nodedb_types::json_to_msgpack_or_empty(doc))) + .collect(); + sort_rows(&mut keyed, sort_keys)?; + let mut slots: Vec> = rows.into_iter().map(Some).collect(); + keyed + .iter() + .map(|(position, _)| { + position + .parse::() + .ok() + .and_then(|i| slots.get_mut(i).and_then(Option::take)) + .ok_or_else(|| crate::Error::Internal { + detail: format!("sort_decoded_rows: row position {position} is not a live row"), + }) + }) + .collect() +} + /// Zero-decode path: every key names a stored field, so ordering is decided /// straight from the msgpack bytes. fn sort_rows_by_column( @@ -177,6 +212,21 @@ mod tests { assert_eq!(order, vec!["b", "c", "a"], "ASC by val: 10, 20, 30"); } + /// Decoded rows sort on a field the window pass appended, and keep their + /// document ids. + #[test] + fn sort_decoded_rows_orders_by_an_appended_column() { + let rows = vec![ + ("a".to_string(), serde_json::json!({"id": "a", "rn": 3})), + ("b".to_string(), serde_json::json!({"id": "b", "rn": 1})), + ("c".to_string(), serde_json::json!({"id": "c", "rn": 2})), + ]; + let sorted = sort_decoded_rows(rows, &[col("rn", true)]).expect("sort_decoded_rows"); + let order: Vec<&str> = sorted.iter().map(|(id, _)| id.as_str()).collect(); + assert_eq!(order, vec!["b", "c", "a"]); + assert_eq!(sorted[0].1, serde_json::json!({"id": "b", "rn": 1})); + } + #[test] fn sort_by_int_field_desc() { let mut rows = rows_with_vals(&[("a", 30), ("b", 10), ("c", 20)]); diff --git a/nodedb/src/data/executor/handlers/document/sort/mod.rs b/nodedb/src/data/executor/handlers/document/sort/mod.rs index 0413ea871..545f85efd 100644 --- a/nodedb/src/data/executor/handlers/document/sort/mod.rs +++ b/nodedb/src/data/executor/handlers/document/sort/mod.rs @@ -6,4 +6,4 @@ pub(in crate::data::executor) mod compare; pub(in crate::data::executor) mod external; pub(in crate::data::executor) mod in_memory; -pub(in crate::data::executor) use in_memory::sort_rows; +pub(in crate::data::executor) use in_memory::{sort_decoded_rows, sort_rows}; diff --git a/nodedb/src/data/executor/handlers/generated.rs b/nodedb/src/data/executor/handlers/generated.rs index aefa9be6a..27efc70d2 100644 --- a/nodedb/src/data/executor/handlers/generated.rs +++ b/nodedb/src/data/executor/handlers/generated.rs @@ -30,10 +30,7 @@ pub fn evaluate_generated_columns( // A generated column's expression is write-path-shaped: a // division/modulo-by-zero fails the write instead of silently // materializing NULL into the stored column. - let result = spec - .expr - .eval(&doc_val) - .map_err(|_e| ErrorCode::DivisionByZero)?; + let result = spec.expr.eval(&doc_val).map_err(ErrorCode::from)?; let computed = serde_json::Value::from(result); if let Some(obj) = doc.as_object_mut() { obj.insert(spec.name.clone(), computed); diff --git a/nodedb/src/data/executor/handlers/grouping_sets_exec.rs b/nodedb/src/data/executor/handlers/grouping_sets_exec.rs index 1a3690fda..f209f9062 100644 --- a/nodedb/src/data/executor/handlers/grouping_sets_exec.rs +++ b/nodedb/src/data/executor/handlers/grouping_sets_exec.rs @@ -152,16 +152,16 @@ pub(super) fn execute_grouping_sets( match ScanFilter::all_match_binary_indexed(&filter_predicates, raw, &idx) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return core.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return core.response_error(task, ErrorCode::from(e)); } } } else { match ScanFilter::all_match_binary(&filter_predicates, raw) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return core.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return core.response_error(task, ErrorCode::from(e)); } } } @@ -169,15 +169,14 @@ pub(super) fn execute_grouping_sets( let group_key = match msgpack_scan::build_group_key(raw, &active_specs) { Ok(k) => k, - Err(_e) => return core.response_error(task, ErrorCode::DivisionByZero), + Err(e) => return core.response_error(task, ErrorCode::from(e)), }; - if groups + if let Err(e) = groups .entry(group_key) .or_insert_with(|| GroupState::new(&real_agg_slice)) .feed(&real_agg_slice, raw) - .is_err() { - return core.response_error(task, ErrorCode::DivisionByZero); + return core.response_error(task, ErrorCode::from(e)); } } @@ -188,13 +187,13 @@ pub(super) fn execute_grouping_sets( for raw in &owned_docs { match ScanFilter::all_match_binary(&filter_predicates, raw) { Ok(true) => { - if grand.feed(&real_agg_slice, raw).is_err() { - return core.response_error(task, ErrorCode::DivisionByZero); + if let Err(e) = grand.feed(&real_agg_slice, raw) { + return core.response_error(task, ErrorCode::from(e)); } } Ok(false) => {} - Err(_e) => { - return core.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return core.response_error(task, ErrorCode::from(e)); } } } @@ -262,26 +261,26 @@ pub(super) fn execute_grouping_sets( // Apply HAVING. if !having_predicates.is_empty() { - // `Vec::retain`'s closure must return `bool`, so a division/modulo- - // by-zero in a HAVING predicate is captured via this `Cell` - // side-channel and checked once the retain finishes. - let predicate_err: std::cell::Cell> = - std::cell::Cell::new(None); + // `Vec::retain`'s closure must return `bool`, so an evaluation error + // in a HAVING predicate is captured via this side-channel and checked + // once the retain finishes. + let predicate_err: std::cell::RefCell> = + std::cell::RefCell::new(None); all_rows.retain(|row| { - if predicate_err.get().is_some() { + if predicate_err.borrow().is_some() { return true; } let mp = nodedb_types::json_to_msgpack_or_empty(row); match ScanFilter::all_match_binary(&having_predicates, &mp) { Ok(keep) => keep, Err(e) => { - predicate_err.set(Some(e)); + predicate_err.replace(Some(e)); true } } }); - if predicate_err.take().is_some() { - return core.response_error(task, ErrorCode::DivisionByZero); + if let Some(e) = predicate_err.take() { + return core.response_error(task, ErrorCode::from(e)); } } diff --git a/nodedb/src/data/executor/handlers/join/grace_probe.rs b/nodedb/src/data/executor/handlers/join/grace_probe.rs index 89c701d00..6d51fb015 100644 --- a/nodedb/src/data/executor/handlers/join/grace_probe.rs +++ b/nodedb/src/data/executor/handlers/join/grace_probe.rs @@ -66,9 +66,9 @@ impl CoreLoop { if batch.is_empty() { return Ok(()); } - // A residual-ON-predicate div/modulo-by-zero propagates out of the - // flush closure; the caller's `?` converts it to - // `crate::Error::DivisionByZero` (SQLSTATE 22012). + // A residual-ON-predicate evaluation error propagates out of the + // flush closure; the caller's `?` converts it to its typed + // `crate::Error`. probe_rows_into( &ProbeParams { probe_docs: batch, diff --git a/nodedb/src/data/executor/handlers/join/hash_handlers.rs b/nodedb/src/data/executor/handlers/join/hash_handlers.rs index 388174c2b..bfe2df6c6 100644 --- a/nodedb/src/data/executor/handlers/join/hash_handlers.rs +++ b/nodedb/src/data/executor/handlers/join/hash_handlers.rs @@ -465,9 +465,9 @@ impl CoreLoop { emit_unmatched_right: true, }) { Ok(r) => r, - // Div/modulo-by-zero in a residual ON predicate surfaces to the - // client as SQLSTATE 22012. - Err(_e) => return self.response_error(join.task, ErrorCode::DivisionByZero), + // An evaluation error in a residual ON predicate reaches the + // client under its own SQLSTATE. + Err(e) => return self.response_error(join.task, ErrorCode::from(e)), }; if enforce_output_budget && results.len() >= probe_limit { diff --git a/nodedb/src/data/executor/handlers/join/nested_loop.rs b/nodedb/src/data/executor/handlers/join/nested_loop.rs index e0e6db89a..a48a9ebe9 100644 --- a/nodedb/src/data/executor/handlers/join/nested_loop.rs +++ b/nodedb/src/data/executor/handlers/join/nested_loop.rs @@ -171,8 +171,8 @@ impl CoreLoop { &merged, ) { Ok(b) => b, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } }; diff --git a/nodedb/src/data/executor/handlers/kv/predicate/matches.rs b/nodedb/src/data/executor/handlers/kv/predicate/matches.rs index 8b7d07fac..356c47633 100644 --- a/nodedb/src/data/executor/handlers/kv/predicate/matches.rs +++ b/nodedb/src/data/executor/handlers/kv/predicate/matches.rs @@ -65,7 +65,7 @@ impl CoreLoop { match ScanFilter::all_match_binary(&predicates, &row) { Ok(true) => {} Ok(false) => continue, - Err(_e) => return Err(ErrorCode::DivisionByZero), + Err(e) => return Err(ErrorCode::from(e)), } } matched.push((key, value)); diff --git a/nodedb/src/data/executor/handlers/kv/scan.rs b/nodedb/src/data/executor/handlers/kv/scan.rs index c947aefd1..939ee634d 100644 --- a/nodedb/src/data/executor/handlers/kv/scan.rs +++ b/nodedb/src/data/executor/handlers/kv/scan.rs @@ -148,8 +148,8 @@ impl CoreLoop { ) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } } diff --git a/nodedb/src/data/executor/handlers/point/update/post_image.rs b/nodedb/src/data/executor/handlers/point/update/post_image.rs index 969e1ce5d..f8e33afc0 100644 --- a/nodedb/src/data/executor/handlers/point/update/post_image.rs +++ b/nodedb/src/data/executor/handlers/point/update/post_image.rs @@ -209,8 +209,8 @@ impl CoreLoop { UpdateValue::Expr(expr) => { let result: nodedb_types::Value = match expr.eval(&eval_doc) { Ok(v) => v, - // Division/modulo by zero fails the statement. - Err(_e) => return Err(ErrorCode::DivisionByZero), + // An evaluation error fails the statement. + Err(e) => return Err(ErrorCode::from(e)), }; let json: serde_json::Value = result.into(); json diff --git a/nodedb/src/data/executor/handlers/provider_scan.rs b/nodedb/src/data/executor/handlers/provider_scan.rs index 799f3acf1..38adb27dd 100644 --- a/nodedb/src/data/executor/handlers/provider_scan.rs +++ b/nodedb/src/data/executor/handlers/provider_scan.rs @@ -78,25 +78,25 @@ impl CoreLoop { } }; if !predicates.is_empty() { - // `Vec::retain`'s closure must return `bool`, so a division/ - // modulo-by-zero is captured via this `Cell` side-channel - // and checked once the retain finishes. - let predicate_err: std::cell::Cell> = - std::cell::Cell::new(None); + // `Vec::retain`'s closure must return `bool`, so an + // evaluation error is captured via this side-channel and + // checked once the retain finishes. + let predicate_err: std::cell::RefCell> = + std::cell::RefCell::new(None); rows.retain(|row| { - if predicate_err.get().is_some() { + if predicate_err.borrow().is_some() { return true; } match ScanFilter::all_match_binary(&predicates, row) { Ok(keep) => keep, Err(e) => { - predicate_err.set(Some(e)); + predicate_err.replace(Some(e)); true } } }); - if predicate_err.take().is_some() { - return self.response_error(task, ErrorCode::DivisionByZero); + if let Some(e) = predicate_err.take() { + return self.response_error(task, ErrorCode::from(e)); } } } diff --git a/nodedb/src/data/executor/handlers/recursive.rs b/nodedb/src/data/executor/handlers/recursive.rs index 2a554da56..574cbb96b 100644 --- a/nodedb/src/data/executor/handlers/recursive.rs +++ b/nodedb/src/data/executor/handlers/recursive.rs @@ -151,8 +151,8 @@ impl CoreLoop { match ScanFilter::all_match_binary(&base_preds, &mp) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } let key = if distinct { @@ -205,8 +205,8 @@ impl CoreLoop { match ScanFilter::all_match_binary(&recursive_preds, &mp) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } @@ -261,8 +261,8 @@ impl CoreLoop { match ScanFilter::all_match_binary(&recursive_preds, &mp) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } let key = if distinct { diff --git a/nodedb/src/data/executor/handlers/spatial/full_scan.rs b/nodedb/src/data/executor/handlers/spatial/full_scan.rs index 5da855a10..66c1e590c 100644 --- a/nodedb/src/data/executor/handlers/spatial/full_scan.rs +++ b/nodedb/src/data/executor/handlers/spatial/full_scan.rs @@ -103,15 +103,15 @@ impl CoreLoop { match ScanFilter::all_match_value(attr_filters, &doc) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } match ScanFilter::all_match_value(rls_filters, &doc) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } diff --git a/nodedb/src/data/executor/handlers/spatial/rtree_scan.rs b/nodedb/src/data/executor/handlers/spatial/rtree_scan.rs index 1c2b09e53..d57118eef 100644 --- a/nodedb/src/data/executor/handlers/spatial/rtree_scan.rs +++ b/nodedb/src/data/executor/handlers/spatial/rtree_scan.rs @@ -298,15 +298,15 @@ impl CoreLoop { match ScanFilter::all_match_value(&attr_filters, &doc) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } match ScanFilter::all_match_value(&row_level_filters, &doc) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } diff --git a/nodedb/src/data/executor/handlers/timeseries/raw_scan/scan.rs b/nodedb/src/data/executor/handlers/timeseries/raw_scan/scan.rs index 824d7f204..055efd4ab 100644 --- a/nodedb/src/data/executor/handlers/timeseries/raw_scan/scan.rs +++ b/nodedb/src/data/executor/handlers/timeseries/raw_scan/scan.rs @@ -167,8 +167,8 @@ impl CoreLoop { ) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return self.response_error(task, ErrorCode::DivisionByZero); + Err(e) => { + return self.response_error(task, ErrorCode::from(e)); } } } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs index 3584e5dfb..b54ade4c4 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_bulk_delete.rs @@ -107,8 +107,8 @@ impl CoreLoop { } }; self.merge_overlay_into_scan(txn_id, &coll_key, &mut rows, &matches); - if predicate_err.take().is_some() { - return self.response_error(task, ErrorCode::DivisionByZero); + if let Some(e) = predicate_err.take() { + return self.response_error(task, ErrorCode::from(e)); } } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs index 53aea9f10..ad66f9d13 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_columnar_dml.rs @@ -322,20 +322,18 @@ impl CoreLoop { match row_matches_filters(&row, &schema, &filter_predicates) { Ok(true) => {} Ok(false) => continue, - Err(_e) => { - return Err(self.response_error(task, ErrorCode::DivisionByZero)); + Err(e) => { + return Err(self.response_error(task, ErrorCode::from(e))); } } } // No computed columns on this path (`&[]` below), so this - // can never actually raise `DivisionByZero` today — handled - // uniformly with every other `row_to_projected_value` caller - // instead of assuming that invariant with an `unwrap`. + // cannot raise an evaluation error today. It is handled like + // every other `row_to_projected_value` caller instead of + // assuming that invariant with an `unwrap`. let obj = match row_to_projected_value(&row, &schema, &[], &[], false) { Ok(v) => v, - Err(_e) => { - return Err(self.response_error(task, ErrorCode::DivisionByZero)); - } + Err(e) => return Err(self.response_error(task, e)), }; matched.push((surrogate, row, obj)); } diff --git a/nodedb/src/data/executor/handlers/vector_direct_targets.rs b/nodedb/src/data/executor/handlers/vector_direct_targets.rs index 50b2daa72..463a61cd4 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_targets.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_targets.rs @@ -46,7 +46,7 @@ pub(in crate::data::executor) fn vector_sidecar_matches( return Ok(true); } let (_id, mp) = sparse_row_to_doc(key, sidecar, SparseBodyFormatRef::VectorSidecar); - ScanFilter::all_match_binary(filters, &mp).map_err(|_| ErrorCode::DivisionByZero) + ScanFilter::all_match_binary(filters, &mp).map_err(ErrorCode::from) } impl CoreLoop { diff --git a/nodedb/src/error/types.rs b/nodedb/src/error/types.rs index 19f13e65b..d6fd3a1e7 100644 --- a/nodedb/src/error/types.rs +++ b/nodedb/src/error/types.rs @@ -343,6 +343,13 @@ pub enum Error { #[error("division by zero")] DivisionByZero, + /// A function received an argument it cannot compute on: vectors of + /// different dimensions, an argument of the wrong type, a malformed + /// JSONPath. `detail` names the function and the value. Rendered as + /// SQLSTATE `22000` (data_exception) at the pgwire layer. + #[error("{detail}")] + DataException { detail: String }, + /// A LIMIT/OFFSET/FETCH bound did not resolve to `[0, usize::MAX]`. /// The pgwire layer renders it as SQLSTATE `2201W`. #[error("invalid {clause} value: {value}")] diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index afd32e72b..1025c0dd1 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -172,6 +172,7 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { Error::AmbiguousColumn { column } => NodeDbError::ambiguous_column(column.clone()), Error::UnknownStrictField { column, .. } => NodeDbError::undefined_column(column.clone()), Error::DivisionByZero => NodeDbError::division_by_zero(), + Error::DataException { detail } => NodeDbError::data_exception(detail.clone()), Error::InvalidLimitValue { clause, value } => { NodeDbError::invalid_limit_value(*clause, value.clone()) } diff --git a/nodedb/src/error_from.rs b/nodedb/src/error_from.rs index 8ba3b07a3..6ac3de35e 100644 --- a/nodedb/src/error_from.rs +++ b/nodedb/src/error_from.rs @@ -16,14 +16,19 @@ impl From for Error { } } -/// `EvalError` has exactly one variant today (`DivisionByZero`); the -/// `match` is exhaustive rather than a `_ =>` fallback so -/// a future evaluator error is forced to pick its own `crate::Error` -/// mapping instead of silently inheriting this one. +/// The `match` is exhaustive rather than a `_ =>` fallback, so a new +/// evaluator error picks its own `crate::Error` mapping instead of silently +/// inheriting one. impl From for Error { fn from(e: nodedb_query::EvalError) -> Self { match e { nodedb_query::EvalError::DivisionByZero => Self::DivisionByZero, + nodedb_query::EvalError::UnknownFunction { name } => Self::UndefinedFunction { name }, + e @ (nodedb_query::EvalError::VectorDimensionMismatch { .. } + | nodedb_query::EvalError::ArgumentType { .. } + | nodedb_query::EvalError::InvalidJsonPath { .. }) => Self::DataException { + detail: e.to_string(), + }, } } } diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index 0f6ec6b12..71f3019da 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -146,6 +146,8 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { ErrorCode::UndefinedColumn { column } => NodeDbError::undefined_column(column), ErrorCode::Unsupported { detail } => NodeDbError::bad_request(detail), ErrorCode::DivisionByZero => NodeDbError::division_by_zero(), + ErrorCode::UndefinedFunction { name } => NodeDbError::undefined_function(name), + ErrorCode::DataException { detail } => NodeDbError::data_exception(detail), // Nothing was enqueued, and the same request succeeds once capacity // frees: the retryable overload class. ErrorCode::DispatchCapacity { reason } => NodeDbError::server_overload(reason), diff --git a/nodedb/src/query/materialized_sum_delta.rs b/nodedb/src/query/materialized_sum_delta.rs index dd33fec31..809ce9315 100644 --- a/nodedb/src/query/materialized_sum_delta.rs +++ b/nodedb/src/query/materialized_sum_delta.rs @@ -45,10 +45,7 @@ pub fn binding_amount( doc: &serde_json::Value, ) -> crate::Result { let row = nodedb_types::Value::from(doc.clone()); - let evaluated = binding - .value_expr - .eval(&row) - .map_err(|_e| crate::Error::DivisionByZero)?; + let evaluated = binding.value_expr.eval(&row).map_err(crate::Error::from)?; Ok(json_to_decimal(&serde_json::Value::from(evaluated)).unwrap_or(Decimal::ZERO)) } diff --git a/nodedb/src/storage/cold_filter.rs b/nodedb/src/storage/cold_filter.rs index 42bd5d20d..930112ee8 100644 --- a/nodedb/src/storage/cold_filter.rs +++ b/nodedb/src/storage/cold_filter.rs @@ -158,17 +158,15 @@ pub fn read_parquet_filtered( detail: format!("read batches: {e}"), })?; - // Check the side-channel AFTER the reader finishes: a division/modulo- - // by-zero row was already excluded from every batch's mask above (never - // surfaced as a mis-filtered result), so `reader.collect()` succeeding - // does not mean the read was clean — it means the predicate closure - // never got a chance to return an `Err` at all. This is the typed - // `crate::Error::DivisionByZero` the pre-fix `ArrowError` conversion - // above lost. + // Check the side-channel AFTER the reader finishes: a row whose + // predicate raised an evaluation error was already excluded from every + // batch's mask above (never surfaced as a mis-filtered result), so + // `reader.collect()` succeeding does not mean the read was clean. The + // recorded error is returned as its own typed `crate::Error`. if let Ok(mut slot) = predicate_err.lock() - && slot.take().is_some() + && let Some(e) = slot.take() { - return Err(crate::Error::DivisionByZero); + return Err(crate::Error::from(e)); } Ok(batches) diff --git a/nodedb/tests/wire/cases/expr_predicate_like_array.rs b/nodedb/tests/wire/cases/expr_predicate_like_array.rs new file mode 100644 index 000000000..6086e842a --- /dev/null +++ b/nodedb/tests/wire/cases/expr_predicate_like_array.rs @@ -0,0 +1,88 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `LIKE` / `ILIKE` and `ARRAY[...]` inside expression predicates. +//! +//! A `LIKE` that sits inside a larger expression (an `OR` with another +//! comparison, a function call on its operand) is evaluated row by row as a +//! `like` / `ilike` call. An `ARRAY[...]` with a column element is built per +//! row. Both return the matching rows, never an empty result. + +use crate::harness::TestServer; + +async fn seeded(name: &str) -> TestServer { + let srv = TestServer::start().await; + srv.exec(&format!( + "CREATE COLLECTION {name} WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + for (id, who, n) in [ + ("a1", "Alice", 1), + ("a2", "amy", 3), + ("b1", "bob", 9), + ("c1", "Carl", 2), + ] { + srv.exec(&format!( + "INSERT INTO {name} {{ id: '{id}', name: '{who}', n: {n} }}" + )) + .await + .unwrap(); + } + srv +} + +async fn ids(srv: &TestServer, sql: &str) -> Vec { + srv.query_rows(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")) + .into_iter() + .map(|row| row[0].clone()) + .collect() +} + +#[tokio::test] +async fn like_inside_an_or_predicate_matches_rows() { + let srv = seeded("like_or").await; + assert_eq!( + ids( + &srv, + "SELECT id FROM like_or WHERE lower(name) LIKE 'a%' OR n > 5 ORDER BY id" + ) + .await, + vec!["a1", "a2", "b1"] + ); +} + +#[tokio::test] +async fn not_like_and_ilike_inside_expression_predicates() { + let srv = seeded("like_not").await; + assert_eq!( + ids( + &srv, + "SELECT id FROM like_not WHERE lower(name) NOT LIKE 'a%' AND n < 5 ORDER BY id" + ) + .await, + vec!["c1"] + ); + assert_eq!( + ids( + &srv, + "SELECT id FROM like_not WHERE name ILIKE 'A%' OR n > 100 ORDER BY id" + ) + .await, + vec!["a1", "a2"] + ); +} + +#[tokio::test] +async fn an_array_with_a_column_element_is_built_per_row() { + let srv = seeded("array_col").await; + assert_eq!( + ids( + &srv, + "SELECT id FROM array_col WHERE array_contains(ARRAY[n, 100], 9) ORDER BY id" + ) + .await, + vec!["b1"] + ); +} diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 137ff2da7..7c447056f 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -77,6 +77,7 @@ mod engine_surface_sparse_vector; mod engine_surface_spatial; mod engine_surface_vector; mod event_write_op_tagging; +mod expr_predicate_like_array; mod fts_read_body_format; mod function_security_e2e; mod graph_analytics_authorization; @@ -147,6 +148,7 @@ mod restart_refused_write_not_resurrected; mod rls_policy_database_scope; mod role_assignment_transaction; mod router_misroute_literals; +mod row_functions_and_window_order; mod scalar_aggregate_empty_input; mod scalar_aggregate_multicore_merge; mod schema_visibility_barrier; diff --git a/nodedb/tests/wire/cases/row_functions_and_window_order.rs b/nodedb/tests/wire/cases/row_functions_and_window_order.rs new file mode 100644 index 000000000..3441045c0 --- /dev/null +++ b/nodedb/tests/wire/cases/row_functions_and_window_order.rs @@ -0,0 +1,226 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-row SQL functions and window output on a document collection. +//! +//! - A window column is computed by the window pass only. ORDER BY and the +//! SELECT list read it as a column, so ORDER BY can name the window alias. +//! - `doc_get`, `doc_exists`, `doc_array_contains` and `vector_distance` +//! evaluate per row in a projection and in WHERE. +//! - A vector dimension mismatch, a non-vector operand and a malformed +//! JSONPath fail with SQLSTATE `22000`, never a silent `NULL`. +//! - An index-owned search function in a row filter is refused at plan time. + +use crate::harness::TestServer; + +async fn rows(srv: &TestServer, sql: &str) -> Vec> { + srv.query_rows(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")) +} + +async fn collection(srv: &TestServer, name: &str) { + srv.exec(&format!( + "CREATE COLLECTION {name} WITH (engine='document_schemaless')" + )) + .await + .unwrap(); +} + +#[tokio::test] +async fn order_by_a_window_alias_sorts_on_the_window_column() { + let srv = TestServer::start().await; + collection(&srv, "win_order").await; + for id in ["w1", "w2", "w3"] { + srv.exec(&format!("INSERT INTO win_order {{ id: '{id}', n: 1 }}")) + .await + .unwrap(); + } + + assert_eq!( + rows( + &srv, + "SELECT id, ROW_NUMBER() OVER (ORDER BY id DESC) AS rn FROM win_order ORDER BY rn", + ) + .await, + vec![ + vec!["w3".to_string(), "1".to_string()], + vec!["w2".to_string(), "2".to_string()], + vec!["w1".to_string(), "3".to_string()], + ] + ); + + assert_eq!( + rows( + &srv, + "SELECT id, ROW_NUMBER() OVER (ORDER BY id) AS rn FROM win_order \ + ORDER BY rn DESC LIMIT 2", + ) + .await, + vec![ + vec!["w3".to_string(), "3".to_string()], + vec!["w2".to_string(), "2".to_string()], + ] + ); +} + +async fn seeded_events(name: &str) -> TestServer { + let srv = TestServer::start().await; + collection(&srv, name).await; + srv.exec(&format!( + "INSERT INTO {name} {{ id: 'e1', payload: {{ user: {{ name: 'ada', email: 'a@x' }}, \ + tags: ['important', 'ops'] }} }}" + )) + .await + .unwrap(); + srv.exec(&format!( + "INSERT INTO {name} {{ id: 'e2', payload: {{ user: {{ name: 'bob' }}, tags: ['ops'] }} }}" + )) + .await + .unwrap(); + srv +} + +#[tokio::test] +async fn doc_get_projects_a_nested_field() { + let srv = seeded_events("ev_get").await; + assert_eq!( + rows( + &srv, + "SELECT id, doc_get(payload, '$.user.name') AS name, \ + doc_get(payload, '$.user.email', 'none') AS email FROM ev_get ORDER BY id", + ) + .await, + vec![ + vec!["e1".to_string(), "ada".to_string(), "a@x".to_string()], + vec!["e2".to_string(), "bob".to_string(), "none".to_string()], + ] + ); +} + +#[tokio::test] +async fn doc_exists_and_doc_array_contains_filter_rows() { + let srv = seeded_events("ev_where").await; + let ids = + |r: Vec>| -> Vec { r.into_iter().map(|row| row[0].clone()).collect() }; + + assert_eq!( + ids(rows( + &srv, + "SELECT id FROM ev_where WHERE doc_exists(payload, '$.user.email') ORDER BY id", + ) + .await), + vec!["e1"] + ); + assert_eq!( + ids(rows( + &srv, + "SELECT id FROM ev_where WHERE doc_array_contains(payload, '$.tags', 'ops') \ + ORDER BY id", + ) + .await), + vec!["e1", "e2"] + ); + assert_eq!( + ids(rows( + &srv, + "SELECT id FROM ev_where \ + WHERE doc_array_contains(payload, '$.tags', 'important') ORDER BY id", + ) + .await), + vec!["e1"] + ); +} + +/// Without a vector-search ORDER BY, `vector_distance` evaluates per row as +/// the squared L2 distance a vector search reports. +#[tokio::test] +async fn vector_distance_in_a_projection_evaluates_per_row() { + let srv = TestServer::start().await; + collection(&srv, "vd_rows").await; + srv.exec("INSERT INTO vd_rows { id: 'v1', emb: [3.0, 4.0] }") + .await + .unwrap(); + + let r = rows( + &srv, + "SELECT id, vector_distance(emb, ARRAY[0.0, 0.0]) AS d FROM vd_rows", + ) + .await; + assert_eq!(r.len(), 1, "one row expected, got {r:?}"); + assert_eq!(r[0][0], "v1"); + let d: f64 = r[0][1] + .parse() + .unwrap_or_else(|_| panic!("distance must be numeric, got {r:?}")); + assert!((d - 25.0).abs() < 1e-9, "expected 25, got {d}"); +} + +/// Operands of different dimensions fail the row with `22000`, naming both +/// dimensions, in the vector engine's message shape. +#[tokio::test] +async fn vector_dimension_mismatch_is_a_data_exception() { + let srv = TestServer::start().await; + collection(&srv, "vd_mismatch").await; + srv.exec("INSERT INTO vd_mismatch { id: 'v1', emb: [3.0, 4.0] }") + .await + .unwrap(); + + let error = srv + .query_text("SELECT id, vector_distance(emb, ARRAY[0.0, 0.0, 0.0]) AS d FROM vd_mismatch") + .await + .expect_err("a dimension mismatch must fail"); + assert!( + error.contains("22000") && error.contains("vector dimension mismatch: expected 2, got 3"), + "expected 22000 naming both dimensions, got: {error}" + ); +} + +/// A non-vector operand fails with `22000`, naming its position and type. +#[tokio::test] +async fn a_non_vector_operand_is_a_data_exception() { + let srv = TestServer::start().await; + collection(&srv, "vd_badarg").await; + srv.exec("INSERT INTO vd_badarg { id: 'v1', emb: 7 }") + .await + .unwrap(); + + let error = srv + .query_text("SELECT id, vector_distance(emb, ARRAY[0.0]) AS d FROM vd_badarg") + .await + .expect_err("a non-vector operand must fail"); + assert!( + error.contains("22000") && error.contains("argument 1") && error.contains("got int"), + "expected 22000 naming argument 1 and its type, got: {error}" + ); +} + +/// A malformed JSONPath fails with `22000`. A missing path returns the +/// default, as `doc_get_projects_a_nested_field` shows. +#[tokio::test] +async fn a_malformed_json_path_is_a_data_exception() { + let srv = seeded_events("ev_badpath").await; + let error = srv + .query_text("SELECT id, doc_get(payload, '$..name') AS n FROM ev_badpath") + .await + .expect_err("a malformed JSONPath must fail"); + assert!( + error.contains("22000") && error.contains("invalid JSONPath"), + "expected 22000 for the malformed path, got: {error}" + ); +} + +/// `bm25_score` reads the full-text index and has no per-row value, so a +/// comparison on it in WHERE is refused with `0A000` at plan time, even on +/// an empty collection. +#[tokio::test] +async fn a_search_score_in_a_row_filter_is_refused() { + let srv = TestServer::start().await; + collection(&srv, "fts_refuse").await; + let error = srv + .query_text("SELECT id FROM fts_refuse WHERE bm25_score(body, 'rust') > 1.0") + .await + .expect_err("bm25_score in a row filter must be refused"); + assert!( + error.contains("0A000") && error.contains("bm25_score"), + "expected 0A000 naming bm25_score, got: {error}" + ); +} diff --git a/nodedb/tests/wire/cases/sql_search_subquery_composition.rs b/nodedb/tests/wire/cases/sql_search_subquery_composition.rs index 14cf6dc6d..584f79d50 100644 --- a/nodedb/tests/wire/cases/sql_search_subquery_composition.rs +++ b/nodedb/tests/wire/cases/sql_search_subquery_composition.rs @@ -524,6 +524,28 @@ async fn a_sparse_where_trigger_refuses_the_distance_cell() { ); } +/// The same WHERE `sparse_score(...)` comparison without the outer column +/// reference stays in a row filter. `sparse_score` reads the sparse index +/// and has no per-row value, so the planner refuses the statement with +/// `0A000` naming the function, whether or not the collection holds rows. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_sparse_where_comparison_is_refused_at_plan_time() { + let server = TestServer::start().await; + server + .exec("CREATE TABLE sp_sparse_refuse (id TEXT PRIMARY KEY, terms SPARSEVECTOR)") + .await + .unwrap(); + + let error = server + .query_text("SELECT id FROM sp_sparse_refuse WHERE sparse_score(terms, '{3: 1.0}') > 0.1") + .await + .expect_err("sparse_score has no per-row value in a row filter"); + assert!( + error.contains("0A000") && error.contains("sparse_score"), + "expected 0A000 naming sparse_score, got: {error}" + ); +} + /// A one-argument `vector_distance` does not route to a search /// (`order_by/triggers.rs` returns `Ok(None)` below two arguments), so no /// cell is declared and `s.distance` must refuse `42703`. From 0060af6ac26daa39bb3609391b4c8479ddd03eef Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 01:46:49 +0800 Subject: [PATCH 44/64] feat(vector): fail on dimension mismatch instead of panicking Replace assert_eq! panics across the vector engine (flat index, HNSW, IVF-PQ, Vamana, PQ/SQ8 quantization, matryoshka, multivec, NAViX, SIEVE) with typed VectorError results. Split DimensionMismatch (bad caller input) from a new StoredDimensionMismatch (corrupt or foreign stored data) so the two are classified differently: an input error maps to SQLSTATE 22000 (data exception) and never takes the core down, while a stored-data mismatch keeps fail-stop handling. Add InvalidFilterBitmap for pre-filter bitmaps that fail to decode and InvalidInput for index build/train calls with unusable parameters. Split IVF-PQ search out of vector_search_exec.rs into its own handler module. Add a generated_always SQLSTATE and a shared constraint_sqlstate mapping so RejectedConstraint kinds (not_null, unique, generated_always, fk_missing, rls_policy, permission_denied) each keep their own error class instead of collapsing to unique_violation. --- nodedb-types/src/error/sqlstate.rs | 5 + nodedb-vector/src/adaptive_filter.rs | 41 +- nodedb-vector/src/codec_index/build.rs | 16 +- nodedb-vector/src/codec_index/graph.rs | 2 +- nodedb-vector/src/codec_index/search.rs | 74 ++- nodedb-vector/src/collection/checkpoint.rs | 21 +- nodedb-vector/src/collection/codec_build.rs | 51 +- .../src/collection/codec_dispatch.rs | 101 +++- nodedb-vector/src/collection/lifecycle.rs | 27 +- .../src/collection/lifecycle_compact.rs | 4 +- .../src/collection/lifecycle_insert_ops.rs | 73 ++- nodedb-vector/src/collection/quantize.rs | 15 +- nodedb-vector/src/collection/rollback.rs | 29 +- nodedb-vector/src/collection/search.rs | 528 +++++++++--------- nodedb-vector/src/delta/compaction.rs | 2 +- nodedb-vector/src/delta/index.rs | 54 +- nodedb-vector/src/dtype/cast.rs | 103 ++-- nodedb-vector/src/error.rs | 26 + nodedb-vector/src/flat.rs | 282 ++++------ nodedb-vector/src/hnsw/build.rs | 4 +- nodedb-vector/src/hnsw/checkpoint.rs | 4 +- nodedb-vector/src/hnsw/graph/index/backing.rs | 4 +- nodedb-vector/src/hnsw/graph/index/state.rs | 8 +- nodedb-vector/src/hnsw/graph/index/vectors.rs | 4 +- nodedb-vector/src/hnsw/search.rs | 222 +++++--- nodedb-vector/src/ivf.rs | 167 ++++-- nodedb-vector/src/matryoshka.rs | 86 ++- nodedb-vector/src/multivec/meta_embed.rs | 55 +- nodedb-vector/src/multivec/plaid.rs | 27 +- nodedb-vector/src/navix/acorn.rs | 45 +- nodedb-vector/src/navix/traversal.rs | 60 +- nodedb-vector/src/quantize/pq.rs | 154 +++-- nodedb-vector/src/quantize/pq_codec.rs | 2 +- nodedb-vector/src/quantize/sq8.rs | 69 ++- nodedb-vector/src/quantize/sq8_codec.rs | 2 +- nodedb-vector/src/rerank/codecs/pq.rs | 3 +- nodedb-vector/src/rerank/codecs/sq8.rs | 16 +- nodedb-vector/src/sieve/collection.rs | 2 +- nodedb-vector/src/sieve/router.rs | 50 +- nodedb-vector/src/vamana/build.rs | 48 +- nodedb-vector/src/vamana/search.rs | 110 +++- .../cases/collection_bitmap_filter.rs | 14 +- .../cases/collection_checkpoint_tombstones.rs | 8 +- .../cases/collection_compact_doc_map.rs | 9 +- .../cases/collection_pq_config.rs | 2 +- .../cases/hnsw_checkpoint_encryption.rs | 4 +- .../cases/quantize_kmeans_distribution.rs | 10 +- .../control/server/pgwire/types/error_map.rs | 13 +- .../src/control/server/shared/ddl/sqlstate.rs | 71 ++- .../control/checkpoint_durable_lsn.rs | 3 +- .../src/data/executor/handlers/generated.rs | 11 +- .../src/data/executor/handlers/graph_rag.rs | 16 +- nodedb/src/data/executor/handlers/mod.rs | 1 + .../handlers/point/apply_put/vector/put.rs | 90 +-- .../handlers/snapshot/restore/engines.rs | 16 +- .../snapshot/restore/tenant_snapshot.rs | 6 +- .../executor/handlers/text_search_hybrid.rs | 28 +- .../executor/handlers/text_search_triple.rs | 28 +- .../handlers/transaction/resolve/entry.rs | 2 +- .../transaction/stage_write/stage_vector.rs | 10 +- .../handlers/transaction/undo/apply.rs | 8 +- .../handlers/transaction/undo/rollback.rs | 2 +- .../handlers/transaction/undo/vector_write.rs | 26 +- nodedb/src/data/executor/handlers/vector.rs | 45 +- .../handlers/vector_direct_resolve/apply.rs | 24 +- .../executor/handlers/vector_direct_row.rs | 18 +- .../data/executor/handlers/vector_multi.rs | 39 +- .../handlers/vector_multi_search_exec.rs | 42 +- .../data/executor/handlers/vector_params.rs | 17 +- .../data/executor/handlers/vector_search.rs | 46 +- .../executor/handlers/vector_search_exec.rs | 113 ++-- .../executor/handlers/vector_search_ivf.rs | 89 +++ .../data/executor/handlers/vector_sparse.rs | 10 +- .../data/executor/handlers/vector_write.rs | 17 +- .../data/executor/vector_checkpoint/load.rs | 4 +- .../data/executor/vector_checkpoint/write.rs | 3 +- nodedb/src/data/executor/vector_string.rs | 34 +- nodedb/src/data/executor/wal_replay_vector.rs | 36 +- .../executor/wal_replay_vector_extended.rs | 3 +- nodedb/src/error_from.rs | 38 +- nodedb/tests/wire/cases/mod.rs | 1 + .../wire/cases/vector_dimension_errors.rs | 87 +++ 82 files changed, 2321 insertions(+), 1319 deletions(-) create mode 100644 nodedb/src/data/executor/handlers/vector_search_ivf.rs create mode 100644 nodedb/tests/wire/cases/vector_dimension_errors.rs diff --git a/nodedb-types/src/error/sqlstate.rs b/nodedb-types/src/error/sqlstate.rs index 422884d27..afb953658 100644 --- a/nodedb-types/src/error/sqlstate.rs +++ b/nodedb-types/src/error/sqlstate.rs @@ -78,6 +78,10 @@ pub const INVALID_PARAMETER_VALUE: &str = "22023"; /// `23000` — `integrity_constraint_violation` (generic) pub const INTEGRITY_CONSTRAINT_VIOLATION: &str = "23000"; +/// `428C9` — `generated_always` (a write names a generated column; its +/// value is computed from other columns) +pub const GENERATED_ALWAYS: &str = "428C9"; + /// `23502` — `not_null_violation` pub const NOT_NULL_VIOLATION: &str = "23502"; @@ -362,6 +366,7 @@ mod tests { INVALID_LIMIT_VALUE, INVALID_TEXT_REPRESENTATION, INTEGRITY_CONSTRAINT_VIOLATION, + GENERATED_ALWAYS, NOT_NULL_VIOLATION, FOREIGN_KEY_VIOLATION, UNIQUE_VIOLATION, diff --git a/nodedb-vector/src/adaptive_filter.rs b/nodedb-vector/src/adaptive_filter.rs index 9b3c59430..767367ee8 100644 --- a/nodedb-vector/src/adaptive_filter.rs +++ b/nodedb-vector/src/adaptive_filter.rs @@ -66,6 +66,10 @@ pub fn select_strategy(selectivity: f64, thresholds: &FilterThresholds) -> Filte } /// Execute adaptive filtered search on an HNSW index. +/// +/// A query without the index dimension fails with +/// [`VectorError::DimensionMismatch`](crate::error::VectorError::DimensionMismatch) +/// under every strategy. pub fn adaptive_search( index: &HnswIndex, query: &[f32], @@ -73,7 +77,8 @@ pub fn adaptive_search( ef: usize, bitmap: &RoaringBitmap, thresholds: &FilterThresholds, -) -> Vec { +) -> Result, crate::error::VectorError> { + crate::error::check_dim(index.dim(), query.len())?; let total = index.len(); let selectivity = estimate_selectivity(bitmap, total); let strategy = select_strategy(selectivity, thresholds); @@ -82,13 +87,13 @@ pub fn adaptive_search( FilterStrategy::PreFilter => index.search_filtered(query, top_k, ef, bitmap), FilterStrategy::PostFilter { over_fetch_factor } => { let fetch_k = top_k * over_fetch_factor; - let results = index.search(query, fetch_k, ef.max(fetch_k)); + let results = index.search(query, fetch_k, ef.max(fetch_k))?; let mut filtered: Vec = results .into_iter() .filter(|r| bitmap.contains(r.id)) .collect(); filtered.truncate(top_k); - filtered + Ok(filtered) } FilterStrategy::BruteForceMatching => { let metric = index.params().metric; @@ -119,7 +124,7 @@ pub fn adaptive_search( .partial_cmp(&b.distance) .unwrap_or(std::cmp::Ordering::Equal) }); - results + Ok(results) } } } @@ -179,7 +184,8 @@ mod tests { bitmap.insert(i); } - let results = adaptive_search(&idx, &[505.0, 0.0, 0.0], 3, 64, &bitmap, &thresholds); + let results = + adaptive_search(&idx, &[505.0, 0.0, 0.0], 3, 64, &bitmap, &thresholds).unwrap(); assert_eq!(results.len(), 3); for r in &results { assert!(bitmap.contains(r.id), "got filtered-out id {}", r.id); @@ -197,7 +203,8 @@ mod tests { bitmap.insert(i); } - let results = adaptive_search(&idx, &[100.0, 0.0, 0.0], 5, 64, &bitmap, &thresholds); + let results = + adaptive_search(&idx, &[100.0, 0.0, 0.0], 5, 64, &bitmap, &thresholds).unwrap(); assert_eq!(results.len(), 5); for r in &results { assert!(bitmap.contains(r.id)); @@ -213,4 +220,26 @@ mod tests { let sel = estimate_selectivity(&bitmap, 1000); assert!((sel - 0.9).abs() < 0.01); } + + /// Every strategy refuses a query of the wrong dimension, including the + /// brute-force path that never touches the graph. + #[test] + fn wrong_dimension_query_is_a_typed_error() { + let idx = build_test_index(); + let thresholds = FilterThresholds::default(); + for n in [5u32, 500, 1000] { + let bitmap: RoaringBitmap = (0..n).collect(); + let result = adaptive_search(&idx, &[1.0, 0.0], 3, 64, &bitmap, &thresholds); + assert!( + matches!( + result, + Err(crate::error::VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + ), + "{result:?}" + ); + } + } } diff --git a/nodedb-vector/src/codec_index/build.rs b/nodedb-vector/src/codec_index/build.rs index 8234f90a4..7f5c63825 100644 --- a/nodedb-vector/src/codec_index/build.rs +++ b/nodedb-vector/src/codec_index/build.rs @@ -11,6 +11,7 @@ use std::collections::{BinaryHeap, HashSet}; use nodedb_codec::vector_quant::codec::VectorCodec; use super::graph::{HnswCodecIndex, NodeC}; +use crate::error::{VectorError, check_dim}; /// Ordered pair for priority queues (dist, node_idx in `nodes` vec). #[derive(Clone, Copy, PartialEq)] @@ -41,8 +42,10 @@ impl HnswCodecIndex { /// Insert a vector with the given caller-supplied `id`. /// /// Encodes `v` via `codec.encode`, assigns a random layer, and runs the - /// standard HNSW neighbour-selection algorithm. - pub fn insert(&mut self, id: u32, v: &[f32]) { + /// standard HNSW neighbour-selection algorithm. A vector without the + /// index dimension fails with [`VectorError::DimensionMismatch`]. + pub fn insert(&mut self, id: u32, v: &[f32]) -> Result<(), VectorError> { + check_dim(self.dim, v.len())?; let quantized = self.codec.encode(v); let node_layer = self.random_layer(); @@ -64,7 +67,7 @@ impl HnswCodecIndex { // First node: it becomes the entry point. self.entry_point = Some(new_idx); self.max_layer = node_layer; - return; + return Ok(()); }; // Phase 1: greedy descent from max_layer down to node_layer + 1. @@ -124,6 +127,7 @@ impl HnswCodecIndex { self.entry_point = Some(new_idx); self.max_layer = node_layer; } + Ok(()) } /// Greedy descent: starting at `ep_idx`, find the single nearest node to @@ -262,14 +266,14 @@ mod tests { .map(|i| (0..dim).map(|d| (i * dim + d) as f32 * 0.1).collect()) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, dim) + Sq8Codec::calibrate(&refs, dim).unwrap() } #[test] fn insert_sets_entry_point() { let codec = make_sq8(4, 10); let mut idx: HnswCodecIndex = HnswCodecIndex::new(4, 8, 50, codec, 1); - idx.insert(0, &[0.1, 0.2, 0.3, 0.4]); + idx.insert(0, &[0.1, 0.2, 0.3, 0.4]).unwrap(); assert!(idx.entry_point.is_some()); assert_eq!(idx.len(), 1); } @@ -280,7 +284,7 @@ mod tests { let mut idx: HnswCodecIndex = HnswCodecIndex::new(4, 8, 50, codec, 42); for i in 0..20u32 { let v: Vec = (0..4).map(|d| (i as usize * 4 + d) as f32).collect(); - idx.insert(i, &v); + idx.insert(i, &v).unwrap(); } assert_eq!(idx.len(), 20); assert!(idx.entry_point.is_some()); diff --git a/nodedb-vector/src/codec_index/graph.rs b/nodedb-vector/src/codec_index/graph.rs index df4baf961..3cd0edc61 100644 --- a/nodedb-vector/src/codec_index/graph.rs +++ b/nodedb-vector/src/codec_index/graph.rs @@ -145,7 +145,7 @@ mod tests { .map(|i| (0..dim).map(|d| (i * dim + d) as f32 * 0.1).collect()) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, dim) + Sq8Codec::calibrate(&refs, dim).unwrap() } #[test] diff --git a/nodedb-vector/src/codec_index/search.rs b/nodedb-vector/src/codec_index/search.rs index befa7d46b..51b3a04d8 100644 --- a/nodedb-vector/src/codec_index/search.rs +++ b/nodedb-vector/src/codec_index/search.rs @@ -13,6 +13,7 @@ use std::collections::{BinaryHeap, HashSet}; use nodedb_codec::vector_quant::codec::VectorCodec; use super::graph::HnswCodecIndex; +use crate::error::{VectorError, check_dim}; /// A single result from a codec-index search. #[derive(Debug, Clone)] @@ -54,13 +55,21 @@ impl HnswCodecIndex { /// `ef_search` controls the beam width at layer 0 (must be >= k). /// /// The returned results are sorted ascending by `exact_asymmetric_distance`. - pub fn search(&self, query: &[f32], k: usize, ef_search: usize) -> Vec { + /// A query without the index dimension fails with + /// [`VectorError::DimensionMismatch`]. + pub fn search( + &self, + query: &[f32], + k: usize, + ef_search: usize, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; if self.is_empty() { - return Vec::new(); + return Ok(Vec::new()); } let Some(ep) = self.entry_point else { - return Vec::new(); + return Ok(Vec::new()); }; let ef = ef_search.max(k); @@ -93,10 +102,10 @@ impl HnswCodecIndex { reranked.sort_unstable_by(|a, b| a.0.total_cmp(&b.0)); reranked.truncate(k); - reranked + Ok(reranked .into_iter() .map(|(distance, id)| CodecSearchResult { id, distance }) - .collect() + .collect()) } /// Greedy single-nearest descent at `layer` using the pre-encoded query. @@ -229,14 +238,14 @@ mod tests { let mut state = 0xDEAD_BEEF_u64; let vecs: Vec> = (0..n).map(|_| rand_vec(&mut state, dim)).collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, dim); + let codec = Sq8Codec::calibrate(&refs, dim).unwrap(); let mut idx: HnswCodecIndex = HnswCodecIndex::new(dim, 8, 100, codec, 7); for (i, v) in vecs.iter().enumerate() { - idx.insert(i as u32, v); + idx.insert(i as u32, v).unwrap(); } // Query with vector 17 — top-1 should return id 17. let query = vecs[17].clone(); - let results = idx.search(&query, 1, 50); + let results = idx.search(&query, 1, 50).unwrap(); assert_eq!(results.len(), 1, "expected 1 result"); assert_eq!( results[0].id, 17, @@ -262,7 +271,7 @@ mod tests { let codec = RaBitQCodec::calibrate(&refs, dim, 0xABCD_1234); let mut idx: HnswCodecIndex = HnswCodecIndex::new(dim, 8, 150, codec, 99); for (i, v) in vecs.iter().enumerate() { - idx.insert(i as u32, v); + idx.insert(i as u32, v).unwrap(); } let n_queries = 10usize; @@ -272,7 +281,7 @@ mod tests { let query = rand_vec(&mut state, dim); let truth: std::collections::HashSet = ground_truth(&vecs, &query, k).into_iter().collect(); - let results = idx.search(&query, k, k * 4); + let results = idx.search(&query, k, k * 4).unwrap(); let found: std::collections::HashSet = results.iter().map(|r| r.id).collect(); total_hits += found.intersection(&truth).count(); total += k; @@ -300,7 +309,7 @@ mod tests { let codec = BbqCodec::calibrate(&refs, dim, 3); let mut idx: HnswCodecIndex = HnswCodecIndex::new(dim, 8, 150, codec, 42); for (i, v) in vecs.iter().enumerate() { - idx.insert(i as u32, v); + idx.insert(i as u32, v).unwrap(); } let n_queries = 10usize; @@ -310,7 +319,7 @@ mod tests { let query = rand_vec(&mut state, dim); let truth: std::collections::HashSet = ground_truth(&vecs, &query, k).into_iter().collect(); - let results = idx.search(&query, k, k * 4); + let results = idx.search(&query, k, k * 4).unwrap(); let found: std::collections::HashSet = results.iter().map(|r| r.id).collect(); total_hits += found.intersection(&truth).count(); total += k; @@ -332,10 +341,10 @@ mod tests { let codec = { let vecs: Vec> = (0..5).map(|i| vec![i as f32; 4]).collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, 4) + Sq8Codec::calibrate(&refs, 4).unwrap() }; let idx: HnswCodecIndex = HnswCodecIndex::new(4, 8, 50, codec, 1); - let results = idx.search(&[0.0, 0.0, 0.0, 0.0], 5, 20); + let results = idx.search(&[0.0, 0.0, 0.0, 0.0], 5, 20).unwrap(); assert!(results.is_empty(), "empty index must return no results"); } @@ -344,12 +353,43 @@ mod tests { let dim = 4; let vecs = [vec![1.0f32, 2.0, 3.0, 4.0]]; let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, dim); + let codec = Sq8Codec::calibrate(&refs, dim).unwrap(); let mut idx: HnswCodecIndex = HnswCodecIndex::new(dim, 8, 50, codec, 5); - idx.insert(0, &vecs[0]); + idx.insert(0, &vecs[0]).unwrap(); // Query with a completely different vector. - let results = idx.search(&[10.0, 20.0, 30.0, 40.0], 1, 10); + let results = idx.search(&[10.0, 20.0, 30.0, 40.0], 1, 10).unwrap(); assert_eq!(results.len(), 1, "single-node index must return 1 result"); assert_eq!(results[0].id, 0, "the only node must be returned"); } + + #[test] + fn wrong_dimension_is_a_typed_error() { + use crate::error::VectorError; + let vecs: Vec> = (0..5).map(|i| vec![i as f32; 4]).collect(); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let codec = Sq8Codec::calibrate(&refs, 4).unwrap(); + let mut idx: HnswCodecIndex = HnswCodecIndex::new(4, 8, 50, codec, 1); + assert!(matches!( + idx.search(&[0.0; 3], 5, 20), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + idx.insert(0, &vecs[0]).unwrap(); + assert!(matches!( + idx.search(&[0.0; 5], 5, 20), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 5 + }) + )); + assert!(matches!( + idx.insert(1, &[0.0; 2]), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 2 + }) + )); + } } diff --git a/nodedb-vector/src/collection/checkpoint.rs b/nodedb-vector/src/collection/checkpoint.rs index 3ff84e40d..18cb4a289 100644 --- a/nodedb-vector/src/collection/checkpoint.rs +++ b/nodedb-vector/src/collection/checkpoint.rs @@ -309,11 +309,14 @@ impl VectorCollection { let mut growing = FlatIndex::new(snap.dim, metric); for (i, v) in snap.growing_vectors.iter().enumerate() { let deleted = snap.growing_deleted.get(i).copied().unwrap_or(false); - if deleted { - growing.insert_tombstoned(v.clone()); + let inserted = if deleted { + growing.insert_tombstoned(v.clone()) } else { - growing.insert(v.clone()); - } + growing.insert(v.clone()) + }; + inserted.map_err(|e| VectorError::CheckpointDeserializationError { + detail: format!("growing-segment replay insert: {e}"), + })?; } let mut sealed = Vec::with_capacity(snap.sealed_segments.len()); @@ -503,7 +506,7 @@ mod tests { for (d, slot) in v.iter_mut().enumerate() { *slot = ((i as f32) * 0.01 + (d as f32) * 0.1).sin(); } - coll.insert(v); + coll.insert(v).unwrap(); } let req = coll.seal("sq8_test").expect("seal produced request"); let mut idx = HnswIndex::new(req.dim, req.params.clone()); @@ -551,14 +554,14 @@ mod tests { }, ); for i in 0..50u32 { - coll.insert(vec![i as f32, 0.0, 0.0]); + coll.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } let bytes = coll.checkpoint_to_bytes(None).unwrap(); let restored = VectorCollection::from_checkpoint(&bytes, None, test_memory()).unwrap(); assert_eq!(restored.len(), 50); assert_eq!(restored.dim(), 3); - let results = restored.search(&[25.0, 0.0, 0.0], 1, 64); + let results = restored.search(&[25.0, 0.0, 0.0], 1, 64).unwrap(); assert_eq!(results[0].id, 25); } @@ -579,7 +582,7 @@ mod tests { ..HnswParams::default() }, ); - coll.insert(vec![1.0, 0.0, 0.0]); + coll.insert(vec![1.0, 0.0, 0.0]).unwrap(); coll.note_checkpoint_lsn(42); assert_eq!(coll.applied_wal_lsn(), 42); assert_eq!( @@ -634,7 +637,7 @@ mod tests { coll.payload .add_index("category".to_string(), PayloadIndexKind::Equality); for i in 0u32..10 { - let node_id = coll.insert(vec![i as f32, 0.0, 0.0]); + let node_id = coll.insert(vec![i as f32, 0.0, 0.0]).unwrap(); let mut fields = HashMap::new(); let cat = if i % 2 == 0 { "A" } else { "B" }; fields.insert("category".to_string(), Value::String(cat.to_string())); diff --git a/nodedb-vector/src/collection/codec_build.rs b/nodedb-vector/src/collection/codec_build.rs index dde9c6841..e683c1c71 100644 --- a/nodedb-vector/src/collection/codec_build.rs +++ b/nodedb-vector/src/collection/codec_build.rs @@ -8,57 +8,46 @@ use super::codec_dispatch::{CollectionCodec, build_collection_codec}; use super::lifecycle::VectorCollection; -impl VectorCollection { - /// Collect all live FP32 vectors from every segment (growing, building, - /// and sealed) in insertion order. Used to train the collection-level - /// codec-dispatch index. - pub(crate) fn gather_all_vectors_fp32(&self) -> Vec> { - let total = self.len(); - let mut out = Vec::with_capacity(total); - - for i in 0..self.growing.len() as u32 { - if let Some(v) = self.growing.get_vector(i) { - out.push(v.to_vec()); - } - } - - for seg in &self.building { - for i in 0..seg.flat.len() as u32 { - if let Some(v) = seg.flat.get_vector(i) { - out.push(v.to_vec()); - } - } - } +use crate::error::VectorError; +impl VectorCollection { + /// Every live FP32 vector of the sealed segments, keyed by its global + /// vector id. The collection-level codec index covers the sealed + /// segments only: search reads the growing and building segments by + /// brute force beside it. + pub(crate) fn gather_sealed_vectors_fp32(&self) -> Vec<(u32, Vec)> { + let mut out = Vec::new(); for seg in &self.sealed { let n = seg.index.len(); for i in 0..n as u32 { if !seg.index.is_deleted(i) && let Some(v) = seg.index.get_vector(i) { - out.push(v.to_vec()); + out.push((seg.base_id + i, v.to_vec())); } } } - out } - /// Build a codec-dispatched index over all current vectors using the + /// Build a codec-dispatched index over the sealed vectors using the /// requested quantization. Replaces any existing dispatch index for /// this collection. Idempotent. /// - /// Returns a reference to the new index, or `None` if the quantization - /// tag is not supported (falls back to per-segment Sq8/PQ paths) or there - /// are no vectors to train on. - pub fn build_codec_dispatch(&mut self, quantization: &str) -> Option<&CollectionCodec> { - let vectors = self.gather_all_vectors_fp32(); + /// Returns a reference to the new index, or `Ok(None)` if the + /// quantization tag is not supported (falls back to per-segment Sq8/PQ + /// paths) or there are no vectors to train on. + pub fn build_codec_dispatch( + &mut self, + quantization: &str, + ) -> Result, VectorError> { + let vectors = self.gather_sealed_vectors_fp32(); let dim = self.dim; let m = self.params.m; let ef_construction = self.params.ef_construction; let seed = 42_u64; self.codec_dispatch = - build_collection_codec(quantization, &vectors, dim, m, ef_construction, seed); - self.codec_dispatch.as_ref() + build_collection_codec(quantization, &vectors, dim, m, ef_construction, seed)?; + Ok(self.codec_dispatch.as_ref()) } } diff --git a/nodedb-vector/src/collection/codec_dispatch.rs b/nodedb-vector/src/collection/codec_dispatch.rs index 6c574c2cc..0da1379ef 100644 --- a/nodedb-vector/src/collection/codec_dispatch.rs +++ b/nodedb-vector/src/collection/codec_dispatch.rs @@ -8,6 +8,7 @@ use nodedb_codec::vector_quant::bbq::BbqCodec; use nodedb_codec::vector_quant::rabitq::RaBitQCodec; use crate::codec_index::HnswCodecIndex; +use crate::error::VectorError; /// One built codec-index per collection (other than Sq8). Variants match /// the publicly-selectable quantization choices that route through @@ -26,7 +27,7 @@ impl CollectionCodec { query: &[f32], k: usize, ef_search: usize, - ) -> Vec { + ) -> Result, VectorError> { match self { Self::RaBitQ(idx) => idx.search(query, k, ef_search), Self::Bbq(idx) => idx.search(query, k, ef_search), @@ -34,7 +35,7 @@ impl CollectionCodec { } /// Forwarding `insert`. - pub fn insert(&mut self, id: u32, v: &[f32]) { + pub fn insert(&mut self, id: u32, v: &[f32]) -> Result<(), VectorError> { match self { Self::RaBitQ(idx) => idx.insert(id, v), Self::Bbq(idx) => idx.insert(id, v), @@ -64,38 +65,47 @@ impl CollectionCodec { /// Build a `CollectionCodec` from a quantization tag and training vectors. /// -/// Returns `None` for unsupported or unrecognised tags (e.g. "sq8", "pq", -/// "none") — those variants use separate per-segment code paths. +/// Returns `Ok(None)` for unsupported or unrecognised tags (e.g. "sq8", +/// "pq", "none") — those variants use separate per-segment code paths — and +/// for an empty vector set. A vector without `dim` components fails with +/// [`VectorError::DimensionMismatch`]. +/// +/// Each entry is `(id, vector)`; the index reports `id` in its results, so +/// it must be the collection's global vector id. pub fn build_collection_codec( quantization: &str, - vectors: &[Vec], + vectors: &[(u32, Vec)], dim: usize, m: usize, ef_construction: usize, seed: u64, -) -> Option { +) -> Result, VectorError> { if vectors.is_empty() { - return None; + return Ok(None); + } + // Calibration reads every vector as `dim` components; check before it. + for (_, v) in vectors { + crate::error::check_dim(dim, v.len())?; } - let refs: Vec<&[f32]> = vectors.iter().map(|v| v.as_slice()).collect(); + let refs: Vec<&[f32]> = vectors.iter().map(|(_, v)| v.as_slice()).collect(); match quantization { "rabitq" => { let codec = RaBitQCodec::calibrate(&refs, dim, seed); let mut idx = HnswCodecIndex::new(dim, m, ef_construction, codec, seed); - for (i, v) in vectors.iter().enumerate() { - idx.insert(i as u32, v); + for (id, v) in vectors { + idx.insert(*id, v)?; } - Some(CollectionCodec::RaBitQ(idx)) + Ok(Some(CollectionCodec::RaBitQ(idx))) } "bbq" => { let codec = BbqCodec::calibrate(&refs, dim, 3); let mut idx = HnswCodecIndex::new(dim, m, ef_construction, codec, seed); - for (i, v) in vectors.iter().enumerate() { - idx.insert(i as u32, v); + for (id, v) in vectors { + idx.insert(*id, v)?; } - Some(CollectionCodec::Bbq(idx)) + Ok(Some(CollectionCodec::Bbq(idx))) } - _ => None, + _ => Ok(None), } } @@ -103,16 +113,19 @@ pub fn build_collection_codec( mod tests { use super::*; - fn make_vectors(n: usize, dim: usize) -> Vec> { + fn make_vectors(n: usize, dim: usize) -> Vec<(u32, Vec)> { (0..n) - .map(|i| (0..dim).map(|d| (i * dim + d) as f32 * 0.01).collect()) + .map(|i| { + let v = (0..dim).map(|d| (i * dim + d) as f32 * 0.01).collect(); + (i as u32, v) + }) .collect() } #[test] fn build_rabitq_returns_some() { let vecs = make_vectors(50, 8); - let result = build_collection_codec("rabitq", &vecs, 8, 16, 100, 42); + let result = build_collection_codec("rabitq", &vecs, 8, 16, 100, 42).unwrap(); assert!( matches!(result, Some(CollectionCodec::RaBitQ(_))), "expected RaBitQ variant" @@ -122,7 +135,7 @@ mod tests { #[test] fn build_bbq_returns_some() { let vecs = make_vectors(50, 8); - let result = build_collection_codec("bbq", &vecs, 8, 16, 100, 42); + let result = build_collection_codec("bbq", &vecs, 8, 16, 100, 42).unwrap(); assert!( matches!(result, Some(CollectionCodec::Bbq(_))), "expected Bbq variant" @@ -132,14 +145,14 @@ mod tests { #[test] fn unknown_codec_returns_none() { let vecs = make_vectors(50, 8); - let result = build_collection_codec("unknown_codec", &vecs, 8, 16, 100, 42); + let result = build_collection_codec("unknown_codec", &vecs, 8, 16, 100, 42).unwrap(); assert!(result.is_none(), "unknown codec should return None"); } #[test] fn sq8_tag_returns_none() { let vecs = make_vectors(50, 8); - let result = build_collection_codec("sq8", &vecs, 8, 16, 100, 42); + let result = build_collection_codec("sq8", &vecs, 8, 16, 100, 42).unwrap(); assert!( result.is_none(), "sq8 tag should fall through to per-segment path" @@ -148,14 +161,16 @@ mod tests { #[test] fn empty_vectors_returns_none() { - let result = build_collection_codec("rabitq", &[], 8, 16, 100, 42); + let result = build_collection_codec("rabitq", &[], 8, 16, 100, 42).unwrap(); assert!(result.is_none(), "empty vectors should return None"); } #[test] fn len_and_is_empty() { let vecs = make_vectors(20, 4); - let codec = build_collection_codec("bbq", &vecs, 4, 8, 50, 1).unwrap(); + let codec = build_collection_codec("bbq", &vecs, 4, 8, 50, 1) + .unwrap() + .unwrap(); assert_eq!(codec.len(), 20); assert!(!codec.is_empty()); } @@ -163,9 +178,45 @@ mod tests { #[test] fn quantization_tag() { let vecs = make_vectors(10, 4); - let rabitq = build_collection_codec("rabitq", &vecs, 4, 8, 50, 1).unwrap(); + let rabitq = build_collection_codec("rabitq", &vecs, 4, 8, 50, 1) + .unwrap() + .unwrap(); assert_eq!(rabitq.quantization(), "rabitq"); - let bbq = build_collection_codec("bbq", &vecs, 4, 8, 50, 1).unwrap(); + let bbq = build_collection_codec("bbq", &vecs, 4, 8, 50, 1) + .unwrap() + .unwrap(); assert_eq!(bbq.quantization(), "bbq"); } + + #[test] + fn wrong_dimension_query_is_a_typed_error() { + let vecs = make_vectors(20, 4); + for tag in ["rabitq", "bbq"] { + let mut codec = build_collection_codec(tag, &vecs, 4, 8, 50, 1) + .unwrap() + .unwrap(); + assert!(matches!( + codec.search(&[0.0; 3], 5, 20), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + assert!(matches!( + codec.insert(99, &[0.0; 5]), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 5 + }) + )); + } + let short = vec![(0_u32, vec![0.0_f32; 3])]; + assert!(matches!( + build_collection_codec("rabitq", &short, 4, 8, 50, 1), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + } } diff --git a/nodedb-vector/src/collection/lifecycle.rs b/nodedb-vector/src/collection/lifecycle.rs index d27fe109b..ce93c7ecd 100644 --- a/nodedb-vector/src/collection/lifecycle.rs +++ b/nodedb-vector/src/collection/lifecycle.rs @@ -260,10 +260,12 @@ impl VectorCollection { .position(|b| b.segment_id == segment_id) { let building = self.building.remove(pos); - let use_codec_dispatch = matches!( - self.quantization, - VectorQuantization::RaBitQ | VectorQuantization::Bbq - ); + let codec_dispatch_tag = match self.quantization { + VectorQuantization::RaBitQ => Some("rabitq"), + VectorQuantization::Bbq => Some("bbq"), + _ => None, + }; + let use_codec_dispatch = codec_dispatch_tag.is_some(); let use_pq = !use_codec_dispatch && self.index_config.index_type == IndexType::HnswPq; let (sq8, pq) = if use_codec_dispatch { (None, None) @@ -287,15 +289,14 @@ impl VectorCollection { mmap_vectors, }); - if use_codec_dispatch { - let tag = match self.quantization { - VectorQuantization::RaBitQ => "rabitq", - VectorQuantization::Bbq => "bbq", - _ => unreachable!( - "invariant: use_codec_dispatch is only true for RaBitQ and Bbq quantization variants" - ), - }; - self.build_codec_dispatch(tag); + if let Some(tag) = codec_dispatch_tag { + let built = self.build_codec_dispatch(tag).map(|_| ()); + if let Err(e) = built { + // Without the codec index the sealed segments are searched + // by their own HNSW graphs, which answer the same queries. + tracing::error!(error = %e, tag, "codec-dispatch build failed; searching sealed segments directly"); + self.codec_dispatch = None; + } } } } diff --git a/nodedb-vector/src/collection/lifecycle_compact.rs b/nodedb-vector/src/collection/lifecycle_compact.rs index 7d7dc13bc..0343f73a6 100644 --- a/nodedb-vector/src/collection/lifecycle_compact.rs +++ b/nodedb-vector/src/collection/lifecycle_compact.rs @@ -171,7 +171,7 @@ mod tests { fields.insert("owner".to_string(), Value::String("a".into())); for (i, v) in [[1.0, 0.0], [0.0, 1.0], [1.0, 1.0]].into_iter().enumerate() { let s = Surrogate::new(i as u32 + 1); - let id = coll.insert_with_surrogate(v.to_vec(), s); + let id = coll.insert_with_surrogate(v.to_vec(), s).unwrap(); coll.payload.insert_row(id, &fields); } assert!( @@ -204,7 +204,7 @@ mod tests { assert!(hits.is_empty(), "payload rows cleared"); let s = Surrogate::new(42); - let id = coll.insert_with_surrogate(vec![0.5, 0.5], s); + let id = coll.insert_with_surrogate(vec![0.5, 0.5], s).unwrap(); assert_eq!(coll.local_for_surrogate(s), Some(id)); assert_eq!(coll.live_count(), 1); } diff --git a/nodedb-vector/src/collection/lifecycle_insert_ops.rs b/nodedb-vector/src/collection/lifecycle_insert_ops.rs index f962f1508..886d1c78a 100644 --- a/nodedb-vector/src/collection/lifecycle_insert_ops.rs +++ b/nodedb-vector/src/collection/lifecycle_insert_ops.rs @@ -5,14 +5,19 @@ use nodedb_types::Surrogate; use super::lifecycle::VectorCollection; +use crate::error::{VectorError, check_dim}; impl VectorCollection { /// Insert a vector. Returns the global vector ID. - pub fn insert(&mut self, vector: Vec) -> u32 { + /// + /// A vector without the collection dimension fails with + /// [`VectorError::DimensionMismatch`] and changes nothing. + pub fn insert(&mut self, vector: Vec) -> Result { + check_dim(self.dim, vector.len())?; let id = self.next_id; - self.growing.insert(vector); + self.growing.insert(vector)?; self.next_id += 1; - id + Ok(id) } /// Insert a vector with an associated surrogate. The surrogate is @@ -24,31 +29,69 @@ impl VectorCollection { /// re-insert never leaves an unreachable node scoring in searches. /// The caller owns the payload bitmap entries of the old node and /// removes them with [`Self::local_for_surrogate`] before this call. - pub fn insert_with_surrogate(&mut self, vector: Vec, surrogate: Surrogate) -> u32 { + /// + /// A vector without the collection dimension fails with + /// [`VectorError::DimensionMismatch`] before the old binding is touched. + pub fn insert_with_surrogate( + &mut self, + vector: Vec, + surrogate: Surrogate, + ) -> Result { + check_dim(self.dim, vector.len())?; if surrogate != Surrogate::ZERO && let Some(old) = self.surrogate_to_local.get(&surrogate).copied() { self.delete_inner(old); self.surrogate_map.remove(&old); } - let id = self.insert(vector); + let id = self.insert(vector)?; if surrogate != Surrogate::ZERO { self.surrogate_map.insert(id, surrogate); self.surrogate_to_local.insert(surrogate, id); } - id + Ok(id) + } + + /// Insert a batch of vectors, the `i`-th bound to `surrogates[i]` + /// ([`Surrogate::ZERO`] when the slice is shorter). Returns the global ids + /// in batch order. + /// + /// Every vector is checked against the collection dimension before any is + /// inserted, so a mismatch fails with [`VectorError::DimensionMismatch`] + /// and inserts none. + pub fn insert_batch_with_surrogates( + &mut self, + vectors: &[Vec], + surrogates: &[Surrogate], + ) -> Result, VectorError> { + for v in vectors { + check_dim(self.dim, v.len())?; + } + let mut ids = Vec::with_capacity(vectors.len()); + for (i, v) in vectors.iter().enumerate() { + let surrogate = surrogates.get(i).copied().unwrap_or(Surrogate::ZERO); + ids.push(self.insert_with_surrogate(v.clone(), surrogate)?); + } + Ok(ids) } /// Insert multiple vectors for a single document (ColBERT-style). /// All N vectors are bound to the same `document_surrogate`. + /// + /// Every vector is checked against the collection dimension before any + /// is inserted, so a mismatch fails with + /// [`VectorError::DimensionMismatch`] and inserts none. pub fn insert_multi_vector( &mut self, vectors: &[&[f32]], document_surrogate: Surrogate, - ) -> Vec { + ) -> Result, VectorError> { + for v in vectors { + check_dim(self.dim, v.len())?; + } let mut ids = Vec::with_capacity(vectors.len()); for &v in vectors { - let id = self.insert(v.to_vec()); + let id = self.insert(v.to_vec())?; if document_surrogate != Surrogate::ZERO { self.surrogate_map.insert(id, document_surrogate); } @@ -57,7 +100,7 @@ impl VectorCollection { if document_surrogate != Surrogate::ZERO { self.multi_doc_map.insert(document_surrogate, ids.clone()); } - ids + Ok(ids) } /// Delete all vectors belonging to a multi-vector document. @@ -233,8 +276,8 @@ mod tests { fn re_insert_under_the_same_surrogate_leaves_one_live_node() { let mut coll = collection(); let s = Surrogate::new(7); - let first = coll.insert_with_surrogate(vec![1.0, 0.0], s); - let second = coll.insert_with_surrogate(vec![0.0, 1.0], s); + let first = coll.insert_with_surrogate(vec![1.0, 0.0], s).unwrap(); + let second = coll.insert_with_surrogate(vec![0.0, 1.0], s).unwrap(); assert_ne!(first, second); assert_eq!(coll.live_count(), 1, "the old node must be tombstoned"); assert_eq!(coll.local_for_surrogate(s), Some(second)); @@ -246,10 +289,10 @@ mod tests { fn delete_then_insert_under_the_same_surrogate_leaves_one_live_node() { let mut coll = collection(); let s = Surrogate::new(9); - let first = coll.insert_with_surrogate(vec![1.0, 0.0], s); + let first = coll.insert_with_surrogate(vec![1.0, 0.0], s).unwrap(); assert!(coll.delete_by_surrogate(s)); assert_eq!(coll.local_for_surrogate(s), None); - let second = coll.insert_with_surrogate(vec![0.0, 1.0], s); + let second = coll.insert_with_surrogate(vec![0.0, 1.0], s).unwrap(); assert_ne!(first, second); assert_eq!(coll.live_count(), 1); assert_eq!(coll.local_for_surrogate(s), Some(second)); @@ -260,7 +303,7 @@ mod tests { fn delete_by_surrogate_is_idempotent() { let mut coll = collection(); let s = Surrogate::new(3); - coll.insert_with_surrogate(vec![1.0, 0.0], s); + coll.insert_with_surrogate(vec![1.0, 0.0], s).unwrap(); assert!(coll.delete_by_surrogate(s)); assert!(!coll.delete_by_surrogate(s)); assert_eq!(coll.live_count(), 0); @@ -270,7 +313,7 @@ mod tests { fn vector_for_surrogate_reads_the_growing_segment_and_hides_deletes() { let mut coll = collection(); let s = Surrogate::new(11); - coll.insert_with_surrogate(vec![0.5, 0.25], s); + coll.insert_with_surrogate(vec![0.5, 0.25], s).unwrap(); assert_eq!(coll.vector_for_surrogate(s), Some(vec![0.5, 0.25])); assert!(coll.delete_by_surrogate(s)); assert_eq!(coll.vector_for_surrogate(s), None); diff --git a/nodedb-vector/src/collection/quantize.rs b/nodedb-vector/src/collection/quantize.rs index e3a6c4eba..09bbda007 100644 --- a/nodedb-vector/src/collection/quantize.rs +++ b/nodedb-vector/src/collection/quantize.rs @@ -70,7 +70,9 @@ impl VectorCollection { return None; } - let codec = Sq8Codec::calibrate(&refs, dim); + // The refs are live vectors of this index, so calibration fails only + // for a zero dimension, which has nothing to quantize. + let codec = Sq8Codec::calibrate(&refs, dim).ok()?; let mut data = Vec::with_capacity(dim * n); for i in 0..n { @@ -109,7 +111,16 @@ impl VectorCollection { } let refs_slices: Vec<&[f32]> = refs.iter().map(|v| v.as_slice()).collect(); let k = 256usize.min(refs.len()); - let codec = PqCodec::train(&refs_slices, dim, pq_m, k, 20, memory); + // A PQ shape the codec cannot train (a codebook over its byte limit, + // a dimension over its decode limit) leaves the segment on plain + // HNSW, which answers the same queries exactly. + let codec = match PqCodec::train(&refs_slices, dim, pq_m, k, 20, memory) { + Ok(codec) => codec, + Err(e) => { + tracing::warn!(error = %e, dim, pq_m, "PQ training refused; segment stays unquantized"); + return None; + } + }; let codes = codec.encode_batch(&refs_slices).ok()?; Some((codec, codes)) } diff --git a/nodedb-vector/src/collection/rollback.rs b/nodedb-vector/src/collection/rollback.rs index 49c7fe015..8da0ae116 100644 --- a/nodedb-vector/src/collection/rollback.rs +++ b/nodedb-vector/src/collection/rollback.rs @@ -188,13 +188,16 @@ mod tests { #[test] fn a_rolled_back_insert_returns_the_id_counter_and_the_binding() { let mut coll = collection(); - coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(1)); + coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(1)) + .unwrap(); let mark = coll.write_mark(&[Surrogate::new(2)], &[]); - coll.insert_with_surrogate(vec![0.3, 0.4], Surrogate::new(2)); + coll.insert_with_surrogate(vec![0.3, 0.4], Surrogate::new(2)) + .unwrap(); assert!(coll.roll_back_to(mark)); assert_eq!(coll.local_for_surrogate(Surrogate::new(2)), None); assert_eq!( - coll.insert_with_surrogate(vec![0.5, 0.6], Surrogate::new(3)), + coll.insert_with_surrogate(vec![0.5, 0.6], Surrogate::new(3)) + .unwrap(), 1, "the next insert takes the id the rolled-back insert took" ); @@ -203,9 +206,12 @@ mod tests { #[test] fn a_rolled_back_rebind_restores_the_replaced_node() { let mut coll = collection(); - let first = coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(7)); + let first = coll + .insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(7)) + .unwrap(); let mark = coll.write_mark(&[Surrogate::new(7)], &[]); - coll.insert_with_surrogate(vec![0.3, 0.4], Surrogate::new(7)); + coll.insert_with_surrogate(vec![0.3, 0.4], Surrogate::new(7)) + .unwrap(); assert!(!coll.is_live(first)); assert!(coll.roll_back_to(mark)); assert!(coll.is_live(first)); @@ -215,7 +221,9 @@ mod tests { #[test] fn a_rolled_back_delete_restores_the_node_and_its_binding() { let mut coll = collection(); - let id = coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(4)); + let id = coll + .insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(4)) + .unwrap(); let mark = coll.write_mark(&[], &[id]); coll.delete(id); assert!(coll.roll_back_to(mark)); @@ -227,7 +235,8 @@ mod tests { fn a_seal_after_the_mark_refuses_the_rollback() { let mut coll = VectorCollection::with_seal_threshold(2, HnswParams::default(), 1); let mark = coll.write_mark(&[Surrogate::new(1)], &[]); - coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(1)); + coll.insert_with_surrogate(vec![0.1, 0.2], Surrogate::new(1)) + .unwrap(); assert!(coll.seal("k").is_some()); assert!(!coll.roll_back_to(mark)); } @@ -235,10 +244,10 @@ mod tests { #[test] fn a_detached_empty_collection_continues_the_id_counter() { let mut coll = collection(); - coll.insert(vec![0.1, 0.2]); - coll.insert(vec![0.3, 0.4]); + coll.insert(vec![0.1, 0.2]).unwrap(); + coll.insert(vec![0.3, 0.4]).unwrap(); let mut fresh = coll.detached_empty(); assert_eq!(fresh.live_count(), 0); - assert_eq!(fresh.insert(vec![0.5, 0.6]), 2); + assert_eq!(fresh.insert(vec![0.5, 0.6]).unwrap(), 2); } } diff --git a/nodedb-vector/src/collection/search.rs b/nodedb-vector/src/collection/search.rs index 32e4a0087..57474a2a5 100644 --- a/nodedb-vector/src/collection/search.rs +++ b/nodedb-vector/src/collection/search.rs @@ -10,7 +10,7 @@ //! never silently dropped. use crate::distance::{DistanceMetric, distance}; -use crate::error::VectorError; +use crate::error::{VectorError, check_dim}; use crate::hnsw::SearchResult; use super::lifecycle::VectorCollection; @@ -50,7 +50,7 @@ fn quantized_search( metric: DistanceMetric, ) -> Result, VectorError> { let rerank_k = top_k.saturating_mul(3).max(20); - let hnsw_candidates = seg.index.search(query, rerank_k, ef); + let hnsw_candidates = seg.index.search(query, rerank_k, ef)?; // Phase 1: rank candidates by quantized distance. let mut scored: Vec<(u32, f32)> = if let Some((codec, codes)) = &seg.pq { @@ -118,92 +118,92 @@ fn quantized_search( Ok(reranked) } -impl VectorCollection { - /// Search across all segments, merging results by distance. - pub fn search(&self, query: &[f32], top_k: usize, ef: usize) -> Vec { - // Codec-dispatch fast path: if a collection-level HnswCodecIndex has - // been built (RaBitQ or BBQ), use it exclusively for sealed-segment - // results and fall back to the growing/building flat segments only. - if let Some(ref dispatch) = self.codec_dispatch { - let mut all: Vec = Vec::new(); - - let codec_results = dispatch.search(query, top_k, ef); - for r in codec_results { - all.push(SearchResult { - id: r.id, - distance: r.distance, - }); - } - - // Growing segment (brute-force, not yet in codec index). - let growing_results = self.growing.search(query, top_k); - for mut r in growing_results { - r.id += self.growing_base_id; - all.push(r); - } +/// Search one sealed segment: through its quantized codec when it has one, +/// else its HNSW graph. A codec pass that exceeds the memory budget falls +/// back to the HNSW graph, which answers the same query from FP32 vectors. +/// Every other error fails the search. +fn search_sealed( + seg: &SealedSegment, + query: &[f32], + top_k: usize, + ef: usize, + metric: DistanceMetric, +) -> Result, VectorError> { + if seg.pq.is_none() && seg.sq8.is_none() { + return seg.index.search(query, top_k, ef); + } + match quantized_search(seg, query, top_k, ef, metric) { + Ok(results) => Ok(results), + Err(VectorError::BudgetExhausted(e)) => { + tracing::warn!(error = %e, "quantized search over budget; searching the HNSW graph"); + seg.index.search(query, top_k, ef) + } + Err(e) => Err(e), + } +} - // Building segments (brute-force while codec index rebuilds). - for seg in &self.building { - let results = seg.flat.search(query, top_k); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } - } +/// Shift segment-local result ids to global ids and append them. +fn push_shifted(all: &mut Vec, results: Vec, base_id: u32) { + all.extend(results.into_iter().map(|mut r| { + r.id += base_id; + r + })); +} - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - return all; - } +/// Order merged results by distance and keep the `top_k` nearest. +fn finish(mut all: Vec, top_k: usize) -> Vec { + all.sort_by(|a, b| { + a.distance + .partial_cmp(&b.distance) + .unwrap_or(std::cmp::Ordering::Equal) + }); + all.truncate(top_k); + all +} +impl VectorCollection { + /// Search across all segments, merging results by distance. + /// + /// A query without the collection dimension fails with + /// [`VectorError::DimensionMismatch`] before any segment is read. + pub fn search( + &self, + query: &[f32], + top_k: usize, + ef: usize, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); - // Search growing segment (brute-force). - let growing_results = self.growing.search(query, top_k); - for mut r in growing_results { - r.id += self.growing_base_id; - all.push(r); - } - - // Search sealed segments. - for seg in &self.sealed { - let results = if seg.pq.is_some() || seg.sq8.is_some() { - match quantized_search(seg, query, top_k, ef, self.params.metric) { - Ok(r) => r, - Err(e) => { - tracing::warn!(error = %e, "quantized_search budget exhausted; skipping segment"); - seg.index.search(query, top_k, ef) - } - } - } else { - seg.index.search(query, top_k, ef) - }; - for mut r in results { - r.id += seg.base_id; - all.push(r); + // Codec-dispatch fast path: a collection-level HnswCodecIndex (RaBitQ + // or BBQ) answers for the sealed segments; the growing and building + // segments are read by brute force beside it. + if let Some(ref dispatch) = self.codec_dispatch { + all.extend( + dispatch + .search(query, top_k, ef)? + .into_iter() + .map(|r| SearchResult { + id: r.id, + distance: r.distance, + }), + ); + } else { + for seg in &self.sealed { + let results = search_sealed(seg, query, top_k, ef, self.params.metric)?; + push_shifted(&mut all, results, seg.base_id); } } - // Search building segments (brute-force while HNSW builds). + push_shifted( + &mut all, + self.growing.search(query, top_k)?, + self.growing_base_id, + ); for seg in &self.building { - let results = seg.flat.search(query, top_k); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } + push_shifted(&mut all, seg.flat.search(query, top_k)?, seg.base_id); } - - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - all + Ok(finish(all, top_k)) } /// Search across all segments using an explicit metric override. @@ -212,88 +212,53 @@ impl VectorCollection { /// during candidate reranking. Growing and building segments apply it exactly /// via brute-force. The HNSW graph structure was built with the collection /// metric; using a different metric affects the scoring but not graph traversal. + /// A codec-dispatch index scores with the collection metric. pub fn search_with_metric( &self, query: &[f32], top_k: usize, ef: usize, metric: DistanceMetric, - ) -> Vec { - // Codec-dispatch fast path: codec dispatch does not yet support per-query - // metric override — fall through to the non-codec path which does. - // When a codec index is active, we search only the growing/building - // segments with the override and add codec results with collection metric - // (approximate cross-metric search for the codec-indexed segments). - if let Some(ref dispatch) = self.codec_dispatch { - let mut all: Vec = Vec::new(); - let codec_results = dispatch.search(query, top_k, ef); - for r in codec_results { - all.push(SearchResult { - id: r.id, - distance: r.distance, - }); - } - for mut r in self.growing.search_with_metric(query, top_k, metric) { - r.id += self.growing_base_id; - all.push(r); - } - for seg in &self.building { - for mut r in seg.flat.search_with_metric(query, top_k, metric) { - r.id += seg.base_id; - all.push(r); - } - } - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - return all; - } - + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); - for mut r in self.growing.search_with_metric(query, top_k, metric) { - r.id += self.growing_base_id; - all.push(r); - } - - for seg in &self.sealed { - let results = if seg.pq.is_some() || seg.sq8.is_some() { - match quantized_search(seg, query, top_k, ef, metric) { - Ok(r) => r, - Err(e) => { - tracing::warn!(error = %e, "quantized_search budget exhausted; skipping segment"); - seg.index.search(query, top_k, ef) - } - } - } else { - seg.index.search(query, top_k, ef) - }; - for mut r in results { - r.id += seg.base_id; - all.push(r); + if let Some(ref dispatch) = self.codec_dispatch { + all.extend( + dispatch + .search(query, top_k, ef)? + .into_iter() + .map(|r| SearchResult { + id: r.id, + distance: r.distance, + }), + ); + } else { + for seg in &self.sealed { + let results = search_sealed(seg, query, top_k, ef, metric)?; + push_shifted(&mut all, results, seg.base_id); } } + push_shifted( + &mut all, + self.growing.search_with_metric(query, top_k, metric)?, + self.growing_base_id, + ); for seg in &self.building { - for mut r in seg.flat.search_with_metric(query, top_k, metric) { - r.id += seg.base_id; - all.push(r); - } + push_shifted( + &mut all, + seg.flat.search_with_metric(query, top_k, metric)?, + seg.base_id, + ); } - - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - all + Ok(finish(all, top_k)) } /// Search with a pre-filter bitmap (byte-array format) and explicit metric override. + /// + /// Bitmap bytes that do not decode fail with + /// [`VectorError::InvalidFilterBitmap`]. pub fn search_with_bitmap_bytes_and_metric( &self, query: &[f32], @@ -301,103 +266,88 @@ impl VectorCollection { ef: usize, bitmap: &[u8], metric: DistanceMetric, - ) -> Vec { + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); - let growing_results = self.growing.search_filtered_offset_with_metric( - query, - top_k, - bitmap, + push_shifted( + &mut all, + self.growing.search_filtered_offset_with_metric( + query, + top_k, + bitmap, + self.growing_base_id, + metric, + )?, self.growing_base_id, - metric, ); - for mut r in growing_results { - r.id += self.growing_base_id; - all.push(r); - } for seg in &self.sealed { - let results = + let mut results = seg.index - .search_with_bitmap_bytes_offset(query, top_k, ef, bitmap, seg.base_id); - for mut r in results { - // Rerank with the requested metric using the stored FP32 vector. - if let Some(v) = seg.index.get_vector(r.id.wrapping_sub(seg.base_id)) { - r.distance = crate::distance::distance(query, v, metric); + .search_with_bitmap_bytes_offset(query, top_k, ef, bitmap, seg.base_id)?; + // Rerank with the requested metric using the stored FP32 vector. + for r in &mut results { + if let Some(v) = seg.index.get_vector(r.id) { + r.distance = distance(query, v, metric); } - r.id += seg.base_id; - all.push(r); } + push_shifted(&mut all, results, seg.base_id); } for seg in &self.building { - let results = seg.flat.search_filtered_offset_with_metric( - query, - top_k, - bitmap, + push_shifted( + &mut all, + seg.flat.search_filtered_offset_with_metric( + query, + top_k, + bitmap, + seg.base_id, + metric, + )?, seg.base_id, - metric, ); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } } - - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - all + Ok(finish(all, top_k)) } /// Search with a pre-filter bitmap (byte-array format). + /// + /// Bitmap bytes that do not decode fail with + /// [`VectorError::InvalidFilterBitmap`]. pub fn search_with_bitmap_bytes( &self, query: &[f32], top_k: usize, ef: usize, bitmap: &[u8], - ) -> Vec { + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); - let growing_results = + push_shifted( + &mut all, self.growing - .search_filtered_offset(query, top_k, bitmap, self.growing_base_id); - for mut r in growing_results { - r.id += self.growing_base_id; - all.push(r); - } - + .search_filtered_offset(query, top_k, bitmap, self.growing_base_id)?, + self.growing_base_id, + ); for seg in &self.sealed { - let results = + push_shifted( + &mut all, seg.index - .search_with_bitmap_bytes_offset(query, top_k, ef, bitmap, seg.base_id); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } + .search_with_bitmap_bytes_offset(query, top_k, ef, bitmap, seg.base_id)?, + seg.base_id, + ); } - for seg in &self.building { - let results = seg - .flat - .search_filtered_offset(query, top_k, bitmap, seg.base_id); - for mut r in results { - r.id += seg.base_id; - all.push(r); - } + push_shifted( + &mut all, + seg.flat + .search_filtered_offset(query, top_k, bitmap, seg.base_id)?, + seg.base_id, + ); } - - all.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - all.truncate(top_k); - all + Ok(finish(all, top_k)) } /// Search with a structured payload predicate. @@ -418,23 +368,24 @@ impl VectorCollection { top_k: usize, ef: usize, predicate: &FilterPredicate, - ) -> (Vec, bool) { + ) -> Result<(Vec, bool), VectorError> { match self.payload.pre_filter(predicate) { Some(bm) => { // Serialize the bitmap to the byte format expected by // `search_with_bitmap_bytes`. let mut bm_bytes = Vec::new(); if bm.serialize_into(&mut bm_bytes).is_ok() { - let results = self.search_with_bitmap_bytes(query, top_k, ef, &bm_bytes); - (results, true) + let results = self.search_with_bitmap_bytes(query, top_k, ef, &bm_bytes)?; + Ok((results, true)) } else { - // Serialization failure: fall back to unfiltered search. - (self.search(query, top_k, ef), false) + // Serialization failure: unfiltered search, and the + // caller applies the predicate as a post-filter. + Ok((self.search(query, top_k, ef)?, false)) } } None => { // Un-indexed field present: full scan, caller must post-filter. - (self.search(query, top_k, ef), false) + Ok((self.search(query, top_k, ef)?, false)) } } } @@ -462,10 +413,10 @@ mod tests { fn insert_and_search() { let mut coll = make_collection(); for i in 0..100u32 { - coll.insert(vec![i as f32, 0.0, 0.0]); + coll.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } assert_eq!(coll.len(), 100); - let results = coll.search(&[50.0, 0.0, 0.0], 3, 64); + let results = coll.search(&[50.0, 0.0, 0.0], 3, 64).unwrap(); assert_eq!(results.len(), 3); assert_eq!(results[0].id, 50); } @@ -474,7 +425,7 @@ mod tests { fn seal_moves_to_building() { let mut coll = VectorCollection::new(2, HnswParams::default()); for i in 0..DEFAULT_SEAL_THRESHOLD { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } assert!(coll.needs_seal()); @@ -483,7 +434,7 @@ mod tests { assert_eq!(coll.building.len(), 1); assert_eq!(coll.growing.len(), 0); - let results = coll.search(&[100.0, 0.0], 1, 64); + let results = coll.search(&[100.0, 0.0], 1, 64).unwrap(); assert!(!results.is_empty()); } @@ -491,7 +442,7 @@ mod tests { fn complete_build_promotes_to_sealed() { let mut coll = VectorCollection::new(2, HnswParams::default()); for i in 0..100 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let req = coll.seal("test").unwrap(); @@ -504,7 +455,7 @@ mod tests { assert_eq!(coll.building.len(), 0); assert_eq!(coll.sealed.len(), 1); - let results = coll.search(&[50.0, 0.0], 3, 64); + let results = coll.search(&[50.0, 0.0], 3, 64).unwrap(); assert!(!results.is_empty()); } @@ -519,7 +470,7 @@ mod tests { ); for i in 0..100 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let req = coll.seal("test").unwrap(); let mut idx = HnswIndex::new(2, req.params); @@ -529,10 +480,10 @@ mod tests { coll.complete_build(req.segment_id, idx, test_memory()); for i in 100..200 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } - let results = coll.search(&[150.0, 0.0], 3, 64); + let results = coll.search(&[150.0, 0.0], 3, 64).unwrap(); assert_eq!(results.len(), 3); assert_eq!(results[0].id, 150); } @@ -541,12 +492,12 @@ mod tests { fn delete_across_segments() { let mut coll = VectorCollection::new(2, HnswParams::default()); for i in 0..10 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } assert!(coll.delete(5)); assert_eq!(coll.live_count(), 9); - let results = coll.search(&[5.0, 0.0], 10, 64); + let results = coll.search(&[5.0, 0.0], 10, 64).unwrap(); assert!(results.iter().all(|r| r.id != 5)); } @@ -561,7 +512,7 @@ mod tests { }, ); for i in 0..n { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let req = coll.seal("seg").unwrap(); let mut idx = HnswIndex::new(req.dim, req.params); @@ -583,7 +534,7 @@ mod tests { .filter_map(|i| sealed.index.get_vector(i as u32).map(|v| v.to_vec())) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, dim); + let codec = Sq8Codec::calibrate(&refs, dim).unwrap(); let sq8_data: Vec = vecs.iter().flat_map(|v| codec.quantize(v)).collect(); sealed.sq8 = Some((codec, sq8_data)); } @@ -593,7 +544,7 @@ mod tests { let mut coll = make_sealed_collection(200); attach_sq8(&mut coll); - let results = coll.search(&[100.0, 0.0], 5, 64); + let results = coll.search(&[100.0, 0.0], 5, 64).unwrap(); assert!(!results.is_empty(), "expected non-empty results"); assert_eq!( results[0].id, 100, @@ -612,8 +563,8 @@ mod tests { let query = [250.0f32, 0.0]; let top_k = 5; - let plain_results = coll_plain.search(&query, top_k, 64); - let sq8_results = coll_sq8.search(&query, top_k, 64); + let plain_results = coll_plain.search(&query, top_k, 64).unwrap(); + let sq8_results = coll_sq8.search(&query, top_k, 64).unwrap(); let plain_ids: std::collections::HashSet = plain_results.iter().map(|r| r.id).collect(); @@ -626,11 +577,11 @@ mod tests { ); } - #[test] - fn codec_dispatch_bbq_search_returns_results_and_stats_report_bbq() { - let dim = 4; + /// A collection whose first 50 vectors (`[i, 0, 0, 0]`) sit in one + /// sealed segment, with an empty growing segment. + fn sealed_dim4_collection() -> VectorCollection { let mut coll = VectorCollection::new( - dim, + 4, HnswParams { metric: DistanceMetric::L2, m: 8, @@ -638,14 +589,24 @@ mod tests { ..HnswParams::default() }, ); - - // Insert 50 vectors: vector i = [i as f32, 0, 0, 0]. for i in 0u32..50 { - coll.insert(vec![i as f32, 0.0, 0.0, 0.0]); + coll.insert(vec![i as f32, 0.0, 0.0, 0.0]).unwrap(); } + let req = coll.seal("codec").unwrap(); + let mut idx = HnswIndex::new(req.dim, req.params); + for v in &req.vectors { + idx.insert(v.clone()).unwrap(); + } + coll.complete_build(req.segment_id, idx, test_memory()); + coll + } - // Build the collection-level BBQ dispatch index over current vectors. - let dispatch = coll.build_codec_dispatch("bbq"); + #[test] + fn codec_dispatch_bbq_search_returns_results_and_stats_report_bbq() { + let mut coll = sealed_dim4_collection(); + + // Build the collection-level BBQ dispatch index over the sealed vectors. + let dispatch = coll.build_codec_dispatch("bbq").unwrap(); assert!( dispatch.is_some(), "build_codec_dispatch(bbq) should return Some" @@ -653,7 +614,7 @@ mod tests { // Query near id=25. let query = [25.0f32, 0.0, 0.0, 0.0]; - let results = coll.search(&query, 5, 32); + let results = coll.search(&query, 5, 32).unwrap(); assert!( !results.is_empty(), "BBQ codec-dispatch search should return results" @@ -670,22 +631,10 @@ mod tests { #[test] fn codec_dispatch_rabitq_search_non_empty() { - let dim = 4; - let mut coll = VectorCollection::new( - dim, - HnswParams { - metric: DistanceMetric::L2, - m: 8, - ef_construction: 50, - ..HnswParams::default() - }, - ); - for i in 0u32..50 { - coll.insert(vec![i as f32, 0.0, 0.0, 0.0]); - } - coll.build_codec_dispatch("rabitq").unwrap(); + let mut coll = sealed_dim4_collection(); + coll.build_codec_dispatch("rabitq").unwrap().unwrap(); - let results = coll.search(&[10.0, 0.0, 0.0, 0.0], 3, 32); + let results = coll.search(&[10.0, 0.0, 0.0, 0.0], 3, 32).unwrap(); assert!( !results.is_empty(), "RaBitQ dispatch search should return results" @@ -698,6 +647,71 @@ mod tests { ); } + /// The codec index covers the sealed segments under their global ids; + /// the growing segment is read beside it. A growing vector is found under + /// its own id, once. + #[test] + fn codec_dispatch_keeps_global_ids_and_reads_growing_once() { + let mut coll = sealed_dim4_collection(); + coll.build_codec_dispatch("bbq").unwrap().unwrap(); + let far = coll.insert(vec![1000.0, 0.0, 0.0, 0.0]).unwrap(); + assert_eq!(far, 50, "the first growing vector takes the next global id"); + + let results = coll.search(&[1000.0, 0.0, 0.0, 0.0], 3, 32).unwrap(); + assert_eq!(results[0].id, far); + assert_eq!( + results.iter().filter(|r| r.id == far).count(), + 1, + "{results:?}" + ); + assert!(results.iter().all(|r| r.id <= far), "{results:?}"); + } + + /// Every collection search entry point refuses a query of the wrong + /// dimension, on each segment kind and on the codec-dispatch path. + #[test] + fn wrong_dimension_query_is_a_typed_error() { + use crate::error::VectorError; + let mut coll = sealed_dim4_collection(); + coll.insert(vec![1.0, 0.0, 0.0, 0.0]).unwrap(); + let bytes = { + let bm: roaring::RoaringBitmap = (0..51u32).collect(); + let mut out = Vec::new(); + bm.serialize_into(&mut out).unwrap(); + out + }; + let short = [1.0_f32, 0.0]; + let check = |coll: &VectorCollection| { + for result in [ + coll.search(&short, 3, 32), + coll.search_with_metric(&short, 3, 32, DistanceMetric::Cosine), + coll.search_with_bitmap_bytes(&short, 3, 32, &bytes), + coll.search_with_bitmap_bytes_and_metric(&short, 3, 32, &bytes, DistanceMetric::L2), + ] { + assert!( + matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 4, + got: 2 + }) + ), + "{result:?}" + ); + } + }; + check(&coll); + coll.build_codec_dispatch("rabitq").unwrap().unwrap(); + check(&coll); + assert!(matches!( + coll.insert(vec![1.0; 3]), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + } + #[test] fn sq8_search_does_not_scan_all_vectors() { // This test validates correctness of the SQ8 search path for a large @@ -708,7 +722,7 @@ mod tests { let mut coll = make_sealed_collection(2000); attach_sq8(&mut coll); - let results = coll.search(&[1000.0, 0.0], 5, 64); + let results = coll.search(&[1000.0, 0.0], 5, 64).unwrap(); assert!(!results.is_empty(), "expected non-empty results"); assert_eq!( results[0].id, 1000, diff --git a/nodedb-vector/src/delta/compaction.rs b/nodedb-vector/src/delta/compaction.rs index 7bcd01973..5676e8281 100644 --- a/nodedb-vector/src/delta/compaction.rs +++ b/nodedb-vector/src/delta/compaction.rs @@ -199,7 +199,7 @@ mod tests { let mut delta = DeltaIndex::new(3, 32); for i in 10u32..15 { let v = vec![i as f32, 1.0, 0.0]; - delta.insert(i, v); + delta.insert(i, v).unwrap(); } assert_eq!(delta.fresh_len(), 5); diff --git a/nodedb-vector/src/delta/index.rs b/nodedb-vector/src/delta/index.rs index 2b0832dee..c37a7c0ce 100644 --- a/nodedb-vector/src/delta/index.rs +++ b/nodedb-vector/src/delta/index.rs @@ -9,6 +9,7 @@ use std::collections::HashSet; use crate::distance::distance; +use crate::error::{VectorError, check_dim}; use nodedb_types::vector_distance::DistanceMetric; /// Secondary in-memory index that absorbs fresh inserts before they are @@ -38,9 +39,12 @@ impl DeltaIndex { } /// Stage a fresh insert. Does not deduplicate — callers must ensure IDs - /// are unique across the delta and the main HNSW. - pub fn insert(&mut self, id: u32, vector: Vec) { + /// are unique across the delta and the main HNSW. A vector without the + /// index dimension fails with [`VectorError::DimensionMismatch`]. + pub fn insert(&mut self, id: u32, vector: Vec) -> Result<(), VectorError> { + check_dim(self.dim, vector.len())?; self.fresh.push((id, vector)); + Ok(()) } /// Mark `id` as tombstoned. It will be excluded from `search` results @@ -60,10 +64,17 @@ impl DeltaIndex { } /// Brute-force scan over fresh vectors (excluding tombstones), returning - /// the top-`k` results sorted ascending by distance. - pub fn search(&self, query: &[f32], k: usize, metric: DistanceMetric) -> Vec<(u32, f32)> { + /// the top-`k` results sorted ascending by distance. A query without + /// the index dimension fails with [`VectorError::DimensionMismatch`]. + pub fn search( + &self, + query: &[f32], + k: usize, + metric: DistanceMetric, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; if k == 0 { - return Vec::new(); + return Ok(Vec::new()); } let mut scored: Vec<(u32, f32)> = self @@ -83,7 +94,7 @@ impl DeltaIndex { scored.sort_unstable_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)); - scored + Ok(scored) } /// Drain all staged fresh vectors for patching into the main HNSW. @@ -113,7 +124,7 @@ mod tests { let mut d = DeltaIndex::new(3, 16); for i in 0u32..10 { let v = vec![i as f32, 0.0, 0.0]; - d.insert(i, v); + d.insert(i, v).unwrap(); } d } @@ -122,7 +133,7 @@ mod tests { fn top_k_returns_nearest() { let d = make_delta(); let query = [0.0f32, 0.0, 0.0]; - let results = d.search(&query, 3, DistanceMetric::L2); + let results = d.search(&query, 3, DistanceMetric::L2).unwrap(); assert_eq!(results.len(), 3); // Nearest to [0,0,0] with L2^2 are ids 0,1,2 assert_eq!(results[0].0, 0); @@ -135,7 +146,7 @@ mod tests { let mut d = make_delta(); d.tombstone(0); let query = [0.0f32, 0.0, 0.0]; - let results = d.search(&query, 3, DistanceMetric::L2); + let results = d.search(&query, 3, DistanceMetric::L2).unwrap(); assert!(results.iter().all(|(id, _)| *id != 0)); } @@ -143,10 +154,10 @@ mod tests { fn is_full_triggers_at_threshold() { let mut d = DeltaIndex::new(3, 3); assert!(!d.is_full()); - d.insert(0, vec![0.0, 0.0, 0.0]); - d.insert(1, vec![1.0, 0.0, 0.0]); + d.insert(0, vec![0.0, 0.0, 0.0]).unwrap(); + d.insert(1, vec![1.0, 0.0, 0.0]).unwrap(); assert!(!d.is_full()); - d.insert(2, vec![2.0, 0.0, 0.0]); + d.insert(2, vec![2.0, 0.0, 0.0]).unwrap(); assert!(d.is_full()); } @@ -168,4 +179,23 @@ mod tests { assert!(!d.is_tombstoned(3)); assert!(!d.is_tombstoned(7)); } + + #[test] + fn wrong_dimension_is_a_typed_error() { + let mut d = DeltaIndex::new(3, 8); + assert!(matches!( + d.insert(0, vec![0.0; 2]), + Err(VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + )); + assert!(matches!( + d.search(&[0.0; 4], 1, DistanceMetric::L2), + Err(VectorError::DimensionMismatch { + expected: 3, + got: 4 + }) + )); + } } diff --git a/nodedb-vector/src/dtype/cast.rs b/nodedb-vector/src/dtype/cast.rs index 22455052f..b6b76797c 100644 --- a/nodedb-vector/src/dtype/cast.rs +++ b/nodedb-vector/src/dtype/cast.rs @@ -26,6 +26,9 @@ pub enum DtypeError { expected: usize, actual: usize, }, + /// A storage dtype this build has no decoder for. + #[error("no f32 decoder for vector storage dtype {dtype}")] + Unsupported { dtype: VectorStorageDtype }, } /// Verify that `bytes.len() == dtype.bytes_for_dim(dim)`. @@ -100,9 +103,9 @@ pub fn cast_to_f32( .collect(); Ok(out) } - // `VectorStorageDtype` is #[non_exhaustive]; this arm is required by - // the compiler but unreachable with any currently-defined variant. - _ => unreachable!("unrecognised VectorStorageDtype variant in cast_to_f32"), + // `VectorStorageDtype` is #[non_exhaustive]: a dtype added upstream + // without a decoder here fails the read instead of the process. + _ => Err(DtypeError::Unsupported { dtype }), } } @@ -241,55 +244,55 @@ mod tests { #[test] fn bad_byte_len_f32_mismatch() { let err = cast_to_f32(&[0u8; 7], VectorStorageDtype::F32, 2).unwrap_err(); - match err { - DtypeError::BadByteLen { - dtype, - dim, - expected, - actual, - } => { - assert_eq!(dtype, VectorStorageDtype::F32); - assert_eq!(dim, 2); - assert_eq!(expected, 8); - assert_eq!(actual, 7); - } - } + let DtypeError::BadByteLen { + dtype, + dim, + expected, + actual, + } = err + else { + panic!("expected BadByteLen, got {err:?}"); + }; + assert_eq!(dtype, VectorStorageDtype::F32); + assert_eq!(dim, 2); + assert_eq!(expected, 8); + assert_eq!(actual, 7); } #[test] fn bad_byte_len_f16_odd_byte_count() { let err = cast_to_f32(&[0u8; 3], VectorStorageDtype::F16, 2).unwrap_err(); - match err { - DtypeError::BadByteLen { - dtype, - dim, - expected, - actual, - } => { - assert_eq!(dtype, VectorStorageDtype::F16); - assert_eq!(dim, 2); - assert_eq!(expected, 4); - assert_eq!(actual, 3); - } - } + let DtypeError::BadByteLen { + dtype, + dim, + expected, + actual, + } = err + else { + panic!("expected BadByteLen, got {err:?}"); + }; + assert_eq!(dtype, VectorStorageDtype::F16); + assert_eq!(dim, 2); + assert_eq!(expected, 4); + assert_eq!(actual, 3); } #[test] fn bad_byte_len_bf16_mismatch() { let err = cast_to_f32(&[0u8; 5], VectorStorageDtype::BF16, 3).unwrap_err(); - match err { - DtypeError::BadByteLen { - dtype, - dim, - expected, - actual, - } => { - assert_eq!(dtype, VectorStorageDtype::BF16); - assert_eq!(dim, 3); - assert_eq!(expected, 6); - assert_eq!(actual, 5); - } - } + let DtypeError::BadByteLen { + dtype, + dim, + expected, + actual, + } = err + else { + panic!("expected BadByteLen, got {err:?}"); + }; + assert_eq!(dtype, VectorStorageDtype::BF16); + assert_eq!(dim, 3); + assert_eq!(expected, 6); + assert_eq!(actual, 5); } // ── validate_byte_len independently ────────────────────────────────────── @@ -304,13 +307,13 @@ mod tests { fn validate_byte_len_off_by_one_fails() { let bytes = [0u8; 11]; // should be 12 let err = validate_byte_len(&bytes, VectorStorageDtype::F32, 3).unwrap_err(); - match err { - DtypeError::BadByteLen { - expected, actual, .. - } => { - assert_eq!(expected, 12); - assert_eq!(actual, 11); - } - } + let DtypeError::BadByteLen { + expected, actual, .. + } = err + else { + panic!("expected BadByteLen, got {err:?}"); + }; + assert_eq!(expected, 12); + assert_eq!(actual, 11); } } diff --git a/nodedb-vector/src/error.rs b/nodedb-vector/src/error.rs index 41c233bfe..2bc53d2b9 100644 --- a/nodedb-vector/src/error.rs +++ b/nodedb-vector/src/error.rs @@ -10,8 +10,24 @@ use nodedb_mem::MemError; pub enum VectorError { #[error("memory budget exhausted: {0}")] BudgetExhausted(#[from] MemError), + /// An input vector — a search query or an inserted vector — has a + /// different dimension from the index. The caller's input is wrong; the + /// index is intact. #[error("vector dimension mismatch: expected {expected}, got {got}")] DimensionMismatch { expected: usize, got: usize }, + /// Stored data — a PQ code, a segment backing, a materialized node + /// vector — disagrees with the index dimension. The stored data is + /// corrupt or belongs to another index. + #[error("stored vector data has dimension {got}, index expects {expected}")] + StoredDimensionMismatch { expected: usize, got: usize }, + /// A serialized pre-filter bitmap does not decode. Searching without it + /// would return rows the filter excludes, so the search fails instead. + #[error("vector search filter bitmap does not decode: {detail}")] + InvalidFilterBitmap { detail: String }, + /// An index build or training call received input it cannot use: an + /// empty training set, a zero dimension, or parameters that do not fit. + #[error("invalid vector index input: {detail}")] + InvalidInput { detail: String }, /// A node's vector could not be materialized: the node is out of range, or /// its local storage is empty and no segment backing supplies the data. /// @@ -58,3 +74,13 @@ pub enum VectorError { #[error("vector segment I/O error: {0}")] SegmentIo(#[from] std::io::Error), } + +/// Check that an input vector of length `got` fits an index of dimension +/// `expected`. +pub fn check_dim(expected: usize, got: usize) -> Result<(), VectorError> { + if expected == got { + Ok(()) + } else { + Err(VectorError::DimensionMismatch { expected, got }) + } +} diff --git a/nodedb-vector/src/flat.rs b/nodedb-vector/src/flat.rs index 047c01e68..16dbfc265 100644 --- a/nodedb-vector/src/flat.rs +++ b/nodedb-vector/src/flat.rs @@ -12,7 +12,9 @@ use roaring::RoaringBitmap; use crate::distance::{DistanceMetric, distance}; +use crate::error::{VectorError, check_dim}; use crate::hnsw::SearchResult; +use crate::hnsw::search::decode_filter_bitmap; /// Default threshold below which collections use flat index instead of HNSW. pub const DEFAULT_FLAT_INDEX_THRESHOLD: usize = 10_000; @@ -41,20 +43,16 @@ impl FlatIndex { } } - /// Insert a vector. Returns the assigned vector ID. - pub fn insert(&mut self, vector: Vec) -> u32 { - assert_eq!( - vector.len(), - self.dim, - "dimension mismatch: expected {}, got {}", - self.dim, - vector.len() - ); + /// Insert a vector. Returns the assigned vector ID, or + /// [`VectorError::DimensionMismatch`] when `vector` does not have the + /// index dimension. + pub fn insert(&mut self, vector: Vec) -> Result { + check_dim(self.dim, vector.len())?; let id = self.len() as u32; self.data.extend_from_slice(&vector); self.deleted.push(false); self.live_count += 1; - id + Ok(id) } /// Soft-delete a vector by ID. @@ -92,83 +90,22 @@ impl FlatIndex { query: &[f32], top_k: usize, metric: DistanceMetric, - ) -> Vec { - assert_eq!(query.len(), self.dim); - let n = self.len(); - if n == 0 || top_k == 0 { - return Vec::new(); - } - - let mut candidates: Vec = Vec::with_capacity(n.min(top_k * 2)); - for i in 0..n { - if self.deleted[i] { - continue; - } - let start = i * self.dim; - let vec_slice = &self.data[start..start + self.dim]; - let dist = distance(query, vec_slice, metric); - candidates.push(SearchResult { - id: i as u32, - distance: dist, - }); - } - - if candidates.len() > top_k { - candidates.select_nth_unstable_by(top_k, |a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates.truncate(top_k); - } - candidates.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates + ) -> Result, VectorError> { + self.scan(query, top_k, metric, None) } /// Brute-force k-NN search. Exact results — no approximation. - pub fn search(&self, query: &[f32], top_k: usize) -> Vec { - assert_eq!(query.len(), self.dim); - let n = self.len(); - if n == 0 || top_k == 0 { - return Vec::new(); - } - - let mut candidates: Vec = Vec::with_capacity(n.min(top_k * 2)); - for i in 0..n { - if self.deleted[i] { - continue; - } - let start = i * self.dim; - let vec_slice = &self.data[start..start + self.dim]; - let dist = distance(query, vec_slice, self.metric); - candidates.push(SearchResult { - id: i as u32, - distance: dist, - }); - } - - if candidates.len() > top_k { - candidates.select_nth_unstable_by(top_k, |a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates.truncate(top_k); - } - candidates.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates + pub fn search(&self, query: &[f32], top_k: usize) -> Result, VectorError> { + self.scan(query, top_k, self.metric, None) } /// Search with a pre-filter bitmap (byte-array format). - pub fn search_filtered(&self, query: &[f32], top_k: usize, bitmap: &[u8]) -> Vec { + pub fn search_filtered( + &self, + query: &[f32], + top_k: usize, + bitmap: &[u8], + ) -> Result, VectorError> { self.search_filtered_offset(query, top_k, bitmap, 0) } @@ -180,89 +117,57 @@ impl FlatIndex { bitmap: &[u8], id_offset: u32, metric: DistanceMetric, - ) -> Vec { - assert_eq!(query.len(), self.dim); - let n = self.len(); - if n == 0 || top_k == 0 { - return Vec::new(); - } - - let parsed = RoaringBitmap::deserialize_from(bitmap).ok(); - - let mut candidates: Vec = Vec::with_capacity(top_k * 2); - for i in 0..n { - if self.deleted[i] { - continue; - } - if let Some(ref bm) = parsed { - let global = (i as u32).saturating_add(id_offset); - if !bm.contains(global) { - continue; - } - } - let start = i * self.dim; - let vec_slice = &self.data[start..start + self.dim]; - let dist = distance(query, vec_slice, metric); - candidates.push(SearchResult { - id: i as u32, - distance: dist, - }); - } - - if candidates.len() > top_k { - candidates.select_nth_unstable_by(top_k, |a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates.truncate(top_k); - } - candidates.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates + ) -> Result, VectorError> { + let filter = decode_filter_bitmap(bitmap)?; + self.scan(query, top_k, metric, Some((&filter, id_offset))) } /// Search with a pre-filter bitmap applying a global id offset. /// /// `bitmap` is a serialized `RoaringBitmap` (matching the HNSW filter /// format). Bit `i + id_offset` tests local id `i`. Used by multi-segment - /// collections where the bitmap holds GLOBAL vector ids. If the bytes - /// fail to deserialize, the search degrades to unfiltered. + /// collections where the bitmap holds GLOBAL vector ids. Bytes that do + /// not decode fail with [`VectorError::InvalidFilterBitmap`]. pub fn search_filtered_offset( &self, query: &[f32], top_k: usize, bitmap: &[u8], id_offset: u32, - ) -> Vec { - assert_eq!(query.len(), self.dim); + ) -> Result, VectorError> { + self.search_filtered_offset_with_metric(query, top_k, bitmap, id_offset, self.metric) + } + + /// Exact scan of every live vector under `metric`, restricted to ids + /// whose `local + offset` is in the filter when one is given. + fn scan( + &self, + query: &[f32], + top_k: usize, + metric: DistanceMetric, + filter: Option<(&RoaringBitmap, u32)>, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; let n = self.len(); if n == 0 || top_k == 0 { - return Vec::new(); + return Ok(Vec::new()); } - let parsed = RoaringBitmap::deserialize_from(bitmap).ok(); - - let mut candidates: Vec = Vec::with_capacity(top_k * 2); + let mut candidates: Vec = Vec::with_capacity(n.min(top_k * 2)); for i in 0..n { if self.deleted[i] { continue; } - if let Some(ref bm) = parsed { - let global = (i as u32).saturating_add(id_offset); - if !bm.contains(global) { - continue; - } + if let Some((bitmap, id_offset)) = filter + && !bitmap.contains((i as u32).saturating_add(id_offset)) + { + continue; } let start = i * self.dim; let vec_slice = &self.data[start..start + self.dim]; - let dist = distance(query, vec_slice, self.metric); candidates.push(SearchResult { id: i as u32, - distance: dist, + distance: distance(query, vec_slice, metric), }); } @@ -279,7 +184,7 @@ impl FlatIndex { .partial_cmp(&b.distance) .unwrap_or(std::cmp::Ordering::Equal) }); - candidates + Ok(candidates) } pub fn len(&self) -> usize { @@ -337,19 +242,13 @@ impl FlatIndex { } /// Insert a vector that is already tombstoned (for checkpoint restore). - pub fn insert_tombstoned(&mut self, vector: Vec) -> u32 { - assert_eq!( - vector.len(), - self.dim, - "dimension mismatch: expected {}, got {}", - self.dim, - vector.len() - ); + pub fn insert_tombstoned(&mut self, vector: Vec) -> Result { + check_dim(self.dim, vector.len())?; let id = self.len() as u32; self.data.extend_from_slice(&vector); self.deleted.push(true); // No live_count increment — it's dead on arrival. - id + Ok(id) } pub fn dim(&self) -> usize { @@ -373,12 +272,12 @@ mod tests { fn insert_and_search() { let mut idx = FlatIndex::new(3, DistanceMetric::L2); for i in 0..100u32 { - idx.insert(vec![i as f32, 0.0, 0.0]); + idx.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } assert_eq!(idx.len(), 100); assert_eq!(idx.live_count(), 100); - let results = idx.search(&[50.0, 0.0, 0.0], 3); + let results = idx.search(&[50.0, 0.0, 0.0], 3).unwrap(); assert_eq!(results.len(), 3); assert_eq!(results[0].id, 50); assert!(results[0].distance < 0.01); @@ -387,14 +286,14 @@ mod tests { #[test] fn delete_excludes_from_search() { let mut idx = FlatIndex::new(2, DistanceMetric::L2); - idx.insert(vec![0.0, 0.0]); - idx.insert(vec![1.0, 0.0]); - idx.insert(vec![2.0, 0.0]); + idx.insert(vec![0.0, 0.0]).unwrap(); + idx.insert(vec![1.0, 0.0]).unwrap(); + idx.insert(vec![2.0, 0.0]).unwrap(); assert!(idx.delete(1)); assert_eq!(idx.live_count(), 2); - let results = idx.search(&[1.0, 0.0], 3); + let results = idx.search(&[1.0, 0.0], 3).unwrap(); assert_eq!(results.len(), 2); assert!(results.iter().all(|r| r.id != 1)); } @@ -402,11 +301,11 @@ mod tests { #[test] fn exact_results() { let mut idx = FlatIndex::new(2, DistanceMetric::Cosine); - idx.insert(vec![1.0, 0.0]); - idx.insert(vec![0.0, 1.0]); - idx.insert(vec![1.0, 1.0]); + idx.insert(vec![1.0, 0.0]).unwrap(); + idx.insert(vec![0.0, 1.0]).unwrap(); + idx.insert(vec![1.0, 1.0]).unwrap(); - let results = idx.search(&[1.0, 0.0], 1); + let results = idx.search(&[1.0, 0.0], 1).unwrap(); assert_eq!(results.len(), 1); assert_eq!(results[0].id, 0); } @@ -414,7 +313,7 @@ mod tests { #[test] fn empty_search() { let idx = FlatIndex::new(3, DistanceMetric::L2); - let results = idx.search(&[1.0, 0.0, 0.0], 5); + let results = idx.search(&[1.0, 0.0, 0.0], 5).unwrap(); assert!(results.is_empty()); } @@ -422,12 +321,67 @@ mod tests { fn filtered_search() { let mut idx = FlatIndex::new(2, DistanceMetric::L2); for i in 0..8u32 { - idx.insert(vec![i as f32, 0.0]); + idx.insert(vec![i as f32, 0.0]).unwrap(); } - let bitmap = vec![0b11001100u8]; - let results = idx.search_filtered(&[3.0, 0.0], 2, &bitmap); + let filter: RoaringBitmap = [2u32, 3, 6, 7].into_iter().collect(); + let mut bitmap = Vec::new(); + filter.serialize_into(&mut bitmap).unwrap(); + let results = idx.search_filtered(&[4.0, 0.0], 2, &bitmap).unwrap(); assert_eq!(results.len(), 2); assert_eq!(results[0].id, 3); - assert_eq!(results[1].id, 2); + assert!(results.iter().all(|r| filter.contains(r.id)), "{results:?}"); + } + + #[test] + fn wrong_dimension_is_a_typed_error() { + let mut idx = FlatIndex::new(2, DistanceMetric::L2); + assert!(matches!( + idx.insert(vec![1.0, 2.0, 3.0]), + Err(VectorError::DimensionMismatch { + expected: 2, + got: 3 + }) + )); + assert!(matches!( + idx.insert_tombstoned(vec![1.0]), + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + )); + idx.insert(vec![1.0, 0.0]).unwrap(); + let filter: RoaringBitmap = [0u32].into_iter().collect(); + let mut bitmap = Vec::new(); + filter.serialize_into(&mut bitmap).unwrap(); + let short = [1.0_f32]; + for result in [ + idx.search(&short, 1), + idx.search_with_metric(&short, 1, DistanceMetric::Cosine), + idx.search_filtered(&short, 1, &bitmap), + idx.search_filtered_offset(&short, 1, &bitmap, 0), + idx.search_filtered_offset_with_metric(&short, 1, &bitmap, 0, DistanceMetric::L2), + ] { + assert!( + matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + ), + "{result:?}" + ); + } + } + + #[test] + fn undecodable_filter_bitmap_is_a_typed_error() { + let mut idx = FlatIndex::new(2, DistanceMetric::L2); + idx.insert(vec![1.0, 0.0]).unwrap(); + let result = idx.search_filtered(&[1.0, 0.0], 1, &[0b1100_1100]); + assert!( + matches!(result, Err(VectorError::InvalidFilterBitmap { .. })), + "{result:?}" + ); } } diff --git a/nodedb-vector/src/hnsw/build.rs b/nodedb-vector/src/hnsw/build.rs index 5c5862bc9..a93ee761a 100644 --- a/nodedb-vector/src/hnsw/build.rs +++ b/nodedb-vector/src/hnsw/build.rs @@ -227,7 +227,7 @@ mod tests { for target in 0..20u32 { let query = idx.get_vector(target).unwrap().to_vec(); - let results = idx.search(&query, 1, 32); + let results = idx.search(&query, 1, 32).unwrap(); assert_eq!(results[0].id, target, "node {target} not reachable"); } } @@ -247,7 +247,7 @@ mod tests { for target_old_id in (1..20u32).step_by(2) { let query = vec![target_old_id as f32, 0.0, 0.0]; - let results = idx.search(&query, 1, 32); + let results = idx.search(&query, 1, 32).unwrap(); assert!(!results.is_empty()); let found_vec = idx.get_vector(results[0].id).unwrap(); assert_eq!(found_vec[0], target_old_id as f32); diff --git a/nodedb-vector/src/hnsw/checkpoint.rs b/nodedb-vector/src/hnsw/checkpoint.rs index e90d2dd1f..ec6766eb3 100644 --- a/nodedb-vector/src/hnsw/checkpoint.rs +++ b/nodedb-vector/src/hnsw/checkpoint.rs @@ -235,8 +235,8 @@ mod tests { assert_eq!(restored.max_layer(), idx.max_layer()); let query = vec![1.0, 2.0, 3.0]; - let orig_results = idx.search(&query, 5, 32); - let rest_results = restored.search(&query, 5, 32); + let orig_results = idx.search(&query, 5, 32).unwrap(); + let rest_results = restored.search(&query, 5, 32).unwrap(); assert_eq!(orig_results.len(), rest_results.len()); for (a, b) in orig_results.iter().zip(rest_results.iter()) { assert_eq!(a.id, b.id); diff --git a/nodedb-vector/src/hnsw/graph/index/backing.rs b/nodedb-vector/src/hnsw/graph/index/backing.rs index 4ecaadb9f..fb5d35cdd 100644 --- a/nodedb-vector/src/hnsw/graph/index/backing.rs +++ b/nodedb-vector/src/hnsw/graph/index/backing.rs @@ -38,7 +38,7 @@ impl HnswIndex { /// segment is unusable — rebuild the index from the authoritative vectors, /// or leave the collection unloaded", never as something to ignore. /// - /// - [`VectorError::DimensionMismatch`] if the backing's `dim()` is not this + /// - [`VectorError::StoredDimensionMismatch`] if the backing's `dim()` is not this /// index's `dim`. /// - [`VectorError::VectorUnavailable`] if the backing holds fewer vectors /// than the index has nodes, or if a node that needs the backing has no @@ -49,7 +49,7 @@ impl HnswIndex { b: Arc, ) -> Result<&mut Self, VectorError> { if b.dim() != self.dim { - return Err(VectorError::DimensionMismatch { + return Err(VectorError::StoredDimensionMismatch { expected: self.dim, got: b.dim(), }); diff --git a/nodedb-vector/src/hnsw/graph/index/state.rs b/nodedb-vector/src/hnsw/graph/index/state.rs index 0743dd12b..ae2c62440 100644 --- a/nodedb-vector/src/hnsw/graph/index/state.rs +++ b/nodedb-vector/src/hnsw/graph/index/state.rs @@ -191,7 +191,7 @@ mod tests { for i in 0..10u32 { idx.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } - let results = idx.search(&[5.0, 0.0, 0.0], 3, 32); + let results = idx.search(&[5.0, 0.0, 0.0], 3, 32).unwrap(); assert_eq!(results.len(), 3); // Results must be in monotonically non-decreasing distance order. for w in results.windows(2) { @@ -210,7 +210,7 @@ mod tests { for i in 0..10u32 { idx.insert(vec![i as f32, 0.0, 0.0]).unwrap(); } - let results = idx.search(&[5.0, 0.0, 0.0], 3, 32); + let results = idx.search(&[5.0, 0.0, 0.0], 3, 32).unwrap(); assert_eq!(results.len(), 3); for w in results.windows(2) { assert!( @@ -376,7 +376,7 @@ mod tests { dim: 4, served: vec![vec![0.0; 4], vec![0.0; 4]], })), - Err(VectorError::DimensionMismatch { .. }) + Err(VectorError::StoredDimensionMismatch { .. }) ), "a backing with the wrong dim must be refused" ); @@ -405,7 +405,7 @@ mod tests { .unwrap() .expect("graph checkpoint must be recognized"); - let results = idx.search(&[1.0, 2.0, 3.0], 5, 16); + let results = idx.search(&[1.0, 2.0, 3.0], 5, 16).unwrap(); for r in &results { assert!( r.distance.is_infinite(), diff --git a/nodedb-vector/src/hnsw/graph/index/vectors.rs b/nodedb-vector/src/hnsw/graph/index/vectors.rs index d41faf70c..bb3e64cfa 100644 --- a/nodedb-vector/src/hnsw/graph/index/vectors.rs +++ b/nodedb-vector/src/hnsw/graph/index/vectors.rs @@ -96,7 +96,7 @@ impl HnswIndex { /// storage is empty and no backing provides the vector. /// - [`VectorError::VectorDecodeFailed`] if dtype-encoded bytes cannot be /// decoded to f32. - /// - [`VectorError::DimensionMismatch`] if the materialized vector's length + /// - [`VectorError::StoredDimensionMismatch`] if the materialized vector's length /// is not `self.dim`. pub fn materialize_vector(&self, id: u32) -> Result, VectorError> { let node = self @@ -121,7 +121,7 @@ impl HnswIndex { None => self.backing_vector(id)?, }; if vector.len() != self.dim { - return Err(VectorError::DimensionMismatch { + return Err(VectorError::StoredDimensionMismatch { expected: self.dim, got: vector.len(), }); diff --git a/nodedb-vector/src/hnsw/search.rs b/nodedb-vector/src/hnsw/search.rs index 4cd4bfa40..3b82e9fbd 100644 --- a/nodedb-vector/src/hnsw/search.rs +++ b/nodedb-vector/src/hnsw/search.rs @@ -32,48 +32,27 @@ fn prefetch_t0(ptr: *const u8) { use roaring::RoaringBitmap; use crate::dtype::cast_from_f32; +use crate::error::{VectorError, check_dim}; use crate::hnsw::graph::{Candidate, HnswIndex, SearchResult}; +/// Maximum beam width to prevent runaway search cost. +const MAX_EF: usize = 8192; + impl HnswIndex { /// K-NN search: find the `k` closest vectors to `query`. /// /// `ef` controls the search beam width (higher = better recall, slower). /// Must be >= k. Typical values: ef = 2*k to 10*k. - pub fn search(&self, query: &[f32], k: usize, ef: usize) -> Vec { - assert_eq!(query.len(), self.dim, "query dimension mismatch"); - if self.is_empty() { - return Vec::new(); - } - - /// Maximum beam width to prevent runaway search cost. - const MAX_EF: usize = 8192; - let ef = ef.max(k).min(MAX_EF); - let Some(ep) = self.entry_point else { - return Vec::new(); - }; - - let query_bytes = cast_from_f32(query, self.params.dtype); - - // Phase 1: Greedy descent from top layer to layer 1. - let mut current_ep = ep; - for layer in (1..=self.max_layer).rev() { - let results = search_layer(self, &query_bytes, current_ep, 1, layer, None, 0); - if let Some(nearest) = results.first() { - current_ep = nearest.id; - } - } - - // Phase 2: Beam search at layer 0. - let results = search_layer(self, &query_bytes, current_ep, ef, 0, None, 0); - - results - .into_iter() - .take(k) - .map(|c| SearchResult { - id: c.id, - distance: c.dist, - }) - .collect() + /// + /// Returns [`VectorError::DimensionMismatch`] when `query` does not have + /// the index dimension. + pub fn search( + &self, + query: &[f32], + k: usize, + ef: usize, + ) -> Result, VectorError> { + self.search_inner(query, k, ef.max(k).min(MAX_EF), None, 0) } /// Filtered K-NN search with Roaring bitmap pre-filtering. @@ -83,7 +62,7 @@ impl HnswIndex { k: usize, ef: usize, filter: &RoaringBitmap, - ) -> Vec { + ) -> Result, VectorError> { self.search_filtered_offset(query, k, ef, filter, 0) } @@ -99,19 +78,60 @@ impl HnswIndex { ef: usize, filter: &RoaringBitmap, id_offset: u32, - ) -> Vec { - assert_eq!(query.len(), self.dim, "query dimension mismatch"); + ) -> Result, VectorError> { + self.search_inner(query, k, ef.max(k), Some(filter), id_offset) + } + + /// Deserialize a Roaring bitmap from bytes and perform filtered search. + pub fn search_with_bitmap_bytes( + &self, + query: &[f32], + k: usize, + ef: usize, + bitmap_bytes: &[u8], + ) -> Result, VectorError> { + self.search_with_bitmap_bytes_offset(query, k, ef, bitmap_bytes, 0) + } + + /// Deserialize a Roaring bitmap and search with an ID offset applied + /// before testing membership. See `search_filtered_offset` for rationale. + /// + /// Bytes that do not decode fail with + /// [`VectorError::InvalidFilterBitmap`]: an unfiltered search would + /// return rows the filter excludes. + pub fn search_with_bitmap_bytes_offset( + &self, + query: &[f32], + k: usize, + ef: usize, + bitmap_bytes: &[u8], + id_offset: u32, + ) -> Result, VectorError> { + let bitmap = decode_filter_bitmap(bitmap_bytes)?; + self.search_filtered_offset(query, k, ef, &bitmap, id_offset) + } + + /// Greedy descent to layer 1, then a beam search of width `ef` at layer + /// 0, restricted to `filter` when one is given. + fn search_inner( + &self, + query: &[f32], + k: usize, + ef: usize, + filter: Option<&RoaringBitmap>, + id_offset: u32, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; if self.is_empty() { - return Vec::new(); + return Ok(Vec::new()); } - - let ef = ef.max(k); let Some(ep) = self.entry_point else { - return Vec::new(); + return Ok(Vec::new()); }; let query_bytes = cast_from_f32(query, self.params.dtype); + // Phase 1: Greedy descent from top layer to layer 1. let mut current_ep = ep; for layer in (1..=self.max_layer).rev() { let results = search_layer(self, &query_bytes, current_ep, 1, layer, None, 0); @@ -120,52 +140,25 @@ impl HnswIndex { } } - let results = search_layer( - self, - &query_bytes, - current_ep, - ef, - 0, - Some(filter), - id_offset, - ); + // Phase 2: Beam search at layer 0. + let results = search_layer(self, &query_bytes, current_ep, ef, 0, filter, id_offset); - results + Ok(results .into_iter() .take(k) .map(|c| SearchResult { id: c.id, distance: c.dist, }) - .collect() - } - - /// Deserialize a Roaring bitmap from bytes and perform filtered search. - pub fn search_with_bitmap_bytes( - &self, - query: &[f32], - k: usize, - ef: usize, - bitmap_bytes: &[u8], - ) -> Vec { - self.search_with_bitmap_bytes_offset(query, k, ef, bitmap_bytes, 0) + .collect()) } +} - /// Deserialize a Roaring bitmap and search with an ID offset applied - /// before testing membership. See `search_filtered_offset` for rationale. - pub fn search_with_bitmap_bytes_offset( - &self, - query: &[f32], - k: usize, - ef: usize, - bitmap_bytes: &[u8], - id_offset: u32, - ) -> Vec { - match RoaringBitmap::deserialize_from(bitmap_bytes) { - Ok(bitmap) => self.search_filtered_offset(query, k, ef, &bitmap, id_offset), - Err(_) => self.search(query, k, ef), - } - } +/// Decode a serialized Roaring pre-filter bitmap. +pub fn decode_filter_bitmap(bytes: &[u8]) -> Result { + RoaringBitmap::deserialize_from(bytes).map_err(|e| VectorError::InvalidFilterBitmap { + detail: e.to_string(), + }) } /// Unified HNSW beam search on a single layer with optional pre-filter. @@ -313,7 +306,7 @@ mod tests { #[test] fn search_empty_index() { let idx = HnswIndex::new(3, HnswParams::default()); - let results = idx.search(&[1.0, 2.0, 3.0], 5, 50); + let results = idx.search(&[1.0, 2.0, 3.0], 5, 50).unwrap(); assert!(results.is_empty()); } @@ -331,7 +324,7 @@ mod tests { 1, ); idx.insert(vec![1.0, 0.0]).unwrap(); - let results = idx.search(&[1.0, 0.0], 1, 10); + let results = idx.search(&[1.0, 0.0], 1, 10).unwrap(); assert_eq!(results.len(), 1); assert_eq!(results[0].id, 0); assert!(results[0].distance < 1e-6); @@ -341,7 +334,7 @@ mod tests { fn search_finds_exact_match() { let idx = build_index(50, 3); let query = idx.get_vector(25).unwrap().to_vec(); - let results = idx.search(&query, 1, 50); + let results = idx.search(&query, 1, 50).unwrap(); assert_eq!(results.len(), 1); assert_eq!(results[0].id, 25); assert!(results[0].distance < 1e-6); @@ -351,7 +344,7 @@ mod tests { fn search_returns_sorted_by_distance() { let idx = build_index(100, 4); let query = vec![50.0, 50.0, 50.0, 50.0]; - let results = idx.search(&query, 10, 64); + let results = idx.search(&query, 10, 64).unwrap(); assert_eq!(results.len(), 10); for w in results.windows(2) { assert!(w[0].distance <= w[1].distance); @@ -361,7 +354,7 @@ mod tests { #[test] fn search_k_larger_than_index() { let idx = build_index(5, 2); - let results = idx.search(&[0.0, 0.0], 20, 50); + let results = idx.search(&[0.0, 0.0], 20, 50).unwrap(); assert_eq!(results.len(), 5); } @@ -369,7 +362,7 @@ mod tests { fn search_recall_at_10() { let idx = build_index(500, 3); let query = vec![100.0, 100.0, 100.0]; - let results = idx.search(&query, 10, 128); + let results = idx.search(&query, 10, 128).unwrap(); let mut truth: Vec<(u32, f32)> = (0..500) .map(|i| { @@ -390,7 +383,7 @@ mod tests { fn search_excludes_tombstoned() { let mut idx = build_index(20, 3); idx.delete(0); - let results = idx.search(&[0.0, 0.0, 0.0], 5, 32); + let results = idx.search(&[0.0, 0.0, 0.0], 5, 32).unwrap(); for r in &results { assert_ne!(r.id, 0, "tombstoned node appeared in results"); } @@ -403,7 +396,9 @@ mod tests { for i in (0..50u32).step_by(2) { filter.insert(i); } - let results = idx.search_filtered(&[0.0, 0.0, 0.0], 5, 64, &filter); + let results = idx + .search_filtered(&[0.0, 0.0, 0.0], 5, 64, &filter) + .unwrap(); assert_eq!(results.len(), 5); for r in &results { assert!(r.id % 2 == 0, "got odd id {}", r.id); @@ -414,7 +409,9 @@ mod tests { fn search_filtered_empty_returns_empty() { let idx = build_index(20, 3); let filter = RoaringBitmap::new(); - let results = idx.search_filtered(&[0.0, 0.0, 0.0], 5, 64, &filter); + let results = idx + .search_filtered(&[0.0, 0.0, 0.0], 5, 64, &filter) + .unwrap(); assert!(results.is_empty()); } @@ -427,9 +424,56 @@ mod tests { } let mut bytes = Vec::new(); filter.serialize_into(&mut bytes).unwrap(); - let results = idx.search_with_bitmap_bytes(&[0.0, 0.0, 0.0], 5, 32, &bytes); + let results = idx + .search_with_bitmap_bytes(&[0.0, 0.0, 0.0], 5, 32, &bytes) + .unwrap(); for r in &results { assert!(r.id < 25, "got filtered-out node {}", r.id); } } + + /// A query of the wrong dimension is a typed error on every entry point, + /// on an empty index and a populated one alike. + #[test] + fn wrong_dimension_query_is_a_typed_error() { + use crate::error::VectorError; + let empty = HnswIndex::new(3, HnswParams::default()); + let idx = build_index(20, 3); + let filter: RoaringBitmap = (0..20u32).collect(); + let mut bytes = Vec::new(); + filter.serialize_into(&mut bytes).unwrap(); + let short = [0.0_f32, 0.0]; + for result in [ + empty.search(&short, 5, 32), + idx.search(&short, 5, 32), + idx.search_filtered(&short, 5, 32, &filter), + idx.search_filtered_offset(&short, 5, 32, &filter, 0), + idx.search_with_bitmap_bytes(&short, 5, 32, &bytes), + idx.search_with_bitmap_bytes_offset(&short, 5, 32, &bytes, 0), + ] { + assert!( + matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + ), + "{result:?}" + ); + } + } + + /// Filter bytes that do not decode fail the search: an unfiltered + /// search would return rows the filter excludes. + #[test] + fn undecodable_filter_bitmap_is_a_typed_error() { + use crate::error::VectorError; + let idx = build_index(20, 3); + let result = idx.search_with_bitmap_bytes(&[0.0, 0.0, 0.0], 5, 32, &[0xff, 0x01]); + assert!( + matches!(result, Err(VectorError::InvalidFilterBitmap { .. })), + "{result:?}" + ); + } } diff --git a/nodedb-vector/src/ivf.rs b/nodedb-vector/src/ivf.rs index 26993fd28..2d30272fb 100644 --- a/nodedb-vector/src/ivf.rs +++ b/nodedb-vector/src/ivf.rs @@ -8,6 +8,7 @@ use nodedb_mem::ScopedMemory; use crate::distance::{DistanceMetric, distance}; +use crate::error::{VectorError, check_dim}; use crate::hnsw::SearchResult; use crate::quantize::pq::PqCodec; @@ -67,15 +68,27 @@ impl IvfPqIndex { /// Train the index from a set of vectors, tracking PQ codebook /// allocations against `memory`. - pub fn train(&mut self, vectors: &[&[f32]], memory: ScopedMemory) { - assert!(!vectors.is_empty()); - assert!(self.dim > 0); - assert!( - self.dim.is_multiple_of(self.params.pq_m), - "dim {} must be divisible by pq_m {}", - self.dim, - self.params.pq_m - ); + /// + /// An empty set, a zero dimension, or a `pq_m` that does not divide the + /// dimension fails with [`VectorError::InvalidInput`]; a vector without + /// the index dimension fails with [`VectorError::DimensionMismatch`]. + pub fn train(&mut self, vectors: &[&[f32]], memory: ScopedMemory) -> Result<(), VectorError> { + if vectors.is_empty() { + return Err(VectorError::InvalidInput { + detail: "IVF-PQ training needs at least one vector".into(), + }); + } + if self.dim == 0 || self.params.pq_m == 0 || !self.dim.is_multiple_of(self.params.pq_m) { + return Err(VectorError::InvalidInput { + detail: format!( + "IVF-PQ dimension {} must be non-zero and divisible by pq_m {}", + self.dim, self.params.pq_m + ), + }); + } + for v in vectors { + check_dim(self.dim, v.len())?; + } let n_cells = self.params.n_cells.min(vectors.len()); self.centroids = kmeans_centroids(vectors, self.dim, n_cells, 20); @@ -99,16 +112,22 @@ impl IvfPqIndex { self.params.pq_k, 20, memory, - )); + )?); + Ok(()) } /// Add a vector to the index. Returns the assigned ID. - pub fn add(&mut self, vector: &[f32]) -> u32 { - assert_eq!(vector.len(), self.dim); - let pq = self - .pq - .as_ref() - .expect("index must be trained before add()"); + /// + /// A vector without the index dimension fails with + /// [`VectorError::DimensionMismatch`]; an untrained index fails with + /// [`VectorError::InvalidInput`]. + pub fn add(&mut self, vector: &[f32]) -> Result { + check_dim(self.dim, vector.len())?; + let Some(pq) = self.pq.as_ref() else { + return Err(VectorError::InvalidInput { + detail: "IVF-PQ index must be trained before add".into(), + }); + }; let cell = self.nearest_centroid(vector); let residual: Vec = vector @@ -120,7 +139,7 @@ impl IvfPqIndex { let id = self.count; self.cells[cell].push((id, code)); self.count += 1; - id + Ok(id) } /// Whether the index holds a trained codebook. @@ -145,23 +164,27 @@ impl IvfPqIndex { self.count = self.count.min(count); } - /// Batch add vectors. - pub fn add_batch(&mut self, vectors: &[&[f32]]) { + /// Batch add vectors. Stops at the first vector that fails to add. + pub fn add_batch(&mut self, vectors: &[&[f32]]) -> Result<(), VectorError> { for v in vectors { - self.add(v); + self.add(v)?; } + Ok(()) } /// Search: find top-k nearest neighbors. - pub fn search(&self, query: &[f32], top_k: usize) -> Vec { - assert_eq!(query.len(), self.dim); + /// + /// A query without the index dimension fails with + /// [`VectorError::DimensionMismatch`]. A distance table that exceeds the + /// memory budget fails the search: skipping its cell would drop results. + pub fn search(&self, query: &[f32], top_k: usize) -> Result, VectorError> { + check_dim(self.dim, query.len())?; if self.centroids.is_empty() || self.count == 0 { - return Vec::new(); + return Ok(Vec::new()); } - let pq = match &self.pq { - Some(p) => p, - None => return Vec::new(), + let Some(pq) = &self.pq else { + return Ok(Vec::new()); }; let nprobe = self.params.nprobe.min(self.centroids.len()); @@ -181,13 +204,7 @@ impl IvfPqIndex { .zip(&self.centroids[cell_idx]) .map(|(q, c)| q - c) .collect(); - let table = match pq.build_distance_table(&residual_query) { - Ok(t) => t, - Err(e) => { - tracing::warn!(error = %e, "IVF PQ build_distance_table budget exhausted; skipping cell"); - continue; - } - }; + let table = pq.build_distance_table(&residual_query)?; for (id, code) in &self.cells[cell_idx] { let dist = pq.asymmetric_distance(&table, code); @@ -211,7 +228,7 @@ impl IvfPqIndex { .partial_cmp(&b.distance) .unwrap_or(std::cmp::Ordering::Equal) }); - candidates + Ok(candidates) } fn nearest_centroid(&self, vector: &[f32]) -> usize { @@ -280,8 +297,8 @@ fn kmeans_centroids(data: &[&[f32]], dim: usize, k: usize, max_iter: usize) -> V } chosen }; - centroids.push(data[next_idx].to_vec()); - let last = centroids.last().expect("just pushed"); + let last = data[next_idx]; + centroids.push(last.to_vec()); for (i, point) in data.iter().enumerate() { let d = distance(point, last, DistanceMetric::L2); if d < min_dists[i] { @@ -357,13 +374,13 @@ mod tests { metric: DistanceMetric::L2, }, ); - idx.train(&refs, test_memory()); - idx.add_batch(&refs); + idx.train(&refs, test_memory()).unwrap(); + idx.add_batch(&refs).unwrap(); assert_eq!(idx.len(), 1000); let query = &vecs[500]; - let results = idx.search(query, 5); + let results = idx.search(query, 5).unwrap(); assert_eq!(results.len(), 5); assert!( results.iter().any(|r| r.id == 500), @@ -386,17 +403,22 @@ mod tests { let vecs = make_vectors(64, 8); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); let mut idx = IvfPqIndex::new(8, small_params()); - idx.train(&refs, test_memory()); - idx.add_batch(&refs[..40]); + idx.train(&refs, test_memory()).unwrap(); + idx.add_batch(&refs[..40]).unwrap(); let mark = idx.len() as u32; - let before: Vec = idx.search(&vecs[5], 40).iter().map(|r| r.id).collect(); + let before: Vec = idx + .search(&vecs[5], 40) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); - idx.add_batch(&refs[40..]); + idx.add_batch(&refs[40..]).unwrap(); idx.roll_back_to(mark, true); assert_eq!(idx.len(), 40); assert!(idx.is_trained(), "the training the index held stays"); - let after = idx.search(&vecs[5], 64); + let after = idx.search(&vecs[5], 64).unwrap(); assert!( after.iter().all(|r| r.id < mark), "no vector added after the mark is found" @@ -405,7 +427,7 @@ mod tests { assert_eq!(after_ids, before, "the search reads as before the adds"); // The next add takes the first id past the mark again. - assert_eq!(idx.add(&vecs[63]), mark); + assert_eq!(idx.add(&vecs[63]).unwrap(), mark); } #[test] @@ -413,20 +435,67 @@ mod tests { let vecs = make_vectors(16, 8); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); let mut idx = IvfPqIndex::new(8, small_params()); - idx.train(&refs, test_memory()); - idx.add_batch(&refs); + idx.train(&refs, test_memory()).unwrap(); + idx.add_batch(&refs).unwrap(); idx.roll_back_to(0, false); assert!(idx.is_empty()); assert!(!idx.is_trained()); assert_eq!(idx.n_cells(), 0); - assert!(idx.search(&vecs[0], 5).is_empty()); + assert!(idx.search(&vecs[0], 5).unwrap().is_empty()); } #[test] fn empty_index() { let idx = IvfPqIndex::new(8, IvfPqParams::default()); - assert!(idx.search(&[0.0; 8], 5).is_empty()); + assert!(idx.search(&[0.0; 8], 5).unwrap().is_empty()); + } + + #[test] + fn wrong_dimension_is_a_typed_error() { + let vecs: Vec> = (0..32) + .map(|i| (0..8).map(|d| ((i * 8 + d) % 17) as f32).collect()) + .collect(); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new( + 8, + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 2, + metric: DistanceMetric::L2, + }, + ); + assert!(matches!( + idx.add(&[0.0; 8]), + Err(VectorError::InvalidInput { .. }) + )); + idx.train(&refs, test_memory()).unwrap(); + idx.add_batch(&refs).unwrap(); + assert!(matches!( + idx.search(&[0.0; 3], 5), + Err(VectorError::DimensionMismatch { + expected: 8, + got: 3 + }) + )); + assert!(matches!( + idx.add(&[0.0; 3]), + Err(VectorError::DimensionMismatch { + expected: 8, + got: 3 + }) + )); + let short = [0.0_f32; 3]; + let mut untrained = IvfPqIndex::new(8, IvfPqParams::default()); + assert!(matches!( + untrained.train(&[&short], test_memory()), + Err(VectorError::DimensionMismatch { + expected: 8, + got: 3 + }) + )); } } diff --git a/nodedb-vector/src/matryoshka.rs b/nodedb-vector/src/matryoshka.rs index d581b38a0..760d7f0bc 100644 --- a/nodedb-vector/src/matryoshka.rs +++ b/nodedb-vector/src/matryoshka.rs @@ -22,6 +22,7 @@ use std::collections::BinaryHeap; use crate::distance::distance; +use crate::error::VectorError; use nodedb_types::vector_distance::DistanceMetric; /// Per-collection Matryoshka configuration. @@ -124,21 +125,39 @@ impl Ord for HeapEntry { /// by ascending full-dim distance. /// /// # Notes -/// - `candidates` yields `(id, full_dim_vector)` pairs. Vectors shorter than -/// `full_dim` are accepted; truncation clips to available length. +/// - `candidates` yields `(id, vector)` pairs; each vector carries at least +/// `full_dim` components, and components past `full_dim` are ignored. /// - When `coarse_dim == full_dim` the method degenerates to a single-pass /// top-k scan with one distance call per candidate (no duplicated work). +/// +/// # Errors +/// - [`VectorError::InvalidInput`] when `coarse_dim` exceeds `full_dim`. +/// - [`VectorError::DimensionMismatch`] when `query` has fewer than +/// `full_dim` components. +/// - [`VectorError::StoredDimensionMismatch`] when a candidate vector has +/// fewer than `full_dim` components. pub fn matryoshka_search<'a, I>( candidates: I, query: &[f32], options: &MatryoshkaSearchOptions, metric: DistanceMetric, -) -> Vec<(u32, f32)> +) -> Result, VectorError> where I: Iterator, { let coarse = options.coarse_dim as usize; let full = options.full_dim as usize; + if coarse > full { + return Err(VectorError::InvalidInput { + detail: format!("Matryoshka coarse dimension {coarse} exceeds full dimension {full}"), + }); + } + if query.len() < full { + return Err(VectorError::DimensionMismatch { + expected: full, + got: query.len(), + }); + } let pool_size = (options.oversample as usize).max(1) * options.k.max(1); let query_coarse = truncate(query, coarse); @@ -151,6 +170,12 @@ where let mut survivor_vecs: Vec> = Vec::with_capacity(pool_size); for (id, vec) in candidates { + if vec.len() < full { + return Err(VectorError::StoredDimensionMismatch { + expected: full, + got: vec.len(), + }); + } let vec_coarse = truncate(vec, coarse); let d = distance(query_coarse, vec_coarse, metric); @@ -163,7 +188,7 @@ where if should_insert { let vec_idx = survivor_vecs.len(); - survivor_vecs.push(vec[..full.min(vec.len())].to_vec()); + survivor_vecs.push(truncate(vec, full).to_vec()); coarse_heap.push(HeapEntry { dist: d, @@ -193,7 +218,7 @@ where reranked.sort_unstable_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)); reranked.truncate(options.k); - reranked + Ok(reranked) } #[cfg(test)] @@ -288,7 +313,7 @@ mod tests { k: 10, }; - let results = matryoshka_search(candidates, &query, &opts, DistanceMetric::L2); + let results = matryoshka_search(candidates, &query, &opts, DistanceMetric::L2).unwrap(); assert_eq!(results.len(), 10, "expected exactly k=10 results"); } @@ -317,7 +342,7 @@ mod tests { oversample: 1, k: 10, }; - let mrl = matryoshka_search(candidates, &query, &opts, DistanceMetric::L2); + let mrl = matryoshka_search(candidates, &query, &opts, DistanceMetric::L2).unwrap(); // Same IDs in same order. let direct_ids: Vec = direct.iter().map(|(id, _)| *id).collect(); @@ -327,4 +352,51 @@ mod tests { "coarse==full should equal direct search" ); } + + #[test] + fn short_query_or_candidate_is_a_typed_error() { + let vecs = make_vecs(4, 8); + let opts = MatryoshkaSearchOptions { + coarse_dim: 4, + full_dim: 8, + oversample: 2, + k: 2, + }; + let candidates = vecs + .iter() + .enumerate() + .map(|(i, v)| (i as u32, v.as_slice())); + assert!(matches!( + matryoshka_search(candidates, &[0.0; 6], &opts, DistanceMetric::L2), + Err(VectorError::DimensionMismatch { + expected: 8, + got: 6 + }) + )); + + let short = [0.0_f32; 5]; + let candidates = std::iter::once((0_u32, &short[..])); + assert!(matches!( + matryoshka_search(candidates, &[0.0; 8], &opts, DistanceMetric::L2), + Err(VectorError::StoredDimensionMismatch { + expected: 8, + got: 5 + }) + )); + + let inverted = MatryoshkaSearchOptions { + coarse_dim: 16, + full_dim: 8, + oversample: 2, + k: 2, + }; + let candidates = vecs + .iter() + .enumerate() + .map(|(i, v)| (i as u32, v.as_slice())); + assert!(matches!( + matryoshka_search(candidates, &[0.0; 8], &inverted, DistanceMetric::L2), + Err(VectorError::InvalidInput { .. }) + )); + } } diff --git a/nodedb-vector/src/multivec/meta_embed.rs b/nodedb-vector/src/multivec/meta_embed.rs index c1ef5ea56..67e896e2c 100644 --- a/nodedb-vector/src/multivec/meta_embed.rs +++ b/nodedb-vector/src/multivec/meta_embed.rs @@ -17,6 +17,7 @@ use nodedb_types::vector_distance::DistanceMetric; use super::plaid::PlaidPruner; use super::scoring::budgeted_maxsim; use super::storage::MultiVectorStore; +use crate::error::{VectorError, check_dim}; /// Search a `MultiVectorStore` using budgeted MaxSim with optional PLAID /// candidate pruning. @@ -30,7 +31,9 @@ use super::storage::MultiVectorStore; /// * `metric` — distance metric (Cosine recommended for MetaEmbed). /// /// # Returns -/// A `Vec<(doc_id, score)>` sorted descending by score, length ≤ `k`. +/// A `Vec<(doc_id, score)>` sorted descending by score, length ≤ `k`, or +/// [`VectorError::DimensionMismatch`] when a query vector does not have the +/// store dimension. pub fn meta_embed_search( store: &MultiVectorStore, plaid: Option<&PlaidPruner>, @@ -38,9 +41,12 @@ pub fn meta_embed_search( budget: u8, k: usize, metric: DistanceMetric, -) -> Vec<(u32, f32)> { +) -> Result, VectorError> { + for v in query { + check_dim(store.dim, v.len())?; + } if k == 0 || query.is_empty() { - return Vec::new(); + return Ok(Vec::new()); } // Effective budget: 0 means use all query vectors. @@ -52,7 +58,7 @@ pub fn meta_embed_search( // Determine candidate set. let candidate_ids: Vec = match plaid { - Some(pruner) => pruner.candidates(query), + Some(pruner) => pruner.candidates(query)?, None => store.iter().map(|doc| doc.doc_id).collect(), }; @@ -70,7 +76,7 @@ pub fn meta_embed_search( // Sort descending by score. scored.sort_unstable_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal)); scored.truncate(k); - scored + Ok(scored) } // --------------------------------------------------------------------------- @@ -108,7 +114,8 @@ mod tests { fn search_returns_at_most_k_results() { let store = build_store(10, 4, 2); let query = vec![vec![1.0f32, 0.0, 0.0, 0.0]]; - let results = meta_embed_search(&store, None, &query, 2, 3, DistanceMetric::Cosine); + let results = + meta_embed_search(&store, None, &query, 2, 3, DistanceMetric::Cosine).unwrap(); assert!(results.len() <= 3); } @@ -116,7 +123,8 @@ mod tests { fn search_results_sorted_descending() { let store = build_store(8, 4, 2); let query = vec![vec![1.0f32, 0.0, 0.0, 0.0]]; - let results = meta_embed_search(&store, None, &query, 2, 8, DistanceMetric::Cosine); + let results = + meta_embed_search(&store, None, &query, 2, 8, DistanceMetric::Cosine).unwrap(); for w in results.windows(2) { assert!(w[0].1 >= w[1].1, "not sorted: {:?}", results); } @@ -130,9 +138,10 @@ mod tests { let pruner = PlaidPruner::train(&store, 3, 10, 99); let query = vec![vec![1.0f32, 0.0f32]]; - let unfiltered = meta_embed_search(&store, None, &query, 2, 9, DistanceMetric::Cosine); + let unfiltered = + meta_embed_search(&store, None, &query, 2, 9, DistanceMetric::Cosine).unwrap(); let filtered = - meta_embed_search(&store, Some(&pruner), &query, 2, 9, DistanceMetric::Cosine); + meta_embed_search(&store, Some(&pruner), &query, 2, 9, DistanceMetric::Cosine).unwrap(); let unfiltered_ids: std::collections::HashSet = unfiltered.iter().map(|(id, _)| *id).collect(); @@ -148,7 +157,7 @@ mod tests { #[test] fn search_empty_query_returns_empty() { let store = build_store(5, 4, 2); - let results = meta_embed_search(&store, None, &[], 2, 5, DistanceMetric::Cosine); + let results = meta_embed_search(&store, None, &[], 2, 5, DistanceMetric::Cosine).unwrap(); assert!(results.is_empty()); } @@ -156,7 +165,8 @@ mod tests { fn search_k_zero_returns_empty() { let store = build_store(5, 4, 2); let query = vec![vec![1.0f32, 0.0, 0.0, 0.0]]; - let results = meta_embed_search(&store, None, &query, 2, 0, DistanceMetric::Cosine); + let results = + meta_embed_search(&store, None, &query, 2, 0, DistanceMetric::Cosine).unwrap(); assert!(results.is_empty()); } @@ -166,8 +176,29 @@ mod tests { // Query is also in direction 0 — doc 0 should rank first. let store = build_store(4, 4, 1); let query = vec![vec![1.0f32, 0.0, 0.0, 0.0]]; - let results = meta_embed_search(&store, None, &query, 1, 1, DistanceMetric::Cosine); + let results = + meta_embed_search(&store, None, &query, 1, 1, DistanceMetric::Cosine).unwrap(); assert_eq!(results.len(), 1); assert_eq!(results[0].0, 0, "expected doc_id=0 to be top result"); } + + #[test] + fn wrong_dimension_query_is_a_typed_error() { + let store = build_store(4, 4, 2); + let pruner = PlaidPruner::train(&store, 2, 3, 7); + let query = vec![vec![1.0f32, 0.0, 0.0]]; + for plaid in [None, Some(&pruner)] { + let result = meta_embed_search(&store, plaid, &query, 1, 2, DistanceMetric::Cosine); + assert!( + matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + ), + "{result:?}" + ); + } + } } diff --git a/nodedb-vector/src/multivec/plaid.rs b/nodedb-vector/src/multivec/plaid.rs index f2ac46fb7..7a49dc5ea 100644 --- a/nodedb-vector/src/multivec/plaid.rs +++ b/nodedb-vector/src/multivec/plaid.rs @@ -13,6 +13,7 @@ use std::collections::{HashMap, HashSet}; use crate::distance::scalar::scalar_distance; +use crate::error::{VectorError, check_dim}; use nodedb_types::vector_distance::DistanceMetric; use super::storage::MultiVectorStore; @@ -258,9 +259,18 @@ impl PlaidPruner { /// /// The query centroid bag is the set of nearest centroids for each query /// vector. - pub fn candidates(&self, query: &[Vec]) -> Vec { - if self.centroids.is_empty() || query.is_empty() { - return Vec::new(); + /// + /// A query vector without the centroid dimension fails with + /// [`VectorError::DimensionMismatch`]. + pub fn candidates(&self, query: &[Vec]) -> Result, VectorError> { + let Some(first) = self.centroids.first() else { + return Ok(Vec::new()); + }; + for v in query { + check_dim(first.len(), v.len())?; + } + if query.is_empty() { + return Ok(Vec::new()); } // Build query centroid bag. @@ -277,11 +287,12 @@ impl PlaidPruner { .collect(); // Collect docs that share at least one centroid with the query. - self.doc_centroids + Ok(self + .doc_centroids .iter() .filter(|(_, doc_ids)| doc_ids.iter().any(|id| query_bag.contains(id))) .map(|(&doc_id, _)| doc_id) - .collect() + .collect()) } } @@ -351,7 +362,7 @@ mod tests { // A query near cluster A should return at least some candidates. let query = vec![vec![0.0f32, 0.0f32]]; - let cands = pruner.candidates(&query); + let cands = pruner.candidates(&query).unwrap(); assert!(!cands.is_empty(), "expected at least one candidate"); } @@ -361,7 +372,7 @@ mod tests { let store = MultiVectorStore::new(2, MultiVecMode::PerToken); let pruner = PlaidPruner::train(&store, 3, 5, 1); let query = vec![vec![0.0f32, 0.0f32]]; - assert!(pruner.candidates(&query).is_empty()); + assert!(pruner.candidates(&query).unwrap().is_empty()); } #[test] @@ -375,7 +386,7 @@ mod tests { vec![10.0f32, 0.0f32], vec![0.0f32, 10.0f32], ]; - let mut cands = pruner.candidates(&query); + let mut cands = pruner.candidates(&query).unwrap(); cands.sort_unstable(); cands.dedup(); assert_eq!(cands.len(), 9, "all docs should be candidates: {:?}", cands); diff --git a/nodedb-vector/src/navix/acorn.rs b/nodedb-vector/src/navix/acorn.rs index 6ec7f8444..2ea98cc76 100644 --- a/nodedb-vector/src/navix/acorn.rs +++ b/nodedb-vector/src/navix/acorn.rs @@ -17,6 +17,7 @@ mod inner { use roaring::RoaringBitmap; use crate::distance::distance; + use crate::error::{VectorError, check_dim}; use crate::hnsw::graph::{Candidate, HnswIndex}; use crate::navix::traversal::SearchResult; @@ -43,26 +44,34 @@ mod inner { /// ACORN-1 filtered search. /// /// Uses static 2-hop expansion: when a 1-hop neighbor is not in `allowed`, - /// expand to its 2-hop neighbors unconditionally. + /// expand to its 2-hop neighbors unconditionally. A query without the + /// index dimension fails with [`VectorError::DimensionMismatch`]. pub fn acorn_search( index: &HnswIndex, query: &[f32], options: &AcornSearchOptions, metric: nodedb_types::vector_distance::DistanceMetric, - ) -> Vec { + ) -> Result, VectorError> { + check_dim(index.dim(), query.len())?; if index.is_empty() || options.allowed.is_empty() || options.k == 0 { - return Vec::new(); + return Ok(Vec::new()); } let total = index.len(); let global_sel = options.allowed.len() as f64 / total as f64; if global_sel < options.brute_force_threshold { - return brute_force_on_allowed(index, query, options.k, &options.allowed, metric); + return Ok(brute_force_on_allowed( + index, + query, + options.k, + &options.allowed, + metric, + )); } let Some(ep) = index.entry_point() else { - return Vec::new(); + return Ok(Vec::new()); }; // Phase 1: greedy descent (unfiltered) to find best layer-0 entry. @@ -78,14 +87,14 @@ mod inner { let ef = options.ef_search.max(options.k); let results = acorn_search_layer_0(index, query, current_ep, ef, &options.allowed, metric); - results + Ok(results .into_iter() .take(options.k) .map(|c| SearchResult { id: c.id, distance: c.dist, }) - .collect() + .collect()) } /// Minimal greedy single-layer descent used for Phase-1 layer navigation. @@ -298,7 +307,7 @@ mod inner { brute_force_threshold: 0.001, }; - let res = acorn_search(&idx, &query, &opts, DistanceMetric::L2); + let res = acorn_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); assert!(!res.is_empty()); for r in &res { assert!( @@ -309,6 +318,24 @@ mod inner { } } + #[test] + fn acorn_wrong_dimension_query_is_a_typed_error() { + let idx = build_index(20); + let opts = AcornSearchOptions { + k: 3, + ef_search: 64, + allowed: (0..20u32).collect(), + brute_force_threshold: 0.001, + }; + assert!(matches!( + acorn_search(&idx, &[1.0], &opts, DistanceMetric::L2), + Err(VectorError::DimensionMismatch { + expected: 3, + got: 1 + }) + )); + } + /// Very low selectivity (1 ID out of 20) — result must be that single ID. #[test] fn acorn_single_allowed_id() { @@ -325,7 +352,7 @@ mod inner { brute_force_threshold: 0.001, }; - let res = acorn_search(&idx, &query, &opts, DistanceMetric::L2); + let res = acorn_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); assert!(res.len() <= 1); if let Some(r) = res.first() { assert_eq!(r.id, 7); diff --git a/nodedb-vector/src/navix/traversal.rs b/nodedb-vector/src/navix/traversal.rs index 257d68648..016ad8b9d 100644 --- a/nodedb-vector/src/navix/traversal.rs +++ b/nodedb-vector/src/navix/traversal.rs @@ -15,6 +15,7 @@ use std::collections::{BinaryHeap, HashSet}; use roaring::RoaringBitmap; use crate::distance::distance; +use crate::error::{VectorError, check_dim}; use crate::hnsw::graph::{Candidate, HnswIndex}; use crate::navix::selectivity::{NavixHeuristic, local_selectivity_at, pick_heuristic}; @@ -59,28 +60,38 @@ impl Default for NavixSearchOptions { /// Returns up to `options.k` nearest vectors from `index` to `query`, where /// candidate IDs must be present in `options.allowed`. /// +/// Returns an empty Vec when the index is empty or `options.allowed` is empty. +/// /// # Errors /// -/// Returns an empty Vec when the index is empty or `options.allowed` is empty. +/// [`VectorError::DimensionMismatch`] when `query` does not have the index +/// dimension. pub fn navix_search( index: &HnswIndex, query: &[f32], options: &NavixSearchOptions, metric: nodedb_types::vector_distance::DistanceMetric, -) -> Vec { +) -> Result, VectorError> { + check_dim(index.dim(), query.len())?; if index.is_empty() || options.allowed.is_empty() || options.k == 0 { - return Vec::new(); + return Ok(Vec::new()); } let total = index.len(); let global_sel = options.allowed.len() as f64 / total as f64; if global_sel < options.brute_force_threshold { - return brute_force_on_allowed(index, query, options.k, &options.allowed, metric); + return Ok(brute_force_on_allowed( + index, + query, + options.k, + &options.allowed, + metric, + )); } let Some(ep) = index.entry_point() else { - return Vec::new(); + return Ok(Vec::new()); }; // Phase 1: greedy descent from max_layer to layer 1 (unfiltered, as in @@ -97,14 +108,14 @@ pub fn navix_search( let ef = options.ef_search.max(options.k); let results = navix_search_layer_0(index, query, current_ep, ef, &options.allowed, metric); - results + Ok(results .into_iter() .take(options.k) .map(|c| SearchResult { id: c.id, distance: c.dist, }) - .collect() + .collect()) } // ── Internal helpers ────────────────────────────────────────────────────────── @@ -513,8 +524,8 @@ mod tests { brute_force_threshold: 0.001, }; - let navix_res = navix_search(&idx, &query, &opts, DistanceMetric::L2); - let hnsw_res = idx.search(&query, 5, 64); + let navix_res = navix_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); + let hnsw_res = idx.search(&query, 5, 64).unwrap(); assert!(!navix_res.is_empty()); // The best result should be id=10 (exact match) in both cases. @@ -536,7 +547,7 @@ mod tests { brute_force_threshold: 0.001, }; - let res = navix_search(&idx, &query, &opts, DistanceMetric::L2); + let res = navix_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); // With only one allowed ID, we get at most 1 result. assert!(res.len() <= 1); if let Some(r) = res.first() { @@ -562,7 +573,7 @@ mod tests { brute_force_threshold: 0.001, }; - let res = navix_search(&idx, &query, &opts, DistanceMetric::L2); + let res = navix_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); assert!(!res.is_empty()); for r in &res { assert!( @@ -593,7 +604,7 @@ mod tests { brute_force_threshold: 0.5, }; - let res = navix_search(&idx, &query, &opts, DistanceMetric::L2); + let res = navix_search(&idx, &query, &opts, DistanceMetric::L2).unwrap(); // Manual brute-force reference. let mut manual: Vec<(u32, f32)> = allowed @@ -634,7 +645,7 @@ mod tests { allowed, brute_force_threshold: 0.001, }; - let res = navix_search(&idx, &[1.0, 0.0, 0.0], &opts, DistanceMetric::L2); + let res = navix_search(&idx, &[1.0, 0.0, 0.0], &opts, DistanceMetric::L2).unwrap(); assert!(res.is_empty()); } @@ -648,7 +659,28 @@ mod tests { allowed: RoaringBitmap::new(), brute_force_threshold: 0.001, }; - let res = navix_search(&idx, &[5.0, 0.0, 0.0], &opts, DistanceMetric::L2); + let res = navix_search(&idx, &[5.0, 0.0, 0.0], &opts, DistanceMetric::L2).unwrap(); assert!(res.is_empty()); } + + #[test] + fn wrong_dimension_query_is_a_typed_error() { + let idx = build_index(20); + let opts = NavixSearchOptions { + k: 3, + allowed: (0..20u32).collect(), + ..NavixSearchOptions::default() + }; + let result = navix_search(&idx, &[1.0, 0.0], &opts, DistanceMetric::L2); + assert!( + matches!( + result, + Err(crate::error::VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + ), + "{result:?}" + ); + } } diff --git a/nodedb-vector/src/quantize/pq.rs b/nodedb-vector/src/quantize/pq.rs index dd143bfad..865cec455 100644 --- a/nodedb-vector/src/quantize/pq.rs +++ b/nodedb-vector/src/quantize/pq.rs @@ -19,7 +19,7 @@ use std::mem::size_of; use nodedb_mem::{ReservationToken, ScopedMemory}; use nodedb_types::decode_bounds::checked_decode_capacity; -use crate::error::VectorError; +use crate::error::{VectorError, check_dim}; /// Hard ceiling for a decoded PQ vector. This bounds corrupted persisted /// configuration even when the codec has no scoped memory handle attached. @@ -99,24 +99,39 @@ impl PqCodec { k: usize, max_iter: usize, memory: ScopedMemory, - ) -> Self { - assert!(!vectors.is_empty()); - assert!( - dim > 0 - && dim <= MAX_PQ_DECODE_DIM - && m > 0 - && k > 0 - && k <= usize::from(u8::MAX) + 1 - && k <= vectors.len() - ); - assert!( - dim.is_multiple_of(m), - "dim ({dim}) must be divisible by m ({m})" - ); - + ) -> Result { + let invalid = |detail: String| Err(VectorError::InvalidInput { detail }); + if vectors.is_empty() { + return invalid("PQ training needs at least one vector".into()); + } + if dim == 0 || dim > MAX_PQ_DECODE_DIM { + return invalid(format!( + "PQ dimension {dim} is outside 1..={MAX_PQ_DECODE_DIM}" + )); + } + if m == 0 || !dim.is_multiple_of(m) { + return invalid(format!("PQ dimension {dim} must be divisible by m ({m})")); + } + if k == 0 || k > usize::from(u8::MAX) + 1 || k > vectors.len() { + return invalid(format!( + "PQ centroid count {k} must be in 1..=256 and at most the {} training vectors", + vectors.len() + )); + } let sub_dim = dim / m; - let codebook_bytes = pq_codebook_allocation_bytes(m, k, sub_dim); - assert!(codebook_bytes.is_some_and(|bytes| bytes <= MAX_PQ_CODEBOOK_BYTES)); + if pq_codebook_allocation_bytes(m, k, sub_dim) + .is_none_or(|bytes| bytes > MAX_PQ_CODEBOOK_BYTES) + { + return invalid(format!( + "PQ codebook for m={m}, k={k}, sub-dimension {sub_dim} exceeds \ + {MAX_PQ_CODEBOOK_BYTES} bytes" + )); + } + // Parameters are checked before any vector is read. + for v in vectors { + check_dim(dim, v.len())?; + } + let mut codebooks = Vec::with_capacity(m); for sub in 0..m { @@ -131,14 +146,14 @@ impl PqCodec { codebooks.push(centroids); } - Self { + Ok(Self { dim, m, k, sub_dim, codebooks, memory, - } + }) } /// Encode a vector: for each subvector, find the nearest centroid index. @@ -147,8 +162,11 @@ impl PqCodec { /// intentionally skipped here to avoid atomic overhead on every candidate /// during search; use [`encode_batch`] for bulk encoding with budget /// enforcement. + /// + /// Precondition: `vector.len() == self.dim`. This per-candidate hot path + /// does not re-check it; every caller checks the dimension first + /// (`encode_batch`, the IVF-PQ `add`, and the codec-index entry points). pub fn encode(&self, vector: &[f32]) -> Vec { - debug_assert_eq!(vector.len(), self.dim); let mut code = Vec::with_capacity(self.m); for sub in 0..self.m { let offset = sub * self.sub_dim; @@ -165,6 +183,9 @@ impl PqCodec { /// before allocating the output buffer. The guard is released at /// the end of this call — the buffer itself remains alive. pub fn encode_batch(&self, vectors: &[&[f32]]) -> Result, VectorError> { + for v in vectors { + check_dim(self.dim, v.len())?; + } let capacity = self.m * vectors.len(); let _g = try_reserve_or_skip(&self.memory, capacity * size_of::())?; let mut out = Vec::with_capacity(capacity); @@ -183,7 +204,7 @@ impl PqCodec { /// Charges `m * k * size_of::()` bytes to the bound budget (if set) /// before allocating the table. pub fn build_distance_table(&self, query: &[f32]) -> Result>, VectorError> { - debug_assert_eq!(query.len(), self.dim); + check_dim(self.dim, query.len())?; let total_bytes = self.m * self.k * size_of::(); let _g = try_reserve_or_skip(&self.memory, total_bytes)?; let mut table = Vec::with_capacity(self.m); @@ -223,12 +244,12 @@ impl PqCodec { .m .checked_mul(self.sub_dim) .filter(|&value| value == self.dim && value <= MAX_PQ_DECODE_DIM) - .ok_or(VectorError::DimensionMismatch { + .ok_or(VectorError::StoredDimensionMismatch { expected: self.dim, got: 0, })?; if code.len() != self.m { - return Err(VectorError::DimensionMismatch { + return Err(VectorError::StoredDimensionMismatch { expected: self.m, got: code.len(), }); @@ -246,7 +267,7 @@ impl PqCodec { MAX_PQ_DECODE_DIM, MAX_PQ_DECODE_DIM * size_of::(), ) - .ok_or(VectorError::DimensionMismatch { + .ok_or(VectorError::StoredDimensionMismatch { expected: self.dim, got: 0, })?; @@ -421,9 +442,9 @@ fn kmeans(data: &[&[f32]], dim: usize, k: usize, max_iter: usize) -> Vec) { + assert!( + matches!(result, Err(VectorError::InvalidInput { .. })), + "expected InvalidInput" + ); + } + + #[test] + fn train_rejects_a_vector_of_the_wrong_dimension() { + let good = [0.0_f32, 1.0]; + let short = [0.0_f32]; + let result = PqCodec::train(&[&good, &short], 2, 1, 1, 1, test_memory()); + assert!(matches!( + result, + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + )); + } + + #[test] + fn distance_table_rejects_a_query_of_the_wrong_dimension() { + let a = [0.0_f32, 1.0]; + let b = [1.0_f32, 0.0]; + let codec = PqCodec::train(&[&a, &b], 2, 1, 2, 1, test_memory()).unwrap(); + assert!(matches!( + codec.build_distance_table(&[1.0]), + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + )); + assert!(matches!( + codec.encode_batch(&[&[1.0]]), + Err(VectorError::DimensionMismatch { + expected: 2, + got: 1 + }) + )); + } + fn make_clustered_data() -> Vec> { // 4 clusters in 4D space, 50 points each. let mut vecs = Vec::new(); @@ -501,40 +564,49 @@ mod tests { } #[test] - #[should_panic] fn train_rejects_dimension_above_decode_limit() { let vector = [0.0]; - PqCodec::train(&[&vector], MAX_PQ_DECODE_DIM + 1, 1, 1, 1, test_memory()); + assert_invalid(PqCodec::train( + &[&vector], + MAX_PQ_DECODE_DIM + 1, + 1, + 1, + 1, + test_memory(), + )); } #[test] - #[should_panic] fn train_rejects_raw_64_mib_codebook_once_container_overhead_is_counted() { let vector = [0.0]; let vectors = vec![vector.as_slice(); 256]; - PqCodec::train(&vectors, 65_536, 1, 256, 1, test_memory()); + assert_invalid(PqCodec::train(&vectors, 65_536, 1, 256, 1, test_memory())); } #[test] - #[should_panic] fn train_rejects_codebook_above_decode_limit() { let vector = [0.0]; let vectors = vec![vector.as_slice(); 17]; - PqCodec::train(&vectors, MAX_PQ_DECODE_DIM, 1, 17, 1, test_memory()); + assert_invalid(PqCodec::train( + &vectors, + MAX_PQ_DECODE_DIM, + 1, + 17, + 1, + test_memory(), + )); } #[test] - #[should_panic] fn train_rejects_more_centroids_than_training_vectors() { let vector = [0.0]; - PqCodec::train(&[&vector], 1, 1, 2, 1, test_memory()); + assert_invalid(PqCodec::train(&[&vector], 1, 1, 2, 1, test_memory())); } #[test] - #[should_panic] fn train_rejects_centroid_count_above_u8_encoding_range() { let vector = [0.0]; - PqCodec::train(&[&vector], 1, 1, 257, 1, test_memory()); + assert_invalid(PqCodec::train(&[&vector], 1, 1, 257, 1, test_memory())); } #[test] @@ -602,7 +674,7 @@ mod tests { fn encode_decode_roundtrip() { let vecs = make_clustered_data(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()).unwrap(); for v in &vecs { let code = codec.encode(v); @@ -616,7 +688,7 @@ mod tests { fn distance_table_gives_correct_ordering() { let vecs = make_clustered_data(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()).unwrap(); let codes: Vec> = vecs.iter().map(|v| codec.encode(v)).collect(); let query = &[5.0, 5.0, 5.0, 5.0]; @@ -650,7 +722,7 @@ mod tests { fn batch_encode() { let vecs = make_clustered_data(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()).unwrap(); let batch = codec.encode_batch(&refs).unwrap(); assert_eq!(batch.len(), 2 * 200); // M=2, N=200 @@ -661,7 +733,7 @@ mod tests { fn pq_codec_golden_format() { let vecs = make_clustered_data(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 10, test_memory()).unwrap(); let bytes = codec.to_bytes().unwrap(); diff --git a/nodedb-vector/src/quantize/pq_codec.rs b/nodedb-vector/src/quantize/pq_codec.rs index 16b505091..c526aa30c 100644 --- a/nodedb-vector/src/quantize/pq_codec.rs +++ b/nodedb-vector/src/quantize/pq_codec.rs @@ -166,7 +166,7 @@ mod tests { }) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - PqCodec::train(&refs, 4, 2, 8, 10, test_memory()) + PqCodec::train(&refs, 4, 2, 8, 10, test_memory()).unwrap() } /// `encode` round-trip: packed_bits in the UQV must match the raw diff --git a/nodedb-vector/src/quantize/sq8.rs b/nodedb-vector/src/quantize/sq8.rs index d54a2e6e9..b2e10c4d0 100644 --- a/nodedb-vector/src/quantize/sq8.rs +++ b/nodedb-vector/src/quantize/sq8.rs @@ -15,7 +15,7 @@ use serde::{Deserialize, Serialize}; -use crate::error::VectorError; +use crate::error::{VectorError, check_dim}; /// Magic bytes identifying a serialized [`Sq8Codec`] blob. /// @@ -47,15 +47,29 @@ impl Sq8Codec { /// At least 1000 vectors recommended for stable calibration; /// for fewer vectors the bounds may be tight, causing clipping /// on future inserts outside the calibration range. - pub fn calibrate(vectors: &[&[f32]], dim: usize) -> Self { - assert!(!vectors.is_empty(), "cannot calibrate on empty set"); - assert!(dim > 0); + /// + /// An empty set or a zero `dim` fails with [`VectorError::InvalidInput`]; + /// a vector without `dim` components fails with + /// [`VectorError::DimensionMismatch`]. + pub fn calibrate(vectors: &[&[f32]], dim: usize) -> Result { + if vectors.is_empty() { + return Err(VectorError::InvalidInput { + detail: "SQ8 calibration needs at least one vector".into(), + }); + } + if dim == 0 { + return Err(VectorError::InvalidInput { + detail: "SQ8 calibration needs a non-zero dimension".into(), + }); + } + for v in vectors { + check_dim(dim, v.len())?; + } let mut mins = vec![f32::MAX; dim]; let mut maxs = vec![f32::MIN; dim]; for v in vectors { - debug_assert_eq!(v.len(), dim); for d in 0..dim { if v[d] < mins[d] { mins[d] = v[d]; @@ -76,12 +90,24 @@ impl Sq8Codec { } } - Self { + Ok(Self { dim, mins, maxs, scales, inv_scales, + }) + } + + /// A codec over the unit range `[0, 1]` on every dimension, for + /// normalized embeddings before any calibration data exists. + pub fn unit_range(dim: usize) -> Self { + Self { + dim, + mins: vec![0.0; dim], + maxs: vec![1.0; dim], + scales: vec![1.0 / 255.0; dim], + inv_scales: vec![255.0; dim], } } @@ -223,7 +249,7 @@ mod tests { .map(|i| vec![i as f32 * 0.1, (i as f32).sin(), (i as f32).cos()]) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, 3) + Sq8Codec::calibrate(&refs, 3).unwrap() } #[test] @@ -284,7 +310,7 @@ mod tests { fn quantize_dequantize_roundtrip() { let vecs = make_vectors(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, 3); + let codec = Sq8Codec::calibrate(&refs, 3).unwrap(); for v in &vecs { let q = codec.quantize(v); @@ -306,7 +332,7 @@ mod tests { fn asymmetric_l2_close_to_exact() { let vecs = make_vectors(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, 3); + let codec = Sq8Codec::calibrate(&refs, 3).unwrap(); let query = &[5.0, 0.5, -0.5]; for v in &vecs { @@ -330,7 +356,7 @@ mod tests { fn batch_quantize() { let vecs = make_vectors(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, 3); + let codec = Sq8Codec::calibrate(&refs, 3).unwrap(); let batch = codec.quantize_batch(&refs); assert_eq!(batch.len(), 3 * 100); @@ -345,10 +371,31 @@ mod tests { // All vectors have the same value in dimension 0. let vecs: Vec> = (0..10).map(|i| vec![5.0, i as f32]).collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = Sq8Codec::calibrate(&refs, 2); + let codec = Sq8Codec::calibrate(&refs, 2).unwrap(); // Constant dimension should quantize to 0 without NaN/inf. let q = codec.quantize(&[5.0, 3.0]); assert_eq!(q[0], 0); // constant dim } + + #[test] + fn calibrate_rejects_bad_input_with_typed_errors() { + use crate::error::VectorError; + assert!(matches!( + Sq8Codec::calibrate(&[], 3), + Err(VectorError::InvalidInput { .. }) + )); + let v = [1.0_f32, 2.0]; + assert!(matches!( + Sq8Codec::calibrate(&[&v], 0), + Err(VectorError::InvalidInput { .. }) + )); + assert!(matches!( + Sq8Codec::calibrate(&[&v], 3), + Err(VectorError::DimensionMismatch { + expected: 3, + got: 2 + }) + )); + } } diff --git a/nodedb-vector/src/quantize/sq8_codec.rs b/nodedb-vector/src/quantize/sq8_codec.rs index 914fa624e..d9766d251 100644 --- a/nodedb-vector/src/quantize/sq8_codec.rs +++ b/nodedb-vector/src/quantize/sq8_codec.rs @@ -118,7 +118,7 @@ mod tests { .map(|i| vec![i as f32 * 0.1, -(i as f32) * 0.05, 1.0 + i as f32 * 0.02]) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - Sq8Codec::calibrate(&refs, 3) + Sq8Codec::calibrate(&refs, 3).unwrap() } /// `encode` round-trip: packed_bits in the UQV must match the raw diff --git a/nodedb-vector/src/rerank/codecs/pq.rs b/nodedb-vector/src/rerank/codecs/pq.rs index f2a3c3a60..8be53356b 100644 --- a/nodedb-vector/src/rerank/codecs/pq.rs +++ b/nodedb-vector/src/rerank/codecs/pq.rs @@ -227,7 +227,8 @@ impl RerankCodec for PqRerank { self.k, self.max_iter, self.memory.clone(), - ); + ) + .map_err(|e| RerankError::BadInput(format!("pq train: {e}")))?; self.codec = Some(codec); Ok(()) } diff --git a/nodedb-vector/src/rerank/codecs/sq8.rs b/nodedb-vector/src/rerank/codecs/sq8.rs index 5b615607d..53c1e27ff 100644 --- a/nodedb-vector/src/rerank/codecs/sq8.rs +++ b/nodedb-vector/src/rerank/codecs/sq8.rs @@ -40,13 +40,12 @@ impl Sq8Rerank { /// which is suitable for normalized embeddings. For best accuracy call /// `train()` with representative samples before encoding. pub fn new(dim: usize) -> Self { - // Build a minimal calibration over the unit range so encoding is - // functional before train() is called. - let lo = vec![0.0f32; dim]; - let hi = vec![1.0f32; dim]; - let samples: Vec<&[f32]> = vec![lo.as_slice(), hi.as_slice()]; - let codec = Sq8Codec::calibrate(&samples, dim); - Self { codec, dim } + // Calibrated over the unit range so encoding is functional before + // train() is called. + Self { + codec: Sq8Codec::unit_range(dim), + dim, + } } /// Wrap an already-trained `Sq8Codec`. @@ -133,7 +132,8 @@ impl RerankCodec for Sq8Rerank { "sq8 train: empty sample set".to_string(), )); } - self.codec = Sq8Codec::calibrate(samples, self.dim); + self.codec = Sq8Codec::calibrate(samples, self.dim) + .map_err(|e| RerankError::BadInput(format!("sq8 train: {e}")))?; Ok(()) } } diff --git a/nodedb-vector/src/sieve/collection.rs b/nodedb-vector/src/sieve/collection.rs index 3daf70866..ea7c10c9b 100644 --- a/nodedb-vector/src/sieve/collection.rs +++ b/nodedb-vector/src/sieve/collection.rs @@ -148,7 +148,7 @@ mod tests { .expect("build"); let idx = coll.get(&"tenant_id=1".to_string()).unwrap(); - let results = idx.search(&[2.0, 2.0, 2.0], 2, 32); + let results = idx.search(&[2.0, 2.0, 2.0], 2, 32).unwrap(); assert!(!results.is_empty()); } } diff --git a/nodedb-vector/src/sieve/router.rs b/nodedb-vector/src/sieve/router.rs index 063541bde..c3a905e12 100644 --- a/nodedb-vector/src/sieve/router.rs +++ b/nodedb-vector/src/sieve/router.rs @@ -39,7 +39,9 @@ impl<'a> SieveRouter<'a> { /// /// # Returns /// - /// Up to `k` nearest-neighbour results, sorted by ascending distance. + /// Up to `k` nearest-neighbour results, sorted by ascending distance, or + /// [`VectorError::DimensionMismatch`](crate::error::VectorError::DimensionMismatch) + /// when `query` does not have the index dimension. pub fn route( &self, query: &[f32], @@ -48,7 +50,7 @@ impl<'a> SieveRouter<'a> { k: usize, ef_search: usize, metric: DistanceMetric, - ) -> Vec { + ) -> Result, crate::error::VectorError> { // Fast path: subindex hit. if let Some(sig) = predicate_signature && let Some(subindex) = self.collection.get(sig) @@ -63,13 +65,13 @@ impl<'a> SieveRouter<'a> { allowed, brute_force_threshold: 0.001, }; - navix_search(self.fallback, query, &opts, metric) + Ok(navix_search(self.fallback, query, &opts, metric)? .into_iter() .map(|r| SearchResult { id: r.id, distance: r.distance, }) - .collect() + .collect()) } } @@ -124,14 +126,16 @@ mod tests { fallback: &fallback, }; - let results = router.route( - &[2.0, 0.0, 0.0], - Some(&"T".to_string()), - all_allowed(20), // bitmap irrelevant on subindex path - 3, - 32, - DistanceMetric::L2, - ); + let results = router + .route( + &[2.0, 0.0, 0.0], + Some(&"T".to_string()), + all_allowed(20), // bitmap irrelevant on subindex path + 3, + 32, + DistanceMetric::L2, + ) + .unwrap(); assert!(!results.is_empty()); // All result IDs must be within the subindex range [0..5). @@ -152,14 +156,16 @@ mod tests { }; let allowed = all_allowed(20); - let results = router.route( - &[10.0, 0.0, 0.0], - Some(&"unknown_sig".to_string()), - allowed, - 3, - 64, - DistanceMetric::L2, - ); + let results = router + .route( + &[10.0, 0.0, 0.0], + Some(&"unknown_sig".to_string()), + allowed, + 3, + 64, + DistanceMetric::L2, + ) + .unwrap(); assert!(!results.is_empty()); // The nearest vector to [10,0,0] in [0..20] is id=10. @@ -182,7 +188,9 @@ mod tests { }; let allowed = all_allowed(20); - let results = router.route(&[5.0, 0.0, 0.0], None, allowed, 3, 64, DistanceMetric::L2); + let results = router + .route(&[5.0, 0.0, 0.0], None, allowed, 3, 64, DistanceMetric::L2) + .unwrap(); assert!(!results.is_empty()); // Must include id=5 since fallback has all 20 vectors. diff --git a/nodedb-vector/src/vamana/build.rs b/nodedb-vector/src/vamana/build.rs index 56d873d7d..a039b83b3 100644 --- a/nodedb-vector/src/vamana/build.rs +++ b/nodedb-vector/src/vamana/build.rs @@ -15,6 +15,7 @@ use nodedb_codec::vector_quant::codec::VectorCodec; +use crate::error::{VectorError, check_dim}; use crate::vamana::graph::VamanaGraph; use crate::vamana::prune::alpha_prune; @@ -35,10 +36,12 @@ use crate::vamana::prune::alpha_prune; /// * `alpha` — pruning factor (typical: 1.2; must be > 1). /// * `l_build` — beam width during construction (typical: 100). /// -/// # Panics +/// # Errors /// -/// Panics if `vectors`, `ids`, and `quantized` do not all have the same -/// length, or if `vectors` is empty. +/// - [`VectorError::InvalidInput`] when `vectors` is empty, or `vectors`, +/// `ids` and `quantized` differ in length. +/// - [`VectorError::DimensionMismatch`] when a vector's length differs from +/// the first vector's. pub fn build_vamana( vectors: &[Vec], ids: &[u64], @@ -47,21 +50,28 @@ pub fn build_vamana( r: usize, alpha: f32, l_build: usize, -) -> VamanaGraph { - assert_eq!( - vectors.len(), - ids.len(), - "vectors and ids must have equal length" - ); - assert_eq!( - vectors.len(), - quantized.len(), - "vectors and quantized must have equal length" - ); - assert!(!vectors.is_empty(), "cannot build from empty vector set"); +) -> Result { + let Some(first) = vectors.first() else { + return Err(VectorError::InvalidInput { + detail: "Vamana build needs at least one vector".into(), + }); + }; + if vectors.len() != ids.len() || vectors.len() != quantized.len() { + return Err(VectorError::InvalidInput { + detail: format!( + "Vamana build got {} vectors, {} ids and {} quantized vectors; the counts must match", + vectors.len(), + ids.len(), + quantized.len() + ), + }); + } + let dim = first.len(); + for v in vectors { + check_dim(dim, v.len())?; + } let n = vectors.len(); - let dim = vectors[0].len(); let l = l_build.max(r); // --- Construct graph skeleton --- @@ -132,7 +142,7 @@ pub fn build_vamana( graph.set_neighbors(i, pruned); } - graph + Ok(graph) } // ------------------------------------------------------------------ @@ -375,7 +385,7 @@ mod tests { let ids: Vec = (0..n as u64).collect(); let quantized: Vec = vecs.iter().map(|v| codec.encode(v)).collect(); - let graph = build_vamana(&vecs, &ids, &codec, &quantized, 8, 1.2, 20); + let graph = build_vamana(&vecs, &ids, &codec, &quantized, 8, 1.2, 20).unwrap(); assert_eq!(graph.len(), n); @@ -402,7 +412,7 @@ mod tests { let ids: Vec = (0..n as u64).collect(); let quantized: Vec = vecs.iter().map(|v| codec.encode(v)).collect(); - let graph = build_vamana(&vecs, &ids, &codec, &quantized, r, 1.2, 15); + let graph = build_vamana(&vecs, &ids, &codec, &quantized, r, 1.2, 15).unwrap(); for i in 0..n { assert!( diff --git a/nodedb-vector/src/vamana/search.rs b/nodedb-vector/src/vamana/search.rs index 6c228985f..b3e2b77f9 100644 --- a/nodedb-vector/src/vamana/search.rs +++ b/nodedb-vector/src/vamana/search.rs @@ -23,6 +23,7 @@ use std::collections::{BinaryHeap, HashSet}; use nodedb_codec::vector_quant::codec::VectorCodec; use crate::distance::scalar::l2_squared; +use crate::error::{VectorError, check_dim}; use crate::vamana::graph::VamanaGraph; use crate::vamana::node_fetcher::NodeFetcher; @@ -82,6 +83,14 @@ impl Ord for Candidate { /// # Returns /// /// Up to `k` results sorted by ascending distance. +/// +/// The query arrives already prepared by the codec, so its dimension is +/// checked where the caller prepares it; [`rerank`] checks the FP32 query. +/// +/// # Errors +/// +/// [`VectorError::InvalidInput`] when `quantized` does not hold one vector +/// per graph node. pub fn beam_search( graph: &VamanaGraph, query: &C::Query, @@ -90,13 +99,22 @@ pub fn beam_search( fetcher: &mut F, k: usize, l_search: usize, -) -> Vec +) -> Result, VectorError> where C: VectorCodec, F: NodeFetcher, { if graph.is_empty() || quantized.is_empty() { - return Vec::new(); + return Ok(Vec::new()); + } + if quantized.len() != graph.len() { + return Err(VectorError::InvalidInput { + detail: format!( + "Vamana search got {} quantized vectors for {} graph nodes", + quantized.len(), + graph.len() + ), + }); } let l = l_search.max(k); @@ -173,12 +191,13 @@ where }); out.truncate(k); - out.into_iter() + Ok(out + .into_iter() .map(|c| BeamSearchResult { id: graph.external_id(c.idx as usize), distance: c.dist, }) - .collect() + .collect()) } /// Rerank a candidate list using full-precision FP32 vectors from `fetcher`. @@ -192,13 +211,21 @@ where /// them without re-submitting. /// /// This is the "SSD fetch + rerank" step described in the DiskANN paper. +/// +/// # Errors +/// +/// - [`VectorError::DimensionMismatch`] when `query_fp32` does not have the +/// graph dimension. +/// - [`VectorError::StoredDimensionMismatch`] when a fetched vector does not +/// have the graph dimension. pub fn rerank( candidates: Vec, query_fp32: &[f32], fetcher: &mut F, graph: &VamanaGraph, k: usize, -) -> Vec { +) -> Result, VectorError> { + check_dim(graph.dim, query_fp32.len())?; // Build id → internal index map. let id_to_idx: std::collections::HashMap = graph.iter().map(|(idx, node)| (node.id, idx)).collect(); @@ -210,18 +237,25 @@ pub fn rerank( .collect(); fetcher.prefetch_batch(&candidate_indices); - let mut reranked: Vec = candidates - .into_iter() - .filter_map(|c| { - let idx = *id_to_idx.get(&c.id)?; - let vec = fetcher.fetch_fp32(idx as u32)?; - let d = l2_squared(query_fp32, &vec); - Some(BeamSearchResult { - id: c.id, - distance: d, - }) - }) - .collect(); + let mut reranked: Vec = Vec::with_capacity(candidates.len()); + for c in candidates { + let Some(&idx) = id_to_idx.get(&c.id) else { + continue; + }; + let Some(vec) = fetcher.fetch_fp32(idx as u32) else { + continue; + }; + if vec.len() != graph.dim { + return Err(VectorError::StoredDimensionMismatch { + expected: graph.dim, + got: vec.len(), + }); + } + reranked.push(BeamSearchResult { + id: c.id, + distance: l2_squared(query_fp32, &vec), + }); + } reranked.sort_by(|a, b| { a.distance @@ -229,7 +263,7 @@ pub fn rerank( .unwrap_or(std::cmp::Ordering::Equal) }); reranked.truncate(k); - reranked + Ok(reranked) } #[cfg(test)] @@ -293,14 +327,14 @@ mod tests { let ids: Vec = (0..n as u64).collect(); let quantized: Vec = vecs.iter().map(|v| codec.encode(v)).collect(); - let graph = build_vamana(&vecs, &ids, &codec, &quantized, 8, 1.2, 20); + let graph = build_vamana(&vecs, &ids, &codec, &quantized, 8, 1.2, 20).unwrap(); // Query with the vector at index 7; it should be the nearest result. let query_vec = vecs[7].clone(); let query = codec.prepare_query(&query_vec); let mut fetcher = InMemoryFetcher::new(dim, vecs.clone()); - let results = beam_search(&graph, &query, &codec, &quantized, &mut fetcher, 5, 20); + let results = beam_search(&graph, &query, &codec, &quantized, &mut fetcher, 5, 20).unwrap(); assert!( !results.is_empty(), @@ -315,4 +349,40 @@ mod tests { "distance to self must be near zero" ); } + + #[test] + fn wrong_dimension_is_a_typed_error() { + let dim = 4; + let codec = L2Codec; + let vecs = random_vecs(10, dim, 7); + let ids: Vec = (0..10).collect(); + let quantized: Vec = vecs.iter().map(|v| codec.encode(v)).collect(); + let graph = build_vamana(&vecs, &ids, &codec, &quantized, 4, 1.2, 8).unwrap(); + let mut fetcher = InMemoryFetcher::new(dim, vecs.clone()); + let query = codec.prepare_query(&vecs[0]); + let candidates = + beam_search(&graph, &query, &codec, &quantized, &mut fetcher, 3, 8).unwrap(); + assert!(matches!( + rerank(candidates, &[0.0; 3], &mut fetcher, &graph, 3), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 3 + }) + )); + + let mut ragged = vecs.clone(); + ragged[3] = vec![0.0; 2]; + let ragged_q: Vec = ragged.iter().map(|v| codec.encode(v)).collect(); + assert!(matches!( + build_vamana(&ragged, &ids, &codec, &ragged_q, 4, 1.2, 8), + Err(VectorError::DimensionMismatch { + expected: 4, + got: 2 + }) + )); + assert!(matches!( + beam_search(&graph, &query, &codec, &quantized[..5], &mut fetcher, 3, 8), + Err(VectorError::InvalidInput { .. }) + )); + } } diff --git a/nodedb-vector/tests/vector_suite/cases/collection_bitmap_filter.rs b/nodedb-vector/tests/vector_suite/cases/collection_bitmap_filter.rs index 9f7696821..4fdc1bece 100644 --- a/nodedb-vector/tests/vector_suite/cases/collection_bitmap_filter.rs +++ b/nodedb-vector/tests/vector_suite/cases/collection_bitmap_filter.rs @@ -28,7 +28,7 @@ fn params() -> HnswParams { /// so the next inserts land at `base_id == seal_count`. fn seal_one(coll: &mut VectorCollection, count: usize) { for i in 0..count { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let req = coll.seal("k").expect("seal produced request"); let mut idx = HnswIndex::new(req.dim, req.params.clone()); @@ -59,7 +59,9 @@ fn bitmap_filter_targets_second_segment_global_ids() { // segment's bitmap lookup tests local id 25 against a bitmap that // contains global 75 → zero matches. let bytes = bitmap_bytes([75u32]); - let results = coll.search_with_bitmap_bytes(&[75.0, 0.0], 1, 64, &bytes); + let results = coll + .search_with_bitmap_bytes(&[75.0, 0.0], 1, 64, &bytes) + .unwrap(); assert_eq!( results.len(), @@ -79,7 +81,9 @@ fn bitmap_filter_recovers_many_globals_across_segments() { let wanted: Vec = (60..70).collect(); let bytes = bitmap_bytes(wanted.iter().copied()); - let results = coll.search_with_bitmap_bytes(&[65.0, 0.0], 10, 128, &bytes); + let results = coll + .search_with_bitmap_bytes(&[65.0, 0.0], 10, 128, &bytes) + .unwrap(); assert_eq!( results.len(), @@ -107,7 +111,9 @@ fn bitmap_filter_first_segment_still_works() { seal_one(&mut coll, 50); let bytes = bitmap_bytes([10u32, 20, 30]); - let results = coll.search_with_bitmap_bytes(&[20.0, 0.0], 3, 64, &bytes); + let results = coll + .search_with_bitmap_bytes(&[20.0, 0.0], 3, 64, &bytes) + .unwrap(); let got: std::collections::HashSet = results.iter().map(|r| r.id).collect(); let expected: std::collections::HashSet = [10u32, 20, 30].into_iter().collect(); assert_eq!(got, expected); diff --git a/nodedb-vector/tests/vector_suite/cases/collection_checkpoint_tombstones.rs b/nodedb-vector/tests/vector_suite/cases/collection_checkpoint_tombstones.rs index ebf6f1cde..f63c2446d 100644 --- a/nodedb-vector/tests/vector_suite/cases/collection_checkpoint_tombstones.rs +++ b/nodedb-vector/tests/vector_suite/cases/collection_checkpoint_tombstones.rs @@ -34,7 +34,7 @@ fn params() -> HnswParams { fn growing_segment_tombstones_survive_checkpoint_roundtrip() { let mut coll = VectorCollection::new(2, params()); for i in 0..10u32 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } assert!(coll.delete(3), "delete on live growing vector must succeed"); assert!(coll.delete(7), "delete on live growing vector must succeed"); @@ -51,7 +51,7 @@ fn growing_segment_tombstones_survive_checkpoint_roundtrip() { "tombstoned growing-segment vectors resurrected on restore" ); - let results = restored.search(&[3.0, 0.0], 10, 64); + let results = restored.search(&[3.0, 0.0], 10, 64).unwrap(); let ids: std::collections::HashSet = results.iter().map(|r| r.id).collect(); assert!( !ids.contains(&3), @@ -69,7 +69,7 @@ fn building_segment_tombstones_survive_checkpoint_roundtrip() { // snapshot time, exercising the `building_segments` encode path. let mut coll = VectorCollection::with_seal_threshold(2, params(), 20); for i in 0..20u32 { - coll.insert(vec![i as f32, 0.0]); + coll.insert(vec![i as f32, 0.0]).unwrap(); } let _req = coll.seal("k").expect("seal produced request"); // Intentionally do NOT complete the build — vectors now sit in the @@ -89,7 +89,7 @@ fn building_segment_tombstones_survive_checkpoint_roundtrip() { "tombstoned building-segment vectors resurrected on restore" ); - let results = restored.search(&[5.0, 0.0], 20, 64); + let results = restored.search(&[5.0, 0.0], 20, 64).unwrap(); let ids: std::collections::HashSet = results.iter().map(|r| r.id).collect(); assert!( !ids.contains(&5), diff --git a/nodedb-vector/tests/vector_suite/cases/collection_compact_doc_map.rs b/nodedb-vector/tests/vector_suite/cases/collection_compact_doc_map.rs index 65ea25343..bb0d48349 100644 --- a/nodedb-vector/tests/vector_suite/cases/collection_compact_doc_map.rs +++ b/nodedb-vector/tests/vector_suite/cases/collection_compact_doc_map.rs @@ -30,7 +30,8 @@ fn build_collection_with_docs() -> VectorCollection { let mut coll = VectorCollection::with_seal_threshold(2, params(), 6); // Six docs, one vector each. Global ids 0..6, surrogates 1..=6. for i in 0..6u32 { - coll.insert_with_surrogate(vec![i as f32, 0.0], Surrogate::new(i + 1)); + coll.insert_with_surrogate(vec![i as f32, 0.0], Surrogate::new(i + 1)) + .unwrap(); } let req = coll.seal("k").expect("seal produced request"); let mut idx = HnswIndex::new(req.dim, req.params.clone()); @@ -56,7 +57,7 @@ fn surrogate_map_stays_correct_after_compact() { let removed = coll.compact(); assert_eq!(removed, 2, "compact should remove 2 tombstoned nodes"); - let results = coll.search(&[0.0, 0.0], 4, 64); + let results = coll.search(&[0.0, 0.0], 4, 64).unwrap(); let ids: Vec = results.iter().map(|r| r.id).collect(); assert_eq!(ids.len(), 4, "expected 4 live vectors post-compact"); @@ -81,12 +82,12 @@ fn multi_doc_map_stays_correct_after_compact() { let a_vecs: Vec> = (0..3u32).map(|i| vec![i as f32, 0.0]).collect(); let a_refs: Vec<&[f32]> = a_vecs.iter().map(|v| v.as_slice()).collect(); - let a_ids = coll.insert_multi_vector(&a_refs, doc_a); + let a_ids = coll.insert_multi_vector(&a_refs, doc_a).unwrap(); assert_eq!(a_ids, vec![0, 1, 2]); let b_vecs: Vec> = (3..6u32).map(|i| vec![i as f32, 0.0]).collect(); let b_refs: Vec<&[f32]> = b_vecs.iter().map(|v| v.as_slice()).collect(); - let b_ids = coll.insert_multi_vector(&b_refs, doc_b); + let b_ids = coll.insert_multi_vector(&b_refs, doc_b).unwrap(); assert_eq!(b_ids, vec![3, 4, 5]); let req = coll.seal("k").expect("seal produced request"); diff --git a/nodedb-vector/tests/vector_suite/cases/collection_pq_config.rs b/nodedb-vector/tests/vector_suite/cases/collection_pq_config.rs index 512769ea8..c8cfd146e 100644 --- a/nodedb-vector/tests/vector_suite/cases/collection_pq_config.rs +++ b/nodedb-vector/tests/vector_suite/cases/collection_pq_config.rs @@ -40,7 +40,7 @@ fn make_built_collection_with_pq_config() -> VectorCollection { for (d, slot) in v.iter_mut().enumerate() { *slot = ((i as f32) * 0.01 + (d as f32) * 0.1).sin(); } - coll.insert(v); + coll.insert(v).unwrap(); } let req = coll.seal("pq").expect("seal produced request"); let mut idx = HnswIndex::new(req.dim, req.params.clone()); diff --git a/nodedb-vector/tests/vector_suite/cases/hnsw_checkpoint_encryption.rs b/nodedb-vector/tests/vector_suite/cases/hnsw_checkpoint_encryption.rs index d2070269e..999b759d9 100644 --- a/nodedb-vector/tests/vector_suite/cases/hnsw_checkpoint_encryption.rs +++ b/nodedb-vector/tests/vector_suite/cases/hnsw_checkpoint_encryption.rs @@ -30,7 +30,7 @@ fn make_key() -> WalEncryptionKey { fn make_collection() -> VectorCollection { let mut coll = VectorCollection::new(DIM, params()); for i in 0u32..10 { - coll.insert(vec![i as f32, 0.0, 0.0, 0.0]); + coll.insert(vec![i as f32, 0.0, 0.0, 0.0]).unwrap(); } coll } @@ -72,7 +72,7 @@ fn hnsw_checkpoint_encrypted_at_rest() { assert_eq!(restored.dim(), DIM); // Nearest neighbour to [5.0, 0, 0, 0] must be vector id=5. - let results = restored.search(&[5.0, 0.0, 0.0, 0.0], 1, 64); + let results = restored.search(&[5.0, 0.0, 0.0, 0.0], 1, 64).unwrap(); assert!(!results.is_empty(), "search must return a result"); assert_eq!( results[0].id, 5, diff --git a/nodedb-vector/tests/vector_suite/cases/quantize_kmeans_distribution.rs b/nodedb-vector/tests/vector_suite/cases/quantize_kmeans_distribution.rs index b7cf74218..a5c036c6a 100644 --- a/nodedb-vector/tests/vector_suite/cases/quantize_kmeans_distribution.rs +++ b/nodedb-vector/tests/vector_suite/cases/quantize_kmeans_distribution.rs @@ -64,7 +64,7 @@ fn unique_centroid_count(codec: &PqCodec, vectors: &[Vec]) -> usize { fn pq_kmeans_produces_diverse_centroids_on_duplicate_heavy_data() { let vecs = clustered_with_duplicates(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 20, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 20, test_memory()).unwrap(); let unique = unique_centroid_count(&codec, &vecs); assert!( @@ -83,7 +83,7 @@ fn pq_distance_table_separates_duplicates_from_outliers() { // codebook entries alias to one point so all distances look similar. let vecs = clustered_with_duplicates(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let codec = PqCodec::train(&refs, 4, 2, 16, 20, test_memory()); + let codec = PqCodec::train(&refs, 4, 2, 16, 20, test_memory()).unwrap(); let query = [0.0f32, 0.0, 0.0, 0.0]; let table = codec @@ -121,15 +121,15 @@ fn ivf_pq_training_does_not_collapse_on_duplicate_heavy_data() { metric: DistanceMetric::L2, }, ); - idx.train(&refs, test_memory()); + idx.train(&refs, test_memory()).unwrap(); for v in &vecs { - idx.add(v); + idx.add(v).unwrap(); } // Query at the origin. Correct training assigns near-duplicates to // one cell and outliers to another; the nearest result must come // from the duplicate cluster (original indices 0..190). - let results = idx.search(&[0.0, 0.0, 0.0, 0.0], 5); + let results = idx.search(&[0.0, 0.0, 0.0, 0.0], 5).unwrap(); assert!(!results.is_empty(), "IVF-PQ returned no results"); for r in &results { assert!( diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index ea60089bf..ca0c85a26 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -121,14 +121,11 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str ), crate::Error::RejectedConstraint { constraint, detail, .. - } => { - let code = if constraint == "not_null" { - sqlstate::NOT_NULL_VIOLATION - } else { - sqlstate::UNIQUE_VIOLATION - }; - ("ERROR", code, detail.clone()) - } + } => ( + "ERROR", + crate::control::server::shared::ddl::sqlstate::constraint_sqlstate(constraint), + detail.clone(), + ), crate::Error::TxnOverlayMemoryExceeded { .. } => { ("ERROR", sqlstate::PROGRAM_LIMIT_EXCEEDED, err.to_string()) } diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index 234031df9..7a7e63664 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -6,6 +6,24 @@ use nodedb_types::error::sqlstate; use crate::bridge::envelope::ErrorCode; +/// The SQLSTATE for a rejected constraint of kind `constraint`. +/// +/// `not_null` and `unique` keep their specific codes; `generated_always` +/// (a write to a generated column) is `428C9`. A CRDT delta refusal carries +/// its violation kind: `fk_missing` is `23503`, and `rls_policy` / +/// `permission_denied` are `42501`. Every other kind is the generic +/// integrity class `23000`, never `unique_violation`. +pub fn constraint_sqlstate(constraint: &str) -> &'static str { + match constraint { + "not_null" => sqlstate::NOT_NULL_VIOLATION, + "unique" => sqlstate::UNIQUE_VIOLATION, + "generated_always" => sqlstate::GENERATED_ALWAYS, + "fk_missing" => sqlstate::FOREIGN_KEY_VIOLATION, + "rls_policy" | "permission_denied" => sqlstate::INSUFFICIENT_PRIVILEGE, + _ => sqlstate::INTEGRITY_CONSTRAINT_VIOLATION, + } +} + /// Map a Data Plane `ErrorCode` to SQLSTATE. pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, String) { match code { @@ -15,11 +33,7 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, "query cancelled due to deadline".into(), ), ErrorCode::RejectedConstraint { constraint, detail } => { - let code = if constraint == "not_null" { - sqlstate::NOT_NULL_VIOLATION - } else { - sqlstate::UNIQUE_VIOLATION - }; + let code = constraint_sqlstate(constraint); ( "ERROR", code, @@ -246,3 +260,50 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, ), } } + +#[cfg(test)] +mod tests { + use super::*; + + /// Only a unique-key refusal is `23505`; every other constraint kind + /// keeps its own class. + #[test] + fn constraint_kinds_map_to_their_own_sqlstate() { + assert_eq!(constraint_sqlstate("unique"), sqlstate::UNIQUE_VIOLATION); + assert_eq!( + constraint_sqlstate("not_null"), + sqlstate::NOT_NULL_VIOLATION + ); + assert_eq!( + constraint_sqlstate("generated_always"), + sqlstate::GENERATED_ALWAYS + ); + assert_eq!( + constraint_sqlstate("fk_missing"), + sqlstate::FOREIGN_KEY_VIOLATION + ); + assert_eq!( + constraint_sqlstate("rls_policy"), + sqlstate::INSUFFICIENT_PRIVILEGE + ); + assert_eq!( + constraint_sqlstate("crdt_single_document_delta"), + sqlstate::INTEGRITY_CONSTRAINT_VIOLATION + ); + } + + /// A vector of the wrong width is a data exception, not a constraint. + #[test] + fn vector_dimension_mismatch_is_a_data_exception() { + let code = ErrorCode::DataException { + detail: nodedb_vector::error::VectorError::DimensionMismatch { + expected: 3, + got: 2, + } + .to_string(), + }; + let (_, state, message) = error_code_to_sqlstate(&code); + assert_eq!(state, sqlstate::DATA_EXCEPTION); + assert_eq!(message, "vector dimension mismatch: expected 3, got 2"); + } +} diff --git a/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs b/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs index 02deb33f0..9cf117345 100644 --- a/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs +++ b/nodedb/src/data/executor/handlers/control/checkpoint_durable_lsn.rs @@ -694,7 +694,8 @@ mod tests { use crate::engine::vector::hnsw::HnswParams; let mut coll = VectorCollection::new(4, HnswParams::default()); - coll.insert_with_surrogate(vec![0.1, 0.2, 0.3, 0.4], nodedb_types::Surrogate::new(1)); + coll.insert_with_surrogate(vec![0.1, 0.2, 0.3, 0.4], nodedb_types::Surrogate::new(1)) + .unwrap(); core.vector_collections.insert( ( nodedb_types::DatabaseId::DEFAULT, diff --git a/nodedb/src/data/executor/handlers/generated.rs b/nodedb/src/data/executor/handlers/generated.rs index 27efc70d2..85ef96448 100644 --- a/nodedb/src/data/executor/handlers/generated.rs +++ b/nodedb/src/data/executor/handlers/generated.rs @@ -52,11 +52,11 @@ pub fn check_generated_readonly( for (field, _) in update_fields { if specs.iter().any(|s| s.name == *field) { return Err(ErrorCode::RejectedConstraint { - constraint: format!( + constraint: "generated_always".into(), + detail: format!( "cannot UPDATE generated column '{field}': \ generated columns are computed automatically" ), - detail: String::new(), }); } } @@ -119,9 +119,10 @@ fn topological_sort(specs: &[GeneratedColumnSpec]) -> Result, ErrorCo } if order.len() != n { - return Err(ErrorCode::RejectedConstraint { - constraint: "cycle detected in generated column dependencies".into(), - detail: String::new(), + return Err(ErrorCode::Unsupported { + detail: "generated columns whose expressions depend on each other in a cycle \ + are not supported" + .into(), }); } diff --git a/nodedb/src/data/executor/handlers/graph_rag.rs b/nodedb/src/data/executor/handlers/graph_rag.rs index 6cb68b202..a645c2383 100644 --- a/nodedb/src/data/executor/handlers/graph_rag.rs +++ b/nodedb/src/data/executor/handlers/graph_rag.rs @@ -188,12 +188,18 @@ impl CoreLoop { }; // An empty vector leg is an empty score set, not an empty response: // the other fusion legs still rank, and the envelope keeps its shape. - if index.is_empty() { - return Ok((Vec::new(), HashMap::new(), Vec::new())); - } - + // A query of the wrong width fails first, on an empty index too. let ef = vector_top_k.saturating_mul(4).max(64); - let vector_results = index.search(query_vector, vector_top_k, ef); + let vector_results = match super::vector_search::search_vector_leg( + index, + query_vector, + vector_top_k, + ef, + None, + ) { + Ok(results) => results, + Err(code) => return Err(self.response_error(task, code)), + }; if vector_results.is_empty() { return Ok((Vec::new(), HashMap::new(), Vec::new())); diff --git a/nodedb/src/data/executor/handlers/mod.rs b/nodedb/src/data/executor/handlers/mod.rs index 1b91b8d76..dfdefe48c 100644 --- a/nodedb/src/data/executor/handlers/mod.rs +++ b/nodedb/src/data/executor/handlers/mod.rs @@ -96,6 +96,7 @@ pub mod vector_params; pub mod vector_search; mod vector_search_ann; mod vector_search_exec; +mod vector_search_ivf; pub mod vector_sparse; pub mod vector_upsert; pub mod vector_write; diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs index 900f77912..9b9d338d6 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs @@ -7,6 +7,17 @@ use crate::data::executor::vector_string::floats_from_value; use super::types::{VectorFieldInsert, VectorIndexDelta, VectorIndexPutParams}; +/// A document vector field whose width differs from its index: the +/// caller's data error, SQLSTATE `22000`, naming the field. +fn field_dimension_mismatch(field_name: &str, expected: usize, got: usize) -> crate::Error { + crate::Error::DataException { + detail: format!( + "{field_name}: {}", + nodedb_vector::error::VectorError::DimensionMismatch { expected, got } + ), + } +} + impl CoreLoop { /// HNSW vector indexing side-effect: index declared strict-schema /// `Vector(dim)` columns, or (schemaless) fields matched by registered @@ -65,11 +76,11 @@ impl CoreLoop { Self::vector_index_key(database_id, tid, collection, field_name); self.check_vector_width(&index_key, field_name, floats.len())?; if floats.len() != *dim as usize { - return Err(crate::Error::RejectedConstraint { - collection: collection.to_string(), - constraint: format!("vector dimension on '{field_name}'"), - detail: format!("column declares {dim}, got {}", floats.len()), - }); + return Err(field_dimension_mismatch( + field_name, + *dim as usize, + floats.len(), + )); } let params = self .vector_params @@ -89,15 +100,17 @@ impl CoreLoop { if skip { continue; } - if let Some(delta) = self.remove_then_insert_vector_field(VectorFieldInsert { - database_id, - tid, - index_key, - collection, - field_name, - storage_key, - floats, - }) { + if let Some(delta) = + self.remove_then_insert_vector_field(VectorFieldInsert { + database_id, + tid, + index_key, + collection, + field_name, + storage_key, + floats, + })? + { inserts.push(delta); } } @@ -164,15 +177,17 @@ impl CoreLoop { if skip { continue; } - if let Some(delta) = self.remove_then_insert_vector_field(VectorFieldInsert { - database_id, - tid, - index_key: store_key, - collection, - field_name, - storage_key, - floats, - }) { + if let Some(delta) = + self.remove_then_insert_vector_field(VectorFieldInsert { + database_id, + tid, + index_key: store_key, + collection, + field_name, + storage_key, + floats, + })? + { inserts.push(delta); } } @@ -194,22 +209,16 @@ impl CoreLoop { field_name: &str, got: usize, ) -> crate::Result<()> { - let mismatch = |expected: usize, source: &str| crate::Error::RejectedConstraint { - collection: index_key.2.clone(), - constraint: format!("vector dimension on '{field_name}'"), - detail: format!("index {source} {expected}, got {got}"), - }; - if let Some(&declared) = self.declared_dims.get(index_key) && declared != 0 && declared != got { - return Err(mismatch(declared, "declares")); + return Err(field_dimension_mismatch(field_name, declared, got)); } if let Some(existing) = self.vector_collections.get(index_key) && existing.dim() != got { - return Err(mismatch(existing.dim(), "has")); + return Err(field_dimension_mismatch(field_name, existing.dim(), got)); } Ok(()) } @@ -228,13 +237,14 @@ impl CoreLoop { /// Binds the vector node to the document's global surrogate so /// cross-engine identity holds: a search hit resolves back to this row's /// surrogate (and thus its user PK at the response boundary) instead of - /// leaking a headless local node id. Returns `None` if `index_key`'s + /// leaking a headless local node id. Returns `Ok(None)` if `index_key`'s /// `VectorCollection` was somehow absent (defensive — it was just - /// populated via `entry().or_insert_with()` by the caller). + /// populated via `entry().or_insert_with()` by the caller), and the + /// collection's typed error when the vector does not fit it. fn remove_then_insert_vector_field( &mut self, params: VectorFieldInsert<'_>, - ) -> Option { + ) -> crate::Result> { let VectorFieldInsert { database_id, tid, @@ -251,8 +261,10 @@ impl CoreLoop { field_name, storage_key, ); - let coll = self.vector_collections.get_mut(&index_key)?; - let vector_id = coll.insert_with_surrogate(floats, storage_key.surrogate()); + let Some(coll) = self.vector_collections.get_mut(&index_key) else { + return Ok(None); + }; + let vector_id = coll.insert_with_surrogate(floats, storage_key.surrogate())?; self.vector_doc_map.insert( ( index_key.0, @@ -263,13 +275,13 @@ impl CoreLoop { ), vector_id, ); - Some(VectorIndexDelta { + Ok(Some(VectorIndexDelta { index_key, vector_id, collection: collection.to_string(), field: field_name.to_string(), doc_id: storage_key, - }) + })) } } @@ -549,7 +561,7 @@ mod tests { }); assert!( - matches!(res, Err(crate::Error::RejectedConstraint { .. })), + matches!(res, Err(crate::Error::DataException { .. })), "malformed embedding '{bad}' must reject the put" ); assert_eq!( diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs b/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs index 39a6aa825..3d1c4f2b8 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs @@ -35,6 +35,9 @@ impl CoreLoop { )) } + /// Install the snapshot's vectors into their collection. A vector whose + /// width differs from the collection's fails the restore before any of + /// the collection's vectors lands. pub(super) fn restore_vector_collection( &mut self, database_id: u64, @@ -42,9 +45,9 @@ impl CoreLoop { coll_key: &str, vectors: Vec<(u32, Vec, Option)>, replace_mode: bool, - ) { + ) -> crate::Result<()> { if vectors.is_empty() { - return; + return Ok(()); } let dim = vectors[0].1.len(); let map_key = ( @@ -70,9 +73,12 @@ impl CoreLoop { let coll = self.vector_collections.entry(map_key).or_insert_with(|| { crate::engine::vector::collection::VectorCollection::new(dim, params) }); - for (_, data, surrogate) in vectors { - coll.insert_with_surrogate(data, surrogate.unwrap_or(nodedb_types::Surrogate::ZERO)); - } + let (data, surrogates): (Vec>, Vec) = vectors + .into_iter() + .map(|(_, data, surrogate)| (data, surrogate.unwrap_or(nodedb_types::Surrogate::ZERO))) + .unzip(); + coll.insert_batch_with_surrogates(&data, &surrogates)?; + Ok(()) } pub(super) fn restore_kv_table( diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs b/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs index cf2a28f8a..270db83cd 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs @@ -227,13 +227,15 @@ impl CoreLoop { }; let count = vectors.len() as u64; let (database_id, coll_key) = parse_vector_snapshot_key(key, tenant_id); - self.restore_vector_collection( + if let Err(e) = self.restore_vector_collection( database_id, tenant_id, coll_key, vectors, replace_mode, - ); + ) { + return self.response_error(task, e); + } vectors_written += count; } diff --git a/nodedb/src/data/executor/handlers/text_search_hybrid.rs b/nodedb/src/data/executor/handlers/text_search_hybrid.rs index b73fe1d9c..c046f895b 100644 --- a/nodedb/src/data/executor/handlers/text_search_hybrid.rs +++ b/nodedb/src/data/executor/handlers/text_search_hybrid.rs @@ -87,29 +87,25 @@ impl CoreLoop { let index_key = CoreLoop::vector_index_key(task.request.database_id.as_u64(), tid, collection, ""); let vector_collection = self.vector_collections.get(&index_key); - let vector_results = if let Some(index) = vector_collection { - if index.is_empty() { - Vec::new() - } else { + let vector_results = match vector_collection { + Some(index) => { let ef = if ef_search > 0 { ef_search.max(fetch_k) } else { fetch_k.saturating_mul(4).max(64) }; - match filter_bitmap { - Some(surrogate_bm) => { - let mut buf = Vec::with_capacity(surrogate_bm.0.serialized_size()); - if surrogate_bm.0.serialize_into(&mut buf).is_ok() { - index.search_with_bitmap_bytes(query_vector, fetch_k, ef, &buf) - } else { - index.search(query_vector, fetch_k, ef) - } - } - None => index.search(query_vector, fetch_k, ef), + match super::vector_search::search_vector_leg( + index, + query_vector, + fetch_k, + ef, + filter_bitmap, + ) { + Ok(results) => results, + Err(code) => return self.response_error(task, code), } } - } else { - Vec::new() + None => Vec::new(), }; // 2. Text search (no surrogate prefilter for the text leg of hybrid search). diff --git a/nodedb/src/data/executor/handlers/text_search_triple.rs b/nodedb/src/data/executor/handlers/text_search_triple.rs index 3cfbc73fb..600f00da2 100644 --- a/nodedb/src/data/executor/handlers/text_search_triple.rs +++ b/nodedb/src/data/executor/handlers/text_search_triple.rs @@ -94,29 +94,25 @@ impl CoreLoop { let index_key = CoreLoop::vector_index_key(task.request.database_id.as_u64(), tid, collection, ""); let vector_collection = self.vector_collections.get(&index_key); - let vector_results = if let Some(index) = vector_collection { - if index.is_empty() { - Vec::new() - } else { + let vector_results = match vector_collection { + Some(index) => { let ef = if ef_search > 0 { ef_search.max(fetch_k) } else { fetch_k.saturating_mul(4).max(64) }; - match filter_bitmap { - Some(surrogate_bm) => { - let mut buf = Vec::with_capacity(surrogate_bm.0.serialized_size()); - if surrogate_bm.0.serialize_into(&mut buf).is_ok() { - index.search_with_bitmap_bytes(query_vector, fetch_k, ef, &buf) - } else { - index.search(query_vector, fetch_k, ef) - } - } - None => index.search(query_vector, fetch_k, ef), + match super::vector_search::search_vector_leg( + index, + query_vector, + fetch_k, + ef, + filter_bitmap, + ) { + Ok(results) => results, + Err(code) => return self.response_error(task, code), } } - } else { - Vec::new() + None => Vec::new(), }; // 2. BM25 text search. diff --git a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs index 6dc7957c0..8ad69cde2 100644 --- a/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs +++ b/nodedb/src/data/executor/handlers/transaction/resolve/entry.rs @@ -2673,7 +2673,7 @@ mod tests { // Seed a base index with one vector. let key = CoreLoop::vector_index_key(DatabaseId::DEFAULT.as_u64(), TID, "emb", ""); let mut coll = VectorCollection::new(3, HnswParams::default()); - coll.insert(vec![7.0, 7.0, 7.0]); + coll.insert(vec![7.0, 7.0, 7.0]).unwrap(); core.vector_collections.insert(key.clone(), coll); let plan = PhysicalPlan::Vector(VectorOp::Insert { diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_vector.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_vector.rs index f9c53bcdf..0e426fcd7 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_vector.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_vector.rs @@ -277,13 +277,9 @@ impl CoreLoop { if let Some(dim) = declared_dim && dim != spec.dim { - return Err(ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "vector dimension mismatch: collection declares {dim}, got {}", - spec.dim - ), - }); + return Err(crate::data::executor::handlers::vector::dimension_mismatch( + dim, spec.dim, + )); } Ok(CoreLoop::vector_index_key( ctx.database_id, diff --git a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs index a3a71eda4..e84bd5b30 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/apply.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/apply.rs @@ -774,7 +774,9 @@ mod tests { .vector_collections .entry(index_key.clone()) .or_insert_with(|| nodedb_vector::VectorCollection::new(2, Default::default())); - let vector_id = coll.insert_with_surrogate(vec![1.0, 2.0], nodedb_types::Surrogate::ZERO); + let vector_id = coll + .insert_with_surrogate(vec![1.0, 2.0], nodedb_types::Surrogate::ZERO) + .unwrap(); // Seed as though the forward `apply_point_put_vector_indexes` insert had // run: it populates `vector_doc_map` alongside the HNSW insert. @@ -813,7 +815,9 @@ mod tests { .vector_collections .entry(index_key.clone()) .or_insert_with(|| nodedb_vector::VectorCollection::new(2, Default::default())); - let vector_id = coll.insert_with_surrogate(vec![3.0, 4.0], nodedb_types::Surrogate::ZERO); + let vector_id = coll + .insert_with_surrogate(vec![3.0, 4.0], nodedb_types::Surrogate::ZERO) + .unwrap(); coll.delete(vector_id); // The forward delete cascade already removed the reverse-map entry (as diff --git a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs index 17cffe40d..97e632e5a 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/rollback.rs @@ -422,7 +422,7 @@ mod tests { fn vector_searchable(core: &Core) -> bool { core.vector_collections .get(&vector_key()) - .map(|coll| !coll.search(&[1.0, 2.0, 3.0], 1, 16).is_empty()) + .map(|coll| !coll.search(&[1.0, 2.0, 3.0], 1, 16).unwrap().is_empty()) .unwrap_or(false) } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs b/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs index 058b3e58b..80f115147 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs @@ -258,16 +258,18 @@ mod tests { let vecs = vectors(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); let mut index = IvfPqIndex::new(4, params()); - index.train( - &refs, - nodedb_mem::ScopedMemory::new( - crate::data::executor::core_loop::test_governor(), - DatabaseId::DEFAULT, - TenantId::new(TID), - nodedb_mem::EngineId::Vector, - ), - ); - index.add_batch(&refs[..4]); + index + .train( + &refs, + nodedb_mem::ScopedMemory::new( + crate::data::executor::core_loop::test_governor(), + DatabaseId::DEFAULT, + TenantId::new(TID), + nodedb_mem::EngineId::Vector, + ), + ) + .unwrap(); + index.add_batch(&refs[..4]).unwrap(); core.ivf_indexes.insert(key.clone(), index); let undo = core @@ -281,7 +283,7 @@ mod tests { }) .expect("capture undo"); if let Some(index) = core.ivf_indexes.get_mut(&key) { - index.add_batch(&refs[4..]); + index.add_batch(&refs[4..]).unwrap(); } let UndoEntry::VectorWrite(undo) = undo else { panic!("a vector write captures a VectorWrite undo"); @@ -293,7 +295,7 @@ mod tests { assert_eq!(index.len(), 4); assert!(index.is_trained()); assert!( - index.search(&vecs[6], 8).iter().all(|r| r.id < 4), + index.search(&vecs[6], 8).unwrap().iter().all(|r| r.id < 4), "no vector the write added is found" ); } diff --git a/nodedb/src/data/executor/handlers/vector.rs b/nodedb/src/data/executor/handlers/vector.rs index 528cca889..f94361ce3 100644 --- a/nodedb/src/data/executor/handlers/vector.rs +++ b/nodedb/src/data/executor/handlers/vector.rs @@ -59,6 +59,14 @@ pub(in crate::data::executor) struct VectorInsertInner<'a> { pub surrogate: Surrogate, } +/// A vector whose dimension differs from the index's: the caller's data +/// error, SQLSTATE `22000`, in the vector engine's message shape. +pub(in crate::data::executor) fn dimension_mismatch(expected: usize, got: usize) -> ErrorCode { + ErrorCode::DataException { + detail: nodedb_vector::error::VectorError::DimensionMismatch { expected, got }.to_string(), + } +} + impl CoreLoop { /// Get or create a vector collection, validating dimension compatibility. pub(in crate::data::executor) fn get_or_create_vector_index( @@ -79,22 +87,13 @@ impl CoreLoop { && declared != 0 && declared != dim { - return Err(ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!("dimension mismatch: index declares {declared}, got {dim}"), - }); + return Err(dimension_mismatch(declared, dim)); } if let Some(existing) = self.vector_collections.get(&index_key) && existing.dim() != dim { - return Err(ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "dimension mismatch: index has {}, got {dim}", - existing.dim() - ), - }); + return Err(dimension_mismatch(existing.dim(), dim)); } let core_id = self.core_id; let params = self @@ -196,16 +195,7 @@ impl CoreLoop { surrogate, } = args; if vector.len() != dim { - return self.response_error( - task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "vector dimension mismatch: expected {dim}, got {}", - vector.len() - ), - }, - ); + return self.response_error(task, dimension_mismatch(dim, vector.len())); } let database_id = task.request.database_id.as_u64(); let index_key = CoreLoop::vector_index_key(database_id, tid, collection, field_name); @@ -224,7 +214,9 @@ impl CoreLoop { let defer_seal = self.recording_redo_undo(); match self.get_or_create_vector_index(database_id, tid, collection, dim, field_name) { Ok(collection_ref) => { - collection_ref.insert_with_surrogate(vector.to_vec(), surrogate); + if let Err(e) = collection_ref.insert_with_surrogate(vector.to_vec(), surrogate) { + return self.response_error(task, crate::Error::from(e)); + } let seal_key = CoreLoop::vector_build_key(&index_key); if !defer_seal && collection_ref.needs_seal() @@ -287,10 +279,15 @@ impl CoreLoop { index_key.1, nodedb_mem::EngineId::Vector, ); - ivf.train(&refs, memory); + if let Err(e) = ivf.train(&refs, memory) { + return self.response_error(task, crate::Error::from(e)); + } } - let vector_id = ivf.add(vector); + let vector_id = match ivf.add(vector) { + Ok(id) => id, + Err(e) => return self.response_error(task, crate::Error::from(e)), + }; // Register surrogate mapping using the actual IVF-assigned vector ID. if surrogate != Surrogate::ZERO { diff --git a/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs b/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs index 0334166d6..94013765e 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_resolve/apply.rs @@ -135,23 +135,17 @@ impl CoreLoop { && declared != 0 && declared != vector.len() { - return Err(ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "dimension mismatch: index declares {declared}, got {}", - vector.len() - ), - }); + return Err(super::super::vector::dimension_mismatch( + declared, + vector.len(), + )); } match width { Some(first) if first != vector.len() => { - return Err(ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "vector dimension mismatch: the write's first vector has {first}, got {}", - vector.len() - ), - }); + return Err(super::super::vector::dimension_mismatch( + first, + vector.len(), + )); } Some(_) => {} None => width = Some(vector.len()), @@ -710,7 +704,7 @@ mod tests { assert!( matches!( resp.error_code.as_deref(), - Some(ErrorCode::RejectedConstraint { .. }) + Some(ErrorCode::DataException { .. }) ), "got {:?}", resp.error_code diff --git a/nodedb/src/data/executor/handlers/vector_direct_row.rs b/nodedb/src/data/executor/handlers/vector_direct_row.rs index a1421222c..a11c12e92 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_row.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_row.rs @@ -76,20 +76,12 @@ impl CoreLoop { return Ok(None); }; if existing.dim() != spec.dim { - return Err(ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "vector dimension mismatch: index has {}, got {}", - existing.dim(), - spec.dim - ), - }); + return Err(super::vector::dimension_mismatch(existing.dim(), spec.dim)); } let existing_dtype = existing.params().dtype; if existing_dtype != spec.storage_dtype { - return Err(ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( + return Err(ErrorCode::DataException { + detail: format!( "vector storage_dtype mismatch: index has {existing_dtype}, got {}; \ dtype is immutable after collection creation", spec.storage_dtype @@ -252,7 +244,9 @@ impl CoreLoop { detail: format!("vector index for '{collection}' vanished during a direct write"), }); }; - let node_id = coll.insert_with_surrogate(vector.to_vec(), surrogate); + let node_id = coll + .insert_with_surrogate(vector.to_vec(), surrogate) + .map_err(|e| ErrorCode::from(crate::Error::from(e)))?; coll.payload.insert_row(node_id, fields); let key = StorageKey::for_surrogate(surrogate); diff --git a/nodedb/src/data/executor/handlers/vector_multi.rs b/nodedb/src/data/executor/handlers/vector_multi.rs index 8c9f9cfa6..c386f9dee 100644 --- a/nodedb/src/data/executor/handlers/vector_multi.rs +++ b/nodedb/src/data/executor/handlers/vector_multi.rs @@ -64,19 +64,17 @@ impl CoreLoop { if count == 0 || dim == 0 { return self.response_error( task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: "multi-vector count and dim must be > 0".into(), + ErrorCode::DataException { + detail: "multi-vector count and dim must be > 0".into(), }, ); } if vectors_flat.len() != count * dim { return self.response_error( task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "data length mismatch: expected {} ({}×{}), got {}", + ErrorCode::DataException { + detail: format!( + "multi-vector data length mismatch: expected {} ({}×{}), got {}", count * dim, count, dim, @@ -93,16 +91,8 @@ impl CoreLoop { if let Some(existing) = self.vector_collections.get(&index_key) && existing.dim() != dim { - return self.response_error( - task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "dimension mismatch: index has {}, got {dim}", - existing.dim() - ), - }, - ); + return self + .response_error(task, super::vector::dimension_mismatch(existing.dim(), dim)); } // Get or create the vector collection. @@ -134,7 +124,10 @@ impl CoreLoop { coll.delete_multi_vector(document_surrogate); // Insert all vectors with shared surrogate. - let ids = coll.insert_multi_vector(&vector_slices, document_surrogate); + let ids = match coll.insert_multi_vector(&vector_slices, document_surrogate) { + Ok(ids) => ids, + Err(e) => return self.response_error(task, crate::Error::from(e)), + }; // Auto-seal if needed. let seal_key = CoreLoop::vector_build_key(&index_key); @@ -230,9 +223,8 @@ impl CoreLoop { None => { return self.response_error( task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( + ErrorCode::DataException { + detail: format!( "unknown score mode '{mode_str}'; supported: max_sim, avg_sim, sum_sim" ), }, @@ -260,7 +252,10 @@ impl CoreLoop { over_fetch.saturating_mul(2).max(64) }; - let candidates = coll.search(query_vector, over_fetch, ef); + let candidates = match coll.search(query_vector, over_fetch, ef) { + Ok(candidates) => candidates, + Err(e) => return self.response_error(task, crate::Error::from(e)), + }; // Group by surrogate. For distance metrics where lower = better // (L2, cosine) we convert similarity = 1 / (1 + distance) so diff --git a/nodedb/src/data/executor/handlers/vector_multi_search_exec.rs b/nodedb/src/data/executor/handlers/vector_multi_search_exec.rs index d24a850a6..32eea63b0 100644 --- a/nodedb/src/data/executor/handlers/vector_multi_search_exec.rs +++ b/nodedb/src/data/executor/handlers/vector_multi_search_exec.rs @@ -12,7 +12,6 @@ use tracing::debug; use super::vector_search::{ VectorMultiSearchParams, build_search_hit, effective_ef, encode_hits_response, - surrogate_bitmap_to_global_ids, }; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; @@ -52,33 +51,46 @@ impl CoreLoop { }; let mut all_results: Vec> = Vec::new(); + // The width of a field index the query could not be compared with. + // Fields of other widths are skipped; when no field has the query's + // width, the query is the caller's data error. + let mut other_width: Option = None; + let mut any_field_of_width = false; for (key, coll) in &self.vector_collections { if key.0 != db || key.1 != tenant_id { continue; } if key == &plain_key || key.2.starts_with(&field_prefix) { - if coll.is_empty() || coll.dim() != query_vector.len() { + if coll.dim() != query_vector.len() { + other_width = Some(coll.dim()); + continue; + } + any_field_of_width = true; + if coll.is_empty() { continue; } let ef = effective_ef(ef_search, fetch_k); - let results = match filter_bitmap { - Some(surrogate_bm) => { - let local_bm = surrogate_bitmap_to_global_ids(coll, surrogate_bm); - let mut buf = Vec::with_capacity(local_bm.serialized_size()); - if local_bm.serialize_into(&mut buf).is_ok() { - coll.search_with_bitmap_bytes(query_vector, fetch_k, ef, &buf) - } else { - coll.search(query_vector, fetch_k, ef) - } - } - None => coll.search(query_vector, fetch_k, ef), - }; - all_results.push(results); + match super::vector_search::search_vector_leg( + coll, + query_vector, + fetch_k, + ef, + filter_bitmap, + ) { + Ok(results) => all_results.push(results), + Err(code) => return self.response_error(task, code), + } } } if all_results.is_empty() { + if !any_field_of_width && let Some(width) = other_width { + return self.response_error( + task, + super::vector::dimension_mismatch(width, query_vector.len()), + ); + } return self.response_error(task, ErrorCode::NotFound); } diff --git a/nodedb/src/data/executor/handlers/vector_params.rs b/nodedb/src/data/executor/handlers/vector_params.rs index 47d75362e..31965b24d 100644 --- a/nodedb/src/data/executor/handlers/vector_params.rs +++ b/nodedb/src/data/executor/handlers/vector_params.rs @@ -38,9 +38,10 @@ impl CoreLoop { if self.vector_collections.contains_key(&index_key) { return self.response_error( task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: "cannot change index params after creation; drop and recreate the collection".into(), + ErrorCode::Unsupported { + detail: "changing vector index params after the index holds vectors is not \ + supported; drop and recreate the collection" + .into(), }, ); } @@ -84,9 +85,8 @@ impl CoreLoop { _ => { return self.response_error( task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( + ErrorCode::DataException { + detail: format!( "unknown metric '{resolved_metric_str}'; supported: l2, cosine, inner_product, manhattan, chebyshev, hamming, jaccard, pearson" ), }, @@ -105,9 +105,8 @@ impl CoreLoop { None => { return self.response_error( task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( + ErrorCode::DataException { + detail: format!( "unknown index_type '{index_type}'; supported: hnsw, hnsw_pq, ivf_pq" ), }, diff --git a/nodedb/src/data/executor/handlers/vector_search.rs b/nodedb/src/data/executor/handlers/vector_search.rs index 5f71da3b4..a4bf6fa9a 100644 --- a/nodedb/src/data/executor/handlers/vector_search.rs +++ b/nodedb/src/data/executor/handlers/vector_search.rs @@ -75,6 +75,45 @@ pub(super) fn surrogate_bitmap_to_global_ids( local_bm } +/// Search one vector leg of a fused query (hybrid, triple): the whole index, +/// or only the rows whose surrogates are in `filter_bitmap`. +/// +/// The surrogate filter is translated to the index's node ids before the +/// search. A query of the wrong width is the caller's data error (`22000`) +/// even on an empty index. A filter that cannot be serialized fails the +/// search: searching without it would return rows it excludes. +pub(super) fn search_vector_leg( + index: &VectorCollection, + query_vector: &[f32], + fetch_k: usize, + ef: usize, + filter_bitmap: Option<&nodedb_types::SurrogateBitmap>, +) -> Result, ErrorCode> { + if index.dim() != query_vector.len() { + return Err(super::vector::dimension_mismatch( + index.dim(), + query_vector.len(), + )); + } + if index.is_empty() { + return Ok(Vec::new()); + } + let searched = match filter_bitmap { + Some(surrogate_bm) => { + let local_bm = surrogate_bitmap_to_global_ids(index, surrogate_bm); + let mut buf = Vec::with_capacity(local_bm.serialized_size()); + local_bm + .serialize_into(&mut buf) + .map_err(|e| ErrorCode::Internal { + detail: format!("vector search filter bitmap serialization: {e}"), + })?; + index.search_with_bitmap_bytes(query_vector, fetch_k, ef, &buf) + } + None => index.search(query_vector, fetch_k, ef), + }; + searched.map_err(|e| ErrorCode::from(crate::Error::from(e))) +} + /// Encode search hits and return response. pub(super) fn encode_hits_response( core: &CoreLoop, @@ -174,7 +213,8 @@ mod tests { let mut coll = VectorCollection::new(1, HnswParams::default()); for i in 0..n { let surrogate = Surrogate(i as u32 + 1); - coll.insert_with_surrogate(vec![i as f32], surrogate); + coll.insert_with_surrogate(vec![i as f32], surrogate) + .unwrap(); } coll } @@ -256,7 +296,7 @@ mod tests { local_bm.serialize_into(&mut buf).unwrap(); // Search for nearest neighbours — all results must be even surrogates. - let results = coll.search_with_bitmap_bytes(&[10.0], 5, 64, &buf); + let results = coll.search_with_bitmap_bytes(&[10.0], 5, 64, &buf).unwrap(); assert!(!results.is_empty(), "expected at least one result"); for r in &results { @@ -307,7 +347,7 @@ mod tests { let mut buf = Vec::new(); local_bm.serialize_into(&mut buf).unwrap(); - let results = coll.search_with_bitmap_bytes(&[5.0], 5, 64, &buf); + let results = coll.search_with_bitmap_bytes(&[5.0], 5, 64, &buf).unwrap(); assert!(results.is_empty(), "empty bitmap should yield no results"); } } diff --git a/nodedb/src/data/executor/handlers/vector_search_exec.rs b/nodedb/src/data/executor/handlers/vector_search_exec.rs index f37291d64..04bb2b62d 100644 --- a/nodedb/src/data/executor/handlers/vector_search_exec.rs +++ b/nodedb/src/data/executor/handlers/vector_search_exec.rs @@ -10,23 +10,11 @@ use super::vector_search::{ surrogate_bitmap_to_global_ids, }; use super::vector_search_ann::{ResolvedAnnOptions, apply_ann_options, quantization_matches}; +use super::vector_search_ivf::SearchIvfParams; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; -/// Parameters for [`CoreLoop::search_ivf`]. -struct SearchIvfParams<'a> { - task: &'a ExecutionTask, - tid: u64, - collection: &'a str, - index_key: &'a (nodedb_types::DatabaseId, crate::types::TenantId, String), - ivf: &'a crate::engine::vector::ivf::IvfPqIndex, - query_vector: &'a [f32], - top_k: usize, - filter_bitmap: Option<&'a nodedb_types::SurrogateBitmap>, - rls_filters: &'a [u8], -} - impl CoreLoop { /// Fetch the document body via the sparse engine (keyed by /// surrogate-hex) and attach it to the hit. Used both by the RLS path @@ -200,6 +188,14 @@ impl CoreLoop { } return self.response_error(task, ErrorCode::NotFound); }; + // The index width is fixed even while it holds no vector, so a + // query of another width fails here too. + if collection_ref.dim() != query_vector.len() { + return self.response_error( + task, + super::vector::dimension_mismatch(collection_ref.dim(), query_vector.len()), + ); + } if collection_ref.is_empty() { if let Some(txn_id) = task.request.txn_id { return staged_only(self, txn_id); @@ -289,23 +285,35 @@ impl CoreLoop { (None, None) => None, }; - let results = match combined_bm { + // A filter that cannot be serialized fails the search: searching + // without it would return rows the filter excludes. + let searched = match combined_bm { Some(local_bm) => { let mut buf = Vec::with_capacity(local_bm.serialized_size()); - if local_bm.serialize_into(&mut buf).is_ok() { - collection_ref.search_with_bitmap_bytes_and_metric( - query_vector, - fetch_k, - ef, - &buf, - metric, - ) - } else { - collection_ref.search_with_metric(query_vector, fetch_k, ef, metric) + if let Err(e) = local_bm.serialize_into(&mut buf) { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("vector search filter bitmap serialization: {e}"), + }, + ); } + collection_ref.search_with_bitmap_bytes_and_metric( + query_vector, + fetch_k, + ef, + &buf, + metric, + ) } None => collection_ref.search_with_metric(query_vector, fetch_k, ef, metric), }; + // A query of the wrong dimension is the caller's data error (22000); + // the core keeps serving. + let results = match searched { + Ok(results) => results, + Err(e) => return self.response_error(task, crate::Error::from(e)), + }; // Pure-vector fast path: projection contains only id/distance. // Skip the sparse-store body fetch entirely. @@ -422,61 +430,4 @@ impl CoreLoop { } encode_hits_response(self, task, &hits) } - - /// Search an IVF-PQ index with optional bitmap post-filtering. - fn search_ivf(&self, params: SearchIvfParams<'_>) -> Response { - let SearchIvfParams { - task, - tid, - collection, - index_key, - ivf, - query_vector, - top_k, - filter_bitmap, - rls_filters, - } = params; - if ivf.is_empty() { - return super::vector_search::empty_hits_response(self, task); - } - let fetch_k = if filter_bitmap.is_some() || !rls_filters.is_empty() { - top_k * self.query_tuning.bitmap_over_fetch_factor.max(2) - } else { - top_k - }; - let results = ivf.search(query_vector, fetch_k); - let surrogate_source = self.vector_collections.get(index_key); - - let mut hits: Vec<_> = results - .iter() - .map(|r| build_search_hit(surrogate_source, r.id, r.distance)) - .collect(); - - if let Some(surrogate_bm) = filter_bitmap { - // Bitmap is a set of surrogates: keep only bound hits whose - // surrogate is in the bitmap. A headless hit has none, so it - // never survives a surrogate-bitmap filter. - hits.retain(|h| { - h.id.storage_key() - .is_some_and(|key| surrogate_bm.contains(key.surrogate())) - }); - } - if !rls_filters.is_empty() { - // CP-side translator runs the predicate; DP only attaches body. - hits = hits - .into_iter() - .map(|h| { - self.attach_body(task.request.database_id.as_u64(), tid, collection, true, h) - }) - .collect(); - } else { - hits.truncate(top_k); - } - - if let Some(ref m) = self.metrics { - m.record_vector_search(0); - m.record_query_by_engine("vector"); - } - encode_hits_response(self, task, &hits) - } } diff --git a/nodedb/src/data/executor/handlers/vector_search_ivf.rs b/nodedb/src/data/executor/handlers/vector_search_ivf.rs new file mode 100644 index 000000000..f1701f062 --- /dev/null +++ b/nodedb/src/data/executor/handlers/vector_search_ivf.rs @@ -0,0 +1,89 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! IVF-PQ search for `CoreLoop::execute_vector_search`. + +use super::vector_search::{build_search_hit, encode_hits_response}; +use crate::bridge::envelope::Response; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; + +/// Parameters for [`CoreLoop::search_ivf`]. +pub(super) struct SearchIvfParams<'a> { + pub task: &'a ExecutionTask, + pub tid: u64, + pub collection: &'a str, + pub index_key: &'a (nodedb_types::DatabaseId, crate::types::TenantId, String), + pub ivf: &'a crate::engine::vector::ivf::IvfPqIndex, + pub query_vector: &'a [f32], + pub top_k: usize, + pub filter_bitmap: Option<&'a nodedb_types::SurrogateBitmap>, + pub rls_filters: &'a [u8], +} + +impl CoreLoop { + /// Search an IVF-PQ index with optional bitmap post-filtering. + pub(super) fn search_ivf(&self, params: SearchIvfParams<'_>) -> Response { + let SearchIvfParams { + task, + tid, + collection, + index_key, + ivf, + query_vector, + top_k, + filter_bitmap, + rls_filters, + } = params; + if ivf.dim() != query_vector.len() { + return self.response_error( + task, + super::vector::dimension_mismatch(ivf.dim(), query_vector.len()), + ); + } + if ivf.is_empty() { + return super::vector_search::empty_hits_response(self, task); + } + let fetch_k = if filter_bitmap.is_some() || !rls_filters.is_empty() { + top_k * self.query_tuning.bitmap_over_fetch_factor.max(2) + } else { + top_k + }; + let results = match ivf.search(query_vector, fetch_k) { + Ok(results) => results, + Err(e) => return self.response_error(task, crate::Error::from(e)), + }; + let surrogate_source = self.vector_collections.get(index_key); + + let mut hits: Vec<_> = results + .iter() + .map(|r| build_search_hit(surrogate_source, r.id, r.distance)) + .collect(); + + if let Some(surrogate_bm) = filter_bitmap { + // Bitmap is a set of surrogates: keep only bound hits whose + // surrogate is in the bitmap. A headless hit has none, so it + // never survives a surrogate-bitmap filter. + hits.retain(|h| { + h.id.storage_key() + .is_some_and(|key| surrogate_bm.contains(key.surrogate())) + }); + } + if !rls_filters.is_empty() { + // CP-side translator runs the predicate; DP only attaches body. + hits = hits + .into_iter() + .map(|h| { + self.attach_body(task.request.database_id.as_u64(), tid, collection, true, h) + }) + .collect(); + } else { + hits.truncate(top_k); + } + + if let Some(ref m) = self.metrics { + m.record_vector_search(0); + m.record_query_by_engine("vector"); + } + encode_hits_response(self, task, &hits) + } +} diff --git a/nodedb/src/data/executor/handlers/vector_sparse.rs b/nodedb/src/data/executor/handlers/vector_sparse.rs index d85d7a1ff..d42753b00 100644 --- a/nodedb/src/data/executor/handlers/vector_sparse.rs +++ b/nodedb/src/data/executor/handlers/vector_sparse.rs @@ -73,9 +73,8 @@ impl CoreLoop { Err(e) => { return self.response_error( task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: e.to_string(), + ErrorCode::DataException { + detail: e.to_string(), }, ); } @@ -133,9 +132,8 @@ impl CoreLoop { Err(e) => { return self.response_error( task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: e.to_string(), + ErrorCode::DataException { + detail: e.to_string(), }, ); } diff --git a/nodedb/src/data/executor/handlers/vector_write.rs b/nodedb/src/data/executor/handlers/vector_write.rs index 8dd929008..3257f84ed 100644 --- a/nodedb/src/data/executor/handlers/vector_write.rs +++ b/nodedb/src/data/executor/handlers/vector_write.rs @@ -30,25 +30,15 @@ impl CoreLoop { // Every vector is checked before any is inserted, so a dimension // refusal applies nothing. if let Some(bad) = vectors.iter().find(|vector| vector.len() != dim) { - return self.response_error( - task, - ErrorCode::RejectedConstraint { - detail: String::new(), - constraint: format!( - "dimension mismatch in batch: expected {dim}, got {}", - bad.len() - ), - }, - ); + return self.response_error(task, super::vector::dimension_mismatch(dim, bad.len())); } let index_key = CoreLoop::vector_index_key(database_id, tid, collection, ""); // A committed-redo install seals once the whole record landed. let defer_seal = self.recording_redo_undo(); match self.get_or_create_vector_index(database_id, tid, collection, dim, "") { Ok(collection_ref) => { - for (i, vector) in vectors.iter().enumerate() { - let s = surrogates.get(i).copied().unwrap_or(Surrogate::ZERO); - collection_ref.insert_with_surrogate(vector.clone(), s); + if let Err(e) = collection_ref.insert_batch_with_surrogates(vectors, surrogates) { + return self.response_error(task, crate::Error::from(e)); } let seal_key = CoreLoop::vector_build_key(&index_key); if !defer_seal @@ -287,6 +277,7 @@ mod tests { .get_or_create_vector_index(0, 1, "docs", 2, "") .expect("create index"); coll.insert_with_surrogate(vec![1.0, 2.0], surrogate) + .unwrap() }; // The internal node id must differ from the surrogate for this test // to actually distinguish the two key spaces. diff --git a/nodedb/src/data/executor/vector_checkpoint/load.rs b/nodedb/src/data/executor/vector_checkpoint/load.rs index 58cc4c97f..d43a75ea6 100644 --- a/nodedb/src/data/executor/vector_checkpoint/load.rs +++ b/nodedb/src/data/executor/vector_checkpoint/load.rs @@ -250,12 +250,12 @@ mod tests { .applied_prefix .observe_outcome_floor(Lsn::new(10)); let mut collection = VectorCollection::new(3, HnswParams::default()); - collection.insert(vec![0.0, 0.0, 1.0]); + collection.insert(vec![0.0, 0.0, 1.0]).unwrap(); core.vector_collections.insert(key.clone(), collection); core.floors.applied_prefix.note_applied(Lsn::new(30)); core.checkpoint_vector_indexes().expect("checkpoint"); if let Some(collection) = core.vector_collections.get_mut(&key) { - collection.insert(vec![1.0, 0.0, 0.0]); + collection.insert(vec![1.0, 0.0, 0.0]).unwrap(); } core.floors.applied_prefix.note_applied(Lsn::new(20)); core.vector_collections.get(&key).map(|c| c.len()) diff --git a/nodedb/src/data/executor/vector_checkpoint/write.rs b/nodedb/src/data/executor/vector_checkpoint/write.rs index 8f1233571..e6917b884 100644 --- a/nodedb/src/data/executor/vector_checkpoint/write.rs +++ b/nodedb/src/data/executor/vector_checkpoint/write.rs @@ -166,7 +166,8 @@ mod tests { fn collection_with_one_vector() -> VectorCollection { let mut coll = VectorCollection::new(4, HnswParams::default()); - coll.insert_with_surrogate(vec![0.1, 0.2, 0.3, 0.4], Surrogate::new(1)); + coll.insert_with_surrogate(vec![0.1, 0.2, 0.3, 0.4], Surrogate::new(1)) + .unwrap(); coll } diff --git a/nodedb/src/data/executor/vector_string.rs b/nodedb/src/data/executor/vector_string.rs index 39b437974..8c2186a29 100644 --- a/nodedb/src/data/executor/vector_string.rs +++ b/nodedb/src/data/executor/vector_string.rs @@ -82,11 +82,11 @@ fn parse_float_list(inner: &str) -> Option> { Some(floats) } +/// A field value that is not a usable vector: the caller's data error, +/// SQLSTATE `22000`, naming the collection and field. fn vector_error(collection: &str, field_name: &str, detail: String) -> crate::Error { - crate::Error::RejectedConstraint { - collection: collection.to_string(), - constraint: format!("vector field '{field_name}'"), - detail, + crate::Error::DataException { + detail: format!("vector field '{field_name}' in '{collection}': {detail}"), } } @@ -272,7 +272,7 @@ mod tests { for bad in ["not-json", "7", "[0.1, \"bad\"]", "{}", "[nan]", "[1 2 3]"] { let res = floats_from_value("c", "embedding", &String(bad.to_string())); assert!( - matches!(res, Err(crate::Error::RejectedConstraint { .. })), + matches!(res, Err(crate::Error::DataException { .. })), "String({bad:?}) must be rejected, got {res:?}" ); } @@ -282,7 +282,7 @@ mod tests { &Array(vec![Float(0.1), String("bad".to_string())]), ); assert!( - matches!(res, Err(crate::Error::RejectedConstraint { .. })), + matches!(res, Err(crate::Error::DataException { .. })), "array with a non-numeric element must be rejected" ); } @@ -293,13 +293,13 @@ mod tests { for empty in ["", "[]", "[,,]", "[ ]"] { let res = floats_from_value("c", "embedding", &String(empty.to_string())); assert!( - matches!(res, Err(crate::Error::RejectedConstraint { .. })), + matches!(res, Err(crate::Error::DataException { .. })), "String({empty:?}) must be rejected as an empty vector, got {res:?}" ); } let res = floats_from_value("c", "embedding", &Array(vec![])); assert!( - matches!(res, Err(crate::Error::RejectedConstraint { .. })), + matches!(res, Err(crate::Error::DataException { .. })), "an empty array must be rejected as an empty vector" ); } @@ -312,19 +312,17 @@ mod tests { &nodedb_types::Value::String("not-json".to_string()), ); match res { - Err(crate::Error::RejectedConstraint { - collection, - constraint, - detail, - }) => { - assert_eq!(collection, "docs", "error must name the collection"); + Err(crate::Error::DataException { detail }) => { assert!( - constraint.contains("embedding"), - "error must name the field, got {constraint:?}" + detail.contains("'docs'"), + "error must name the collection: {detail}" + ); + assert!( + detail.contains("'embedding'"), + "error must name the field: {detail}" ); - assert!(!detail.is_empty(), "error must carry the offending input"); } - other => panic!("expected RejectedConstraint, got {other:?}"), + other => panic!("expected DataException, got {other:?}"), } } } diff --git a/nodedb/src/data/executor/wal_replay_vector.rs b/nodedb/src/data/executor/wal_replay_vector.rs index bcd168f68..43e2088a4 100644 --- a/nodedb/src/data/executor/wal_replay_vector.rs +++ b/nodedb/src/data/executor/wal_replay_vector.rs @@ -300,7 +300,17 @@ impl CoreLoop { // `SurrogateBind` replay path. Engine inserts here are // local-id-only and bind to `Surrogate::ZERO`. let _ = doc_id; - index.insert_with_surrogate(vector, nodedb_types::Surrogate::ZERO); + if let Err(e) = + index.insert_with_surrogate(vector, nodedb_types::Surrogate::ZERO) + { + self.replay_record_rejected( + "vector", + record_lsn, + None, + &format!("vector record for '{collection}': {e}"), + ); + continue; + } inserted += 1; } else if let Ok((collection, vector, dim)) = zerompk::from_msgpack::<(String, Vec, usize)>(&record.payload) @@ -370,7 +380,15 @@ impl CoreLoop { ); continue; } - index.insert(vector); + if let Err(e) = index.insert(vector) { + self.replay_record_rejected( + "vector", + record_lsn, + None, + &format!("vector record for '{collection}': {e}"), + ); + continue; + } inserted += 1; } else if let Ok((collection, vectors, dim)) = zerompk::from_msgpack::<(String, Vec>, usize)>(&record.payload) @@ -413,8 +431,16 @@ impl CoreLoop { .vector_collections .entry(index_key) .or_insert_with(|| VectorCollection::new(dim, params)); - for vector in vectors { - index.insert(vector); + // Checked as a whole before any vector lands, so a record + // holding one vector of another width applies nothing. + if let Err(e) = index.insert_batch_with_surrogates(&vectors, &[]) { + self.replay_record_rejected( + "vector", + record_lsn, + None, + &format!("vector batch record for '{collection}': {e}"), + ); + continue; } inserted += 1; } @@ -524,7 +550,7 @@ mod tests { ) { let dim = vector.len(); let mut coll = VectorCollection::new(dim, HnswParams::default()); - coll.insert(vector); + coll.insert(vector).unwrap(); let key = CoreLoop::vector_index_key(0, tenant_id, collection, ""); core.vector_collections.insert(key, coll); core.floors.replay_floors.vector.set(stamp); diff --git a/nodedb/src/data/executor/wal_replay_vector_extended.rs b/nodedb/src/data/executor/wal_replay_vector_extended.rs index 0c9a96e7e..14ad54613 100644 --- a/nodedb/src/data/executor/wal_replay_vector_extended.rs +++ b/nodedb/src/data/executor/wal_replay_vector_extended.rs @@ -570,7 +570,8 @@ mod tests { // A restored checkpoint holding the first write, whose stamp names its // LSN. The second write is the WAL tail the checkpoint does not hold. let mut coll = VectorCollection::new(3, HnswParams::default()); - coll.insert_with_surrogate(vec![1.0, 2.0, 3.0], Surrogate::new(1)); + coll.insert_with_surrogate(vec![1.0, 2.0, 3.0], Surrogate::new(1)) + .unwrap(); h.core.vector_collections.insert(du_index_key(), coll); h.core.floors.replay_floors.vector.set( crate::data::executor::applied_prefix::ReplayStamp::through(records[0].header.lsn), diff --git a/nodedb/src/error_from.rs b/nodedb/src/error_from.rs index 6ac3de35e..b9a7b1420 100644 --- a/nodedb/src/error_from.rs +++ b/nodedb/src/error_from.rs @@ -124,11 +124,18 @@ impl From for Error { } impl From for Error { - /// Checkpoint failures fail-stop because replay history may be truncated. + /// An input vector of the wrong dimension, or index input the engine + /// cannot use, is the caller's data error: SQLSTATE `22000`. Checkpoint + /// and stored-data failures fail-stop because replay history may be + /// truncated. fn from(e: nodedb_vector::error::VectorError) -> Self { use nodedb_vector::error::VectorError as Ve; let detail = e.to_string(); match e { + Ve::DimensionMismatch { .. } | Ve::InvalidInput { .. } => { + Self::DataException { detail } + } + Ve::InvalidFilterBitmap { .. } => Self::Internal { detail }, Ve::BudgetExhausted(_) => Self::MemoryExhausted { engine: "vector".to_string(), }, @@ -136,7 +143,7 @@ impl From for Error { engine: "vector".to_string(), detail, }, - Ve::DimensionMismatch { .. } + Ve::StoredDimensionMismatch { .. } | Ve::UnsupportedVersion { .. } | Ve::InvalidMagic | Ve::DeserializationFailed(_) @@ -445,4 +452,31 @@ mod tests { other => panic!("expected Error::DataPlane, got {other:?}"), } } + + /// An input vector of the wrong width is the caller's data error; stored + /// data of the wrong width is corruption. + #[test] + fn vector_dimension_errors_classify_by_source() { + use nodedb_vector::error::VectorError; + let input: Error = VectorError::DimensionMismatch { + expected: 3, + got: 2, + } + .into(); + match input { + Error::DataException { detail } => { + assert_eq!(detail, "vector dimension mismatch: expected 3, got 2"); + } + other => panic!("expected Error::DataException, got {other:?}"), + } + let stored: Error = VectorError::StoredDimensionMismatch { + expected: 3, + got: 2, + } + .into(); + assert!( + matches!(stored, Error::SegmentCorrupted { .. }), + "{stored:?}" + ); + } } diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 7c447056f..7238f714d 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -322,6 +322,7 @@ mod truncate_engine_conformance; mod truncate_engine_conformance_columnar_family; mod txn_ddl_commit_registry_sync; mod user_transaction; +mod vector_dimension_errors; mod vector_index_bulk_delete_reindex; mod vector_index_bulk_update_reindex; mod vector_index_merge_reindex; diff --git a/nodedb/tests/wire/cases/vector_dimension_errors.rs b/nodedb/tests/wire/cases/vector_dimension_errors.rs new file mode 100644 index 000000000..5b4b2bdb2 --- /dev/null +++ b/nodedb/tests/wire/cases/vector_dimension_errors.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A vector of the wrong dimension is the caller's data error, SQLSTATE +//! `22000`, on every vector path. +//! +//! - A search whose query width differs from the index fails with `22000` +//! instead of panicking the Data Plane core, and the server keeps serving: +//! the next search on the same collection succeeds. +//! - An insert whose vector width differs from the index fails with `22000`, +//! not `23505` (`unique_violation`), and inserts nothing. + +use crate::harness::TestServer; + +/// A collection holding four 3-wide vectors under a `DIM 3` index. +async fn seeded(name: &str) -> TestServer { + let srv = TestServer::start().await; + srv.exec(&format!( + "CREATE COLLECTION {name} WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + srv.exec(&format!( + "CREATE VECTOR INDEX idx_{name} ON {name} (embedding) METRIC l2 DIM 3" + )) + .await + .unwrap(); + for (id, emb) in [ + ("v1", "[1.0, 0.0, 0.0]"), + ("v2", "[0.0, 1.0, 0.0]"), + ("v3", "[0.0, 0.0, 1.0]"), + ("v4", "[0.7, 0.7, 0.0]"), + ] { + srv.exec(&format!( + "INSERT INTO {name} {{ id: '{id}', embedding: {emb} }}" + )) + .await + .unwrap(); + } + srv +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn wrong_dimension_search_is_22000_and_the_server_keeps_serving() { + let srv = seeded("vd_search").await; + + let error = srv + .query_text( + "SELECT id FROM vd_search \ + ORDER BY vector_distance(embedding, ARRAY[1.0, 0.0]) LIMIT 2", + ) + .await + .expect_err("a 2-wide query against a 3-wide index must fail"); + assert!( + error.contains("22000") && error.contains("vector dimension mismatch: expected 3, got 2"), + "expected 22000 naming both widths, got: {error}" + ); + + let rows = srv + .query_rows( + "SELECT id FROM vd_search \ + ORDER BY vector_distance(embedding, ARRAY[1.0, 0.0, 0.0]) LIMIT 2", + ) + .await + .expect("the next search on the same collection succeeds"); + assert_eq!(rows.first().map(|r| r[0].as_str()), Some("v1"), "{rows:?}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn wrong_dimension_insert_is_22000_and_inserts_nothing() { + let srv = seeded("vd_insert").await; + + let error = srv + .exec("INSERT INTO vd_insert { id: 'bad', embedding: [0.5, 0.5] }") + .await + .expect_err("a 2-wide vector must not enter a 3-wide index"); + assert!( + error.contains("22000") && error.contains("vector dimension mismatch"), + "expected 22000, got: {error}" + ); + assert!(!error.contains("23505"), "not a unique violation: {error}"); + + let rows = srv + .query_rows("SELECT id FROM vd_insert WHERE id = 'bad'") + .await + .expect("read after the refused insert"); + assert!(rows.is_empty(), "the refused row is not stored: {rows:?}"); +} From 021b5c2e43fca156fd7154f82569f306bf134b6d Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 04:06:08 +0800 Subject: [PATCH 45/64] fix(query): fail search legs on error instead of folding to empty FTS query analysis, corpus stats, facet counting, and staged spatial row decoding used to swallow storage/analyzer errors with ok()/ unwrap_or_default(), silently degrading hybrid, triple, graph-RAG, and facet searches to partial results. They now propagate the underlying error through the search response instead. --- nodedb-vector/src/{ivf.rs => ivf/index.rs} | 0 nodedb/src/data/executor/handlers/facet.rs | 69 +++++++++--------- .../executor/handlers/graph_rag_triple.rs | 30 ++++---- .../src/data/executor/handlers/text_search.rs | 9 +-- .../executor/handlers/text_search_hybrid.rs | 30 ++++---- .../executor/handlers/text_search_scan.rs | 27 +------ .../executor/handlers/text_search_triple.rs | 30 ++++---- .../handlers/transaction/overlay/fts_merge.rs | 70 +++++++++---------- .../handlers/transaction/overlay/fts_score.rs | 16 ++--- .../transaction/overlay/spatial_merge.rs | 29 ++++---- nodedb/tests/wire/cases/sql_hybrid_search.rs | 27 +++++++ .../tests/wire/cases/sql_three_source_rrf.rs | 28 ++++++++ 12 files changed, 194 insertions(+), 171 deletions(-) rename nodedb-vector/src/{ivf.rs => ivf/index.rs} (100%) diff --git a/nodedb-vector/src/ivf.rs b/nodedb-vector/src/ivf/index.rs similarity index 100% rename from nodedb-vector/src/ivf.rs rename to nodedb-vector/src/ivf/index.rs diff --git a/nodedb/src/data/executor/handlers/facet.rs b/nodedb/src/data/executor/handlers/facet.rs index e4f41fe3e..c5088eea0 100644 --- a/nodedb/src/data/executor/handlers/facet.rs +++ b/nodedb/src/data/executor/handlers/facet.rs @@ -57,14 +57,7 @@ impl CoreLoop { &filters, ) { Ok(ids) => ids, - Err(e) => { - return self.response_error( - task, - ErrorCode::Internal { - detail: e.to_string(), - }, - ); - } + Err(e) => return self.response_error(task, e), }; let matching_set: HashSet = @@ -79,14 +72,17 @@ impl CoreLoop { }; for field in fields { - let counts = self.count_facet_field( + let counts = match self.count_facet_field( task.request.database_id.as_u64(), tid, collection, field, &matching_set, &matching_ids, - ); + ) { + Ok(counts) => counts, + Err(e) => return self.response_error(task, e), + }; let facet_values: Vec = counts .into_iter() .take(effective_limit) @@ -135,7 +131,8 @@ impl CoreLoop { /// Count distinct values for a single facet field, filtered to matching documents. /// /// Tries index-backed counting first (O(index_entries)), falls back to - /// document-scan counting (O(matching_docs)). + /// document-scan counting (O(matching_docs)). A storage read or value + /// transcoding error fails the count: a partial count is a wrong count. fn count_facet_field( &self, database_id: u64, @@ -144,17 +141,17 @@ impl CoreLoop { field: &str, matching_set: &HashSet, matching_ids: &[nodedb_types::StorageKey], - ) -> Vec<(String, usize)> { + ) -> crate::Result> { // Fast path: index-backed counting with filtered doc set. - if let Ok(groups) = self.sparse.scan_index_groups_filtered( + let groups = self.sparse.scan_index_groups_filtered( database_id, tid, collection, field, matching_set, - ) && !groups.is_empty() - { - return groups; + )?; + if !groups.is_empty() { + return Ok(groups); } // Fallback: scan matching documents, extract field from msgpack, count. @@ -172,29 +169,33 @@ impl CoreLoop { ); let mut counts: HashMap = HashMap::new(); for key in matching_ids { - if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, key) { + if let Some(bytes) = self.sparse.get(database_id, tid, collection, key)? { let mp = crate::data::executor::scan_normalize::sparse_body_to_msgpack( &bytes, body_format.as_format_ref(), ); if let Some((start, end)) = nodedb_query::msgpack_scan::extract_field(&mp, 0, field) { - let value_str = if let Some(s) = - nodedb_query::msgpack_scan::read_str(&mp, start) - { - s.to_string() - } else if let Some(i) = nodedb_query::msgpack_scan::read_i64(&mp, start) { - i.to_string() - } else if let Some(f) = nodedb_query::msgpack_scan::read_f64(&mp, start) { - f.to_string() - } else if let Some(b) = nodedb_query::msgpack_scan::read_bool(&mp, start) { - b.to_string() - } else if nodedb_query::msgpack_scan::read_null(&mp, start) { - continue; - } else { - // Complex value — stringify via transcoder. - nodedb_types::msgpack_to_json_string(&mp[start..end]).unwrap_or_default() - }; + let value_str = + if let Some(s) = nodedb_query::msgpack_scan::read_str(&mp, start) { + s.to_string() + } else if let Some(i) = nodedb_query::msgpack_scan::read_i64(&mp, start) { + i.to_string() + } else if let Some(f) = nodedb_query::msgpack_scan::read_f64(&mp, start) { + f.to_string() + } else if let Some(b) = nodedb_query::msgpack_scan::read_bool(&mp, start) { + b.to_string() + } else if nodedb_query::msgpack_scan::read_null(&mp, start) { + continue; + } else { + // Complex value — stringify via transcoder. + nodedb_types::msgpack_to_json_string(&mp[start..end]).map_err(|e| { + crate::Error::Serialization { + format: "msgpack".to_string(), + detail: format!("facet value of '{field}' in {key}: {e}"), + } + })? + }; *counts.entry(value_str).or_default() += 1; } } @@ -202,7 +203,7 @@ impl CoreLoop { let mut result: Vec<(String, usize)> = counts.into_iter().collect(); result.sort_by_key(|r| std::cmp::Reverse(r.1)); // Count descending. - result + Ok(result) } } diff --git a/nodedb/src/data/executor/handlers/graph_rag_triple.rs b/nodedb/src/data/executor/handlers/graph_rag_triple.rs index 11b0799b2..138e2de9e 100644 --- a/nodedb/src/data/executor/handlers/graph_rag_triple.rs +++ b/nodedb/src/data/executor/handlers/graph_rag_triple.rs @@ -90,21 +90,21 @@ impl CoreLoop { // BM25 text search. let fetch_k = final_top_k.saturating_mul(3).max(20); - let text_results = self - .inverted - .search( - task.request.database_id.as_u64(), - tid_typed, - collection, - FtsSearchParams { - query: bm25_query, - top_k: fetch_k, - fuzzy_enabled: true, - mode: QueryMode::And, - prefilter: None, - }, - ) - .unwrap_or_default(); + let text_results = match self.inverted.search( + task.request.database_id.as_u64(), + tid_typed, + collection, + FtsSearchParams { + query: bm25_query, + top_k: fetch_k, + fuzzy_enabled: true, + mode: QueryMode::And, + prefilter: None, + }, + ) { + Ok(results) => results, + Err(e) => return self.response_error(task, e), + }; // Graph expansion seeded directly by the vector hits' surrogates. let expansion = self.expand_graph(GraphExpansionParams { diff --git a/nodedb/src/data/executor/handlers/text_search.rs b/nodedb/src/data/executor/handlers/text_search.rs index af1defba8..6fd7d2919 100644 --- a/nodedb/src/data/executor/handlers/text_search.rs +++ b/nodedb/src/data/executor/handlers/text_search.rs @@ -96,14 +96,7 @@ impl CoreLoop { }, ) { Ok(r) => r, - Err(e) => { - return self.response_error( - task, - ErrorCode::Internal { - detail: e.to_string(), - }, - ); - } + Err(e) => return self.response_error(task, e), }; // Read-your-own-writes for FTS: fold this transaction's staged diff --git a/nodedb/src/data/executor/handlers/text_search_hybrid.rs b/nodedb/src/data/executor/handlers/text_search_hybrid.rs index c046f895b..f2876b67c 100644 --- a/nodedb/src/data/executor/handlers/text_search_hybrid.rs +++ b/nodedb/src/data/executor/handlers/text_search_hybrid.rs @@ -109,21 +109,21 @@ impl CoreLoop { }; // 2. Text search (no surrogate prefilter for the text leg of hybrid search). - let text_results = self - .inverted - .search( - task.request.database_id.as_u64(), - tenant_id, - collection, - FtsSearchParams { - query: query_text, - top_k: fetch_k, - fuzzy_enabled: fuzzy, - mode: QueryMode::And, - prefilter: None, - }, - ) - .unwrap_or_default(); + let text_results = match self.inverted.search( + task.request.database_id.as_u64(), + tenant_id, + collection, + FtsSearchParams { + query: query_text, + top_k: fetch_k, + fuzzy_enabled: fuzzy, + mode: QueryMode::And, + prefilter: None, + }, + ) { + Ok(results) => results, + Err(e) => return self.response_error(task, e), + }; // 3. Build ranked lists for weighted RRF. // Higher weight → lower k → steeper rank discount → more influence. diff --git a/nodedb/src/data/executor/handlers/text_search_scan.rs b/nodedb/src/data/executor/handlers/text_search_scan.rs index 4932c7ef1..8e98ce8ea 100644 --- a/nodedb/src/data/executor/handlers/text_search_scan.rs +++ b/nodedb/src/data/executor/handlers/text_search_scan.rs @@ -60,14 +60,7 @@ impl CoreLoop { }, ) { Ok(r) => r, - Err(e) => { - return self.response_error( - task, - ErrorCode::Internal { - detail: e.to_string(), - }, - ); - } + Err(e) => return self.response_error(task, e), }; // Read-your-own-writes for FTS phrase search: fold staged document @@ -162,14 +155,7 @@ impl CoreLoop { }, ) { Ok(hits) => hits.into_iter().map(|h| (h.doc_id, h.score)).collect(), - Err(e) => { - return self.response_error( - task, - ErrorCode::Internal { - detail: e.to_string(), - }, - ); - } + Err(e) => return self.response_error(task, e), }; // Read-your-own-writes for FTS: fold staged document bodies into @@ -207,14 +193,7 @@ impl CoreLoop { ); let mut docs: Vec<(StorageKey, Vec)> = match scan_result { Ok(d) => d, - Err(e) => { - return self.response_error( - task, - ErrorCode::Internal { - detail: e.to_string(), - }, - ); - } + Err(e) => return self.response_error(task, e), }; // Read-your-own-writes: fold staged document bodies into the row diff --git a/nodedb/src/data/executor/handlers/text_search_triple.rs b/nodedb/src/data/executor/handlers/text_search_triple.rs index 600f00da2..5ab3cda69 100644 --- a/nodedb/src/data/executor/handlers/text_search_triple.rs +++ b/nodedb/src/data/executor/handlers/text_search_triple.rs @@ -116,21 +116,21 @@ impl CoreLoop { }; // 2. BM25 text search. - let text_results = self - .inverted - .search( - task.request.database_id.as_u64(), - tenant_id, - collection, - FtsSearchParams { - query: query_text, - top_k: fetch_k, - fuzzy_enabled: fuzzy, - mode: QueryMode::And, - prefilter: None, - }, - ) - .unwrap_or_default(); + let text_results = match self.inverted.search( + task.request.database_id.as_u64(), + tenant_id, + collection, + FtsSearchParams { + query: query_text, + top_k: fetch_k, + fuzzy_enabled: fuzzy, + mode: QueryMode::And, + prefilter: None, + }, + ) { + Ok(results) => results, + Err(e) => return self.response_error(task, e), + }; // 3. Graph BFS from seed node. // The seed is named by the query itself, so it resolves to a surrogate diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/fts_merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/fts_merge.rs index 0ba7183f4..52edf7587 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/fts_merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/fts_merge.rs @@ -93,13 +93,8 @@ impl CoreLoop { base_results.clear(); } - let Some((positive_terms, negative_terms)) = - self.analyze_query_terms(database_id.as_u64(), tid, collection, query) - else { - // An invalid query already failed the base search with an - // error before this merge could run — nothing to fold in. - return Ok(()); - }; + let (positive_terms, negative_terms) = + self.analyze_query_terms(database_id.as_u64(), tid, collection, query)?; if positive_terms.is_empty() { // No positive terms to score staged docs against — but staged // tombstones still hide base rows. @@ -109,7 +104,7 @@ impl CoreLoop { let config_key = (database_id, tid, collection.to_string()); let bm25_params = Bm25Params::default(); - let ctx = self.staged_score_ctx(database_id, tid, collection, &config_key, &bm25_params); + let ctx = self.staged_score_ctx(database_id, tid, collection, &config_key, &bm25_params)?; let mut seen: HashMap = base_results .iter() @@ -202,16 +197,14 @@ impl CoreLoop { // (`InvertedIndex::phrase_search`) uses — so the contiguity check // compares stemmed/normalized tokens on both sides. let db_u64 = database_id.as_u64(); - let phrase_terms: Vec = terms - .iter() - .map(|t| { - self.inverted - .analyze_for_collection(db_u64, tid, collection, t) - .ok() - .and_then(|tokens| tokens.into_iter().next()) - .unwrap_or_else(|| t.clone()) - }) - .collect(); + // A term the analyzer drops (a stop word) is matched as written. + let mut phrase_terms: Vec = Vec::with_capacity(terms.len()); + for t in terms { + let tokens = self + .inverted + .analyze_for_collection(db_u64, tid, collection, t)?; + phrase_terms.push(tokens.into_iter().next().unwrap_or_else(|| t.clone())); + } let config_key = (database_id, tid, collection.to_string()); @@ -282,15 +275,12 @@ impl CoreLoop { if overlay.is_truncated(&coll_key) { base_results.clear(); } - let Some((positive_terms, negative_terms)) = - self.analyze_query_terms(database_id.as_u64(), tid, collection, query) - else { - return Ok(()); - }; + let (positive_terms, negative_terms) = + self.analyze_query_terms(database_id.as_u64(), tid, collection, query)?; let config_key = (database_id, tid, collection.to_string()); let bm25_params = Bm25Params::default(); - let ctx = self.staged_score_ctx(database_id, tid, collection, &config_key, &bm25_params); + let ctx = self.staged_score_ctx(database_id, tid, collection, &config_key, &bm25_params)?; for (surrogate, staged) in overlay.iter_for_collection(&coll_key) { match staged { @@ -384,25 +374,31 @@ impl CoreLoop { /// — the same resolution the forward-indexing path and the base search /// use), once per merge call (never per staged document — every staged /// doc in the merge loop is scored against this same pair of term - /// lists). Returns `None` when `query` fails to parse (the base search - /// already surfaced that error). + /// lists). A query that fails to parse fails with `BadRequest`, as the + /// base search does. An analyzer error propagates. fn analyze_query_terms( &self, database_id: u64, tid: TenantId, collection: &str, query: &str, - ) -> Option<(Vec, Vec)> { - let parsed = parse_query(query).ok()?; - let positive_terms = self - .inverted - .analyze_for_collection(database_id, tid, collection, &parsed.positive.join(" ")) - .unwrap_or_default(); - let negative_terms = self - .inverted - .analyze_for_collection(database_id, tid, collection, &parsed.negative.join(" ")) - .unwrap_or_default(); - Some((positive_terms, negative_terms)) + ) -> crate::Result<(Vec, Vec)> { + let parsed = parse_query(query).map_err(|e| crate::Error::BadRequest { + detail: e.to_string(), + })?; + let positive_terms = self.inverted.analyze_for_collection( + database_id, + tid, + collection, + &parsed.positive.join(" "), + )?; + let negative_terms = self.inverted.analyze_for_collection( + database_id, + tid, + collection, + &parsed.negative.join(" "), + )?; + Ok((positive_terms, negative_terms)) } } diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/fts_score.rs b/nodedb/src/data/executor/handlers/transaction/overlay/fts_score.rs index b0824db73..1772a7ebf 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/fts_score.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/fts_score.rs @@ -43,12 +43,11 @@ impl CoreLoop { collection: &'a str, config_key: &'a (DatabaseId, TenantId, String), bm25_params: &'a Bm25Params, - ) -> StagedFtsScoreCtx<'a> { - let (total_docs, avg_doc_len) = self - .inverted - .corpus_stats(database_id.as_u64(), tid, collection) - .unwrap_or((0, 1.0)); - StagedFtsScoreCtx { + ) -> crate::Result> { + let (total_docs, avg_doc_len) = + self.inverted + .corpus_stats(database_id.as_u64(), tid, collection)?; + Ok(StagedFtsScoreCtx { database_id: database_id.as_u64(), tid, collection, @@ -56,7 +55,7 @@ impl CoreLoop { total_docs: total_docs.max(1), avg_doc_len: if avg_doc_len > 0.0 { avg_doc_len } else { 1.0 }, bm25_params, - } + }) } /// Decode a staged body, re-tokenize with the forward-indexing @@ -99,8 +98,7 @@ impl CoreLoop { matched_any = true; let df = self .inverted - .term_df(ctx.database_id, ctx.tid, ctx.collection, term) - .unwrap_or(0) + .term_df(ctx.database_id, ctx.tid, ctx.collection, term)? .max(1); score += bm25_score( tf, diff --git a/nodedb/src/data/executor/handlers/transaction/overlay/spatial_merge.rs b/nodedb/src/data/executor/handlers/transaction/overlay/spatial_merge.rs index ac069a9ce..b25304305 100644 --- a/nodedb/src/data/executor/handlers/transaction/overlay/spatial_merge.rs +++ b/nodedb/src/data/executor/handlers/transaction/overlay/spatial_merge.rs @@ -60,24 +60,26 @@ pub(in crate::data::executor) struct SpatialOverlayMergeParams<'a> { /// Decode a staged spatial-collection overlay body into a full (unprojected) /// `Value::Object`, handling both possible staged shapes (see module doc). -/// Returns `Ok(None)` for a body that fails to decode, or a `Value::Array` -/// staged row whose collection has no known columnar schema (defensively -/// treated as "does not match" rather than surfacing a panic). Returns -/// `Err` only when the row *does* decode but its computed-column projection -/// hits a division/modulo-by-zero — `row_to_projected_value` is called with -/// no computed columns here (`&[]`), so this is currently unreachable, but -/// the `Result` return keeps the signature honest about what -/// `row_to_projected_value` can do. +/// Returns `Ok(None)` for a `Value::Array` staged row whose collection has no +/// known columnar schema, and for a body of any other shape: neither is a +/// row this search can match. A body that does not decode fails with +/// `Serialization`: the transaction staged it, so it is corrupt. A +/// computed-column projection error propagates. fn decode_staged_spatial_row( body: &[u8], schema: Option<&ColumnarSchema>, ) -> crate::Result> { - Ok(match nodedb_types::value_from_msgpack(body).ok() { - Some(Value::Array(row)) => match schema { + let value = + nodedb_types::value_from_msgpack(body).map_err(|e| crate::Error::Serialization { + format: "msgpack".to_string(), + detail: format!("staged spatial row does not decode: {e}"), + })?; + Ok(match value { + Value::Array(row) => match schema { Some(schema) => Some(row_to_projected_value(&row, schema, &[], &[], false)?), None => None, }, - Some(obj @ Value::Object(_)) => Some(obj), + obj @ Value::Object(_) => Some(obj), _ => None, }) } @@ -168,9 +170,8 @@ impl CoreLoop { Some(Staged::Put(body)) => { let doc = match decode_staged_spatial_row(body, schema) { Ok(Some(doc)) => doc, - // A staged body that fails to decode carries no - // usable row: drop it rather than surface stale - // base data. + // A staged row with no matchable shape: drop it + // rather than surface stale base data. Ok(None) => return false, Err(e) => { first_err = Some(e); diff --git a/nodedb/tests/wire/cases/sql_hybrid_search.rs b/nodedb/tests/wire/cases/sql_hybrid_search.rs index 80a51f8ee..33cf8bab3 100644 --- a/nodedb/tests/wire/cases/sql_hybrid_search.rs +++ b/nodedb/tests/wire/cases/sql_hybrid_search.rs @@ -272,3 +272,30 @@ async fn hybrid_search_id_is_user_primary_key_not_surrogate_hex() { ); } } + +// ── A failing text leg fails the hybrid search ────────────────────────────── + +/// A NOT-only text query is refused by the FTS engine. The hybrid search +/// returns that error instead of fusing the vector leg alone. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn hybrid_search_returns_the_text_leg_error() { + let server = TestServer::start().await; + create_hybrid_collection(&server, "hs_text_err").await; + + let err = server + .query_rows( + "SELECT id, \ + rrf_score(\ + vector_distance(embedding, ARRAY[0.1, 0.2, 0.3, 0.4]), \ + bm25_score(content, 'NOT consensus')\ + ) AS score \ + FROM hs_text_err \ + ORDER BY score DESC LIMIT 5", + ) + .await + .expect_err("a hybrid search whose text leg fails must fail"); + assert!( + err.contains("at least one positive term"), + "the error must be the text leg's own error; got: {err}" + ); +} diff --git a/nodedb/tests/wire/cases/sql_three_source_rrf.rs b/nodedb/tests/wire/cases/sql_three_source_rrf.rs index f3cea193f..849de1e8f 100644 --- a/nodedb/tests/wire/cases/sql_three_source_rrf.rs +++ b/nodedb/tests/wire/cases/sql_three_source_rrf.rs @@ -364,3 +364,31 @@ async fn rrf_score_triple_with_two_k_constants_is_rejected() { "inconsistent arity (3 ranks + 2 k values) must return a typed error" ); } + +// ── A failing text leg fails the three-source search ──────────────────────── + +/// A NOT-only text query is refused by the FTS engine. The three-source +/// search returns that error instead of fusing the vector and graph legs. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn rrf_score_triple_returns_the_text_leg_error() { + let server = TestServer::start().await; + create_triple_collection(&server, "t3_text_err").await; + + let err = server + .query_rows( + "SELECT id, \ + rrf_score(\ + vector_distance(embedding, ARRAY[1.0, 0.0, 0.0]), \ + bm25_score(body, 'NOT alpha'), \ + graph_score(id, 'n1', depth => 1, label => 'hop') \ + ) AS score \ + FROM t3_text_err \ + LIMIT 10", + ) + .await + .expect_err("a three-source search whose text leg fails must fail"); + assert!( + err.contains("at least one positive term"), + "the error must be the text leg's own error; got: {err}" + ); +} From 70174e51992f07dff1874255ba49a3f2f0907dd3 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 04:06:29 +0800 Subject: [PATCH 46/64] feat(vector): back IVF-PQ with a live, trainable index The IVF-PQ index type used to build once from a static snapshot and serve reads only. It now buffers inserted vectors, searched exactly, until it holds max(ivf_cells, pq_k) live vectors, trains its k-means cells and PQ codebooks on them, and from then on supports insert, soft-delete, and search with cell-probe plus exact rerank. - nodedb-vector/src/ivf splits into mod.rs, index.rs, kmeans.rs, params.rs, search.rs, and checkpoint.rs; each entry carries a caller-assigned id so a collection can move its vectors under the ids they already have. - nodedb-vector/src/quantize/pq_kmeans.rs carries the k-means implementation shared by PQ codebook training and IVF cell training. - VectorCollection tracks IVF training state (collection/ivf_mode.rs) and settles (trains or reseeds) a filled index on transaction commit instead of only sealing HNSW builds. - The executor wires insert, delete, undo, WAL replay, snapshot restore, compaction, and reindex through the new lifecycle, and vector_settle.rs replaces vector_search_ivf.rs as the settle path. - VectorIndexStats reports IVF training state (threshold, trained, cell count, nprobe) through SHOW VECTOR INDEX. - FTS and spatial search legs used to swallow errors with ok()/ unwrap_or_default() and fold to empty rather than fail; they now propagate the underlying error through the search response. --- nodedb-types/src/lib.rs | 4 +- nodedb-types/src/vector_index_stats.rs | 44 ++ nodedb-vector/src/collection/budget.rs | 3 +- nodedb-vector/src/collection/checkpoint.rs | 17 +- nodedb-vector/src/collection/ivf_mode.rs | 283 +++++++++ nodedb-vector/src/collection/lifecycle.rs | 23 +- .../src/collection/lifecycle_compact.rs | 35 +- .../src/collection/lifecycle_insert_ops.rs | 38 +- nodedb-vector/src/collection/mod.rs | 1 + nodedb-vector/src/collection/rollback.rs | 19 +- nodedb-vector/src/collection/search.rs | 18 +- nodedb-vector/src/collection/stats.rs | 28 +- nodedb-vector/src/index_config.rs | 4 +- nodedb-vector/src/ivf/checkpoint.rs | 165 +++++ nodedb-vector/src/ivf/index.rs | 597 +++++++++--------- nodedb-vector/src/ivf/kmeans.rs | 99 +++ nodedb-vector/src/ivf/mod.rs | 10 + nodedb-vector/src/ivf/params.rs | 40 ++ nodedb-vector/src/ivf/search.rs | 203 ++++++ nodedb-vector/src/quantize/mod.rs | 1 + nodedb-vector/src/quantize/pq.rs | 120 +--- nodedb-vector/src/quantize/pq_kmeans.rs | 115 ++++ .../shared/ddl/neutral/dsl/vector_index.rs | 10 +- .../ddl/neutral/maintenance/vector_index.rs | 13 +- nodedb/src/data/executor/core_loop/open.rs | 1 - nodedb/src/data/executor/core_loop/state.rs | 5 - .../executor/core_loop/vector_index_seed.rs | 5 + .../data/executor/handlers/compact/runner.rs | 9 +- .../data/executor/handlers/control/reindex.rs | 4 + nodedb/src/data/executor/handlers/mod.rs | 2 +- .../handlers/point/apply_put/vector/put.rs | 27 +- nodedb/src/data/executor/handlers/purge.rs | 1 - .../handlers/snapshot/restore/engines.rs | 15 +- .../handlers/transaction/redo_apply/settle.rs | 23 +- .../handlers/transaction/undo/vector_write.rs | 178 +++--- .../handlers/unregister_collection.rs | 2 - nodedb/src/data/executor/handlers/vector.rs | 125 +--- .../executor/handlers/vector_direct_row.rs | 19 +- .../executor/handlers/vector_index_drop.rs | 1 - .../executor/handlers/vector_lifecycle.rs | 34 - .../data/executor/handlers/vector_multi.rs | 41 +- .../handlers/vector_multi_search_exec.rs | 150 ++--- .../data/executor/handlers/vector_params.rs | 5 + .../executor/handlers/vector_search_exec.rs | 38 +- .../executor/handlers/vector_search_ivf.rs | 89 --- .../data/executor/handlers/vector_settle.rs | 170 +++++ .../data/executor/handlers/vector_write.rs | 12 +- nodedb/src/data/executor/wal_replay_vector.rs | 80 +-- .../executor/wal_replay_vector_index_drop.rs | 1 - nodedb/src/data/runtime/boot_replay.rs | 5 + nodedb/tests/wire/cases/mod.rs | 1 + .../wire/cases/vector_ivf_pq_training.rs | 131 ++++ 52 files changed, 2026 insertions(+), 1038 deletions(-) create mode 100644 nodedb-vector/src/collection/ivf_mode.rs create mode 100644 nodedb-vector/src/ivf/checkpoint.rs create mode 100644 nodedb-vector/src/ivf/kmeans.rs create mode 100644 nodedb-vector/src/ivf/mod.rs create mode 100644 nodedb-vector/src/ivf/params.rs create mode 100644 nodedb-vector/src/ivf/search.rs create mode 100644 nodedb-vector/src/quantize/pq_kmeans.rs delete mode 100644 nodedb/src/data/executor/handlers/vector_search_ivf.rs create mode 100644 nodedb/src/data/executor/handlers/vector_settle.rs create mode 100644 nodedb/tests/wire/cases/vector_ivf_pq_training.rs diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index 68f14647e..0f4756d38 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -144,6 +144,8 @@ pub use value::{NotScalar, Value, scalar_to_raw_bytes}; pub use vector_ann::{VectorAnnOptions, VectorQuantization}; pub use vector_dtype::VectorStorageDtype; pub use vector_index_params::StoredVectorIndexParams; -pub use vector_index_stats::{VectorIndexQuantization, VectorIndexStats, VectorIndexType}; +pub use vector_index_stats::{ + VectorIndexQuantization, VectorIndexStats, VectorIndexType, VectorIvfStats, +}; pub use vector_model::{VectorModelEntry, VectorModelMetadata}; pub use volatility::Volatility; diff --git a/nodedb-types/src/vector_index_stats.rs b/nodedb-types/src/vector_index_stats.rs index c1f8221b9..34018636d 100644 --- a/nodedb-types/src/vector_index_stats.rs +++ b/nodedb-types/src/vector_index_stats.rs @@ -126,6 +126,40 @@ pub struct VectorIndexStats { /// bytes. `None` when the collection has no dedicated arena (e.g., it is /// not vector-primary, or the runtime does not support per-arena stats). pub arena_bytes: Option, + /// IVF-PQ training state. `None` unless the index type is `ivf_pq`. + pub ivf: Option, +} + +/// Training state of an IVF-PQ vector index. +/// +/// An IVF-PQ index buffers vectors, searched exactly, until it holds +/// `training_threshold` live vectors. It then trains its cells and PQ +/// codebooks on them and searches through IVF-PQ from then on. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct VectorIvfStats { + /// Live vectors the index needs before it trains: `max(ivf_cells, pq_k)`. + pub training_threshold: usize, + /// Whether training has run. + pub trained: bool, + /// Vectors the training read. `0` before training. + pub trained_on: usize, + /// Unix milliseconds of the training. `0` before training. + pub trained_at_ms: u64, + /// Vectors held by the trained index, live or soft-deleted. + pub indexed_vectors: usize, + /// Voronoi cells of the trained index. `0` before training. + pub cells: usize, + /// Cells probed per query. + pub nprobe: usize, } #[cfg(test)] @@ -155,6 +189,15 @@ mod tests { seal_threshold: 65_536, mmap_segment_count: 1, arena_bytes: Some(4 * 1024 * 1024), + ivf: Some(VectorIvfStats { + training_threshold: 256, + trained: true, + trained_on: 300, + trained_at_ms: 1_700_000_000_000, + indexed_vectors: 310, + cells: 16, + nprobe: 4, + }), }; let bytes = zerompk::to_msgpack_vec(&stats).unwrap(); let restored: VectorIndexStats = zerompk::from_msgpack(&bytes).unwrap(); @@ -162,5 +205,6 @@ mod tests { assert_eq!(restored.live_count, 183_000); assert_eq!(restored.quantization, VectorIndexQuantization::Sq8); assert_eq!(restored.index_type, VectorIndexType::Hnsw); + assert_eq!(restored.ivf, stats.ivf); } } diff --git a/nodedb-vector/src/collection/budget.rs b/nodedb-vector/src/collection/budget.rs index de1850c85..2dfcfa752 100644 --- a/nodedb-vector/src/collection/budget.rs +++ b/nodedb-vector/src/collection/budget.rs @@ -36,7 +36,8 @@ impl VectorCollection { .filter(|s| s.tier == StorageTier::L0Ram) .map(|s| s.index.len() * bytes_per_vector) .sum(); - growing + building + sealed_ram + let ivf = self.ivf.as_ref().map_or(0, |ivf| ivf.memory_bytes()); + growing + building + sealed_ram + ivf } /// Whether the RAM budget is exceeded. diff --git a/nodedb-vector/src/collection/checkpoint.rs b/nodedb-vector/src/collection/checkpoint.rs index 18cb4a289..9f80fedcd 100644 --- a/nodedb-vector/src/collection/checkpoint.rs +++ b/nodedb-vector/src/collection/checkpoint.rs @@ -30,6 +30,7 @@ use crate::distance::DistanceMetric; use crate::error::VectorError; use crate::flat::FlatIndex; use crate::hnsw::{HnswIndex, HnswParams}; +use crate::ivf::IvfPqIndex; use crate::quantize::pq::PqCodec; use crate::quantize::sq8::Sq8Codec; @@ -98,6 +99,12 @@ pub(crate) struct CollectionSnapshot { /// nothing and replay everything, exactly as before. #[serde(default)] pub checkpoint_wal_lsn: u64, + /// Index type and PQ/IVF parameters. Its HNSW parameters are the + /// `params_*` fields above. + pub index_config: crate::index_config::IndexConfig, + /// Encoded trained IVF-PQ index. `None` for a non-IVF collection or one + /// still buffering toward its training threshold. + pub ivf_bytes: Option>, } #[derive(Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack)] @@ -225,6 +232,8 @@ impl VectorCollection { } }, checkpoint_wal_lsn: self.checkpoint_wal_lsn.max(self.applied_wal_lsn), + index_config: self.index_config.clone(), + ivf_bytes: self.ivf.as_ref().map(IvfPqIndex::to_bytes).transpose()?, }; let msgpack = match zerompk::to_msgpack_vec(&snapshot) { Ok(bytes) => bytes, @@ -399,8 +408,13 @@ impl VectorCollection { let index_config = crate::index_config::IndexConfig { hnsw: params.clone(), - ..crate::index_config::IndexConfig::default() + ..snap.index_config }; + let ivf = snap + .ivf_bytes + .as_deref() + .map(|bytes| IvfPqIndex::from_bytes(bytes, memory.clone())) + .transpose()?; Ok(Self { growing, growing_base_id: snap.growing_base_id, @@ -432,6 +446,7 @@ impl VectorCollection { seal_threshold: DEFAULT_SEAL_THRESHOLD, index_config, codec_dispatch: None, + ivf, quantization: quantization_from_tag(snap.quantization_tag), payload: if snap.payload_index_bytes.is_empty() { super::payload_index::PayloadIndexSet::default() diff --git a/nodedb-vector/src/collection/ivf_mode.rs b/nodedb-vector/src/collection/ivf_mode.rs new file mode 100644 index 000000000..7fc1a467c --- /dev/null +++ b/nodedb-vector/src/collection/ivf_mode.rs @@ -0,0 +1,283 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The IVF-PQ mode of a `VectorCollection`. +//! +//! An `IvfPq` collection buffers its vectors in the growing segment until it +//! holds the training threshold, `max(ivf_cells, pq_k)`. The growing segment +//! is searched exactly, checkpointed, and rebuilt by WAL replay, so the +//! buffer is durable and searchable with no extra machinery. It never seals. +//! +//! [`VectorCollection::train_ivf`] trains the IVF centroids and PQ codebooks +//! on every live vector, moves each one into the trained index under its +//! global id, and empties the segments. The move happens in one call on the +//! owning core, so no search sees a vector twice or misses one. Later inserts +//! land in the trained index. + +use nodedb_mem::ScopedMemory; + +use crate::error::VectorError; +use crate::index_config::{IndexConfig, IndexType}; +use crate::ivf::IvfPqIndex; + +use super::lifecycle::VectorCollection; +use super::lifecycle_insert_ops::sealed_vector; + +impl VectorCollection { + /// Whether the collection is configured as an IVF-PQ index. + pub fn is_ivf(&self) -> bool { + self.index_config.index_type == IndexType::IvfPq + } + + /// The full index configuration. + pub fn index_config(&self) -> &IndexConfig { + &self.index_config + } + + /// Replace the index configuration. The HNSW parameters the collection + /// holds stay: its segments were built with them. + pub fn set_index_config(&mut self, config: IndexConfig) { + self.index_config = IndexConfig { + hnsw: self.params.clone(), + ..config + }; + } + + /// Live vectors an `IvfPq` collection needs before it trains. + pub fn ivf_training_threshold(&self) -> usize { + self.index_config.to_ivf_params().training_threshold() + } + + /// Whether an `IvfPq` collection is untrained and holds the training + /// threshold of live vectors. + pub fn needs_ivf_training(&self) -> bool { + self.is_ivf() && self.ivf.is_none() && self.live_count() >= self.ivf_training_threshold() + } + + /// The trained IVF-PQ index, if any. + pub fn ivf_index(&self) -> Option<&IvfPqIndex> { + self.ivf.as_ref() + } + + /// Train the IVF-PQ index on every live vector and move them all into it. + /// `trained_at_ms` stamps the training, in Unix milliseconds. + /// + /// Fails with [`VectorError::InvalidInput`] when the collection is not + /// `IvfPq`, is already trained, or holds fewer live vectors than the + /// training threshold. A training error propagates. Any failure leaves + /// the collection unchanged. + pub fn train_ivf( + &mut self, + memory: ScopedMemory, + trained_at_ms: u64, + ) -> Result<(), VectorError> { + if !self.is_ivf() || self.ivf.is_some() { + return Err(VectorError::InvalidInput { + detail: "IVF-PQ training needs an untrained ivf_pq collection".into(), + }); + } + let live = self.gather_live_vectors(); + let threshold = self.ivf_training_threshold(); + if live.len() < threshold { + return Err(VectorError::InvalidInput { + detail: format!( + "IVF-PQ training needs {threshold} live vectors; the collection holds {}", + live.len() + ), + }); + } + let mut ivf = IvfPqIndex::new(self.dim, self.index_config.to_ivf_params()); + { + let refs: Vec<&[f32]> = live.iter().map(|(_, v)| v.as_slice()).collect(); + ivf.train(&refs, memory)?; + } + for (id, vector) in live { + ivf.insert_with_id(id, vector)?; + } + ivf.set_trained_at_ms(trained_at_ms); + self.clear_segments(); + self.ivf = Some(ivf); + Ok(()) + } + + /// Every live FP32 vector outside the IVF index, keyed by global id. + fn gather_live_vectors(&self) -> Vec<(u32, Vec)> { + let mut out = Vec::with_capacity(self.live_count()); + for seg in &self.sealed { + for local in 0..seg.index.len() as u32 { + if !seg.index.is_deleted(local) + && let Some(v) = sealed_vector(seg, local) + { + out.push((seg.base_id + local, v)); + } + } + } + for seg in &self.building { + for local in 0..seg.flat.len() as u32 { + if let Some(v) = seg.flat.get_vector(local) { + out.push((seg.base_id + local, v.to_vec())); + } + } + } + for local in 0..self.growing.len() as u32 { + if let Some(v) = self.growing.get_vector(local) { + out.push((self.growing_base_id + local, v.to_vec())); + } + } + out + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + use crate::test_support::test_memory; + + const DIM: usize = 8; + + /// An L2 `IvfPq` collection with 4 cells, all probed, and 256 PQ + /// centroids: the threshold is 256 vectors. + fn ivf_collection() -> VectorCollection { + let config = IndexConfig { + hnsw: crate::hnsw::HnswParams { + metric: crate::distance::DistanceMetric::L2, + ..crate::hnsw::HnswParams::default() + }, + index_type: IndexType::IvfPq, + pq_m: 4, + ivf_cells: 4, + ivf_nprobe: 4, + ..IndexConfig::default() + }; + VectorCollection::with_index_config(DIM, config) + } + + /// Distinct per `i`: the first component is `i + 1`. + fn vector(i: usize) -> Vec { + [1, 7, 11, 13, 17, 19, 23, 29] + .iter() + .map(|&m| (if m == 1 { i + 1 } else { i % m + 1 }) as f32) + .collect() + } + + fn all_ids(coll: &VectorCollection, query: &[f32]) -> Vec { + let mut ids: Vec = coll + .search(query, 10_000, 64) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + ids.sort_unstable(); + ids + } + + #[test] + fn below_the_threshold_vectors_wait_in_the_exact_buffer() { + let mut coll = ivf_collection(); + assert_eq!(coll.ivf_training_threshold(), 256); + for i in 0..10 { + coll.insert_with_surrogate(vector(i), Surrogate::new(i as u32 + 1)) + .unwrap(); + } + assert!(!coll.needs_ivf_training()); + assert!(!coll.needs_seal()); + let hit = &coll.search(&vector(3), 1, 64).unwrap()[0]; + assert_eq!(hit.id, 3); + assert_eq!(hit.distance, 0.0, "the buffer is searched exactly"); + } + + #[test] + fn training_moves_every_vector_once_and_later_inserts_follow() { + let mut coll = ivf_collection(); + for i in 0..256 { + coll.insert_with_surrogate(vector(i), Surrogate::new(i as u32 + 1)) + .unwrap(); + } + coll.delete(5); + coll.insert(vector(256)).unwrap(); + assert!(coll.needs_ivf_training()); + let before = all_ids(&coll, &vector(0)); + + coll.train_ivf(test_memory(), 42).unwrap(); + + assert!(!coll.needs_ivf_training()); + assert!(coll.growing_is_empty()); + let ivf = coll.ivf_index().unwrap(); + assert_eq!(ivf.trained_on(), 256); + assert_eq!(ivf.trained_at_ms(), 42); + assert_eq!( + all_ids(&coll, &vector(0)), + before, + "no vector lost or doubled" + ); + + let id = coll + .insert_with_surrogate(vector(300), Surrogate::new(9_999)) + .unwrap(); + assert_eq!(id, 257); + assert!( + coll.growing_is_empty(), + "a trained collection inserts into IVF" + ); + assert_eq!(coll.search(&vector(300), 1, 64).unwrap()[0].id, 257); + assert_eq!( + coll.vector_for_surrogate(Surrogate::new(9_999)), + Some(vector(300)) + ); + assert!(coll.delete_by_surrogate(Surrogate::new(9_999))); + assert!(coll.search(&vector(300), 1, 64).unwrap()[0].id != 257); + } + + #[test] + fn training_below_the_threshold_is_refused() { + let mut coll = ivf_collection(); + coll.insert(vector(0)).unwrap(); + assert!(matches!( + coll.train_ivf(test_memory(), 0), + Err(VectorError::InvalidInput { .. }) + )); + assert!(coll.ivf_index().is_none()); + assert_eq!(coll.live_count(), 1); + } + + #[test] + fn a_trained_collection_survives_a_checkpoint() { + let mut coll = ivf_collection(); + for i in 0..260 { + coll.insert(vector(i)).unwrap(); + } + coll.train_ivf(test_memory(), 7).unwrap(); + coll.insert(vector(400)).unwrap(); + coll.delete(9); + + let bytes = coll.checkpoint_to_bytes(None).unwrap(); + let restored = VectorCollection::from_checkpoint(&bytes, None, test_memory()).unwrap(); + + assert!(restored.is_ivf()); + assert_eq!(restored.ivf_index().unwrap().trained_at_ms(), 7); + assert_eq!(all_ids(&restored, &vector(0)), all_ids(&coll, &vector(0))); + assert_eq!(restored.search(&vector(400), 1, 64).unwrap()[0].id, 260); + } + + #[test] + fn an_untrained_buffer_survives_a_checkpoint_and_trains_after() { + let mut coll = ivf_collection(); + for i in 0..100 { + coll.insert(vector(i)).unwrap(); + } + let bytes = coll.checkpoint_to_bytes(None).unwrap(); + let mut restored = VectorCollection::from_checkpoint(&bytes, None, test_memory()).unwrap(); + assert!(restored.is_ivf()); + assert!(restored.ivf_index().is_none()); + for i in 100..256 { + restored.insert(vector(i)).unwrap(); + } + assert!(restored.needs_ivf_training()); + restored.train_ivf(test_memory(), 1).unwrap(); + assert_eq!( + all_ids(&restored, &vector(0)), + (0..256).collect::>() + ); + } +} diff --git a/nodedb-vector/src/collection/lifecycle.rs b/nodedb-vector/src/collection/lifecycle.rs index ce93c7ecd..8fb7016f8 100644 --- a/nodedb-vector/src/collection/lifecycle.rs +++ b/nodedb-vector/src/collection/lifecycle.rs @@ -22,6 +22,7 @@ use nodedb_types::{Surrogate, VectorQuantization}; use crate::flat::FlatIndex; use crate::hnsw::{HnswIndex, HnswParams}; use crate::index_config::{IndexConfig, IndexType}; +use crate::ivf::IvfPqIndex; use super::codec_dispatch::CollectionCodec; use super::payload_index::PayloadIndexSet; @@ -71,6 +72,12 @@ pub struct VectorCollection { /// Coexists with sealed segments — for codec-dispatched collections the /// per-segment Sq8 builder is skipped and this index is used instead. pub codec_dispatch: Option, + /// Trained IVF-PQ index of an `IvfPq` collection. `None` until the + /// collection holds the training threshold of vectors: until then its + /// vectors wait in the growing segment, searched exactly. Training moves + /// every vector into this index under its global id, and later inserts + /// land here. + pub(crate) ivf: Option, /// Quantization mode requested at collection-creation time. /// /// When `!= None && != Sq8`, each call to `complete_build` additionally @@ -159,6 +166,7 @@ impl VectorCollection { seal_threshold, index_config: config, codec_dispatch: None, + ivf: None, quantization: VectorQuantization::default(), payload: PayloadIndexSet::default(), arena_index: None, @@ -203,14 +211,16 @@ impl VectorCollection { Self::with_seal_threshold(dim, params, DEFAULT_SEAL_THRESHOLD) } - /// Check if the growing segment should be sealed. + /// Check if the growing segment should be sealed. An `IvfPq` collection + /// never seals: its growing segment is the buffer IVF-PQ training reads. pub fn needs_seal(&self) -> bool { - self.growing.len() >= self.seal_threshold + !self.is_ivf() && self.growing.len() >= self.seal_threshold } - /// Seal the growing segment and return a build request. + /// Seal the growing segment and return a build request. `None` for an + /// empty growing segment or an `IvfPq` collection. pub fn seal(&mut self, key: &str) -> Option { - if self.growing.is_empty() { + if self.growing.is_empty() || self.is_ivf() { return None; } @@ -317,7 +327,7 @@ impl VectorCollection { } pub fn len(&self) -> usize { - let mut total = self.growing.len(); + let mut total = self.growing.len() + self.ivf.as_ref().map_or(0, IvfPqIndex::len); for seg in &self.sealed { total += seg.index.len(); } @@ -328,7 +338,8 @@ impl VectorCollection { } pub fn live_count(&self) -> usize { - let mut total = self.growing.live_count(); + let mut total = + self.growing.live_count() + self.ivf.as_ref().map_or(0, IvfPqIndex::live_count); for seg in &self.sealed { total += seg.index.live_count(); } diff --git a/nodedb-vector/src/collection/lifecycle_compact.rs b/nodedb-vector/src/collection/lifecycle_compact.rs index 0343f73a6..55db42cb7 100644 --- a/nodedb-vector/src/collection/lifecycle_compact.rs +++ b/nodedb-vector/src/collection/lifecycle_compact.rs @@ -21,8 +21,24 @@ impl VectorCollection { /// no matching entry in `building` and is ignored, and an mmap file name /// is never reused. The mmap file of each dropped sealed segment is /// removed from disk. + /// + /// An `IvfPq` collection drops its trained index too and buffers again: + /// its next training reads the vectors inserted after the truncate. pub fn truncate(&mut self) -> usize { let dropped = self.live_count(); + self.clear_segments(); + self.ivf = None; + self.surrogate_map.clear(); + self.surrogate_to_local.clear(); + self.multi_doc_map.clear(); + self.payload.clear_rows(); + dropped + } + + /// Empty the growing segment and drop every sealed segment, in-flight + /// build and codec-dispatch index. The next insert keeps the id counter. + /// The mmap file of each dropped sealed segment is removed from disk. + pub(super) fn clear_segments(&mut self) { self.growing = FlatIndex::new(self.dim, self.params.metric); self.growing_base_id = self.next_id; for seg in self.sealed.drain(..) { @@ -41,21 +57,17 @@ impl VectorCollection { } self.building.clear(); self.mmap_segment_count = 0; - self.surrogate_map.clear(); - self.surrogate_to_local.clear(); - self.multi_doc_map.clear(); self.codec_dispatch = None; - self.payload.clear_rows(); - dropped } - /// Compact sealed segments by removing tombstoned nodes. + /// Compact sealed segments and the IVF-PQ index by removing tombstoned + /// nodes. /// /// Rewrites `surrogate_map` and `multi_doc_map` for every sealed /// segment so that global ids continue to resolve to the correct - /// surrogate after local-id renumbering. + /// surrogate after local-id renumbering. IVF-PQ entries keep their ids. pub fn compact(&mut self) -> usize { - let mut total_removed = 0; + let mut total_removed = self.ivf.as_mut().map_or(0, |ivf| ivf.compact()); for seg in &mut self.sealed { let base_id = seg.base_id; let (removed, id_map) = seg.index.compact_with_map(); @@ -125,6 +137,13 @@ impl VectorCollection { pub fn export_snapshot(&self) -> Result, crate::error::VectorError> { let mut result = Vec::new(); + if let Some(ivf) = &self.ivf { + for (vid, data) in ivf.live_vectors() { + let surrogate = self.surrogate_map.get(&vid).copied(); + result.push((vid, data, surrogate)); + } + } + for i in 0..self.growing.len() as u32 { let vid = self.growing_base_id + i; if let Some(data) = self.growing.get_vector(i) { diff --git a/nodedb-vector/src/collection/lifecycle_insert_ops.rs b/nodedb-vector/src/collection/lifecycle_insert_ops.rs index 886d1c78a..3acd15324 100644 --- a/nodedb-vector/src/collection/lifecycle_insert_ops.rs +++ b/nodedb-vector/src/collection/lifecycle_insert_ops.rs @@ -8,14 +8,21 @@ use super::lifecycle::VectorCollection; use crate::error::{VectorError, check_dim}; impl VectorCollection { - /// Insert a vector. Returns the global vector ID. + /// Insert a vector. Returns the global vector ID. A trained IVF-PQ + /// collection inserts into its IVF index, every other collection into + /// the growing segment. /// /// A vector without the collection dimension fails with /// [`VectorError::DimensionMismatch`] and changes nothing. pub fn insert(&mut self, vector: Vec) -> Result { check_dim(self.dim, vector.len())?; let id = self.next_id; - self.growing.insert(vector)?; + match &mut self.ivf { + Some(ivf) => ivf.insert_with_id(id, vector)?, + None => { + self.growing.insert(vector)?; + } + } self.next_id += 1; Ok(id) } @@ -139,6 +146,11 @@ impl VectorCollection { } pub(super) fn delete_inner(&mut self, id: u32) -> bool { + if let Some(ivf) = &mut self.ivf + && ivf.contains(id) + { + return ivf.delete(id); + } if id >= self.growing_base_id { let local = id - self.growing_base_id; if (local as usize) < self.growing.len() { @@ -167,6 +179,11 @@ impl VectorCollection { /// The live FP32 vector stored under global `id`, whichever segment /// holds it. `None` for an unknown or soft-deleted id. pub fn vector_for_id(&self, id: u32) -> Option> { + if let Some(ivf) = &self.ivf + && ivf.contains(id) + { + return ivf.get_vector(id).map(<[f32]>::to_vec); + } if id >= self.growing_base_id { let local = id - self.growing_base_id; if (local as usize) < self.growing.len() { @@ -211,13 +228,16 @@ impl VectorCollection { /// Un-delete a previously soft-deleted vector (for transaction rollback). /// - /// Symmetric to [`Self::delete_inner`]: the vector may live in the growing - /// segment (the common case for a just-inserted vector), a sealed HNSW - /// segment, or an in-flight building segment — reverse the tombstone - /// wherever it landed. Only clearing sealed tombstones (the prior behavior) - /// silently failed to restore growing/building vectors, leaving a - /// rolled-back delete permanently unsearchable. + /// Symmetric to [`Self::delete_inner`]: the vector may live in the IVF + /// index, the growing segment (the common case for a just-inserted + /// vector), a sealed HNSW segment, or an in-flight building segment. The + /// tombstone is reversed wherever it landed. pub fn undelete(&mut self, id: u32) -> bool { + if let Some(ivf) = &mut self.ivf + && ivf.contains(id) + { + return ivf.undelete(id); + } if id >= self.growing_base_id { let local = id - self.growing_base_id; if (local as usize) < self.growing.len() { @@ -247,7 +267,7 @@ impl VectorCollection { /// The FP32 vector at `local` in a sealed segment: the mmap tier when the /// segment lives there, else the HNSW node (decoded from a narrow dtype or /// fetched from the segment backing when the node holds no local copy). -fn sealed_vector(seg: &super::segment::SealedSegment, local: u32) -> Option> { +pub(super) fn sealed_vector(seg: &super::segment::SealedSegment, local: u32) -> Option> { if let Some(mmap) = &seg.mmap_vectors { return mmap.get_vector(local).map(<[f32]>::to_vec); } diff --git a/nodedb-vector/src/collection/mod.rs b/nodedb-vector/src/collection/mod.rs index 28d8b3958..f407bf83d 100644 --- a/nodedb-vector/src/collection/mod.rs +++ b/nodedb-vector/src/collection/mod.rs @@ -4,6 +4,7 @@ pub mod budget; pub mod checkpoint; pub mod codec_build; pub mod codec_dispatch; +pub mod ivf_mode; pub mod lifecycle; pub mod lifecycle_compact; pub mod lifecycle_insert_ops; diff --git a/nodedb-vector/src/collection/rollback.rs b/nodedb-vector/src/collection/rollback.rs index 8da0ae116..7a574d019 100644 --- a/nodedb-vector/src/collection/rollback.rs +++ b/nodedb-vector/src/collection/rollback.rs @@ -10,9 +10,10 @@ //! would have taken without the write. Then it puts every binding and every //! tombstone back. //! -//! The inserted nodes must still sit in the growing segment: a seal between -//! the mark and the rollback moves them out, and the rollback then reports -//! that it cannot restore the mark. +//! The inserted nodes must still sit in the growing segment or the IVF-PQ +//! index: a seal or an IVF-PQ training between the mark and the rollback +//! moves them out, and the rollback then reports that it cannot restore the +//! mark. use nodedb_types::Surrogate; @@ -36,6 +37,11 @@ impl VectorCollection { /// Whether node `id` exists and is not soft-deleted, whichever segment /// holds it. pub fn is_live(&self, id: u32) -> bool { + if let Some(ivf) = &self.ivf + && ivf.contains(id) + { + return !ivf.is_deleted(id); + } if id >= self.growing_base_id { let local = id - self.growing_base_id; if (local as usize) < self.growing.len() { @@ -99,8 +105,8 @@ impl VectorCollection { } /// Put the collection back to `mark`. Returns `false`, changing nothing, - /// when a seal moved the nodes inserted since the mark out of the growing - /// segment. + /// when a seal or an IVF-PQ training moved the nodes inserted since the + /// mark out of the growing segment. pub fn roll_back_to(&mut self, mark: VectorWriteMark) -> bool { if self.growing_base_id != mark.growing_base_id || mark.next_id < self.growing_base_id { return false; @@ -114,6 +120,9 @@ impl VectorCollection { } self.growing .truncate((mark.next_id - self.growing_base_id) as usize); + if let Some(ivf) = &mut self.ivf { + ivf.roll_back_to(mark.next_id); + } self.next_id = mark.next_id; for (surrogate, prior) in mark.bindings { diff --git a/nodedb-vector/src/collection/search.rs b/nodedb-vector/src/collection/search.rs index 57474a2a5..ecff5578f 100644 --- a/nodedb-vector/src/collection/search.rs +++ b/nodedb-vector/src/collection/search.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: Apache-2.0 -//! VectorCollection search: multi-segment merging with SQ8 reranking. +//! VectorCollection search: multi-segment merging with SQ8 reranking. A +//! trained IVF-PQ index answers beside the segments under global ids. //! //! The `search_with_payload_filter` method wires payload bitmap pre-filtering //! into the search path. When all referenced fields in the predicate are @@ -12,6 +13,7 @@ use crate::distance::{DistanceMetric, distance}; use crate::error::{VectorError, check_dim}; use crate::hnsw::SearchResult; +use crate::hnsw::search::decode_filter_bitmap; use super::lifecycle::VectorCollection; use super::payload_index::FilterPredicate; @@ -194,6 +196,9 @@ impl VectorCollection { push_shifted(&mut all, results, seg.base_id); } } + if let Some(ivf) = &self.ivf { + all.extend(ivf.search(query, top_k)?); + } push_shifted( &mut all, @@ -239,6 +244,9 @@ impl VectorCollection { push_shifted(&mut all, results, seg.base_id); } } + if let Some(ivf) = &self.ivf { + all.extend(ivf.search_with(query, top_k, metric, None)?); + } push_shifted( &mut all, @@ -270,6 +278,10 @@ impl VectorCollection { check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); + if let Some(ivf) = &self.ivf { + let filter = decode_filter_bitmap(bitmap)?; + all.extend(ivf.search_with(query, top_k, metric, Some(&filter))?); + } push_shifted( &mut all, self.growing.search_filtered_offset_with_metric( @@ -325,6 +337,10 @@ impl VectorCollection { check_dim(self.dim, query.len())?; let mut all: Vec = Vec::new(); + if let Some(ivf) = &self.ivf { + let filter = decode_filter_bitmap(bitmap)?; + all.extend(ivf.search_with(query, top_k, self.params.metric, Some(&filter))?); + } push_shifted( &mut all, self.growing diff --git a/nodedb-vector/src/collection/stats.rs b/nodedb-vector/src/collection/stats.rs index 226b51c9e..e36f37458 100644 --- a/nodedb-vector/src/collection/stats.rs +++ b/nodedb-vector/src/collection/stats.rs @@ -13,11 +13,13 @@ impl VectorCollection { let sealed_vectors: usize = self.sealed.iter().map(|s| s.index.len()).sum(); let building_vectors: usize = self.building.iter().map(|s| s.flat.len()).sum(); - let tombstone_count: usize = self - .sealed - .iter() - .map(|s| s.index.tombstone_count()) - .sum::() + let ivf_vectors = self.ivf.as_ref().map_or(0, |ivf| ivf.len()); + let tombstone_count: usize = self.ivf.as_ref().map_or(0, |ivf| ivf.tombstone_count()) + + self + .sealed + .iter() + .map(|s| s.index.tombstone_count()) + .sum::() + self.growing.tombstone_count() + self .building @@ -25,7 +27,7 @@ impl VectorCollection { .map(|s| s.flat.tombstone_count()) .sum::(); - let total = growing_vectors + sealed_vectors + building_vectors; + let total = growing_vectors + sealed_vectors + building_vectors + ivf_vectors; let tombstone_ratio = if total > 0 { tombstone_count as f64 / total as f64 } else { @@ -38,7 +40,7 @@ impl VectorCollection { "bbq" => nodedb_types::VectorIndexQuantization::Bbq, _ => nodedb_types::VectorIndexQuantization::None, } - } else if self.sealed.iter().any(|s| s.pq.is_some()) { + } else if self.ivf.is_some() || self.sealed.iter().any(|s| s.pq.is_some()) { nodedb_types::VectorIndexQuantization::Pq } else if self.sealed.iter().any(|s| s.sq8.is_some()) { nodedb_types::VectorIndexQuantization::Sq8 @@ -64,7 +66,8 @@ impl VectorCollection { .sum(); let growing_mem = growing_vectors * self.dim * std::mem::size_of::(); let building_mem = building_vectors * self.dim * std::mem::size_of::(); - let memory_bytes = hnsw_mem + sq8_mem + growing_mem + building_mem; + let ivf_mem = self.ivf.as_ref().map_or(0, |ivf| ivf.memory_bytes()); + let memory_bytes = hnsw_mem + sq8_mem + growing_mem + building_mem + ivf_mem; let disk_bytes: usize = self .sealed @@ -99,6 +102,15 @@ impl VectorCollection { // is always `None` here; callers overwrite it after calling // `stats()` when a dedicated arena handle is available. arena_bytes: None, + ivf: self.is_ivf().then(|| nodedb_types::VectorIvfStats { + training_threshold: self.ivf_training_threshold(), + trained: self.ivf.is_some(), + trained_on: self.ivf.as_ref().map_or(0, |ivf| ivf.trained_on()), + trained_at_ms: self.ivf.as_ref().map_or(0, |ivf| ivf.trained_at_ms()), + indexed_vectors: ivf_vectors, + cells: self.ivf.as_ref().map_or(0, |ivf| ivf.n_cells()), + nprobe: self.index_config.ivf_nprobe, + }), } } } diff --git a/nodedb-vector/src/index_config.rs b/nodedb-vector/src/index_config.rs index 90f4ab6b3..483b268c1 100644 --- a/nodedb-vector/src/index_config.rs +++ b/nodedb-vector/src/index_config.rs @@ -23,7 +23,9 @@ pub enum IndexType { Hnsw, /// HNSW graph with PQ-compressed storage for traversal. HnswPq, - /// IVF-PQ flat index. Lowest memory (~16 bytes/vector), best for >10M vectors. + /// IVF-PQ: vectors buffer, searched exactly, until `max(ivf_cells, pq_k)` + /// are held, then train k-means cells and PQ codebooks. A search probes + /// `ivf_nprobe` cells by PQ distance and reranks by exact distance. IvfPq, } diff --git a/nodedb-vector/src/ivf/checkpoint.rs b/nodedb-vector/src/ivf/checkpoint.rs new file mode 100644 index 000000000..296fb4965 --- /dev/null +++ b/nodedb-vector/src/ivf/checkpoint.rs @@ -0,0 +1,165 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Checkpoint encoding for `IvfPqIndex`: centroids, the PQ codec, every +//! cell's ids, codes and FP32 vectors, the tombstones, and the training stamp. +//! Codes are stored, never recomputed on load. + +use std::collections::HashMap; + +use nodedb_mem::ScopedMemory; +use roaring::RoaringBitmap; + +use crate::error::VectorError; +use crate::quantize::pq::PqCodec; + +use super::index::{IvfCell, IvfPqIndex}; +use super::params::IvfPqParams; + +#[derive(zerompk::ToMessagePack, zerompk::FromMessagePack)] +struct IvfSnapshot { + dim: usize, + params: IvfPqParams, + centroids: Vec>, + pq_bytes: Option>, + cells: Vec, + deleted: Vec, + next_id: u32, + trained_on: usize, + trained_at_ms: u64, +} + +fn corrupt(detail: String) -> VectorError { + VectorError::CheckpointDeserializationError { detail } +} + +impl IvfPqIndex { + /// Encode the whole index as MessagePack. + pub fn to_bytes(&self) -> Result, VectorError> { + let snapshot = IvfSnapshot { + dim: self.dim, + params: self.params.clone(), + centroids: self.centroids.clone(), + pq_bytes: self.pq.as_ref().map(PqCodec::to_bytes).transpose()?, + cells: self.cells.clone(), + deleted: self.deleted.iter().collect(), + next_id: self.next_id, + trained_on: self.trained_on, + trained_at_ms: self.trained_at_ms, + }; + zerompk::to_msgpack_vec(&snapshot).map_err(|e| VectorError::CheckpointSerializationError { + detail: format!("IVF-PQ index encode: {e}"), + }) + } + + /// Decode an index written by [`Self::to_bytes`], charging the PQ codec + /// to `memory`. + /// + /// Fails with [`VectorError::CheckpointDeserializationError`] when the + /// bytes do not decode or the decoded cells disagree with the dimension, + /// the PQ code width, or the centroid count. + pub fn from_bytes(bytes: &[u8], memory: ScopedMemory) -> Result { + let snap: IvfSnapshot = zerompk::from_msgpack(bytes) + .map_err(|e| corrupt(format!("IVF-PQ index decode: {e}")))?; + let pq = snap + .pq_bytes + .as_deref() + .map(|b| PqCodec::from_bytes(b, memory)) + .transpose() + .map_err(|e| corrupt(format!("IVF-PQ codec decode: {e}")))?; + let m = pq.as_ref().map_or(0, |pq| pq.m); + if snap.cells.len() != snap.centroids.len() { + return Err(corrupt(format!( + "IVF-PQ index has {} cells for {} centroids", + snap.cells.len(), + snap.centroids.len() + ))); + } + let mut slots = HashMap::new(); + for (cell_idx, cell) in snap.cells.iter().enumerate() { + let n = cell.ids.len(); + if cell.codes.len() != n * m || cell.vectors.len() != n * snap.dim { + return Err(corrupt(format!( + "IVF-PQ cell {cell_idx} holds {n} ids, {} code bytes and {} vector \ + components; expected {} and {}", + cell.codes.len(), + cell.vectors.len(), + n * m, + n * snap.dim + ))); + } + for (pos, &id) in cell.ids.iter().enumerate() { + if slots.insert(id, (cell_idx as u32, pos as u32)).is_some() { + return Err(corrupt(format!("IVF-PQ index holds vector id {id} twice"))); + } + } + } + let deleted: RoaringBitmap = snap.deleted.into_iter().collect(); + if let Some(id) = deleted.iter().find(|id| !slots.contains_key(id)) { + return Err(corrupt(format!( + "IVF-PQ index tombstones vector id {id} it does not hold" + ))); + } + Ok(Self { + dim: snap.dim, + params: snap.params, + centroids: snap.centroids, + pq, + cells: snap.cells, + slots, + deleted, + next_id: snap.next_id, + trained_on: snap.trained_on, + trained_at_ms: snap.trained_at_ms, + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::distance::DistanceMetric; + use crate::test_support::test_memory; + + #[test] + fn round_trip_keeps_entries_tombstones_and_training() { + let vecs: Vec> = (0..32) + .map(|i| (0..8).map(|d| ((i * 8 + d) as f32) * 0.01).collect()) + .collect(); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new( + 8, + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 4, + metric: DistanceMetric::L2, + }, + ); + idx.train(&refs, test_memory()).unwrap(); + for (i, v) in vecs.iter().enumerate() { + idx.insert_with_id(10 + i as u32, v.clone()).unwrap(); + } + idx.delete(12); + idx.set_trained_at_ms(1_700_000_000_000); + + let restored = IvfPqIndex::from_bytes(&idx.to_bytes().unwrap(), test_memory()).unwrap(); + assert_eq!(restored.len(), 32); + assert!(restored.is_deleted(12)); + assert_eq!(restored.trained_on(), 32); + assert_eq!(restored.trained_at_ms(), 1_700_000_000_000); + assert_eq!(restored.get_vector(20), Some(vecs[10].as_slice())); + assert_eq!( + restored.search(&vecs[5], 3).unwrap()[0].id, + idx.search(&vecs[5], 3).unwrap()[0].id + ); + } + + #[test] + fn garbage_is_a_typed_error() { + assert!(matches!( + IvfPqIndex::from_bytes(b"not an index", test_memory()), + Err(VectorError::CheckpointDeserializationError { .. }) + )); + } +} diff --git a/nodedb-vector/src/ivf/index.rs b/nodedb-vector/src/ivf/index.rs index 2d30272fb..e52c57ac8 100644 --- a/nodedb-vector/src/ivf/index.rs +++ b/nodedb-vector/src/ivf/index.rs @@ -1,60 +1,64 @@ // SPDX-License-Identifier: Apache-2.0 -//! IVF-PQ index for billion-scale datasets. +//! IVF-PQ index: inverted file with product quantization. //! -//! Inverted File with Product Quantization: partition vectors into Voronoi -//! cells using k-means centroids, PQ-compress within cells. +//! k-means centroids partition the vectors into Voronoi cells. Each entry +//! holds a PQ code of its residual against the cell centroid, plus the FP32 +//! vector. A search probes the nearest cells, ranks their entries by PQ +//! distance, and reranks the best of them by exact distance. +//! +//! Entries carry caller-assigned ids, so a collection can move its vectors +//! into the index under the ids they already have. + +use std::collections::HashMap; use nodedb_mem::ScopedMemory; +use roaring::RoaringBitmap; -use crate::distance::{DistanceMetric, distance}; +use crate::distance::distance; use crate::error::{VectorError, check_dim}; -use crate::hnsw::SearchResult; use crate::quantize::pq::PqCodec; -/// IVF-PQ index configuration. -#[derive(Clone)] -pub struct IvfPqParams { - /// Number of Voronoi cells (partitions). Typical: sqrt(N). - pub n_cells: usize, - /// Number of PQ subvectors. Must divide dimension evenly. - pub pq_m: usize, - /// Centroids per PQ subvector (fixed at 256 for u8 encoding). - pub pq_k: usize, - /// Number of cells to probe at query time. Higher = better recall. - pub nprobe: usize, - /// Distance metric. - pub metric: DistanceMetric, -} +use super::kmeans::kmeans_centroids; +use super::params::IvfPqParams; -impl Default for IvfPqParams { - fn default() -> Self { - Self { - n_cells: 256, - pq_m: 8, - pq_k: 256, - nprobe: 16, - metric: DistanceMetric::L2, - } - } +/// k-means iterations for the coarse centroids and the PQ codebooks. +pub(super) const TRAIN_ITERATIONS: usize = 20; + +/// The entries of one Voronoi cell, stored column-wise. Entry `i` has id +/// `ids[i]`, PQ code `codes[i * m .. (i + 1) * m]`, and FP32 vector +/// `vectors[i * dim .. (i + 1) * dim]`. +#[derive(Debug, Clone, Default, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct IvfCell { + pub(super) ids: Vec, + pub(super) codes: Vec, + pub(super) vectors: Vec, } /// IVF-PQ index: inverted file with product quantization. pub struct IvfPqIndex { - dim: usize, - params: IvfPqParams, + pub(super) dim: usize, + pub(super) params: IvfPqParams, /// Coarse centroids: `n_cells` × `dim` FP32 vectors. - centroids: Vec>, - /// PQ codec trained on the dataset. - pq: Option, - /// Per-cell inverted lists: `cells[cell_id]` = list of (vector_id, pq_code). - cells: Vec)>>, - /// Total vectors indexed. - count: u32, + pub(super) centroids: Vec>, + /// PQ codec trained on the residuals. `None` until [`Self::train`]. + pub(super) pq: Option, + pub(super) cells: Vec, + /// Entry id → (cell, position in the cell). + pub(super) slots: HashMap, + /// Soft-deleted entry ids. + pub(super) deleted: RoaringBitmap, + /// Id [`Self::add`] assigns next: one past the highest id ever held. + pub(super) next_id: u32, + /// Vectors the codebooks were trained on. + pub(super) trained_on: usize, + /// Unix milliseconds of the training, as the caller stamped it. `0` when + /// never stamped. + pub(super) trained_at_ms: u64, } impl IvfPqIndex { - /// Create an empty IVF-PQ index. + /// Create an empty, untrained IVF-PQ index. pub fn new(dim: usize, params: IvfPqParams) -> Self { Self { dim, @@ -62,22 +66,36 @@ impl IvfPqIndex { centroids: Vec::new(), pq: None, cells: Vec::new(), - count: 0, + slots: HashMap::new(), + deleted: RoaringBitmap::new(), + next_id: 0, + trained_on: 0, + trained_at_ms: 0, } } - /// Train the index from a set of vectors, tracking PQ codebook - /// allocations against `memory`. + /// Train the coarse centroids and the PQ codebooks on `vectors`, tracking + /// PQ codebook allocations against `memory`. /// - /// An empty set, a zero dimension, or a `pq_m` that does not divide the - /// dimension fails with [`VectorError::InvalidInput`]; a vector without - /// the index dimension fails with [`VectorError::DimensionMismatch`]. + /// Fails with [`VectorError::InvalidInput`] when the set is empty, the + /// dimension is zero or not divisible by `pq_m`, the set is smaller than + /// `pq_k`, or the index already holds vectors (their codes would not + /// match new codebooks). A vector without the index dimension fails with + /// [`VectorError::DimensionMismatch`]. A failed training changes nothing. pub fn train(&mut self, vectors: &[&[f32]], memory: ScopedMemory) -> Result<(), VectorError> { if vectors.is_empty() { return Err(VectorError::InvalidInput { detail: "IVF-PQ training needs at least one vector".into(), }); } + if !self.slots.is_empty() { + return Err(VectorError::InvalidInput { + detail: format!( + "IVF-PQ index already holds {} vectors; train an empty index", + self.slots.len() + ), + }); + } if self.dim == 0 || self.params.pq_m == 0 || !self.dim.is_multiple_of(self.params.pq_m) { return Err(VectorError::InvalidInput { detail: format!( @@ -91,165 +109,188 @@ impl IvfPqIndex { } let n_cells = self.params.n_cells.min(vectors.len()); - self.centroids = kmeans_centroids(vectors, self.dim, n_cells, 20); - self.cells = vec![Vec::new(); self.centroids.len()]; - - let mut residuals: Vec> = Vec::with_capacity(vectors.len()); - for v in vectors { - let cell = self.nearest_centroid(v); - let res: Vec = v - .iter() - .zip(&self.centroids[cell]) - .map(|(a, b)| a - b) - .collect(); - residuals.push(res); - } + let centroids = kmeans_centroids(vectors, self.dim, n_cells, TRAIN_ITERATIONS); + let residuals: Vec> = vectors + .iter() + .map(|v| residual(v, ¢roids[nearest(¢roids, v, &self.params)])) + .collect(); let res_refs: Vec<&[f32]> = residuals.iter().map(|r| r.as_slice()).collect(); - self.pq = Some(PqCodec::train( + let pq = PqCodec::train( &res_refs, self.dim, self.params.pq_m, self.params.pq_k, - 20, + TRAIN_ITERATIONS, memory, - )?); + )?; + + self.cells = vec![IvfCell::default(); centroids.len()]; + self.centroids = centroids; + self.pq = Some(pq); + self.trained_on = vectors.len(); Ok(()) } - /// Add a vector to the index. Returns the assigned ID. + /// Add `vector` under the caller-assigned `id`. /// - /// A vector without the index dimension fails with - /// [`VectorError::DimensionMismatch`]; an untrained index fails with - /// [`VectorError::InvalidInput`]. - pub fn add(&mut self, vector: &[f32]) -> Result { + /// Fails with [`VectorError::DimensionMismatch`] for a vector without the + /// index dimension, and with [`VectorError::InvalidInput`] when the index + /// is untrained or already holds `id`. A failed add changes nothing. + pub fn insert_with_id(&mut self, id: u32, vector: Vec) -> Result<(), VectorError> { check_dim(self.dim, vector.len())?; let Some(pq) = self.pq.as_ref() else { return Err(VectorError::InvalidInput { detail: "IVF-PQ index must be trained before add".into(), }); }; + if self.slots.contains_key(&id) { + return Err(VectorError::InvalidInput { + detail: format!("IVF-PQ index already holds vector id {id}"), + }); + } + let cell_idx = nearest(&self.centroids, &vector, &self.params); + let code = pq.encode(&residual(&vector, &self.centroids[cell_idx])); + let cell = &mut self.cells[cell_idx]; + let pos = cell.ids.len() as u32; + cell.ids.push(id); + cell.codes.extend_from_slice(&code); + cell.vectors.extend_from_slice(&vector); + self.slots.insert(id, (cell_idx as u32, pos)); + self.next_id = self.next_id.max(id.saturating_add(1)); + Ok(()) + } - let cell = self.nearest_centroid(vector); - let residual: Vec = vector - .iter() - .zip(&self.centroids[cell]) - .map(|(a, b)| a - b) - .collect(); - let code = pq.encode(&residual); - let id = self.count; - self.cells[cell].push((id, code)); - self.count += 1; + /// Add a vector under the next free id. Returns the id. + /// + /// Fails as [`Self::insert_with_id`] does. + pub fn add(&mut self, vector: &[f32]) -> Result { + let id = self.next_id; + self.insert_with_id(id, vector.to_vec())?; Ok(id) } + /// Add vectors under consecutive free ids. Stops at the first vector + /// that fails to add. + pub fn add_batch(&mut self, vectors: &[&[f32]]) -> Result<(), VectorError> { + for v in vectors { + self.add(v)?; + } + Ok(()) + } + /// Whether the index holds a trained codebook. pub fn is_trained(&self) -> bool { self.pq.is_some() } - /// Drop every vector added with id `count` or later. When `trained` is - /// `false` the training goes too, as the index held before its first - /// add. A rollback uses it to withdraw the newest adds. - pub fn roll_back_to(&mut self, count: u32, trained: bool) { - if !trained { - self.centroids.clear(); - self.pq = None; - self.cells.clear(); - self.count = 0; - return; - } - for cell in &mut self.cells { - cell.retain(|(id, _)| *id < count); - } - self.count = self.count.min(count); + /// Whether the index holds `id`, live or soft-deleted. + pub fn contains(&self, id: u32) -> bool { + self.slots.contains_key(&id) } - /// Batch add vectors. Stops at the first vector that fails to add. - pub fn add_batch(&mut self, vectors: &[&[f32]]) -> Result<(), VectorError> { - for v in vectors { - self.add(v)?; - } - Ok(()) + /// Whether `id` is held and soft-deleted. + pub fn is_deleted(&self, id: u32) -> bool { + self.deleted.contains(id) } - /// Search: find top-k nearest neighbors. - /// - /// A query without the index dimension fails with - /// [`VectorError::DimensionMismatch`]. A distance table that exceeds the - /// memory budget fails the search: skipping its cell would drop results. - pub fn search(&self, query: &[f32], top_k: usize) -> Result, VectorError> { - check_dim(self.dim, query.len())?; - if self.centroids.is_empty() || self.count == 0 { - return Ok(Vec::new()); - } + /// Soft-delete `id`. `false` when the index does not hold it live. + pub fn delete(&mut self, id: u32) -> bool { + self.contains(id) && self.deleted.insert(id) + } - let Some(pq) = &self.pq else { - return Ok(Vec::new()); - }; + /// Reverse a soft delete of `id`. `false` when `id` was not deleted. + pub fn undelete(&mut self, id: u32) -> bool { + self.deleted.remove(id) + } - let nprobe = self.params.nprobe.min(self.centroids.len()); - let mut centroid_dists: Vec<(usize, f32)> = self - .centroids - .iter() - .enumerate() - .map(|(i, c)| (i, distance(query, c, self.params.metric))) - .collect(); - centroid_dists.sort_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)); - - let mut candidates: Vec = Vec::new(); - - for &(cell_idx, _) in centroid_dists.iter().take(nprobe) { - let residual_query: Vec = query - .iter() - .zip(&self.centroids[cell_idx]) - .map(|(q, c)| q - c) - .collect(); - let table = pq.build_distance_table(&residual_query)?; - - for (id, code) in &self.cells[cell_idx] { - let dist = pq.asymmetric_distance(&table, code); - candidates.push(SearchResult { - id: *id, - distance: dist, - }); - } + /// The FP32 vector of a live `id`. + pub fn get_vector(&self, id: u32) -> Option<&[f32]> { + if self.deleted.contains(id) { + return None; } + let &(cell, pos) = self.slots.get(&id)?; + let start = pos as usize * self.dim; + self.cells + .get(cell as usize)? + .vectors + .get(start..start + self.dim) + } - if candidates.len() > top_k { - candidates.select_nth_unstable_by(top_k, |a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - candidates.truncate(top_k); + /// Every live entry as `(id, vector)`, ordered by id. + pub fn live_vectors(&self) -> Vec<(u32, Vec)> { + let mut out: Vec<(u32, Vec)> = Vec::with_capacity(self.live_count()); + for cell in &self.cells { + for (pos, &id) in cell.ids.iter().enumerate() { + if !self.deleted.contains(id) { + let start = pos * self.dim; + out.push((id, cell.vectors[start..start + self.dim].to_vec())); + } + } } - candidates.sort_by(|a, b| { - a.distance - .partial_cmp(&b.distance) - .unwrap_or(std::cmp::Ordering::Equal) - }); - Ok(candidates) - } - - fn nearest_centroid(&self, vector: &[f32]) -> usize { - let mut best = 0; - let mut best_dist = f32::MAX; - for (i, c) in self.centroids.iter().enumerate() { - let d = distance(vector, c, self.params.metric); - if d < best_dist { - best_dist = d; - best = i; + out.sort_unstable_by_key(|(id, _)| *id); + out + } + + /// Drop every entry with id `next_id` or later, so the next + /// [`Self::add`] takes `next_id` again. The training stays. A rollback + /// uses it to withdraw the newest adds. + pub fn roll_back_to(&mut self, next_id: u32) { + self.retain_entries(|id| id < next_id); + self.deleted.remove_range(next_id..); + self.next_id = self.next_id.min(next_id); + } + + /// Remove every soft-deleted entry. Returns the number removed. Ids of + /// the remaining entries do not change. + pub fn compact(&mut self) -> usize { + let deleted = std::mem::take(&mut self.deleted); + self.retain_entries(|id| !deleted.contains(id)) + } + + /// Keep the entries whose id satisfies `keep` and rebuild the slot map. + /// Returns the number of entries dropped. + fn retain_entries(&mut self, keep: impl Fn(u32) -> bool) -> usize { + let m = self.pq.as_ref().map_or(0, |pq| pq.m); + let dim = self.dim; + let mut removed = 0; + self.slots.clear(); + for (cell_idx, cell) in self.cells.iter_mut().enumerate() { + let mut kept = IvfCell::default(); + for (pos, &id) in cell.ids.iter().enumerate() { + if !keep(id) { + removed += 1; + continue; + } + self.slots + .insert(id, (cell_idx as u32, kept.ids.len() as u32)); + kept.ids.push(id); + kept.codes + .extend_from_slice(&cell.codes[pos * m..(pos + 1) * m]); + kept.vectors + .extend_from_slice(&cell.vectors[pos * dim..(pos + 1) * dim]); } + *cell = kept; } - best + removed } + /// Entries held, live or soft-deleted. pub fn len(&self) -> usize { - self.count as usize + self.slots.len() + } + + /// Entries held and not soft-deleted. + pub fn live_count(&self) -> usize { + self.slots.len() - self.deleted.len() as usize + } + + /// Soft-deleted entries held. + pub fn tombstone_count(&self) -> usize { + self.deleted.len() as usize } pub fn is_empty(&self) -> bool { - self.count == 0 + self.slots.is_empty() } pub fn dim(&self) -> usize { @@ -259,98 +300,67 @@ impl IvfPqIndex { pub fn n_cells(&self) -> usize { self.centroids.len() } -} -fn kmeans_centroids(data: &[&[f32]], dim: usize, k: usize, max_iter: usize) -> Vec> { - let n = data.len(); - let k = k.min(n); - if k == 0 { - return Vec::new(); + pub fn params(&self) -> &IvfPqParams { + &self.params } - let mut centroids: Vec> = vec![data[0].to_vec()]; - let mut min_dists = vec![f32::MAX; n]; + /// Vectors the codebooks were trained on. `0` when untrained. + pub fn trained_on(&self) -> usize { + self.trained_on + } - // Initialize min_dists against the first centroid. - for (i, point) in data.iter().enumerate() { - let d = distance(point, ¢roids[0], DistanceMetric::L2); - if d < min_dists[i] { - min_dists[i] = d; - } + /// Unix milliseconds the caller stamped the training with. + pub fn trained_at_ms(&self) -> u64 { + self.trained_at_ms } - let mut rng = crate::hnsw::Xorshift64::new(0xC0FF_EEDE_ADBE_EF42); - for _ in 1..k { - let total: f64 = min_dists.iter().map(|&d| d as f64).sum(); - let next_idx = if total < f64::EPSILON { - 0 - } else { - let target = rng.next_f64() * total; - let mut acc = 0.0f64; - let mut chosen = n - 1; - for (i, &d) in min_dists.iter().enumerate() { - acc += d as f64; - if acc >= target { - chosen = i; - break; - } - } - chosen - }; - let last = data[next_idx]; - centroids.push(last.to_vec()); - for (i, point) in data.iter().enumerate() { - let d = distance(point, last, DistanceMetric::L2); - if d < min_dists[i] { - min_dists[i] = d; - } - } + /// Stamp the training time, in Unix milliseconds. + pub fn set_trained_at_ms(&mut self, ms: u64) { + self.trained_at_ms = ms; } - let mut assignments = vec![0usize; n]; - for _ in 0..max_iter { - let mut changed = false; - for (i, point) in data.iter().enumerate() { - let mut best = 0; - let mut best_d = f32::MAX; - for (c, centroid) in centroids.iter().enumerate() { - let d = distance(point, centroid, DistanceMetric::L2); - if d < best_d { - best_d = d; - best = c; - } - } - if assignments[i] != best { - assignments[i] = best; - changed = true; - } - } - if !changed { - break; - } - let mut sums = vec![vec![0.0f32; dim]; k]; - let mut counts = vec![0usize; k]; - for (i, point) in data.iter().enumerate() { - let c = assignments[i]; - counts[c] += 1; - for d in 0..dim { - sums[c][d] += point[d]; - } - } - for c in 0..k { - if counts[c] > 0 { - for d in 0..dim { - centroids[c][d] = sums[c][d] / counts[c] as f32; - } - } + /// Approximate heap bytes of the centroids, codes and vectors. + pub fn memory_bytes(&self) -> usize { + let f32_size = std::mem::size_of::(); + let centroids = self.centroids.len() * self.dim * f32_size; + let entries: usize = self + .cells + .iter() + .map(|c| { + c.ids.len() * std::mem::size_of::() + + c.codes.len() + + c.vectors.len() * f32_size + }) + .sum(); + centroids + entries + } +} + +/// Index of the centroid in `centroids` nearest to `vector`. `0` when there +/// are none. +fn nearest(centroids: &[Vec], vector: &[f32], params: &IvfPqParams) -> usize { + let mut best = 0; + let mut best_dist = f32::MAX; + for (i, c) in centroids.iter().enumerate() { + let d = distance(vector, c, params.metric); + if d < best_dist { + best_dist = d; + best = i; } } - centroids + best +} + +/// `vector - centroid`, component-wise. +pub(super) fn residual(vector: &[f32], centroid: &[f32]) -> Vec { + vector.iter().zip(centroid).map(|(a, b)| a - b).collect() } #[cfg(test)] mod tests { use super::*; + use crate::distance::DistanceMetric; use crate::test_support::test_memory; fn make_vectors(n: usize, dim: usize) -> Vec> { @@ -359,35 +369,6 @@ mod tests { .collect() } - #[test] - fn train_and_search() { - let vecs = make_vectors(1000, 16); - let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - - let mut idx = IvfPqIndex::new( - 16, - IvfPqParams { - n_cells: 32, - pq_m: 4, - pq_k: 32, - nprobe: 8, - metric: DistanceMetric::L2, - }, - ); - idx.train(&refs, test_memory()).unwrap(); - idx.add_batch(&refs).unwrap(); - - assert_eq!(idx.len(), 1000); - - let query = &vecs[500]; - let results = idx.search(query, 5).unwrap(); - assert_eq!(results.len(), 5); - assert!( - results.iter().any(|r| r.id == 500), - "exact match not found in top-5" - ); - } - fn small_params() -> IvfPqParams { IvfPqParams { n_cells: 4, @@ -398,12 +379,18 @@ mod tests { } } + fn trained(vecs: &[Vec]) -> IvfPqIndex { + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new(8, small_params()); + idx.train(&refs, test_memory()).unwrap(); + idx + } + #[test] fn rolling_back_withdraws_every_vector_added_after_the_mark() { let vecs = make_vectors(64, 8); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let mut idx = IvfPqIndex::new(8, small_params()); - idx.train(&refs, test_memory()).unwrap(); + let mut idx = trained(&vecs); idx.add_batch(&refs[..40]).unwrap(); let mark = idx.len() as u32; let before: Vec = idx @@ -414,16 +401,17 @@ mod tests { .collect(); idx.add_batch(&refs[40..]).unwrap(); - idx.roll_back_to(mark, true); + idx.roll_back_to(mark); assert_eq!(idx.len(), 40); assert!(idx.is_trained(), "the training the index held stays"); - let after = idx.search(&vecs[5], 64).unwrap(); - assert!( - after.iter().all(|r| r.id < mark), - "no vector added after the mark is found" - ); - let after_ids: Vec = after.iter().map(|r| r.id).collect(); + let after_ids: Vec = idx + .search(&vecs[5], 64) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + assert!(after_ids.iter().all(|id| *id < mark)); assert_eq!(after_ids, before, "the search reads as before the adds"); // The next add takes the first id past the mark again. @@ -431,25 +419,39 @@ mod tests { } #[test] - fn rolling_back_to_an_untrained_mark_drops_the_training() { + fn caller_ids_survive_delete_compact_and_reads() { let vecs = make_vectors(16, 8); - let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let mut idx = IvfPqIndex::new(8, small_params()); - idx.train(&refs, test_memory()).unwrap(); - idx.add_batch(&refs).unwrap(); - - idx.roll_back_to(0, false); - - assert!(idx.is_empty()); - assert!(!idx.is_trained()); - assert_eq!(idx.n_cells(), 0); - assert!(idx.search(&vecs[0], 5).unwrap().is_empty()); + let mut idx = trained(&vecs); + for (i, v) in vecs.iter().enumerate() { + idx.insert_with_id(100 + i as u32, v.clone()).unwrap(); + } + assert!(matches!( + idx.insert_with_id(100, vecs[0].clone()), + Err(VectorError::InvalidInput { .. }) + )); + assert!(idx.delete(103)); + assert!(!idx.delete(103), "a second delete finds nothing live"); + assert!(idx.get_vector(103).is_none()); + assert_eq!(idx.live_count(), 15); + + assert_eq!(idx.compact(), 1); + assert_eq!(idx.len(), 15); + assert!(!idx.contains(103)); + assert_eq!(idx.get_vector(104), Some(vecs[4].as_slice())); + assert_eq!(idx.add(&vecs[0]).unwrap(), 116); } #[test] - fn empty_index() { - let idx = IvfPqIndex::new(8, IvfPqParams::default()); - assert!(idx.search(&[0.0; 8], 5).unwrap().is_empty()); + fn a_trained_index_holding_vectors_refuses_to_retrain() { + let vecs = make_vectors(16, 8); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = trained(&vecs); + idx.add(&vecs[0]).unwrap(); + assert!(matches!( + idx.train(&refs, test_memory()), + Err(VectorError::InvalidInput { .. }) + )); + assert_eq!(idx.len(), 1); } #[test] @@ -458,16 +460,7 @@ mod tests { .map(|i| (0..8).map(|d| ((i * 8 + d) % 17) as f32).collect()) .collect(); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let mut idx = IvfPqIndex::new( - 8, - IvfPqParams { - n_cells: 4, - pq_m: 4, - pq_k: 8, - nprobe: 2, - metric: DistanceMetric::L2, - }, - ); + let mut idx = IvfPqIndex::new(8, small_params()); assert!(matches!( idx.add(&[0.0; 8]), Err(VectorError::InvalidInput { .. }) diff --git a/nodedb-vector/src/ivf/kmeans.rs b/nodedb-vector/src/ivf/kmeans.rs new file mode 100644 index 000000000..31fa432ec --- /dev/null +++ b/nodedb-vector/src/ivf/kmeans.rs @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! k-means++ seeding and Lloyd iterations for the IVF coarse quantizer. + +use crate::distance::{DistanceMetric, distance}; + +/// Up to `k` centroids for `data`, seeded by k-means++ from a fixed seed so +/// the same training set always yields the same centroids. +pub(crate) fn kmeans_centroids( + data: &[&[f32]], + dim: usize, + k: usize, + max_iter: usize, +) -> Vec> { + let n = data.len(); + let k = k.min(n); + if k == 0 { + return Vec::new(); + } + + let mut centroids: Vec> = vec![data[0].to_vec()]; + let mut min_dists = vec![f32::MAX; n]; + + // Initialize min_dists against the first centroid. + for (i, point) in data.iter().enumerate() { + let d = distance(point, ¢roids[0], DistanceMetric::L2); + if d < min_dists[i] { + min_dists[i] = d; + } + } + + let mut rng = crate::hnsw::Xorshift64::new(0xC0FF_EEDE_ADBE_EF42); + for _ in 1..k { + let total: f64 = min_dists.iter().map(|&d| d as f64).sum(); + let next_idx = if total < f64::EPSILON { + 0 + } else { + let target = rng.next_f64() * total; + let mut acc = 0.0f64; + let mut chosen = n - 1; + for (i, &d) in min_dists.iter().enumerate() { + acc += d as f64; + if acc >= target { + chosen = i; + break; + } + } + chosen + }; + let last = data[next_idx]; + centroids.push(last.to_vec()); + for (i, point) in data.iter().enumerate() { + let d = distance(point, last, DistanceMetric::L2); + if d < min_dists[i] { + min_dists[i] = d; + } + } + } + + let mut assignments = vec![0usize; n]; + for _ in 0..max_iter { + let mut changed = false; + for (i, point) in data.iter().enumerate() { + let mut best = 0; + let mut best_d = f32::MAX; + for (c, centroid) in centroids.iter().enumerate() { + let d = distance(point, centroid, DistanceMetric::L2); + if d < best_d { + best_d = d; + best = c; + } + } + if assignments[i] != best { + assignments[i] = best; + changed = true; + } + } + if !changed { + break; + } + let mut sums = vec![vec![0.0f32; dim]; k]; + let mut counts = vec![0usize; k]; + for (i, point) in data.iter().enumerate() { + let c = assignments[i]; + counts[c] += 1; + for d in 0..dim { + sums[c][d] += point[d]; + } + } + for c in 0..k { + if counts[c] > 0 { + for d in 0..dim { + centroids[c][d] = sums[c][d] / counts[c] as f32; + } + } + } + } + centroids +} diff --git a/nodedb-vector/src/ivf/mod.rs b/nodedb-vector/src/ivf/mod.rs new file mode 100644 index 000000000..503cee1cc --- /dev/null +++ b/nodedb-vector/src/ivf/mod.rs @@ -0,0 +1,10 @@ +// SPDX-License-Identifier: Apache-2.0 + +pub mod checkpoint; +pub mod index; +pub mod kmeans; +pub mod params; +pub mod search; + +pub use index::IvfPqIndex; +pub use params::IvfPqParams; diff --git a/nodedb-vector/src/ivf/params.rs b/nodedb-vector/src/ivf/params.rs new file mode 100644 index 000000000..e3d681464 --- /dev/null +++ b/nodedb-vector/src/ivf/params.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! IVF-PQ index parameters. + +use crate::distance::DistanceMetric; + +/// IVF-PQ index configuration. +#[derive(Debug, Clone, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct IvfPqParams { + /// Number of Voronoi cells (partitions). Typical: sqrt(N). + pub n_cells: usize, + /// Number of PQ subvectors. Must divide dimension evenly. + pub pq_m: usize, + /// Centroids per PQ subvector (at most 256 for u8 codes). + pub pq_k: usize, + /// Number of cells to probe at query time. Higher = better recall. + pub nprobe: usize, + /// Distance metric. + pub metric: DistanceMetric, +} + +impl IvfPqParams { + /// Vectors an index needs before it can train: one per coarse cell and + /// one per PQ centroid, since both k-means runs need at least `k` points. + pub fn training_threshold(&self) -> usize { + self.n_cells.max(self.pq_k).max(1) + } +} + +impl Default for IvfPqParams { + fn default() -> Self { + Self { + n_cells: 256, + pq_m: 8, + pq_k: 256, + nprobe: 16, + metric: DistanceMetric::L2, + } + } +} diff --git a/nodedb-vector/src/ivf/search.rs b/nodedb-vector/src/ivf/search.rs new file mode 100644 index 000000000..789a0f948 --- /dev/null +++ b/nodedb-vector/src/ivf/search.rs @@ -0,0 +1,203 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! IVF-PQ search: probe the nearest cells, rank their entries by PQ +//! distance, and rerank the best of them by exact FP32 distance. + +use roaring::RoaringBitmap; + +use crate::distance::{DistanceMetric, distance}; +use crate::error::{VectorError, check_dim}; +use crate::hnsw::SearchResult; + +use super::index::{IvfPqIndex, residual}; + +/// PQ-ranked candidates reranked exactly, per result asked for. +const RERANK_PER_RESULT: usize = 3; +/// Fewest PQ-ranked candidates reranked exactly. +const MIN_RERANK: usize = 20; + +/// One PQ-ranked entry: id, PQ distance, cell, position in the cell. +struct Candidate { + id: u32, + pq_distance: f32, + cell: usize, + pos: usize, +} + +impl IvfPqIndex { + /// The `top_k` live entries nearest to `query` under the index metric. + /// + /// A query without the index dimension fails with + /// [`VectorError::DimensionMismatch`]. A distance table over the memory + /// budget fails the search: skipping its cell would drop results. + pub fn search(&self, query: &[f32], top_k: usize) -> Result, VectorError> { + self.search_with(query, top_k, self.params.metric, None) + } + + /// The `top_k` live entries nearest to `query` under `metric`, keeping + /// only ids in `filter` when one is given. Fails as [`Self::search`]. + pub fn search_with( + &self, + query: &[f32], + top_k: usize, + metric: DistanceMetric, + filter: Option<&RoaringBitmap>, + ) -> Result, VectorError> { + check_dim(self.dim, query.len())?; + let Some(pq) = &self.pq else { + return Ok(Vec::new()); + }; + if top_k == 0 || self.slots.is_empty() { + return Ok(Vec::new()); + } + + let mut probe: Vec<(usize, f32)> = self + .centroids + .iter() + .enumerate() + .map(|(i, c)| (i, distance(query, c, self.params.metric))) + .collect(); + probe.sort_by(|a, b| a.1.total_cmp(&b.1)); + probe.truncate(self.params.nprobe.max(1)); + + let m = pq.m; + let mut candidates: Vec = Vec::new(); + for &(cell_idx, _) in &probe { + let cell = &self.cells[cell_idx]; + if cell.ids.is_empty() { + continue; + } + let table = pq.build_distance_table(&residual(query, &self.centroids[cell_idx]))?; + for (pos, &id) in cell.ids.iter().enumerate() { + if self.deleted.contains(id) || filter.is_some_and(|f| !f.contains(id)) { + continue; + } + let code = &cell.codes[pos * m..(pos + 1) * m]; + candidates.push(Candidate { + id, + pq_distance: pq.asymmetric_distance(&table, code), + cell: cell_idx, + pos, + }); + } + } + + let pool = top_k.saturating_mul(RERANK_PER_RESULT).max(MIN_RERANK); + if candidates.len() > pool { + candidates.select_nth_unstable_by(pool, |a, b| a.pq_distance.total_cmp(&b.pq_distance)); + candidates.truncate(pool); + } + + let dim = self.dim; + let mut results: Vec = candidates + .into_iter() + .map(|c| { + let start = c.pos * dim; + let vector = &self.cells[c.cell].vectors[start..start + dim]; + SearchResult { + id: c.id, + distance: distance(query, vector, metric), + } + }) + .collect(); + results.sort_by(|a, b| a.distance.total_cmp(&b.distance)); + results.truncate(top_k); + Ok(results) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::ivf::IvfPqParams; + use crate::test_support::test_memory; + + fn make_vectors(n: usize, dim: usize) -> Vec> { + (0..n) + .map(|i| (0..dim).map(|d| ((i * dim + d) as f32) * 0.01).collect()) + .collect() + } + + fn index_with(vecs: &[Vec], params: IvfPqParams) -> IvfPqIndex { + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let mut idx = IvfPqIndex::new(vecs[0].len(), params); + idx.train(&refs, test_memory()).unwrap(); + idx.add_batch(&refs).unwrap(); + idx + } + + #[test] + fn train_and_search_finds_the_exact_match_first() { + let vecs = make_vectors(1000, 16); + let idx = index_with( + &vecs, + IvfPqParams { + n_cells: 32, + pq_m: 4, + pq_k: 32, + nprobe: 8, + metric: DistanceMetric::L2, + }, + ); + assert_eq!(idx.len(), 1000); + let results = idx.search(&vecs[500], 5).unwrap(); + assert_eq!(results.len(), 5); + assert_eq!(results[0].id, 500, "the exact rerank puts the match first"); + assert_eq!(results[0].distance, 0.0); + } + + #[test] + fn probing_every_cell_returns_every_live_entry_once() { + let vecs = make_vectors(64, 8); + let mut idx = index_with( + &vecs, + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 4, + metric: DistanceMetric::L2, + }, + ); + idx.delete(7); + let mut ids: Vec = idx + .search(&vecs[0], 100) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + ids.sort_unstable(); + let expected: Vec = (0..64).filter(|id| *id != 7).collect(); + assert_eq!(ids, expected); + } + + #[test] + fn a_filter_keeps_only_its_ids() { + let vecs = make_vectors(64, 8); + let idx = index_with( + &vecs, + IvfPqParams { + n_cells: 4, + pq_m: 4, + pq_k: 8, + nprobe: 4, + metric: DistanceMetric::L2, + }, + ); + let filter: RoaringBitmap = [3u32, 40, 41].into_iter().collect(); + let mut ids: Vec = idx + .search_with(&vecs[0], 10, DistanceMetric::L2, Some(&filter)) + .unwrap() + .iter() + .map(|r| r.id) + .collect(); + ids.sort_unstable(); + assert_eq!(ids, vec![3, 40, 41]); + } + + #[test] + fn an_untrained_index_finds_nothing() { + let idx = IvfPqIndex::new(8, IvfPqParams::default()); + assert!(idx.search(&[0.0; 8], 5).unwrap().is_empty()); + } +} diff --git a/nodedb-vector/src/quantize/mod.rs b/nodedb-vector/src/quantize/mod.rs index 2bd207f56..0a2e03d5b 100644 --- a/nodedb-vector/src/quantize/mod.rs +++ b/nodedb-vector/src/quantize/mod.rs @@ -5,6 +5,7 @@ pub mod binary_codec; pub mod pq; pub mod pq_decode; +pub mod pq_kmeans; pub mod pq_codec; diff --git a/nodedb-vector/src/quantize/pq.rs b/nodedb-vector/src/quantize/pq.rs index 865cec455..ed8a57a90 100644 --- a/nodedb-vector/src/quantize/pq.rs +++ b/nodedb-vector/src/quantize/pq.rs @@ -21,6 +21,8 @@ use nodedb_types::decode_bounds::checked_decode_capacity; use crate::error::{VectorError, check_dim}; +use super::pq_kmeans::{kmeans, l2_sub}; + /// Hard ceiling for a decoded PQ vector. This bounds corrupted persisted /// configuration even when the codec has no scoped memory handle attached. const MAX_PQ_DECODE_DIM: usize = 1_048_576; @@ -300,7 +302,11 @@ impl PqCodec { sub_dim: self.sub_dim, codebooks: self.codebooks.clone(), }; - let payload = zerompk::to_msgpack_vec(&data).unwrap_or_default(); + let payload = zerompk::to_msgpack_vec(&data).map_err(|e| { + VectorError::CheckpointSerializationError { + detail: format!("PQ codec encode: {e}"), + } + })?; let mut out = Vec::with_capacity(7 + payload.len()); out.extend_from_slice(MAGIC); out.push(VERSION); @@ -387,118 +393,6 @@ impl PqCodec { } } -/// L2 squared distance for sub-vectors (used in k-means and encoding). -#[inline] -fn l2_sub(a: &[f32], b: &[f32]) -> f32 { - let mut sum = 0.0f32; - for i in 0..a.len() { - let d = a[i] - b[i]; - sum += d * d; - } - sum -} - -/// Simple k-means clustering for PQ codebook training. -/// -/// Uses proper k-means++ initialization (weighted d² sampling) with a -/// deterministic seed so training is reproducible across runs. -fn kmeans(data: &[&[f32]], dim: usize, k: usize, max_iter: usize) -> Vec> { - let n = data.len(); - if n == 0 || k == 0 { - return Vec::new(); - } - let k = k.min(n); // Can't have more centroids than data points. - - // K-means++ initialization with deterministic xorshift. - let mut rng = crate::hnsw::Xorshift64::new(0xC0FF_EEDE_ADBE_EF42); - - let mut centroids: Vec> = Vec::with_capacity(k); - centroids.push(data[0].to_vec()); - - let mut min_dists = vec![f32::MAX; n]; - // Update against the first centroid. - for (i, point) in data.iter().enumerate() { - let d = l2_sub(point, ¢roids[0]); - if d < min_dists[i] { - min_dists[i] = d; - } - } - - for _ in 1..k { - let total: f64 = min_dists.iter().map(|&d| d as f64).sum(); - let next_idx = if total < f64::EPSILON { - // All points coincide with existing centroids. - 0 - } else { - let target = rng.next_f64() * total; - let mut acc = 0.0f64; - let mut chosen = n - 1; - for (i, &d) in min_dists.iter().enumerate() { - acc += d as f64; - if acc >= target { - chosen = i; - break; - } - } - chosen - }; - let last = data[next_idx]; - centroids.push(last.to_vec()); - // Incrementally update min_dists against the new centroid. - for (i, point) in data.iter().enumerate() { - let d = l2_sub(point, last); - if d < min_dists[i] { - min_dists[i] = d; - } - } - } - - // K-means iterations. - let mut assignments = vec![0usize; n]; - for _ in 0..max_iter { - // Assignment step. - let mut changed = false; - for (i, point) in data.iter().enumerate() { - let mut best = 0; - let mut best_d = f32::MAX; - for (c, centroid) in centroids.iter().enumerate() { - let d = l2_sub(point, centroid); - if d < best_d { - best_d = d; - best = c; - } - } - if assignments[i] != best { - assignments[i] = best; - changed = true; - } - } - if !changed { - break; - } - - // Update step: recompute centroids as means. - let mut sums = vec![vec![0.0f32; dim]; k]; - let mut counts = vec![0usize; k]; - for (i, point) in data.iter().enumerate() { - let c = assignments[i]; - counts[c] += 1; - for d in 0..dim { - sums[c][d] += point[d]; - } - } - for c in 0..k { - if counts[c] > 0 { - for d in 0..dim { - centroids[c][d] = sums[c][d] / counts[c] as f32; - } - } - } - } - - centroids -} - #[cfg(test)] mod tests { use super::*; diff --git a/nodedb-vector/src/quantize/pq_kmeans.rs b/nodedb-vector/src/quantize/pq_kmeans.rs new file mode 100644 index 000000000..6e6d8a5d6 --- /dev/null +++ b/nodedb-vector/src/quantize/pq_kmeans.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Sub-vector distance and k-means for PQ codebook training. + +/// L2 squared distance for sub-vectors (used in k-means and encoding). +#[inline] +pub(crate) fn l2_sub(a: &[f32], b: &[f32]) -> f32 { + let mut sum = 0.0f32; + for i in 0..a.len() { + let d = a[i] - b[i]; + sum += d * d; + } + sum +} + +/// Simple k-means clustering for PQ codebook training. +/// +/// Uses proper k-means++ initialization (weighted d² sampling) with a +/// deterministic seed so training is reproducible across runs. +pub(crate) fn kmeans(data: &[&[f32]], dim: usize, k: usize, max_iter: usize) -> Vec> { + let n = data.len(); + if n == 0 || k == 0 { + return Vec::new(); + } + let k = k.min(n); // Can't have more centroids than data points. + + // K-means++ initialization with deterministic xorshift. + let mut rng = crate::hnsw::Xorshift64::new(0xC0FF_EEDE_ADBE_EF42); + + let mut centroids: Vec> = Vec::with_capacity(k); + centroids.push(data[0].to_vec()); + + let mut min_dists = vec![f32::MAX; n]; + // Update against the first centroid. + for (i, point) in data.iter().enumerate() { + let d = l2_sub(point, ¢roids[0]); + if d < min_dists[i] { + min_dists[i] = d; + } + } + + for _ in 1..k { + let total: f64 = min_dists.iter().map(|&d| d as f64).sum(); + let next_idx = if total < f64::EPSILON { + // All points coincide with existing centroids. + 0 + } else { + let target = rng.next_f64() * total; + let mut acc = 0.0f64; + let mut chosen = n - 1; + for (i, &d) in min_dists.iter().enumerate() { + acc += d as f64; + if acc >= target { + chosen = i; + break; + } + } + chosen + }; + let last = data[next_idx]; + centroids.push(last.to_vec()); + // Incrementally update min_dists against the new centroid. + for (i, point) in data.iter().enumerate() { + let d = l2_sub(point, last); + if d < min_dists[i] { + min_dists[i] = d; + } + } + } + + // K-means iterations. + let mut assignments = vec![0usize; n]; + for _ in 0..max_iter { + // Assignment step. + let mut changed = false; + for (i, point) in data.iter().enumerate() { + let mut best = 0; + let mut best_d = f32::MAX; + for (c, centroid) in centroids.iter().enumerate() { + let d = l2_sub(point, centroid); + if d < best_d { + best_d = d; + best = c; + } + } + if assignments[i] != best { + assignments[i] = best; + changed = true; + } + } + if !changed { + break; + } + + // Update step: recompute centroids as means. + let mut sums = vec![vec![0.0f32; dim]; k]; + let mut counts = vec![0usize; k]; + for (i, point) in data.iter().enumerate() { + let c = assignments[i]; + counts[c] += 1; + for d in 0..dim { + sums[c][d] += point[d]; + } + } + for c in 0..k { + if counts[c] > 0 { + for d in 0..dim { + centroids[c][d] = sums[c][d] / counts[c] as f32; + } + } + } + } + + centroids +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs index 3e4134293..74d67f5a0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs @@ -339,10 +339,16 @@ fn validate(options: &ParsedOptions) -> Result { )); } - if uses_pq && pq_m > 0 && !dim.is_multiple_of(pq_m) { + // An omitted PQ_M takes the engine default, which must divide dim too. + let effective_pq_m = if pq_m > 0 { + pq_m + } else { + nodedb_vector::index_config::DEFAULT_PQ_M + }; + if uses_pq && !dim.is_multiple_of(effective_pq_m) { return Err(ddl_err( "22023", - format!("{CONTEXT}: pq_m ({pq_m}) must divide dim ({dim}) evenly"), + format!("{CONTEXT}: pq_m ({effective_pq_m}) must divide dim ({dim}) evenly"), )); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs index faf24988b..cd7b2a89f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs @@ -73,7 +73,7 @@ pub async fn handle_show_vector_index( let columns = vec!["property".to_string(), "value".to_string()]; - let pairs: Vec<(&str, String)> = vec![ + let mut pairs: Vec<(&str, String)> = vec![ ("dimensions", stats.dimensions.to_string()), ("metric", stats.metric.clone()), ("index_type", stats.index_type.to_string()), @@ -103,6 +103,17 @@ pub async fn handle_show_vector_index( ("seal_threshold", stats.seal_threshold.to_string()), ("mmap_segments", stats.mmap_segment_count.to_string()), ]; + if let Some(ivf) = &stats.ivf { + pairs.extend([ + ("ivf_training_threshold", ivf.training_threshold.to_string()), + ("ivf_trained", ivf.trained.to_string()), + ("ivf_trained_on", ivf.trained_on.to_string()), + ("ivf_trained_at_ms", ivf.trained_at_ms.to_string()), + ("ivf_indexed_vectors", ivf.indexed_vectors.to_string()), + ("ivf_cells", ivf.cells.to_string()), + ("ivf_nprobe", ivf.nprobe.to_string()), + ]); + } let rows: Vec> = pairs .into_iter() diff --git a/nodedb/src/data/executor/core_loop/open.rs b/nodedb/src/data/executor/core_loop/open.rs index be5e98c69..c86cd1dfd 100644 --- a/nodedb/src/data/executor/core_loop/open.rs +++ b/nodedb/src/data/executor/core_loop/open.rs @@ -130,7 +130,6 @@ impl CoreLoop { aggregate_cache: HashMap::new(), maintenance: super::maintenance_state::MaintenanceState::new(), index_configs: HashMap::new(), - ivf_indexes: HashMap::new(), sparse_vector_indexes: HashMap::new(), doc_cache: DocCache::new( nodedb_types::config::tuning::QueryTuning::default().doc_cache_entries, diff --git a/nodedb/src/data/executor/core_loop/state.rs b/nodedb/src/data/executor/core_loop/state.rs index fb4c10a6c..0fd2633f2 100644 --- a/nodedb/src/data/executor/core_loop/state.rs +++ b/nodedb/src/data/executor/core_loop/state.rs @@ -215,11 +215,6 @@ pub struct CoreLoop { pub(in crate::data::executor) index_configs: HashMap<(DatabaseId, TenantId, String), crate::engine::vector::index_config::IndexConfig>, - /// IVF-PQ indexes for collections configured with `index_type = "ivf_pq"`. - /// Key: `(DatabaseId, TenantId, collection_key)` — same shape as `vector_collections`. - pub(in crate::data::executor) ivf_indexes: - HashMap<(DatabaseId, TenantId, String), crate::engine::vector::ivf::IvfPqIndex>, - /// Per-collection sparse vector inverted indexes, keyed by /// (DatabaseId, TenantId, collection, field). /// The field is `"_sparse"` when no named field is specified. diff --git a/nodedb/src/data/executor/core_loop/vector_index_seed.rs b/nodedb/src/data/executor/core_loop/vector_index_seed.rs index fcbcd6097..5fc29cfd1 100644 --- a/nodedb/src/data/executor/core_loop/vector_index_seed.rs +++ b/nodedb/src/data/executor/core_loop/vector_index_seed.rs @@ -46,6 +46,11 @@ impl CoreLoop { self.declared_dims.insert(key.clone(), e.dim); } self.vector_params.insert(key.clone(), params); + // A collection the checkpoint restored takes the catalog's index + // configuration, which is the source of truth for its type. + if let Some(coll) = self.vector_collections.get_mut(&key) { + coll.set_index_config(config.clone()); + } self.index_configs.insert(key, config); } } diff --git a/nodedb/src/data/executor/handlers/compact/runner.rs b/nodedb/src/data/executor/handlers/compact/runner.rs index 0cf9f40cc..0c7fc3f69 100644 --- a/nodedb/src/data/executor/handlers/compact/runner.rs +++ b/nodedb/src/data/executor/handlers/compact/runner.rs @@ -78,16 +78,21 @@ impl CoreLoop { None => continue, }; + // A trained IVF-PQ index holds its own tombstones beside the + // sealed segments', and `compact` removes both. + let ivf = collection.ivf_index(); let total_tombstones: usize = collection .sealed_segments() .iter() .map(|seg| seg.index.tombstone_count()) - .sum(); + .sum::() + + ivf.map_or(0, |ivf| ivf.tombstone_count()); let total_nodes: usize = collection .sealed_segments() .iter() .map(|seg| seg.index.len()) - .sum(); + .sum::() + + ivf.map_or(0, |ivf| ivf.len()); if total_tombstones == 0 { continue; diff --git a/nodedb/src/data/executor/handlers/control/reindex.rs b/nodedb/src/data/executor/handlers/control/reindex.rs index 793df132e..199553d6a 100644 --- a/nodedb/src/data/executor/handlers/control/reindex.rs +++ b/nodedb/src/data/executor/handlers/control/reindex.rs @@ -264,6 +264,10 @@ impl CoreLoop { Some(c) => c, None => continue, }; + // An IVF-PQ collection keeps no HNSW segments to rebuild. + if coll.is_ivf() { + continue; + } let dim = coll.dim(); let params = coll.hnsw_params(); diff --git a/nodedb/src/data/executor/handlers/mod.rs b/nodedb/src/data/executor/handlers/mod.rs index dfdefe48c..62bd550cb 100644 --- a/nodedb/src/data/executor/handlers/mod.rs +++ b/nodedb/src/data/executor/handlers/mod.rs @@ -96,7 +96,7 @@ pub mod vector_params; pub mod vector_search; mod vector_search_ann; mod vector_search_exec; -mod vector_search_ivf; +pub mod vector_settle; pub mod vector_sparse; pub mod vector_upsert; pub mod vector_write; diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs index 9b9d338d6..c81f4107b 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs @@ -82,21 +82,12 @@ impl CoreLoop { floats.len(), )); } - let params = self - .vector_params - .get(&index_key) - .cloned() - .unwrap_or_default(); // Skip a record the restored vector checkpoint holds. Its // stamp names only records applied before the checkpoint, // all of them replayed before the core serves a request, // so a live write is never named. let skip = wal_lsn != 0 && self.vector_replay_skips(wal_lsn); - self.vector_collections - .entry(index_key.clone()) - .or_insert_with(|| { - nodedb_vector::VectorCollection::new(*dim as usize, params) - }); + self.ensure_vector_collection(&index_key, &index_key, *dim as usize)?; if skip { continue; } @@ -159,11 +150,6 @@ impl CoreLoop { }, None => continue, }; - let params = self - .vector_params - .get(params_key) - .cloned() - .unwrap_or_default(); // Use field-qualified key so search can find it. let store_key = Self::vector_index_key(database_id, tid, collection, field_name); @@ -171,9 +157,7 @@ impl CoreLoop { let dim = floats.len(); // Same stamp gate as the strict arm above. let skip = wal_lsn != 0 && self.vector_replay_skips(wal_lsn); - self.vector_collections - .entry(store_key.clone()) - .or_insert_with(|| nodedb_vector::VectorCollection::new(dim, params)); + self.ensure_vector_collection(&store_key, params_key, dim)?; if skip { continue; } @@ -194,6 +178,13 @@ impl CoreLoop { } } + // A committed-redo install trains once the whole record landed, so a + // rollback finds its inserts in the growing segment. + if !self.recording_redo_undo() { + for delta in &inserts { + self.train_ivf_if_ready(&delta.index_key); + } + } Ok(inserts) } diff --git a/nodedb/src/data/executor/handlers/purge.rs b/nodedb/src/data/executor/handlers/purge.rs index 82dfed25a..02994c8cb 100644 --- a/nodedb/src/data/executor/handlers/purge.rs +++ b/nodedb/src/data/executor/handlers/purge.rs @@ -111,7 +111,6 @@ impl CoreLoop { self.vector_collections.retain(|(_, t, _), _| *t != tid_key); self.vector_params.retain(|(_, t, _), _| *t != tid_key); self.index_configs.retain(|(_, t, _), _| *t != tid_key); - self.ivf_indexes.retain(|(_, t, _), _| *t != tid_key); before - self.vector_collections.len() }; diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs b/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs index 3d1c4f2b8..2cc754a33 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore/engines.rs @@ -55,29 +55,20 @@ impl CoreLoop { crate::types::TenantId::new(tenant_id), coll_key.to_string(), ); - let params = self - .vector_params - .get(&map_key) - .cloned() - .unwrap_or_default(); // Raft InstallSnapshot apply (`replace_mode`) must REPLACE the local // collection so the snapshot's vectors are not appended on top of stale // entries. User RESTORE (`!replace_mode`) keeps the prior insert-into- // existing-or-create behavior. if replace_mode { - self.vector_collections.insert( - map_key.clone(), - crate::engine::vector::collection::VectorCollection::new(dim, params.clone()), - ); + self.vector_collections.remove(&map_key); } - let coll = self.vector_collections.entry(map_key).or_insert_with(|| { - crate::engine::vector::collection::VectorCollection::new(dim, params) - }); + let coll = self.ensure_vector_collection(&map_key, &map_key, dim)?; let (data, surrogates): (Vec>, Vec) = vectors .into_iter() .map(|(_, data, surrogate)| (data, surrogate.unwrap_or(nodedb_types::Surrogate::ZERO))) .unzip(); coll.insert_batch_with_surrogates(&data, &surrogates)?; + self.train_ivf_if_ready(&map_key); Ok(()) } diff --git a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs index c22708f4e..90158283a 100644 --- a/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs +++ b/nodedb/src/data/executor/handlers/transaction/redo_apply/settle.rs @@ -62,7 +62,7 @@ impl CoreLoop { ) -> Result<(), ErrorCode> { self.finalize_timeseries_truncates(&scope.undo); self.finalize_vector_truncates(&mut scope.undo); - self.seal_full_vector_collections(); + self.settle_filled_vector_collections(); for key in std::mem::take(&mut scope.columnar_written) { let collection = key.2.clone(); self.flush_columnar_memtable_if_needed(task, &key, &collection) @@ -87,28 +87,17 @@ impl CoreLoop { Ok(()) } - /// Seal every vector collection whose growing segment filled while the - /// install held its seals back. - fn seal_full_vector_collections(&mut self) { + /// Settle every vector collection the install filled while it held its + /// seals and IVF-PQ trainings back. + fn settle_filled_vector_collections(&mut self) { let full: Vec<_> = self .vector_collections .iter() - .filter(|(_, coll)| coll.needs_seal()) + .filter(|(_, coll)| coll.needs_seal() || coll.needs_ivf_training()) .map(|(key, _)| key.clone()) .collect(); for key in full { - let seal_key = CoreLoop::vector_build_key(&key); - if let Some(coll) = self.vector_collections.get_mut(&key) - && let Some(req) = coll.seal(&seal_key) - && let Some(tx) = &self.build_tx - && let Err(e) = tx.send(req) - { - tracing::warn!( - core = self.core_id, - error = %e, - "failed to send HNSW build request" - ); - } + self.settle_vector_collection(&key); } } diff --git a/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs b/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs index 80f115147..34221689f 100644 --- a/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs +++ b/nodedb/src/data/executor/handlers/transaction/undo/vector_write.rs @@ -5,9 +5,9 @@ //! vector-primary row write. //! //! The pre-image is taken before the write: the collection's write mark -//! (`VectorCollection::write_mark`), the IVF-PQ add counter, and for a -//! vector-primary collection the sidecar row and payload bitmap entries of -//! every named row. The undo withdraws every node the write inserted, puts +//! (`VectorCollection::write_mark`, which covers a trained IVF-PQ index too), +//! and for a vector-primary collection the sidecar row and payload bitmap +//! entries of every named row. The undo withdraws every node the write inserted, puts //! every binding and tombstone back, and restores the sidecars and bitmap //! entries. A collection the write created is removed again. @@ -21,13 +21,6 @@ use crate::engine::vector::collection::VectorWriteMark; use super::UndoEntry; -/// The IVF-PQ state of a collection before a write. -pub(in crate::data::executor) struct IvfMark { - /// Vectors the index held. - pub count: u32, - pub trained: bool, -} - /// The pre-image of one vector write. pub(in crate::data::executor) struct VectorWriteUndo { pub index_key: VectorIndexKey, @@ -38,8 +31,6 @@ pub(in crate::data::executor) struct VectorWriteUndo { pub mark: Option, /// Whether the write found no `vector_params` entry for the key. pub params_absent: bool, - /// `None` when the collection had no IVF-PQ index. - pub ivf: Option, /// Sidecar bytes of every named surrogate, `None` when absent. Empty for /// a write that stores no sidecar. pub sidecars: Vec<(Surrogate, Option>)>, @@ -114,10 +105,6 @@ impl CoreLoop { collection: collection.to_string(), mark: coll.map(|coll| coll.write_mark(surrogates, ids)), params_absent: !self.vector_params.contains_key(index_key), - ivf: self.ivf_indexes.get(index_key).map(|ivf| IvfMark { - count: ivf.len() as u32, - trained: ivf.is_trained(), - }), sidecars: sidecar_rows, payload_rows, }))) @@ -135,7 +122,6 @@ impl CoreLoop { collection, mark, params_absent, - ivf, sidecars, payload_rows, } = undo; @@ -177,7 +163,8 @@ impl CoreLoop { }; if !coll.roll_back_to(mark) { return Err(fail(format!( - "vector index {:?} sealed the nodes a rolled-back write inserted", + "vector index {:?} sealed or trained away the nodes a rolled-back \ + write inserted", index_key ))); } @@ -192,16 +179,6 @@ impl CoreLoop { if params_absent { self.vector_params.remove(&index_key); } - match ivf { - Some(IvfMark { count, trained }) => { - if let Some(index) = self.ivf_indexes.get_mut(&index_key) { - index.roll_back_to(count, trained); - } - } - None => { - self.ivf_indexes.remove(&index_key); - } - } for (surrogate, prior) in sidecars { let key = StorageKey::for_surrogate(surrogate); @@ -227,54 +204,49 @@ impl CoreLoop { mod tests { use super::*; use crate::data::executor::core_loop::tests::make_core_with_dir; - use crate::engine::vector::ivf::{IvfPqIndex, IvfPqParams}; + use crate::engine::vector::collection::VectorCollection; + use crate::engine::vector::index_config::{IndexConfig, IndexType}; use crate::types::{DatabaseId, TenantId}; const TID: u64 = 1; + const DIM: usize = 4; - fn params() -> IvfPqParams { - IvfPqParams { - n_cells: 2, - pq_m: 2, - pq_k: 4, - nprobe: 2, - metric: nodedb_vector::DistanceMetric::L2, - } + /// Distinct per `i`: the first component is `i + 1`. + fn vector(i: usize) -> Vec { + vec![ + (i + 1) as f32, + (i % 7 + 1) as f32, + (i % 11 + 1) as f32, + (i % 13 + 1) as f32, + ] } - fn vectors() -> Vec> { - (0..8) - .map(|i| vec![i as f32, (i * 2) as f32, 1.0, 0.5]) - .collect() + fn ivf_collection() -> VectorCollection { + VectorCollection::with_index_config( + DIM, + IndexConfig { + index_type: IndexType::IvfPq, + pq_m: 2, + ivf_cells: 2, + ivf_nprobe: 2, + ..IndexConfig::default() + }, + ) } - /// The IVF-PQ adds of a rolled-back write leave the index: it holds the - /// vectors and the training it held before the write. - #[test] - fn a_rolled_back_write_withdraws_its_ivf_adds() { - let dir = tempfile::tempdir().expect("tempdir"); - let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); - let key: VectorIndexKey = (DatabaseId::DEFAULT, TenantId::new(TID), "docs:".into()); - let vecs = vectors(); - let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); - let mut index = IvfPqIndex::new(4, params()); - index - .train( - &refs, - nodedb_mem::ScopedMemory::new( - crate::data::executor::core_loop::test_governor(), - DatabaseId::DEFAULT, - TenantId::new(TID), - nodedb_mem::EngineId::Vector, - ), - ) - .unwrap(); - index.add_batch(&refs[..4]).unwrap(); - core.ivf_indexes.insert(key.clone(), index); + fn memory() -> nodedb_mem::ScopedMemory { + nodedb_mem::ScopedMemory::new( + crate::data::executor::core_loop::test_governor(), + DatabaseId::DEFAULT, + TenantId::new(TID), + nodedb_mem::EngineId::Vector, + ) + } + fn capture(core: &CoreLoop, key: &VectorIndexKey) -> VectorWriteUndo { let undo = core .capture_vector_write_undo(VectorWriteTarget { - index_key: &key, + index_key: key, tid: TID, collection: "docs", surrogates: &[], @@ -282,49 +254,69 @@ mod tests { sidecars: false, }) .expect("capture undo"); - if let Some(index) = core.ivf_indexes.get_mut(&key) { - index.add_batch(&refs[4..]).unwrap(); - } let UndoEntry::VectorWrite(undo) = undo else { panic!("a vector write captures a VectorWrite undo"); }; - core.apply_undo_vector_write(0, *undo) + *undo + } + + /// The inserts of a rolled-back write leave a trained IVF-PQ index: it + /// holds the vectors and the training it held before the write. + #[test] + fn a_rolled_back_write_withdraws_its_ivf_inserts() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let key: VectorIndexKey = (DatabaseId::DEFAULT, TenantId::new(TID), "docs:".into()); + let mut coll = ivf_collection(); + for i in 0..256 { + coll.insert(vector(i)).unwrap(); + } + coll.train_ivf(memory(), 1).unwrap(); + core.vector_collections.insert(key.clone(), coll); + + let undo = capture(&core, &key); + if let Some(coll) = core.vector_collections.get_mut(&key) { + for i in 256..260 { + coll.insert(vector(i)).unwrap(); + } + } + core.apply_undo_vector_write(0, undo) .expect("undo vector write"); - let index = core.ivf_indexes.get(&key).expect("index stays"); - assert_eq!(index.len(), 4); - assert!(index.is_trained()); + let coll = core + .vector_collections + .get_mut(&key) + .expect("collection stays"); + assert_eq!(coll.live_count(), 256); + assert!(coll.ivf_index().is_some_and(|ivf| ivf.is_trained())); assert!( - index.search(&vecs[6], 8).unwrap().iter().all(|r| r.id < 4), - "no vector the write added is found" + coll.search(&vector(258), 300, 64) + .unwrap() + .iter() + .all(|r| r.id < 256), + "no vector the write inserted is found" + ); + assert_eq!( + coll.insert(vector(300)).unwrap(), + 256, + "the id counter is back" ); } - /// An IVF-PQ index the rolled-back write created is removed. + /// An IVF-PQ collection the rolled-back write created is removed. #[test] - fn a_rolled_back_write_removes_the_ivf_index_it_created() { + fn a_rolled_back_write_removes_the_ivf_collection_it_created() { let dir = tempfile::tempdir().expect("tempdir"); let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); let key: VectorIndexKey = (DatabaseId::DEFAULT, TenantId::new(TID), "docs:".into()); - let undo = core - .capture_vector_write_undo(VectorWriteTarget { - index_key: &key, - tid: TID, - collection: "docs", - surrogates: &[], - ids: &[], - sidecars: false, - }) - .expect("capture undo"); - core.ivf_indexes - .insert(key.clone(), IvfPqIndex::new(4, params())); - let UndoEntry::VectorWrite(undo) = undo else { - panic!("a vector write captures a VectorWrite undo"); - }; - core.apply_undo_vector_write(0, *undo) + let undo = capture(&core, &key); + let mut coll = ivf_collection(); + coll.insert(vector(0)).unwrap(); + core.vector_collections.insert(key.clone(), coll); + core.apply_undo_vector_write(0, undo) .expect("undo vector write"); - assert!(!core.ivf_indexes.contains_key(&key)); + assert!(!core.vector_collections.contains_key(&key)); } } diff --git a/nodedb/src/data/executor/handlers/unregister_collection.rs b/nodedb/src/data/executor/handlers/unregister_collection.rs index 16b68840e..488fc18aa 100644 --- a/nodedb/src/data/executor/handlers/unregister_collection.rs +++ b/nodedb/src/data/executor/handlers/unregister_collection.rs @@ -261,8 +261,6 @@ impl CoreLoop { .retain(|(d, t, c), _| !(*d == db && *t == tid) || keep(c)); self.index_configs .retain(|(d, t, c), _| !(*d == db && *t == tid) || keep(c)); - self.ivf_indexes - .retain(|(d, t, c), _| !(*d == db && *t == tid) || keep(c)); removed }; diff --git a/nodedb/src/data/executor/handlers/vector.rs b/nodedb/src/data/executor/handlers/vector.rs index f94361ce3..dd560013d 100644 --- a/nodedb/src/data/executor/handlers/vector.rs +++ b/nodedb/src/data/executor/handlers/vector.rs @@ -5,13 +5,17 @@ use nodedb_types::Surrogate; use nodedb_types::sync::wire::{AckStatus, SyncProvenance}; -use tracing::{debug, warn}; +use std::collections::hash_map::Entry; + +use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::sync_gate::{SyncAdmit, ack_status_from_admit}; use crate::data::executor::task::ExecutionTask; use crate::engine::vector::collection::VectorCollection; + +use super::vector_settle::{check_ivf_dim, vector_index_config_for}; use crate::types::TenantId; use nodedb_types::DatabaseId; @@ -90,21 +94,23 @@ impl CoreLoop { return Err(dimension_mismatch(declared, dim)); } - if let Some(existing) = self.vector_collections.get(&index_key) - && existing.dim() != dim - { - return Err(dimension_mismatch(existing.dim(), dim)); - } let core_id = self.core_id; - let params = self - .vector_params - .get(&index_key) - .cloned() - .unwrap_or_default(); - Ok(self.vector_collections.entry(index_key).or_insert_with(|| { - debug!(core = core_id, dim, m = params.m, ef = params.ef_construction, ?params.metric, "creating vector collection"); - VectorCollection::new(dim, params) - })) + match self.vector_collections.entry(index_key) { + Entry::Occupied(entry) => { + let existing = entry.into_mut(); + if existing.dim() != dim { + return Err(dimension_mismatch(existing.dim(), dim)); + } + Ok(existing) + } + Entry::Vacant(entry) => { + let config = + vector_index_config_for(&self.index_configs, &self.vector_params, entry.key()); + check_ivf_dim(&config, dim).map_err(ErrorCode::from)?; + debug!(core = core_id, dim, index_type = ?config.index_type, "creating vector collection"); + Ok(entry.insert(VectorCollection::with_index_config(dim, config))) + } + } } pub(in crate::data::executor) fn execute_vector_insert( @@ -200,15 +206,7 @@ impl CoreLoop { let database_id = task.request.database_id.as_u64(); let index_key = CoreLoop::vector_index_key(database_id, tid, collection, field_name); - // Check if this collection uses IVF-PQ index. - if let Some(cfg) = self.index_configs.get(&index_key) - && cfg.index_type == crate::engine::vector::index_config::IndexType::IvfPq - { - let key = index_key.clone(); - return self.ivf_insert(task, tid, &key, vector, dim, surrogate); - } - - // Default: HNSW (with or without PQ). A committed-redo install seals + // A committed-redo install seals, or trains an IVF-PQ collection, // once the whole record landed, so a rollback finds its inserts in // the growing segment. let defer_seal = self.recording_redo_undo(); @@ -217,14 +215,8 @@ impl CoreLoop { if let Err(e) = collection_ref.insert_with_surrogate(vector.to_vec(), surrogate) { return self.response_error(task, crate::Error::from(e)); } - let seal_key = CoreLoop::vector_build_key(&index_key); - if !defer_seal - && collection_ref.needs_seal() - && let Some(req) = collection_ref.seal(&seal_key) - && let Some(tx) = &self.build_tx - && let Err(e) = tx.send(req) - { - warn!(core = self.core_id, error = %e, "failed to send HNSW build request"); + if !defer_seal { + self.settle_vector_collection(&index_key); } self.checkpoint_coordinator.mark_dirty("vector", 1); // Record this write's version so cross-shard OCC read-set @@ -242,75 +234,6 @@ impl CoreLoop { } } - /// Insert into an IVF-PQ index, returning the assigned vector ID. - fn ivf_insert( - &mut self, - task: &ExecutionTask, - tid: u64, - index_key: &(DatabaseId, TenantId, String), - vector: &[f32], - dim: usize, - surrogate: Surrogate, - ) -> Response { - let ivf = self - .ivf_indexes - .entry(index_key.clone()) - .or_insert_with(|| { - let cfg = self - .index_configs - .get(index_key) - .cloned() - .unwrap_or_default(); - let params = cfg.to_ivf_params(); - debug!( - core = self.core_id, - key = %index_key.2, - "creating IVF-PQ index" - ); - crate::engine::vector::ivf::IvfPqIndex::new(dim, params) - }); - - // IVF-PQ requires training before the first insert. - if ivf.n_cells() == 0 { - let refs: Vec<&[f32]> = vec![vector]; - let memory = nodedb_mem::ScopedMemory::new( - self.governor.clone(), - index_key.0, - index_key.1, - nodedb_mem::EngineId::Vector, - ); - if let Err(e) = ivf.train(&refs, memory) { - return self.response_error(task, crate::Error::from(e)); - } - } - - let vector_id = match ivf.add(vector) { - Ok(id) => id, - Err(e) => return self.response_error(task, crate::Error::from(e)), - }; - - // Register surrogate mapping using the actual IVF-assigned vector ID. - if surrogate != Surrogate::ZERO { - let coll = self - .vector_collections - .entry(index_key.clone()) - .or_insert_with(|| VectorCollection::new(dim, Default::default())); - coll.surrogate_map.insert(vector_id, surrogate); - coll.surrogate_to_local.insert(surrogate, vector_id); - } - - self.checkpoint_coordinator.mark_dirty("vector", 1); - // Record this write's version so cross-shard OCC read-set validation - // sees this insert, same as the HNSW insert path above. `ZERO` means - // no surrogate binding was made (headless insert) — floor-only. - if surrogate == Surrogate::ZERO { - self.note_collection_write_lsn(task, &index_key.2); - } else { - self.note_surrogate_write_lsn(task, tid, &index_key.2, surrogate.as_u32()); - } - self.response_ok(task) - } - /// Delete a vector by surrogate (sync inbound path). /// /// Resolves `surrogate → HNSW node_id` via `surrogate_to_local`, then diff --git a/nodedb/src/data/executor/handlers/vector_direct_row.rs b/nodedb/src/data/executor/handlers/vector_direct_row.rs index a11c12e92..08428cb46 100644 --- a/nodedb/src/data/executor/handlers/vector_direct_row.rs +++ b/nodedb/src/data/executor/handlers/vector_direct_row.rs @@ -264,8 +264,8 @@ impl CoreLoop { Ok(()) } - /// Bookkeeping every completed direct write runs once: seal the growing - /// segment when it is full, mark the checkpoint dirty, and record each + /// Bookkeeping every completed direct write runs once: seal a full growing + /// segment or train an IVF-PQ collection at its threshold, mark the checkpoint dirty, and record each /// touched surrogate's write version for cross-shard OCC validation. pub(in crate::data::executor) fn finish_vector_direct_write( &mut self, @@ -275,21 +275,10 @@ impl CoreLoop { collection: &str, surrogates: &[Surrogate], ) { - let seal_key = CoreLoop::vector_build_key(index_key); // A committed-redo install seals once the whole record landed, so a // rollback finds its inserts in the growing segment. - if !self.recording_redo_undo() - && let Some(coll) = self.vector_collections.get_mut(index_key) - && coll.needs_seal() - && let Some(req) = coll.seal(&seal_key) - && let Some(tx) = &self.build_tx - && let Err(e) = tx.send(req) - { - tracing::warn!( - core = self.core_id, - error = %e, - "failed to send HNSW build request" - ); + if !self.recording_redo_undo() { + self.settle_vector_collection(index_key); } self.checkpoint_coordinator.mark_dirty("vector", 1); for surrogate in surrogates { diff --git a/nodedb/src/data/executor/handlers/vector_index_drop.rs b/nodedb/src/data/executor/handlers/vector_index_drop.rs index d9ec51f57..1b720b9f6 100644 --- a/nodedb/src/data/executor/handlers/vector_index_drop.rs +++ b/nodedb/src/data/executor/handlers/vector_index_drop.rs @@ -38,7 +38,6 @@ impl CoreLoop { let (db, tenant, collection_key) = index_key.clone(); let had_index = self.vector_collections.remove(&index_key).is_some(); - self.ivf_indexes.remove(&index_key); self.vector_params.remove(&index_key); self.index_configs.remove(&index_key); self.declared_dims.remove(&index_key); diff --git a/nodedb/src/data/executor/handlers/vector_lifecycle.rs b/nodedb/src/data/executor/handlers/vector_lifecycle.rs index ca7ad4fce..2bbe2f50f 100644 --- a/nodedb/src/data/executor/handlers/vector_lifecycle.rs +++ b/nodedb/src/data/executor/handlers/vector_lifecycle.rs @@ -42,40 +42,6 @@ impl CoreLoop { ); let Some(coll) = self.vector_collections.get(&index_key) else { - // Check IVF index as fallback. - if let Some(ivf) = self.ivf_indexes.get(&index_key) { - let stats = nodedb_types::VectorIndexStats { - sealed_count: 0, - building_count: 0, - growing_vectors: ivf.len(), - sealed_vectors: 0, - live_count: ivf.len(), - tombstone_count: 0, - tombstone_ratio: 0.0, - quantization: nodedb_types::VectorIndexQuantization::Pq, - memory_bytes: 0, - disk_bytes: 0, - build_in_progress: false, - index_type: nodedb_types::VectorIndexType::IvfPq, - hnsw_m: 0, - hnsw_m0: 0, - hnsw_ef_construction: 0, - metric: "l2".into(), - dimensions: ivf.dim(), - seal_threshold: 0, - mmap_segment_count: 0, - arena_bytes: None, - }; - return match zerompk::to_msgpack_vec(&stats) { - Ok(bytes) => self.response_with_payload(task, bytes), - Err(e) => self.response_error( - task, - ErrorCode::Internal { - detail: format!("serialize stats: {e}"), - }, - ), - }; - } return self.response_error(task, ErrorCode::NotFound); }; diff --git a/nodedb/src/data/executor/handlers/vector_multi.rs b/nodedb/src/data/executor/handlers/vector_multi.rs index c386f9dee..ab79841af 100644 --- a/nodedb/src/data/executor/handlers/vector_multi.rs +++ b/nodedb/src/data/executor/handlers/vector_multi.rs @@ -87,33 +87,13 @@ impl CoreLoop { let database_id = task.request.database_id.as_u64(); let index_key = CoreLoop::vector_index_key(database_id, tid, collection, field_name); - // Validate dimension compatibility before taking mutable reference. - if let Some(existing) = self.vector_collections.get(&index_key) - && existing.dim() != dim - { - return self - .response_error(task, super::vector::dimension_mismatch(existing.dim(), dim)); - } - - // Get or create the vector collection. - let core_id = self.core_id; - let params = self - .vector_params - .get(&index_key) - .cloned() - .unwrap_or_default(); // A committed-redo install seals once the whole record landed. let defer_seal = self.recording_redo_undo(); - let coll = self - .vector_collections - .entry(index_key.clone()) - .or_insert_with(|| { - debug!( - core = core_id, - dim, "creating vector collection for multi-vector" - ); - crate::engine::vector::collection::VectorCollection::new(dim, params) - }); + let coll = + match self.get_or_create_vector_index(database_id, tid, collection, dim, field_name) { + Ok(coll) => coll, + Err(err) => return self.response_error(task, err), + }; // Build vector slices from flat data. let vector_slices: Vec<&[f32]> = (0..count) @@ -129,15 +109,8 @@ impl CoreLoop { Err(e) => return self.response_error(task, crate::Error::from(e)), }; - // Auto-seal if needed. - let seal_key = CoreLoop::vector_build_key(&index_key); - if !defer_seal - && coll.needs_seal() - && let Some(req) = coll.seal(&seal_key) - && let Some(tx) = &self.build_tx - && let Err(e) = tx.send(req) - { - warn!(core = self.core_id, error = %e, "failed to send HNSW build after multi-vector insert"); + if !defer_seal { + self.settle_vector_collection(&index_key); } self.checkpoint_coordinator.mark_dirty("vector", ids.len()); diff --git a/nodedb/src/data/executor/handlers/vector_multi_search_exec.rs b/nodedb/src/data/executor/handlers/vector_multi_search_exec.rs index 32eea63b0..c060ab715 100644 --- a/nodedb/src/data/executor/handlers/vector_multi_search_exec.rs +++ b/nodedb/src/data/executor/handlers/vector_multi_search_exec.rs @@ -8,13 +8,19 @@ //! (see `handlers::vector_search_exec::execute_vector_search` for that) -- //! `MultiSearch` staging/merge is an explicitly out-of-scope follow-up. +use nodedb_types::StorageKey; use tracing::debug; +use super::hybrid_key::HybridFusionKey; use super::vector_search::{ VectorMultiSearchParams, build_search_hit, effective_ef, encode_hits_response, }; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::response_codec::VectorSearchHit; +use crate::engine::vector::collection::VectorCollection; +use crate::engine::vector::hnsw::SearchResult; +use crate::query::fusion::{RankedResult, reciprocal_rank_fusion}; impl CoreLoop { /// Multi-vector search: query all named vector fields in a collection, @@ -50,7 +56,8 @@ impl CoreLoop { top_k.saturating_mul(2).max(20) }; - let mut all_results: Vec> = Vec::new(); + // Each searched field's collection with its hits. + let mut all_results: Vec<(&VectorCollection, Vec)> = Vec::new(); // The width of a field index the query could not be compared with. // Fields of other widths are skipped; when no field has the query's // width, the query is the caller's data error. @@ -78,7 +85,7 @@ impl CoreLoop { ef, filter_bitmap, ) { - Ok(results) => all_results.push(results), + Ok(results) => all_results.push((coll, results)), Err(code) => return self.response_error(task, code), } } @@ -94,81 +101,53 @@ impl CoreLoop { return self.response_error(task, ErrorCode::NotFound); } - // Single field — return directly. - if all_results.len() == 1 { - let Some(results) = all_results.into_iter().next() else { - return self.response_error(task, ErrorCode::NotFound); - }; - let doc_source = self.vector_collections.get(&plain_key); - let hits: Vec<_> = results - .iter() - .map(|r| build_search_hit(doc_source, r.id, r.distance)) - .map(|hit| { - self.attach_body( - task.request.database_id.as_u64(), - tid, - collection, - !rls_filters.is_empty(), - hit, - ) - }) - .take(fetch_k) - .collect(); - if let Some(ref m) = self.metrics { - m.record_vector_search(0); - m.record_query_by_engine("vector"); - } - return encode_hits_response(self, task, &hits); - } - - // RRF fusion across fields using shared fusion module. - use crate::query::fusion::{RankedResult, reciprocal_rank_fusion}; - - let ranked_lists: Vec> = all_results - .iter() - .map(|results| { + let attach = !rls_filters.is_empty(); + let hits: crate::Result> = + if let [(coll, results)] = all_results.as_slice() { + // Single field: its own ranking, resolved in its own collection. results + .iter() + .take(fetch_k) + .map(|r| build_search_hit(Some(*coll), r.id, r.distance)) + .map(|hit| self.attach_body(database_id, tid, collection, attach, hit)) + .collect() + } else { + // RRF across fields, fused on each row's surrogate. Every + // field's collection numbers its nodes on its own, so a local + // id names a row only within the collection that ranked it. + let ranked_lists: Vec>> = all_results .iter() .enumerate() - .map(|(rank, r)| RankedResult { - document_id: r.id.to_string(), - rank, - score: r.distance, - source: "vector", + .map(|(field, (coll, results))| { + results + .iter() + .enumerate() + .map(|(rank, r)| RankedResult { + document_id: FieldFusionKey::of(coll, field, r.id), + rank, + score: r.distance, + source: "vector", + }) + .collect() + }) + .collect(); + reciprocal_rank_fusion(&ranked_lists, None, top_k) + .into_iter() + .map(|f| { + let hit = VectorSearchHit { + id: f.document_id.hit_key(), + distance: f.rrf_score as f32, + doc_id: None, + body: None, + }; + self.attach_body(database_id, tid, collection, attach, hit) }) .collect() - }) - .collect(); - - let fused = reciprocal_rank_fusion(&ranked_lists, None, top_k); - - // Surface fused results with surrogate-as-id; CP fills doc_id and - // applies the RLS predicate at the response boundary. - let hits: Vec<_> = fused - .iter() - .filter_map(|f| { - let local_id: u32 = f.document_id.parse().ok()?; - let source = self.vector_collections.get(&plain_key).or_else(|| { - self.vector_collections - .iter() - .filter(|(k, _)| { - k.0 == db - && k.1 == tenant_id - && (k == &&plain_key || k.2.starts_with(&field_prefix)) - }) - .map(|(_, c)| c) - .next() - }); - let hit = build_search_hit(source, local_id, f.rrf_score as f32); - Some(self.attach_body( - task.request.database_id.as_u64(), - tid, - collection, - !rls_filters.is_empty(), - hit, - )) - }) - .collect(); + }; + let hits = match hits { + Ok(hits) => hits, + Err(e) => return self.response_error(task, e), + }; if let Some(ref m) = self.metrics { m.record_vector_search(0); m.record_query_by_engine("vector"); @@ -176,3 +155,30 @@ impl CoreLoop { encode_hits_response(self, task, &hits) } } + +/// The key a multi-field search fuses on. A bound hit fuses under its row's +/// storage key, so the same row ranked by two fields fuses into one result. +/// A headless hit has no row identity and fuses with nothing: its key names +/// the field that ranked it and its local id there. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)] +enum FieldFusionKey { + Bound(StorageKey), + Headless { field: usize, local_id: u32 }, +} + +impl FieldFusionKey { + fn of(collection: &VectorCollection, field: usize, local_id: u32) -> Self { + match collection.get_surrogate(local_id) { + Some(surrogate) => Self::Bound(StorageKey::for_surrogate(surrogate)), + None => Self::Headless { field, local_id }, + } + } + + /// The hit id the response carries. + fn hit_key(self) -> HybridFusionKey { + match self { + Self::Bound(key) => HybridFusionKey::Bound(key), + Self::Headless { local_id, .. } => HybridFusionKey::Headless(local_id), + } + } +} diff --git a/nodedb/src/data/executor/handlers/vector_params.rs b/nodedb/src/data/executor/handlers/vector_params.rs index 31965b24d..d66846275 100644 --- a/nodedb/src/data/executor/handlers/vector_params.rs +++ b/nodedb/src/data/executor/handlers/vector_params.rs @@ -170,6 +170,11 @@ impl CoreLoop { declared_dim: resolved_dim, }; + if resolved_dim > 0 + && let Err(e) = super::vector_settle::check_ivf_dim(&config, resolved_dim) + { + return self.response_error(task, e); + } if resolved_dim > 0 { self.declared_dims.insert(index_key.clone(), resolved_dim); } diff --git a/nodedb/src/data/executor/handlers/vector_search_exec.rs b/nodedb/src/data/executor/handlers/vector_search_exec.rs index 04bb2b62d..674ddce0c 100644 --- a/nodedb/src/data/executor/handlers/vector_search_exec.rs +++ b/nodedb/src/data/executor/handlers/vector_search_exec.rs @@ -10,7 +10,6 @@ use super::vector_search::{ surrogate_bitmap_to_global_ids, }; use super::vector_search_ann::{ResolvedAnnOptions, apply_ann_options, quantization_matches}; -use super::vector_search_ivf::SearchIvfParams; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::task::ExecutionTask; @@ -22,7 +21,8 @@ impl CoreLoop { /// the slow-path SELECT (the Control Plane response translator flattens /// the body's fields into the hit JSON so payload columns surface to /// the client). When `attach == false`, or the hit carries no surrogate - /// binding, the hit is returned unchanged. + /// binding, the hit is returned unchanged. A storage read error fails + /// the call: a hit without its body would skip the RLS predicate check. /// /// The bytes are normalized to a standard msgpack map through the shared /// sparse-body normalizer, resolved from the collection's registered kind. @@ -38,14 +38,14 @@ impl CoreLoop { collection: &str, attach: bool, mut hit: super::super::response_codec::VectorSearchHit, - ) -> super::super::response_codec::VectorSearchHit { + ) -> crate::Result { if !attach { - return hit; + return Ok(hit); } let Some(key) = hit.id.storage_key() else { - return hit; + return Ok(hit); }; - if let Ok(Some(bytes)) = self.sparse.get(database_id, tid, collection, &key) { + if let Some(bytes) = self.sparse.get(database_id, tid, collection, &key)? { let format = self.sparse_body_format( crate::types::DatabaseId::new(database_id), crate::types::TenantId::new(tid), @@ -61,7 +61,7 @@ impl CoreLoop { .into_owned(), ); } - hit + Ok(hit) } pub(in crate::data::executor) fn execute_vector_search( @@ -132,22 +132,8 @@ impl CoreLoop { let database_id = task.request.database_id.as_u64(); let index_key = CoreLoop::vector_index_key(database_id, tid, collection, field_name); - // Check for IVF-PQ index first. - if let Some(ivf) = self.ivf_indexes.get(&index_key) { - return self.search_ivf(SearchIvfParams { - task, - tid, - collection, - index_key: &index_key, - ivf, - query_vector, - top_k, - filter_bitmap, - rls_filters, - }); - } - - // Default: HNSW collection. + // Every index type is one `VectorCollection`: an IVF-PQ collection + // answers from its exact buffer or its trained IVF-PQ index. // If the specific field-named index does not exist, fall back to the // empty-field index. This handles data synced from NodeDB-Lite (which // uses collection-level storage, not named-field storage) being @@ -363,7 +349,7 @@ impl CoreLoop { // OR when RLS filters need them; the CP response translator flattens // the bytes' fields into the hit JSON for client column projection. let attach = !skip_payload_fetch || !rls_filters.is_empty(); - let mut hits: Vec<_> = results + let hits: crate::Result> = results .iter() .map(|r| build_search_hit(Some(collection_ref), r.id, r.distance)) .map(|hit| { @@ -376,6 +362,10 @@ impl CoreLoop { ) }) .collect(); + let mut hits = match hits { + Ok(hits) => hits, + Err(e) => return self.response_error(task, e), + }; let truncate_to = if rls_filters.is_empty() { top_k } else { diff --git a/nodedb/src/data/executor/handlers/vector_search_ivf.rs b/nodedb/src/data/executor/handlers/vector_search_ivf.rs deleted file mode 100644 index f1701f062..000000000 --- a/nodedb/src/data/executor/handlers/vector_search_ivf.rs +++ /dev/null @@ -1,89 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! IVF-PQ search for `CoreLoop::execute_vector_search`. - -use super::vector_search::{build_search_hit, encode_hits_response}; -use crate::bridge::envelope::Response; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; - -/// Parameters for [`CoreLoop::search_ivf`]. -pub(super) struct SearchIvfParams<'a> { - pub task: &'a ExecutionTask, - pub tid: u64, - pub collection: &'a str, - pub index_key: &'a (nodedb_types::DatabaseId, crate::types::TenantId, String), - pub ivf: &'a crate::engine::vector::ivf::IvfPqIndex, - pub query_vector: &'a [f32], - pub top_k: usize, - pub filter_bitmap: Option<&'a nodedb_types::SurrogateBitmap>, - pub rls_filters: &'a [u8], -} - -impl CoreLoop { - /// Search an IVF-PQ index with optional bitmap post-filtering. - pub(super) fn search_ivf(&self, params: SearchIvfParams<'_>) -> Response { - let SearchIvfParams { - task, - tid, - collection, - index_key, - ivf, - query_vector, - top_k, - filter_bitmap, - rls_filters, - } = params; - if ivf.dim() != query_vector.len() { - return self.response_error( - task, - super::vector::dimension_mismatch(ivf.dim(), query_vector.len()), - ); - } - if ivf.is_empty() { - return super::vector_search::empty_hits_response(self, task); - } - let fetch_k = if filter_bitmap.is_some() || !rls_filters.is_empty() { - top_k * self.query_tuning.bitmap_over_fetch_factor.max(2) - } else { - top_k - }; - let results = match ivf.search(query_vector, fetch_k) { - Ok(results) => results, - Err(e) => return self.response_error(task, crate::Error::from(e)), - }; - let surrogate_source = self.vector_collections.get(index_key); - - let mut hits: Vec<_> = results - .iter() - .map(|r| build_search_hit(surrogate_source, r.id, r.distance)) - .collect(); - - if let Some(surrogate_bm) = filter_bitmap { - // Bitmap is a set of surrogates: keep only bound hits whose - // surrogate is in the bitmap. A headless hit has none, so it - // never survives a surrogate-bitmap filter. - hits.retain(|h| { - h.id.storage_key() - .is_some_and(|key| surrogate_bm.contains(key.surrogate())) - }); - } - if !rls_filters.is_empty() { - // CP-side translator runs the predicate; DP only attaches body. - hits = hits - .into_iter() - .map(|h| { - self.attach_body(task.request.database_id.as_u64(), tid, collection, true, h) - }) - .collect(); - } else { - hits.truncate(top_k); - } - - if let Some(ref m) = self.metrics { - m.record_vector_search(0); - m.record_query_by_engine("vector"); - } - encode_hits_response(self, task, &hits) - } -} diff --git a/nodedb/src/data/executor/handlers/vector_settle.rs b/nodedb/src/data/executor/handlers/vector_settle.rs new file mode 100644 index 000000000..e32dd490d --- /dev/null +++ b/nodedb/src/data/executor/handlers/vector_settle.rs @@ -0,0 +1,170 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Vector collection construction and post-write settling. +//! +//! Every vector index, whatever its type, is one `VectorCollection` built +//! from the index configuration `CREATE VECTOR INDEX` set. After a write the +//! collection settles: a full growing segment seals and its HNSW build request +//! goes to `build_tx`, and an IVF-PQ collection that holds its training +//! threshold trains and moves its buffered vectors into the IVF-PQ index. + +use std::collections::HashMap; +use std::collections::hash_map::Entry; + +use tracing::{error, info, warn}; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::vector_direct_row::VectorIndexKey; +use crate::engine::vector::collection::VectorCollection; +use crate::engine::vector::hnsw::HnswParams; +use crate::engine::vector::index_config::{IndexConfig, IndexType}; + +/// The configuration a new collection under `key` is built with: the index +/// configuration when one was set, else the default index type over the +/// HNSW parameters set for `key`. +pub(in crate::data::executor) fn vector_index_config_for( + index_configs: &HashMap, + vector_params: &HashMap, + key: &VectorIndexKey, +) -> IndexConfig { + match index_configs.get(key) { + Some(config) => config.clone(), + None => IndexConfig { + hnsw: vector_params.get(key).cloned().unwrap_or_default(), + ..IndexConfig::default() + }, + } +} + +/// An IVF-PQ index splits each vector into `pq_m` equal subvectors, so its +/// dimension must be a multiple of `pq_m`. Any other index type passes. +pub(in crate::data::executor) fn check_ivf_dim( + config: &IndexConfig, + dim: usize, +) -> crate::Result<()> { + if config.index_type == IndexType::IvfPq + && (config.pq_m == 0 || !dim.is_multiple_of(config.pq_m)) + { + return Err(crate::Error::DataException { + detail: format!( + "an ivf_pq index needs a vector dimension divisible by pq_m {}; got dimension {dim}", + config.pq_m + ), + }); + } + Ok(()) +} + +impl CoreLoop { + /// Create the collection under `key` at `dim` when it does not exist, + /// from the index configuration registered under `config_key`. A + /// schemaless field index registers under the bare collection key and + /// stores under the field-qualified one. Fails as [`check_ivf_dim`] does. + pub(in crate::data::executor) fn ensure_vector_collection( + &mut self, + key: &VectorIndexKey, + config_key: &VectorIndexKey, + dim: usize, + ) -> crate::Result<&mut VectorCollection> { + match self.vector_collections.entry(key.clone()) { + Entry::Occupied(entry) => Ok(entry.into_mut()), + Entry::Vacant(entry) => { + let config = + vector_index_config_for(&self.index_configs, &self.vector_params, config_key); + check_ivf_dim(&config, dim)?; + Ok(entry.insert(VectorCollection::with_index_config(dim, config))) + } + } + } + + /// Settle the collection under `key` after a write: seal a full growing + /// segment and send its HNSW build, or train an IVF-PQ collection that + /// holds its training threshold. + pub(in crate::data::executor) fn settle_vector_collection(&mut self, key: &VectorIndexKey) { + let seal_key = CoreLoop::vector_build_key(key); + let Some(coll) = self.vector_collections.get_mut(key) else { + return; + }; + if coll.needs_seal() + && let Some(req) = coll.seal(&seal_key) + && let Some(tx) = &self.build_tx + && let Err(e) = tx.send(req) + { + warn!(core = self.core_id, error = %e, "failed to send HNSW build request"); + } + self.train_ivf_if_ready(key); + } + + /// Train the collection under `key` when it is an untrained IVF-PQ + /// collection holding its training threshold. + pub(in crate::data::executor) fn train_ivf_if_ready(&mut self, key: &VectorIndexKey) { + if self + .vector_collections + .get(key) + .is_some_and(|coll| coll.needs_ivf_training()) + { + self.train_vector_collection_ivf(key); + } + } + + /// Train every IVF-PQ collection that holds its training threshold. Boot + /// runs it once WAL replay and the store rebuild have restored the + /// buffers, so a collection that crossed its threshold before a restart + /// searches through IVF-PQ again without waiting for a write. + pub fn train_ready_ivf_collections(&mut self) { + let ready: Vec = self + .vector_collections + .iter() + .filter(|(_, coll)| coll.needs_ivf_training()) + .map(|(key, _)| key.clone()) + .collect(); + for key in ready { + self.train_vector_collection_ivf(&key); + } + } + + /// Train the IVF-PQ index of the collection under `key`. + /// + /// The write that crossed the threshold is already applied and durable, + /// so a training error does not fail it. The vectors stay in the exact + /// buffer, search stays correct, and the next write retries the training. + fn train_vector_collection_ivf(&mut self, key: &VectorIndexKey) { + let memory = nodedb_mem::ScopedMemory::new( + self.governor.clone(), + key.0, + key.1, + nodedb_mem::EngineId::Vector, + ); + // no-determinism: the wall clock only stamps the training for + // `SHOW VECTOR INDEX`; no stored state or search depends on it. + let trained_at_ms = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |d| u64::try_from(d.as_millis()).unwrap_or(u64::MAX)); + let Some(coll) = self.vector_collections.get_mut(key) else { + return; + }; + let threshold = coll.ivf_training_threshold(); + match coll.train_ivf(memory, trained_at_ms) { + Ok(()) => { + let trained_on = coll.ivf_index().map_or(0, |ivf| ivf.trained_on()); + info!( + core = self.core_id, + key = %key.2, + threshold, + trained_on, + "IVF-PQ index trained; buffered vectors moved into it" + ); + self.checkpoint_coordinator.mark_dirty("vector", trained_on); + } + Err(e) => { + error!( + core = self.core_id, + key = %key.2, + threshold, + error = %e, + "IVF-PQ training failed; vectors stay in the exact buffer and the next write retries" + ); + } + } + } +} diff --git a/nodedb/src/data/executor/handlers/vector_write.rs b/nodedb/src/data/executor/handlers/vector_write.rs index 3257f84ed..9798bd5b4 100644 --- a/nodedb/src/data/executor/handlers/vector_write.rs +++ b/nodedb/src/data/executor/handlers/vector_write.rs @@ -5,7 +5,7 @@ //! Extracted from `vector.rs` to keep file sizes within the 500-line limit. use nodedb_types::Surrogate; -use tracing::{debug, warn}; +use tracing::debug; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; @@ -40,14 +40,8 @@ impl CoreLoop { if let Err(e) = collection_ref.insert_batch_with_surrogates(vectors, surrogates) { return self.response_error(task, crate::Error::from(e)); } - let seal_key = CoreLoop::vector_build_key(&index_key); - if !defer_seal - && collection_ref.needs_seal() - && let Some(req) = collection_ref.seal(&seal_key) - && let Some(tx) = &self.build_tx - && let Err(e) = tx.send(req) - { - warn!(core = self.core_id, error = %e, "failed to send HNSW build request"); + if !defer_seal { + self.settle_vector_collection(&index_key); } self.checkpoint_coordinator .mark_dirty("vector", vectors.len()); diff --git a/nodedb/src/data/executor/wal_replay_vector.rs b/nodedb/src/data/executor/wal_replay_vector.rs index 43e2088a4..4079e75c0 100644 --- a/nodedb/src/data/executor/wal_replay_vector.rs +++ b/nodedb/src/data/executor/wal_replay_vector.rs @@ -23,8 +23,6 @@ impl CoreLoop { num_cores: usize, tombstones: &nodedb_wal::TombstoneSet, ) { - use crate::engine::vector::collection::VectorCollection; - use crate::engine::vector::hnsw::HnswParams; use nodedb_wal::record::RecordType; let mut inserted = 0usize; @@ -259,22 +257,18 @@ impl CoreLoop { &collection, &field_name, ); - let params = self - .vector_params - .get(&index_key) - .cloned() - .unwrap_or_else(|| { - tracing::debug!( - core = self.core_id, - %collection, - "no VectorParams found during WAL replay; using defaults" + let index = match self.ensure_vector_collection(&index_key, &index_key, dim) { + Ok(index) => index, + Err(e) => { + self.replay_record_rejected( + "vector", + record_lsn, + None, + &format!("vector record for '{collection}': {e}"), ); - HnswParams::default() - }); - let index = self - .vector_collections - .entry(index_key) - .or_insert_with(|| VectorCollection::new(dim, params)); + continue; + } + }; // Unlike the record-internal check above, this compares the // record against a LIVE index whose width the collection // may legitimately have changed since the record was @@ -344,22 +338,18 @@ impl CoreLoop { } let index_key = CoreLoop::vector_index_key(database_id, tenant_id, &collection, ""); - let params = self - .vector_params - .get(&index_key) - .cloned() - .unwrap_or_else(|| { - tracing::debug!( - core = self.core_id, - %collection, - "no VectorParams found during WAL replay; using defaults" + let index = match self.ensure_vector_collection(&index_key, &index_key, dim) { + Ok(index) => index, + Err(e) => { + self.replay_record_rejected( + "vector", + record_lsn, + None, + &format!("vector record for '{collection}': {e}"), ); - HnswParams::default() - }); - let index = self - .vector_collections - .entry(index_key) - .or_insert_with(|| VectorCollection::new(dim, params)); + continue; + } + }; // Unlike the record-internal check above, this compares the // record against a LIVE index whose width the collection // may legitimately have changed since the record was @@ -415,22 +405,18 @@ impl CoreLoop { skipped += 1; continue; } - let params = self - .vector_params - .get(&index_key) - .cloned() - .unwrap_or_else(|| { - tracing::debug!( - core = self.core_id, - %collection, - "no VectorParams found for batch replay; using defaults" + let index = match self.ensure_vector_collection(&index_key, &index_key, dim) { + Ok(index) => index, + Err(e) => { + self.replay_record_rejected( + "vector", + record_lsn, + None, + &format!("vector batch record for '{collection}': {e}"), ); - HnswParams::default() - }); - let index = self - .vector_collections - .entry(index_key) - .or_insert_with(|| VectorCollection::new(dim, params)); + continue; + } + }; // Checked as a whole before any vector lands, so a record // holding one vector of another width applies nothing. if let Err(e) = index.insert_batch_with_surrogates(&vectors, &[]) { diff --git a/nodedb/src/data/executor/wal_replay_vector_index_drop.rs b/nodedb/src/data/executor/wal_replay_vector_index_drop.rs index 1b6fc0866..d4fb928e7 100644 --- a/nodedb/src/data/executor/wal_replay_vector_index_drop.rs +++ b/nodedb/src/data/executor/wal_replay_vector_index_drop.rs @@ -32,7 +32,6 @@ impl CoreLoop { CoreLoop::vector_index_key(database_id, tenant_id, &collection, &field_name); let (db, tenant, _) = index_key.clone(); self.vector_collections.remove(&index_key); - self.ivf_indexes.remove(&index_key); self.vector_params.remove(&index_key); self.index_configs.remove(&index_key); self.declared_dims.remove(&index_key); diff --git a/nodedb/src/data/runtime/boot_replay.rs b/nodedb/src/data/runtime/boot_replay.rs index 0476806f2..a01b54511 100644 --- a/nodedb/src/data/runtime/boot_replay.rs +++ b/nodedb/src/data/runtime/boot_replay.rs @@ -47,6 +47,11 @@ pub(super) fn replay_wal_and_rebuild_indexes( // checkpoint + WAL replay above already restored. core.rebuild_vector_indexes_from_store(vector_index_param_seed); + // Replay applies inserts without settling. An IVF-PQ collection whose + // restored buffer holds its training threshold trains here, so search + // uses IVF-PQ again from the first request. + core.train_ready_ivf_collections(); + // The in-memory R-tree spatial index needs no separate backstop here. // A document collection's geometry is indexed by the same // `apply_point_put_spatial` side-effect on both the live write and the diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 7238f714d..95800e476 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -342,6 +342,7 @@ mod vector_index_txn_update_from_join; mod vector_index_txn_update_from_join_stmt_stage; mod vector_index_update_from_join_reindex; mod vector_index_update_reindex; +mod vector_ivf_pq_training; mod vector_primary_dml; mod vector_primary_fast_path; mod vector_primary_write_rls_predicate; diff --git a/nodedb/tests/wire/cases/vector_ivf_pq_training.rs b/nodedb/tests/wire/cases/vector_ivf_pq_training.rs new file mode 100644 index 000000000..fb8f45cbf --- /dev/null +++ b/nodedb/tests/wire/cases/vector_ivf_pq_training.rs @@ -0,0 +1,131 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! An `ivf_pq` vector index buffers vectors until it holds its training +//! threshold, `max(IVF_CELLS, 256)`, then trains and serves through IVF-PQ. +//! +//! - Below the threshold every insert succeeds and search finds each vector +//! exactly. +//! - Past the threshold search still finds every vector once. +//! - A WAL-only restart keeps every vector searchable and the index trained. +//! - `SHOW VECTOR INDEX` reports the threshold and the training. + +use crate::harness::TestServer; + +const COLL: &str = "ivf_docs"; +/// Vectors held once the threshold is crossed. +const PAST_THRESHOLD: usize = 300; + +/// Distinct per `i`: the first component is `i + 1`. +fn vector(i: usize) -> [f32; 8] { + let mut v = [0.0f32; 8]; + for (slot, m) in v.iter_mut().zip([1usize, 7, 11, 13, 17, 19, 23, 29]) { + *slot = (if m == 1 { i + 1 } else { i % m + 1 }) as f32; + } + v +} + +fn array(v: &[f32; 8]) -> String { + let parts: Vec = v.iter().map(|x| format!("{x:.1}")).collect(); + format!("[{}]", parts.join(", ")) +} + +async fn insert_range(srv: &TestServer, range: std::ops::Range) { + for i in range { + srv.exec(&format!( + "INSERT INTO {COLL} {{ id: 'v{i}', embedding: {} }}", + array(&vector(i)) + )) + .await + .unwrap_or_else(|e| panic!("insert of v{i} must succeed: {e}")); + } +} + +async fn nearest(srv: &TestServer, i: usize) -> String { + let rows = srv + .query_rows(&format!( + "SELECT id FROM {COLL} ORDER BY vector_distance(embedding, ARRAY{}) LIMIT 1", + array(&vector(i)) + )) + .await + .unwrap(); + assert_eq!(rows.len(), 1, "search for v{i} must return a row"); + rows[0][0].clone() +} + +/// Every id the index returns for one wide search, sorted. +async fn all_ids(srv: &TestServer) -> Vec { + let rows = srv + .query_rows(&format!( + "SELECT id FROM {COLL} ORDER BY vector_distance(embedding, ARRAY{}) LIMIT 1000", + array(&vector(0)) + )) + .await + .unwrap(); + let mut ids: Vec = rows.into_iter().map(|r| r[0].clone()).collect(); + ids.sort(); + ids +} + +fn expected_ids(n: usize) -> Vec { + let mut ids: Vec = (0..n).map(|i| format!("v{i}")).collect(); + ids.sort(); + ids +} + +async fn status(srv: &TestServer, property: &str) -> String { + let rows = srv + .query_rows(&format!("SHOW VECTOR INDEX status ON {COLL}.embedding")) + .await + .unwrap(); + rows.iter() + .find(|r| r[0] == property) + .map(|r| r[1].clone()) + .unwrap_or_else(|| panic!("SHOW VECTOR INDEX must report {property}: {rows:?}")) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn ivf_pq_buffers_trains_and_survives_a_restart() { + let srv = TestServer::start().await; + srv.exec(&format!( + "CREATE COLLECTION {COLL} WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + srv.exec(&format!( + "CREATE VECTOR INDEX idx_{COLL} ON {COLL} (embedding) METRIC l2 DIM 8 \ + INDEX_TYPE ivf_pq PQ_M 4 IVF_CELLS 4 IVF_NPROBE 4" + )) + .await + .unwrap(); + + // Below the threshold: the exact buffer answers. + insert_range(&srv, 0..10).await; + for i in 0..10 { + assert_eq!(nearest(&srv, i).await, format!("v{i}")); + } + assert_eq!(status(&srv, "ivf_training_threshold").await, "256"); + assert_eq!(status(&srv, "ivf_trained").await, "false"); + + // Past the threshold: trained, and every vector found once. + insert_range(&srv, 10..PAST_THRESHOLD).await; + assert_eq!(status(&srv, "ivf_trained").await, "true"); + assert_eq!(status(&srv, "ivf_cells").await, "4"); + assert_eq!(all_ids(&srv).await, expected_ids(PAST_THRESHOLD)); + for i in [0, 9, 255, 256, 299] { + assert_eq!(nearest(&srv, i).await, format!("v{i}")); + } + + // WAL-only restart: the buffer replays, trains again, and serves. + let (srv, dir) = srv.take_dir(); + srv.graceful_shutdown().await; + let (srv2, _dir) = TestServer::open_on_path(dir).await; + assert_eq!(status(&srv2, "ivf_trained").await, "true"); + assert_eq!(all_ids(&srv2).await, expected_ids(PAST_THRESHOLD)); + for i in [0, 9, 255, 256, 299] { + assert_eq!(nearest(&srv2, i).await, format!("v{i}")); + } + + // Inserts after the restart land in the trained index. + insert_range(&srv2, PAST_THRESHOLD..PAST_THRESHOLD + 5).await; + assert_eq!(all_ids(&srv2).await, expected_ids(PAST_THRESHOLD + 5)); +} From 9f3a62c1da23d9c2b86fb87b78ea135f9479c102 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 06:25:17 +0800 Subject: [PATCH 47/64] feat(vector): queue HNSW builds on a bounded, non-blocking builder The HNSW builder thread used unbounded channels and blocking sends, so a core could stall on a full completion queue and a failed insert only logged and skipped a vector, silently shifting every later node's id. The builder now takes bounded request/completion queues, builds vectors in order and fails the whole build on the first insert error, and reports success or failure back through a new VectorBuildQueue that backlogs requests the queue can't take yet and drains completions once per tick. Sealing a growing segment, settling after boot, and REINDEX CONCURRENTLY all route through this one queue and its collection.rs::build install path, replacing REINDEX's separate rebuild-on-a-plain-thread codepath and the old blocking send on seal/settle. VectorCollection tracks per-key completed/failed build counts and a configurable seal threshold (VectorTuning), surfaced through SHOW VECTOR INDEX and Prometheus. Fix a duplicate-node bug: a vector write over an already-indexed row left the row's original node live after a subsequent delete, since the delete looked up the surrogate's originally recorded node instead of the node a later write rebound it to. INSERT also stopped double-indexing fields a vector index already covers via the document write. Add ErrorCode::BadRequest so the Data Plane can report a malformed request (SQLSTATE 42601) the same way the Control Plane already does, and route CollectionDeactivated/CalvinSerializationConflict/SourceFrozen/ FeatureNotSupported/CrossCollectionNotColocated through the existing NotFound/ConflictRetry/Unsupported classes instead of falling through to Internal. --- .../src/rpc_codec/data_plane_error.rs | 4 + nodedb-types/src/vector_index_stats.rs | 10 + nodedb-vector/src/builder.rs | 162 ++++++-- nodedb-vector/src/collection/build.rs | 389 ++++++++++++++++++ nodedb-vector/src/collection/checkpoint.rs | 44 +- nodedb-vector/src/collection/lifecycle.rs | 84 ++-- nodedb-vector/src/collection/mod.rs | 3 +- nodedb-vector/src/collection/quantize.rs | 16 +- nodedb-vector/src/collection/segment.rs | 23 +- nodedb-vector/src/collection/stats.rs | 4 + nodedb-vector/src/lib.rs | 2 +- nodedb/src/bootstrap/data_plane.rs | 1 + nodedb/src/bridge/envelope/error_code.rs | 24 +- .../control/cluster/data_plane_error_wire.rs | 2 + .../src/control/metrics/prometheus/engines.rs | 30 ++ nodedb/src/control/metrics/system/fields.rs | 10 + nodedb/src/control/metrics/system/record.rs | 28 ++ .../server/dispatch_utils/write_abort.rs | 3 + .../collection/dml/indexed_vector_fields.rs | 75 ++++ .../ddl/neutral/collection/dml/insert.rs | 18 +- .../shared/ddl/neutral/collection/dml/mod.rs | 1 + .../ddl/neutral/maintenance/vector_index.rs | 3 + .../src/control/server/shared/ddl/sqlstate.rs | 13 + .../data/executor/core_loop/maintenance.rs | 6 + nodedb/src/data/executor/core_loop/mod.rs | 1 + nodedb/src/data/executor/core_loop/open.rs | 4 +- nodedb/src/data/executor/core_loop/state.rs | 11 +- .../executor/core_loop/vector_build_queue.rs | 123 ++++++ .../data/executor/handlers/control/reindex.rs | 100 ++--- .../handlers/control/reindex_apply.rs | 79 +--- nodedb/src/data/executor/handlers/mod.rs | 1 + .../handlers/point/apply_put/vector/put.rs | 47 ++- .../handlers/point/apply_put/vector/remove.rs | 10 +- nodedb/src/data/executor/handlers/vector.rs | 9 +- .../data/executor/handlers/vector_build.rs | 305 ++++++++++++++ .../executor/handlers/vector_lifecycle.rs | 89 +--- .../data/executor/handlers/vector_settle.rs | 44 +- .../vector_checkpoint/build_completions.rs | 143 +++++-- .../data/executor/vector_checkpoint/load.rs | 3 +- nodedb/src/data/runtime/boot_replay.rs | 9 +- nodedb/src/data/runtime/config.rs | 3 + nodedb/src/data/runtime/spawn.rs | 4 + nodedb/src/diag/context/mod.rs | 2 + nodedb/src/diag/context/vector_build.rs | 153 +++++++ nodedb/src/diag/mod.rs | 24 +- nodedb/src/diag/recording/mod.rs | 5 + nodedb/src/diag/recording/vector_build.rs | 94 +++++ nodedb/src/error_from_data_plane.rs | 1 + .../inproc/cases/reindex_vector_concurrent.rs | 46 ++- nodedb/tests/wire/cases/mod.rs | 1 + nodedb/tests/wire/cases/sql_hybrid_search.rs | 21 +- .../tests/wire/cases/sql_three_source_rrf.rs | 4 +- nodedb/tests/wire/cases/vector_hnsw_build.rs | 242 +++++++++++ nodedb/tests/wire/harness/config_toml.rs | 14 + nodedb/tests/wire/harness/lifecycle.rs | 30 ++ 55 files changed, 2134 insertions(+), 443 deletions(-) create mode 100644 nodedb-vector/src/collection/build.rs create mode 100644 nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs create mode 100644 nodedb/src/data/executor/core_loop/vector_build_queue.rs create mode 100644 nodedb/src/data/executor/handlers/vector_build.rs create mode 100644 nodedb/src/diag/context/vector_build.rs create mode 100644 nodedb/src/diag/recording/vector_build.rs create mode 100644 nodedb/tests/wire/cases/vector_hnsw_build.rs diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index b33c57ed3..25865471b 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -156,6 +156,10 @@ pub enum DataPlaneErrorCode { hold: DataPlaneSyncHold, applied_seq: u64, }, + /// The request itself is malformed (SQLSTATE `42601`). + BadRequest { + detail: String, + }, } /// Wire mirror of `nodedb::bridge::envelope::SyncHold`. diff --git a/nodedb-types/src/vector_index_stats.rs b/nodedb-types/src/vector_index_stats.rs index 34018636d..0bbec7e01 100644 --- a/nodedb-types/src/vector_index_stats.rs +++ b/nodedb-types/src/vector_index_stats.rs @@ -128,6 +128,12 @@ pub struct VectorIndexStats { pub arena_bytes: Option, /// IVF-PQ training state. `None` unless the index type is `ivf_pq`. pub ivf: Option, + /// HNSW builds of this index waiting for or running on the builder. + pub builds_queued: usize, + /// HNSW builds of this index installed since the core opened it. + pub builds_completed: u64, + /// HNSW builds of this index that failed since the core opened it. + pub builds_failed: u64, } /// Training state of an IVF-PQ vector index. @@ -198,6 +204,9 @@ mod tests { cells: 16, nprobe: 4, }), + builds_queued: 1, + builds_completed: 2, + builds_failed: 0, }; let bytes = zerompk::to_msgpack_vec(&stats).unwrap(); let restored: VectorIndexStats = zerompk::from_msgpack(&bytes).unwrap(); @@ -206,5 +215,6 @@ mod tests { assert_eq!(restored.quantization, VectorIndexQuantization::Sq8); assert_eq!(restored.index_type, VectorIndexType::Hnsw); assert_eq!(restored.ivf, stats.ivf); + assert_eq!(restored.builds_completed, 2); } } diff --git a/nodedb-vector/src/builder.rs b/nodedb-vector/src/builder.rs index 372119f9b..99b5ba889 100644 --- a/nodedb-vector/src/builder.rs +++ b/nodedb-vector/src/builder.rs @@ -1,9 +1,20 @@ // SPDX-License-Identifier: Apache-2.0 -//! Background HNSW index builder thread. +//! Background HNSW builder thread. //! -//! Each Data Plane core has one builder thread that processes HNSW -//! construction requests sequentially (FIFO). +//! Each Data Plane core owns one builder thread. The core sends build +//! requests with `try_send` and drains finished builds with `try_recv` once +//! per tick, so a build never blocks the core's reactor. Both channels are +//! bounded: +//! +//! - The request queue holds `capacity` requests. When it is full the core +//! keeps the job in its own backlog and sends it on a later tick. The +//! segment stays searchable by brute force meanwhile. +//! - The completion queue holds `capacity` results. When it is full the +//! builder thread waits for the core to drain it. +//! +//! The thread builds requests in FIFO order and stops when the core drops +//! its request sender. use std::sync::mpsc; use std::thread::JoinHandle; @@ -11,18 +22,28 @@ use std::thread::JoinHandle; use tracing::{debug, info, warn}; use crate::collection::{BuildComplete, BuildRequest}; +use crate::error::VectorError; use crate::hnsw::HnswIndex; /// Sender half: TPC core sends build requests to the builder thread. -pub type BuildSender = mpsc::Sender; +pub type BuildSender = mpsc::SyncSender; /// Receiver half: TPC core receives completed builds. pub type CompleteReceiver = mpsc::Receiver; -/// Spawn a background HNSW builder thread for a Data Plane core. -pub fn spawn_builder(core_id: usize) -> (BuildSender, CompleteReceiver, JoinHandle<()>) { - let (request_tx, request_rx) = mpsc::channel::(); - let (complete_tx, complete_rx) = mpsc::channel::(); +/// Build requests a core's builder queue holds. Each request carries a whole +/// segment's vectors, so the bound caps the memory in flight. +pub const BUILD_QUEUE_CAPACITY: usize = 4; + +/// Spawn the HNSW builder thread for Data Plane core `core_id`, with request +/// and completion queues of `capacity` entries each. Fails when the OS +/// refuses the thread. +pub fn spawn_builder( + core_id: usize, + capacity: usize, +) -> std::io::Result<(BuildSender, CompleteReceiver, JoinHandle<()>)> { + let (request_tx, request_rx) = mpsc::sync_channel::(capacity); + let (complete_tx, complete_rx) = mpsc::sync_channel::(capacity); let handle = std::thread::Builder::new() .name(format!("hnsw-builder-{core_id}")) @@ -30,13 +51,16 @@ pub fn spawn_builder(core_id: usize) -> (BuildSender, CompleteReceiver, JoinHand info!(core_id, "HNSW builder thread started"); builder_loop(core_id, request_rx, complete_tx); info!(core_id, "HNSW builder thread stopped"); - }) - .expect("failed to spawn HNSW builder thread"); + })?; - (request_tx, complete_rx, handle) + Ok((request_tx, complete_rx, handle)) } -fn builder_loop(core_id: usize, rx: mpsc::Receiver, tx: mpsc::Sender) { +fn builder_loop( + core_id: usize, + rx: mpsc::Receiver, + tx: mpsc::SyncSender, +) { while let Ok(req) = rx.recv() { debug!( core_id, @@ -46,40 +70,96 @@ fn builder_loop(core_id: usize, rx: mpsc::Receiver, tx: mpsc::Send dim = req.dim, "building HNSW index" ); - let start = std::time::Instant::now(); - let mut index = HnswIndex::with_seed( - req.dim, - req.params, - (core_id as u64 + 1) * 1000 + req.segment_id as u64, - ); + let (key, segment_id, kind) = (req.key.clone(), req.segment_id, req.kind); + let result = build(core_id, req); + match &result { + Ok(index) => info!( + core_id, + key = %key, + segment_id, + vectors = index.len(), + elapsed_ms = start.elapsed().as_millis() as u64, + "HNSW index built" + ), + Err(e) => warn!(core_id, key = %key, segment_id, error = %e, "HNSW build failed"), + } + let complete = BuildComplete { + key, + segment_id, + kind, + result, + }; + if tx.send(complete).is_err() { + warn!(core_id, "builder: core channel closed, stopping"); + break; + } + } +} - for vector in req.vectors { - index - .insert(vector) - .unwrap_or_else(|e| tracing::error!(error = %e, "HNSW insert failed")); +/// Build one graph, inserting every vector in local-id order so node `i` of +/// the graph is vector `i` of the request. The first insert error fails the +/// build: a skipped vector would shift every later id. +fn build(core_id: usize, req: BuildRequest) -> Result { + let seed = (core_id as u64 + 1) * 1000 + u64::from(req.segment_id); + let mut index = HnswIndex::with_seed(req.dim, req.params, seed); + for vector in req.vectors { + index.insert(vector)?; + } + Ok(index) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::collection::BuildKind; + use crate::hnsw::HnswParams; + + fn request(segment_id: u32, vectors: Vec>) -> BuildRequest { + BuildRequest { + key: "k".into(), + segment_id, + kind: BuildKind::Seal, + vectors, + dim: 2, + params: HnswParams::default(), } + } - let elapsed = start.elapsed(); - info!( - core_id, - key = %req.key, - segment_id = req.segment_id, - vectors = index.len(), - elapsed_ms = elapsed.as_millis() as u64, - "HNSW index built" - ); + #[test] + fn builds_keep_one_node_per_vector_and_report_errors() { + let (tx, rx, handle) = spawn_builder(0, 1).unwrap(); + tx.send(request(1, vec![vec![1.0, 0.0], vec![0.0, 1.0]])) + .unwrap(); + tx.send(request(2, vec![vec![1.0, 0.0], vec![1.0]])) + .unwrap(); - if tx - .send(BuildComplete { - key: req.key, - segment_id: req.segment_id, - index, - }) - .is_err() - { - warn!(core_id, "builder: core channel closed, stopping"); - break; + let first = rx.recv().unwrap(); + assert_eq!(first.segment_id, 1); + assert_eq!(first.result.unwrap().len(), 2); + let second = rx.recv().unwrap(); + assert!(matches!( + second.result, + Err(VectorError::DimensionMismatch { .. }) + )); + + drop(tx); + handle.join().unwrap(); + } + + #[test] + fn a_full_request_queue_refuses_without_blocking() { + let (tx, _rx, _handle) = spawn_builder(0, 1).unwrap(); + // The thread holds at most one request in hand and one queued; with + // the completion side undrained, later sends find the queue full. + let mut refused = false; + for id in 0..8 { + if let Err(mpsc::TrySendError::Full(_)) = tx.try_send(request(id, vec![vec![1.0, 0.0]])) + { + refused = true; + break; + } } + assert!(refused, "a bounded queue must refuse once full"); } } diff --git a/nodedb-vector/src/collection/build.rs b/nodedb-vector/src/collection/build.rs new file mode 100644 index 000000000..2fb5209b7 --- /dev/null +++ b/nodedb-vector/src/collection/build.rs @@ -0,0 +1,389 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! HNSW builds for a `VectorCollection`: the requests the owning core sends +//! to its builder thread, and installing the finished graphs. +//! +//! Every request carries one vector per local node id, soft-deleted nodes +//! included, so a built graph keeps the ids its segment had. Installing a +//! build applies the tombstones the segment holds at that moment, so a +//! delete that landed while the build ran is kept. +//! +//! - A seal build promotes a building segment to sealed. +//! - A rebuild replaces a sealed segment in place, from that segment's own +//! vectors, and quantizes it again under the collection's config. + +use nodedb_mem::ScopedMemory; +use nodedb_types::VectorQuantization; + +use crate::error::VectorError; +use crate::hnsw::HnswIndex; +use crate::index_config::IndexType; + +use super::lifecycle::VectorCollection; +use super::lifecycle_insert_ops::sealed_vector; +use super::segment::{BuildKind, BuildRequest, SealedSegment}; + +impl VectorCollection { + /// Segment ids of the building segments, oldest first. + pub fn building_segment_ids(&self) -> Vec { + self.building.iter().map(|b| b.segment_id).collect() + } + + /// A seal build request for building segment `segment_id`, read from its + /// flat vectors. `None` when no building segment has that id. + pub fn build_request_for(&self, key: &str, segment_id: u32) -> Option { + let seg = self.building.iter().find(|b| b.segment_id == segment_id)?; + let vectors = (0..seg.flat.len() as u32) + .filter_map(|i| seg.flat.get_vector_raw(i).map(<[f32]>::to_vec)) + .collect(); + Some(BuildRequest { + key: key.to_string(), + segment_id, + kind: BuildKind::Seal, + vectors, + dim: self.dim, + params: self.params.clone(), + }) + } + + /// Base ids of the non-empty sealed segments, in segment order. + pub fn sealed_base_ids(&self) -> Vec { + self.sealed + .iter() + .filter(|s| !s.index.is_empty()) + .map(|s| s.base_id) + .collect() + } + + /// A rebuild request for the sealed segment at `base_id`, under the + /// current HNSW params, read from that segment only: the growing and + /// building segments keep their own vectors. `None` when no sealed + /// segment starts at `base_id`. + /// + /// Fails with [`VectorError::VectorUnavailable`] when a node's vector + /// cannot be read. + pub fn rebuild_request_for( + &mut self, + key: &str, + base_id: u32, + ) -> Result, VectorError> { + let Some(seg) = self.sealed.iter().find(|s| s.base_id == base_id) else { + return Ok(None); + }; + let len = seg.index.len(); + let mut vectors = Vec::with_capacity(len); + for local in 0..len as u32 { + let v = sealed_vector(seg, local).ok_or(VectorError::VectorUnavailable { + id: base_id + local, + })?; + vectors.push(v); + } + let segment_id = self.next_segment_id; + self.next_segment_id += 1; + Ok(Some(BuildRequest { + key: key.to_string(), + segment_id, + kind: BuildKind::Rebuild { base_id, len }, + vectors, + dim: self.dim, + params: self.params.clone(), + })) + } + + /// Install a seal build: promote building segment `segment_id` to sealed + /// with its current tombstones. Returns `false`, changing nothing, when + /// no building segment has that id (a truncate dropped it) or the graph + /// does not hold one node per vector of the segment. + pub fn complete_build( + &mut self, + segment_id: u32, + index: HnswIndex, + memory: ScopedMemory, + ) -> bool { + let Some(pos) = self + .building + .iter() + .position(|b| b.segment_id == segment_id) + else { + return false; + }; + let mut index = index; + if index.len() != self.building[pos].flat.len() { + tracing::error!( + segment_id, + built = index.len(), + expected = self.building[pos].flat.len(), + "HNSW build does not match its segment; segment stays on brute force" + ); + return false; + } + let building = self.building.remove(pos); + for local in 0..building.flat.len() as u32 { + if building.flat.is_deleted(local) { + index.delete(local); + } + } + let seg = self.sealed_segment_from(segment_id, building.base_id, index, &memory); + self.sealed.push(seg); + self.builds_completed += 1; + self.refresh_codec_dispatch(); + true + } + + /// Install a rebuild of the sealed segment at `base_id`, read when it held + /// `len` nodes. The old segment's tombstones carry over and the segment is + /// quantized again under the collection's config. Returns `false`, + /// changing nothing, when no sealed segment at `base_id` still holds + /// `len` nodes, or the graph does not hold `len` nodes. + pub fn complete_rebuild( + &mut self, + segment_id: u32, + base_id: u32, + len: usize, + index: HnswIndex, + memory: ScopedMemory, + ) -> bool { + let Some(pos) = self + .sealed + .iter() + .position(|s| s.base_id == base_id && s.index.len() == len) + else { + return false; + }; + if index.len() != len { + tracing::error!( + base_id, + built = index.len(), + expected = len, + "HNSW rebuild does not match its segment; the old segment stays" + ); + return false; + } + let mut index = index; + for local in 0..len as u32 { + if self.sealed[pos].index.is_deleted(local) { + index.delete(local); + } + } + let seg = self.sealed_segment_from(segment_id, base_id, index, &memory); + let old = std::mem::replace(&mut self.sealed[pos], seg); + self.drop_sealed(old); + self.builds_completed += 1; + self.refresh_codec_dispatch(); + true + } + + /// Record a build that failed. The segment stays as it was. + pub fn note_build_failed(&mut self) { + self.builds_failed += 1; + } + + /// A sealed segment for `index`: quantized under the collection's config + /// and placed on the storage tier the memory budget allows. + fn sealed_segment_from( + &mut self, + segment_id: u32, + base_id: u32, + index: HnswIndex, + memory: &ScopedMemory, + ) -> SealedSegment { + let use_codec_dispatch = self.codec_dispatch_tag().is_some(); + let use_pq = !use_codec_dispatch && self.index_config.index_type == IndexType::HnswPq; + let (sq8, pq) = if use_codec_dispatch { + (None, None) + } else if use_pq { + ( + None, + Self::build_pq_for_index(&index, self.index_config.pq_m, memory.clone()), + ) + } else { + (Self::build_sq8_for_index(&index), None) + }; + let (tier, mmap_vectors) = self.resolve_tier_for_build(segment_id, base_id, &index, memory); + SealedSegment { + index, + base_id, + sq8, + pq, + tier, + mmap_vectors, + } + } + + /// Drop a replaced sealed segment, removing its mmap file. + fn drop_sealed(&mut self, seg: SealedSegment) { + let mmap_path = seg.mmap_vectors.as_ref().map(|m| m.path().to_path_buf()); + // Unmap before the file goes. + drop(seg); + if let Some(path) = mmap_path { + self.mmap_segment_count = self.mmap_segment_count.saturating_sub(1); + if let Err(e) = std::fs::remove_file(&path) { + tracing::warn!( + path = %path.display(), + error = %e, + "vector rebuild: replaced mmap segment file not removed" + ); + } + } + } + + /// The codec-dispatch tag the collection's quantization selects. + fn codec_dispatch_tag(&self) -> Option<&'static str> { + match self.quantization { + VectorQuantization::RaBitQ => Some("rabitq"), + VectorQuantization::Bbq => Some("bbq"), + _ => None, + } + } + + /// Rebuild the collection-level codec-dispatch index over the sealed + /// segments when the quantization selects one. + fn refresh_codec_dispatch(&mut self) { + let Some(tag) = self.codec_dispatch_tag() else { + return; + }; + if let Err(e) = self.build_codec_dispatch(tag).map(|_| ()) { + // Without the codec index the sealed segments are searched by + // their own HNSW graphs, which answer the same queries. + tracing::error!(error = %e, tag, "codec-dispatch build failed; searching sealed segments directly"); + self.codec_dispatch = None; + } + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use super::*; + use crate::hnsw::HnswParams; + use crate::test_support::test_memory; + + fn vector(i: usize) -> Vec { + vec![(i + 1) as f32, (i % 7 + 1) as f32, (i % 11 + 1) as f32] + } + + fn l2() -> HnswParams { + HnswParams { + metric: crate::distance::DistanceMetric::L2, + ..HnswParams::default() + } + } + + fn build(req: &BuildRequest) -> HnswIndex { + let mut index = HnswIndex::with_seed(req.dim, req.params.clone(), 7); + for v in &req.vectors { + index.insert(v.clone()).unwrap(); + } + index + } + + /// Ten L2 vectors bound to surrogates 1..=10, with a seal threshold of 10. + fn collection() -> VectorCollection { + let mut coll = VectorCollection::with_seal_threshold(3, l2(), 10); + for i in 0..10 { + coll.insert_with_surrogate(vector(i), Surrogate::new(i as u32 + 1)) + .unwrap(); + } + coll + } + + #[test] + fn a_seal_build_keeps_ids_and_the_deletes_made_meanwhile() { + let mut coll = collection(); + coll.delete(2); + let req = coll.seal("k").unwrap(); + assert_eq!(req.vectors.len(), 10, "deleted vectors keep their slot"); + coll.delete(4); + + assert!(coll.complete_build(req.segment_id, build(&req), test_memory())); + + assert!(coll.building.is_empty()); + assert!(!coll.is_live(2) && !coll.is_live(4) && coll.is_live(9)); + assert_eq!(coll.vector_for_id(7), Some(vector(7))); + assert_eq!(coll.search(&vector(7), 1, 64).unwrap()[0].id, 7); + assert_eq!(coll.get_surrogate(7), Some(Surrogate::new(8))); + } + + #[test] + fn a_build_of_the_wrong_size_is_refused() { + let mut coll = collection(); + let req = coll.seal("k").unwrap(); + let mut short = HnswIndex::new(3, l2()); + short.insert(vector(0)).unwrap(); + assert!(!coll.complete_build(req.segment_id, short, test_memory())); + assert_eq!(coll.building.len(), 1, "the segment stays on brute force"); + } + + #[test] + fn a_rebuild_keeps_ids_codes_and_tombstones() { + let mut coll = VectorCollection::with_pq_config(3, l2(), 3); + coll.set_seal_threshold(10); + for i in 0..10 { + coll.insert_with_surrogate(vector(i), Surrogate::new(i as u32 + 1)) + .unwrap(); + } + let req = coll.seal("k").unwrap(); + coll.complete_build(req.segment_id, build(&req), test_memory()); + coll.delete(1); + + let req = coll.rebuild_request_for("k", 0).unwrap().unwrap(); + assert_eq!( + req.kind, + BuildKind::Rebuild { + base_id: 0, + len: 10 + } + ); + coll.delete(6); + let BuildKind::Rebuild { base_id, len } = req.kind else { + panic!("a rebuild request"); + }; + assert!(coll.complete_rebuild(req.segment_id, base_id, len, build(&req), test_memory())); + + assert_eq!(coll.sealed.len(), 1); + assert!( + coll.sealed[0].pq.is_some(), + "the segment is quantized again" + ); + assert!(!coll.is_live(1) && !coll.is_live(6)); + for i in [0usize, 3, 9] { + assert_eq!(coll.search(&vector(i), 1, 64).unwrap()[0].id, i as u32); + assert_eq!( + coll.get_surrogate(i as u32), + Some(Surrogate::new(i as u32 + 1)) + ); + } + } + + #[test] + fn a_rebuild_of_a_compacted_segment_is_refused() { + let mut coll = collection(); + let req = coll.seal("k").unwrap(); + coll.complete_build(req.segment_id, build(&req), test_memory()); + coll.delete(3); + let req = coll.rebuild_request_for("k", 0).unwrap().unwrap(); + assert_eq!(coll.compact(), 1); + let BuildKind::Rebuild { base_id, len } = req.kind else { + panic!("a rebuild request"); + }; + assert!(!coll.complete_rebuild(req.segment_id, base_id, len, build(&req), test_memory())); + assert_eq!(coll.sealed[0].index.len(), 9, "the compacted segment stays"); + } + + #[test] + fn an_unbuilt_segment_survives_a_checkpoint_as_building() { + let mut coll = collection(); + coll.delete(5); + coll.seal("k").unwrap(); + let bytes = coll.checkpoint_to_bytes(None).unwrap(); + let restored = VectorCollection::from_checkpoint(&bytes, None, test_memory()).unwrap(); + + let ids = restored.building_segment_ids(); + assert_eq!(ids.len(), 1); + let req = restored.build_request_for("k", ids[0]).unwrap(); + assert_eq!(req.vectors.len(), 10); + assert!(!restored.is_live(5)); + assert_eq!(restored.search(&vector(8), 1, 64).unwrap()[0].id, 8); + } +} diff --git a/nodedb-vector/src/collection/checkpoint.rs b/nodedb-vector/src/collection/checkpoint.rs index 9f80fedcd..e39985717 100644 --- a/nodedb-vector/src/collection/checkpoint.rs +++ b/nodedb-vector/src/collection/checkpoint.rs @@ -24,7 +24,7 @@ use nodedb_types::{Surrogate, VectorQuantization}; use serde::{Deserialize, Serialize}; use crate::collection::payload_index::PayloadIndexSetSnapshot; -use crate::collection::segment::{DEFAULT_SEAL_THRESHOLD, SealedSegment}; +use crate::collection::segment::{BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment}; use crate::collection::tier::StorageTier; use crate::distance::DistanceMetric; use crate::error::VectorError; @@ -378,34 +378,30 @@ impl VectorCollection { }); } + // A segment sealed but not yet built comes back as a building + // segment, searched by brute force; the owning core queues its build. + let mut next_segment_id = (sealed.len() + 1) as u32; + let mut building = Vec::with_capacity(snap.building_segments.len()); for bs in &snap.building_segments { - let mut index = HnswIndex::new(snap.dim, params.clone()); - for v in &bs.vectors { - index.insert(v.clone()).map_err(|e| { - VectorError::CheckpointDeserializationError { - detail: format!("building-segment replay insert: {e}"), - } + let mut flat = FlatIndex::new(snap.dim, metric); + for (i, v) in bs.vectors.iter().enumerate() { + let inserted = if bs.deleted.get(i).copied().unwrap_or(false) { + flat.insert_tombstoned(v.clone()) + } else { + flat.insert(v.clone()) + }; + inserted.map_err(|e| VectorError::CheckpointDeserializationError { + detail: format!("building-segment replay insert: {e}"), })?; } - // Replay building-segment tombstones onto the HNSW index. - for (i, &dead) in bs.deleted.iter().enumerate() { - if dead { - index.delete(i as u32); - } - } - let sq8 = VectorCollection::build_sq8_for_index(&index); - sealed.push(SealedSegment { - index, + building.push(BuildingSegment { + flat, base_id: bs.base_id, - sq8, - pq: None, - tier: StorageTier::L0Ram, - mmap_vectors: None, + segment_id: next_segment_id, }); + next_segment_id += 1; } - let next_segment_id = (sealed.len() + 1) as u32; - let index_config = crate::index_config::IndexConfig { hnsw: params.clone(), ..snap.index_config @@ -419,7 +415,7 @@ impl VectorCollection { growing, growing_base_id: snap.growing_base_id, sealed, - building: Vec::new(), + building, params, next_id: snap.next_id, next_segment_id, @@ -458,6 +454,8 @@ impl VectorCollection { arena_index: None, checkpoint_wal_lsn: snap.checkpoint_wal_lsn, applied_wal_lsn: snap.checkpoint_wal_lsn, + builds_completed: 0, + builds_failed: 0, }) } } diff --git a/nodedb-vector/src/collection/lifecycle.rs b/nodedb-vector/src/collection/lifecycle.rs index 8fb7016f8..32457e42b 100644 --- a/nodedb-vector/src/collection/lifecycle.rs +++ b/nodedb-vector/src/collection/lifecycle.rs @@ -1,6 +1,8 @@ // SPDX-License-Identifier: Apache-2.0 -//! VectorCollection lifecycle: insert, delete, seal, complete_build, compact. +//! VectorCollection lifecycle: construction, seal, counts and settings. +//! +//! Installing finished builds lives in `build`. //! //! Identity model: every vector inserted into the collection is bound to //! a global `Surrogate` allocated by the Control Plane before the engine @@ -16,17 +18,18 @@ use std::collections::HashMap; -use nodedb_mem::ScopedMemory; use nodedb_types::{Surrogate, VectorQuantization}; use crate::flat::FlatIndex; -use crate::hnsw::{HnswIndex, HnswParams}; -use crate::index_config::{IndexConfig, IndexType}; +use crate::hnsw::HnswParams; +use crate::index_config::IndexConfig; use crate::ivf::IvfPqIndex; use super::codec_dispatch::CollectionCodec; use super::payload_index::PayloadIndexSet; -use super::segment::{BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment}; +use super::segment::{ + BuildKind, BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment, +}; /// Manages all vector segments for a single collection (one index key). /// @@ -118,6 +121,10 @@ pub struct VectorCollection { /// gating its siblings). Folded into the persisted watermark at checkpoint /// save time via `max(checkpoint_wal_lsn, applied_wal_lsn)`. pub(crate) applied_wal_lsn: u64, + /// HNSW builds installed since the collection was opened. + pub(crate) builds_completed: u64, + /// HNSW builds that failed since the collection was opened. + pub(crate) builds_failed: u64, } impl VectorCollection { @@ -172,6 +179,8 @@ impl VectorCollection { arena_index: None, checkpoint_wal_lsn: 0, applied_wal_lsn: 0, + builds_completed: 0, + builds_failed: 0, } } @@ -211,6 +220,12 @@ impl VectorCollection { Self::with_seal_threshold(dim, params, DEFAULT_SEAL_THRESHOLD) } + /// Set the growing-segment size that triggers a seal. A zero threshold + /// is raised to one vector. + pub fn set_seal_threshold(&mut self, threshold: usize) { + self.seal_threshold = threshold.max(1); + } + /// Check if the growing segment should be sealed. An `IvfPq` collection /// never seals: its growing segment is the buffer IVF-PQ training reads. pub fn needs_seal(&self) -> bool { @@ -227,10 +242,12 @@ impl VectorCollection { let segment_id = self.next_segment_id; self.next_segment_id += 1; + // Soft-deleted vectors go too: the built graph keeps every local id, + // and the tombstones are applied when the build is installed. let count = self.growing.len(); let mut vectors = Vec::with_capacity(count); for i in 0..count as u32 { - if let Some(v) = self.growing.get_vector(i) { + if let Some(v) = self.growing.get_vector_raw(i) { vectors.push(v.to_vec()); } } @@ -251,66 +268,13 @@ impl VectorCollection { Some(BuildRequest { key: key.to_string(), segment_id, + kind: BuildKind::Seal, vectors, dim: self.dim, params: self.params.clone(), }) } - /// Accept a completed HNSW build from the background thread. - /// - /// After promoting the segment to sealed, rebuilds the collection-level - /// codec-dispatch index when `self.quantization` is `RaBitQ` or `Bbq`. - /// The rebuild trains over all vectors so the codec index always covers - /// every sealed segment. - pub fn complete_build(&mut self, segment_id: u32, index: HnswIndex, memory: ScopedMemory) { - if let Some(pos) = self - .building - .iter() - .position(|b| b.segment_id == segment_id) - { - let building = self.building.remove(pos); - let codec_dispatch_tag = match self.quantization { - VectorQuantization::RaBitQ => Some("rabitq"), - VectorQuantization::Bbq => Some("bbq"), - _ => None, - }; - let use_codec_dispatch = codec_dispatch_tag.is_some(); - let use_pq = !use_codec_dispatch && self.index_config.index_type == IndexType::HnswPq; - let (sq8, pq) = if use_codec_dispatch { - (None, None) - } else if use_pq { - ( - None, - Self::build_pq_for_index(&index, self.index_config.pq_m, memory.clone()), - ) - } else { - (Self::build_sq8_for_index(&index), None) - }; - let (tier, mmap_vectors) = - self.resolve_tier_for_build(segment_id, building.base_id, &index, &memory); - - self.sealed.push(SealedSegment { - index, - base_id: building.base_id, - sq8, - pq, - tier, - mmap_vectors, - }); - - if let Some(tag) = codec_dispatch_tag { - let built = self.build_codec_dispatch(tag).map(|_| ()); - if let Err(e) = built { - // Without the codec index the sealed segments are searched - // by their own HNSW graphs, which answer the same queries. - tracing::error!(error = %e, tag, "codec-dispatch build failed; searching sealed segments directly"); - self.codec_dispatch = None; - } - } - } - } - /// Access sealed segments (read-only). pub fn sealed_segments(&self) -> &[SealedSegment] { &self.sealed diff --git a/nodedb-vector/src/collection/mod.rs b/nodedb-vector/src/collection/mod.rs index f407bf83d..452daf7e1 100644 --- a/nodedb-vector/src/collection/mod.rs +++ b/nodedb-vector/src/collection/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: Apache-2.0 pub mod budget; +pub mod build; pub mod checkpoint; pub mod codec_build; pub mod codec_dispatch; @@ -21,6 +22,6 @@ pub use lifecycle::VectorCollection; pub use payload_index::{FilterPredicate, PayloadIndex, PayloadIndexKind, PayloadIndexSet}; pub use rollback::VectorWriteMark; pub use segment::{ - BuildComplete, BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment, + BuildComplete, BuildKind, BuildRequest, BuildingSegment, DEFAULT_SEAL_THRESHOLD, SealedSegment, }; pub use tier::StorageTier; diff --git a/nodedb-vector/src/collection/quantize.rs b/nodedb-vector/src/collection/quantize.rs index 09bbda007..ea3e2881d 100644 --- a/nodedb-vector/src/collection/quantize.rs +++ b/nodedb-vector/src/collection/quantize.rs @@ -87,7 +87,7 @@ impl VectorCollection { } /// Train a PQ codec from a built HNSW index's live vectors, tracking - /// codebook allocations against `memory`. + /// codebook allocations against `memory`, and encode every node. pub fn build_pq_for_index( index: &HnswIndex, pq_m: usize, @@ -121,7 +121,19 @@ impl VectorCollection { return None; } }; - let codes = codec.encode_batch(&refs_slices).ok()?; + // One code per local node id, soft-deleted nodes included: search + // reads the code of node `id` at `id * pq_m`. + let zero = vec![0.0f32; dim]; + let all: Vec<&[f32]> = (0..n as u32) + .map(|i| index.get_vector(i).unwrap_or(zero.as_slice())) + .collect(); + let codes = match codec.encode_batch(&all) { + Ok(codes) => codes, + Err(e) => { + tracing::warn!(error = %e, dim, pq_m, "PQ encoding refused; segment stays unquantized"); + return None; + } + }; Some((codec, codes)) } } diff --git a/nodedb-vector/src/collection/segment.rs b/nodedb-vector/src/collection/segment.rs index 048d0b13d..c30feb4eb 100644 --- a/nodedb-vector/src/collection/segment.rs +++ b/nodedb-vector/src/collection/segment.rs @@ -13,10 +13,26 @@ use crate::quantize::sq8::Sq8Codec; /// 64K vectors × 768 dims × 4 bytes = ~192 MiB per segment. pub const DEFAULT_SEAL_THRESHOLD: usize = 65_536; -/// Request to build an HNSW index from sealed vectors (sent to builder thread). +/// What a finished build replaces. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BuildKind { + /// Promote building segment `segment_id` to sealed. + Seal, + /// Replace the sealed segment at `base_id`, which held `len` nodes when + /// its vectors were read. A segment that no longer holds `len` nodes + /// (compaction renumbered it) refuses the result. + Rebuild { base_id: u32, len: usize }, +} + +/// Request to build an HNSW index (sent to the builder thread). +/// +/// `vectors` holds one vector per local node id, soft-deleted nodes +/// included, so the built graph keeps every id. The owning core applies +/// the tombstones when it installs the result. pub struct BuildRequest { pub key: String, pub segment_id: u32, + pub kind: BuildKind, pub vectors: Vec>, pub dim: usize, pub params: HnswParams, @@ -26,7 +42,10 @@ pub struct BuildRequest { pub struct BuildComplete { pub key: String, pub segment_id: u32, - pub index: HnswIndex, + pub kind: BuildKind, + /// The built index, or the error that stopped the build. A failed build + /// leaves the segment as it was. + pub result: Result, } /// A sealed segment whose HNSW index is being built in background. diff --git a/nodedb-vector/src/collection/stats.rs b/nodedb-vector/src/collection/stats.rs index e36f37458..5f88d7693 100644 --- a/nodedb-vector/src/collection/stats.rs +++ b/nodedb-vector/src/collection/stats.rs @@ -111,6 +111,10 @@ impl VectorCollection { cells: self.ivf.as_ref().map_or(0, |ivf| ivf.n_cells()), nprobe: self.index_config.ivf_nprobe, }), + // The owning core fills this from its build queue. + builds_queued: 0, + builds_completed: self.builds_completed, + builds_failed: self.builds_failed, } } } diff --git a/nodedb-vector/src/lib.rs b/nodedb-vector/src/lib.rs index 8495537c1..519da2d53 100644 --- a/nodedb-vector/src/lib.rs +++ b/nodedb-vector/src/lib.rs @@ -75,7 +75,7 @@ pub use adaptive_filter::{ #[cfg(not(target_arch = "wasm32"))] pub use builder::{BuildSender, CompleteReceiver}; #[cfg(not(target_arch = "wasm32"))] -pub use collection::{BuildComplete, BuildRequest, StorageTier, VectorCollection}; +pub use collection::{BuildComplete, BuildKind, BuildRequest, StorageTier, VectorCollection}; pub use flat::FlatIndex; pub use index_config::{IndexConfig, IndexType}; pub use ivf::{IvfPqIndex, IvfPqParams}; diff --git a/nodedb/src/bootstrap/data_plane.rs b/nodedb/src/bootstrap/data_plane.rs index c6c5803ff..f731db951 100644 --- a/nodedb/src/bootstrap/data_plane.rs +++ b/nodedb/src/bootstrap/data_plane.rs @@ -338,6 +338,7 @@ pub fn spawn_data_plane_cores( query: config.tuning.query.clone(), graph: config.tuning.graph.clone(), timeseries: config.tuning.timeseries.clone(), + vector: config.tuning.vector.clone(), checkpoint_interval: std::time::Duration::from_secs(config.checkpoint.interval_secs), }; diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index f70d38443..3493782b5 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -167,6 +167,10 @@ pub enum ErrorCode { /// answers for a task it stopped part way. Surfaces as the same /// query-cancelled error. ExpiredBeforeExecution, + /// The request itself is malformed: an FTS query with no positive term, + /// a value the target cannot hold. The same verdict the Control Plane + /// gives `crate::Error::BadRequest`: SQLSTATE `42601` (syntax_error). + BadRequest { detail: String }, } /// An expression evaluation failure, as the Data Plane reports it. @@ -198,11 +202,14 @@ impl From for ErrorCode { Self::RejectedPrevalidation { reason } } crate::Error::RetryableRefusal { reason } => Self::RetryableRefusal { reason }, - crate::Error::CollectionNotFound { .. } | crate::Error::DocumentNotFound { .. } => { - Self::NotFound - } + crate::Error::CollectionNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::DocumentNotFound { .. } => Self::NotFound, crate::Error::RejectedAuthz { resource, .. } => Self::RejectedAuthz { resource }, - crate::Error::ConflictRetry { .. } => Self::ConflictRetry, + // The Control Plane gives all three `40001` (serialization_failure). + crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::SourceFrozen { .. } => Self::ConflictRetry, crate::Error::FanOutExceeded { .. } => Self::FanOutExceeded, crate::Error::MemoryExhausted { .. } => Self::ResourcesExhausted, crate::Error::Backpressure { .. } => Self::ResourcesExhausted, @@ -281,6 +288,15 @@ impl From for ErrorCode { crate::Error::DivisionByZero => Self::DivisionByZero, crate::Error::UndefinedFunction { name } => Self::UndefinedFunction { name }, crate::Error::DataException { detail } => Self::DataException { detail }, + // `42601` (syntax_error), as the Control Plane gives both. + crate::Error::BadRequest { detail } | crate::Error::PlanError { detail } => { + Self::BadRequest { detail } + } + // `0A000` (feature_not_supported), as the Control Plane gives both. + crate::Error::FeatureNotSupported { detail } => Self::Unsupported { detail }, + unsupported @ crate::Error::CrossCollectionNotColocated { .. } => Self::Unsupported { + detail: unsupported.to_string(), + }, crate::Error::UndefinedColumn { column } => Self::UndefinedColumn { column }, // Same condition an undefined column reports at plan time, raised // here by the strict encoder for a transport the planner never diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 662d7c879..9c5ce4f8d 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -180,6 +180,7 @@ impl From for DataPlaneErrorCode { ErrorCode::DataException { detail } => Self::DataException { detail }, ErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, ErrorCode::ExpiredBeforeExecution => Self::ExpiredBeforeExecution, + ErrorCode::BadRequest { detail } => Self::BadRequest { detail }, } } } @@ -309,6 +310,7 @@ impl From for ErrorCode { DataPlaneErrorCode::DataException { detail } => Self::DataException { detail }, DataPlaneErrorCode::DispatchCapacity { reason } => Self::DispatchCapacity { reason }, DataPlaneErrorCode::ExpiredBeforeExecution => Self::ExpiredBeforeExecution, + DataPlaneErrorCode::BadRequest { detail } => Self::BadRequest { detail }, } } } diff --git a/nodedb/src/control/metrics/prometheus/engines.rs b/nodedb/src/control/metrics/prometheus/engines.rs index bed82ea04..b4b72aa55 100644 --- a/nodedb/src/control/metrics/prometheus/engines.rs +++ b/nodedb/src/control/metrics/prometheus/engines.rs @@ -31,6 +31,36 @@ impl SystemMetrics { "Vectors stored", self.vector_vectors_stored.load(Ordering::Relaxed), ); + counter( + out, + "nodedb_vector_builds_started_total", + "HNSW builds sent to a builder thread", + self.vector_builds_started.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_vector_builds_completed_total", + "HNSW builds installed", + self.vector_builds_completed.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_vector_builds_failed_total", + "HNSW builds that failed", + self.vector_builds_failed.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_vector_builds_deferred_total", + "Times a full builder queue kept an HNSW build waiting", + self.vector_builds_deferred.load(Ordering::Relaxed), + ); + gauge( + out, + "nodedb_vector_build_pending", + "HNSW builds waiting for or running on a builder", + self.vector_build_pending.load(Ordering::Relaxed), + ); self.vector_query_seconds.write_prometheus( out, "nodedb_vector_query_seconds", diff --git a/nodedb/src/control/metrics/system/fields.rs b/nodedb/src/control/metrics/system/fields.rs index c39299cda..a21da10da 100644 --- a/nodedb/src/control/metrics/system/fields.rs +++ b/nodedb/src/control/metrics/system/fields.rs @@ -63,6 +63,16 @@ pub struct SystemMetrics { pub vector_collections: AtomicU64, pub vector_vectors_stored: AtomicU64, pub vector_query_seconds: AtomicHistogram, + /// HNSW builds sent to a builder thread. + pub vector_builds_started: AtomicU64, + /// HNSW builds installed on their core. + pub vector_builds_completed: AtomicU64, + /// HNSW builds that failed or could not be read. + pub vector_builds_failed: AtomicU64, + /// Times a core found its builder queue full and kept the job waiting. + pub vector_builds_deferred: AtomicU64, + /// HNSW builds waiting for or running on a builder, across all cores. + pub vector_build_pending: AtomicU64, pub graph_traversals: AtomicU64, pub graph_nodes: AtomicU64, diff --git a/nodedb/src/control/metrics/system/record.rs b/nodedb/src/control/metrics/system/record.rs index e2eb7efe3..bb8d2d65e 100644 --- a/nodedb/src/control/metrics/system/record.rs +++ b/nodedb/src/control/metrics/system/record.rs @@ -172,6 +172,34 @@ impl SystemMetrics { self.vector_query_seconds.observe(latency_us); } + pub fn record_vector_build_started(&self) { + self.vector_builds_started.fetch_add(1, Ordering::Relaxed); + } + + pub fn record_vector_build_completed(&self) { + self.vector_builds_completed.fetch_add(1, Ordering::Relaxed); + } + + pub fn record_vector_build_failed(&self) { + self.vector_builds_failed.fetch_add(1, Ordering::Relaxed); + } + + pub fn record_vector_build_deferred(&self) { + self.vector_builds_deferred.fetch_add(1, Ordering::Relaxed); + } + + /// Move the cross-core pending-build gauge from one core's previous + /// count `before` to its current count `after`. + pub fn move_vector_build_pending(&self, before: u64, after: u64) { + if after > before { + self.vector_build_pending + .fetch_add(after - before, Ordering::Relaxed); + } else if before > after { + self.vector_build_pending + .fetch_sub(before - after, Ordering::Relaxed); + } + } + pub fn update_vector_stats(&self, collections: u64, vectors: u64) { self.vector_collections .store(collections, Ordering::Relaxed); diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index 1bf261f09..e6ccb3c17 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -132,7 +132,10 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { // * `DuplicateWrite` — the idempotency gate fired because the write // ALREADY applied under the original request; nothing to undo, and // the duplicate record replays to the same state. + // * `BadRequest` — raised by many engine paths, some of them after a + // multi-row plan already wrote rows. ErrorCode::DeadlineExceeded + | ErrorCode::BadRequest { .. } | ErrorCode::RollbackFailed { .. } | ErrorCode::ResourcesExhausted | ErrorCode::Internal { .. } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs new file mode 100644 index 000000000..d978ef904 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The fields of a `{ ... }` INSERT that the document write already indexes +//! into a vector index. +//! +//! The Data Plane indexes a document's vectors when it stores the document: +//! +//! - A strict collection indexes every `VECTOR(n)` column. +//! - Any other collection indexes each field that has its own vector index. +//! When no field has one, a default-field vector index covers `embedding`. +//! +//! The `{ ... }` handler also sends a vector insert for numeric-array +//! fields, so a field no index covers is still searchable. It skips the +//! fields listed here. A second insert of such a field appends a second +//! HNSW node for the same row. + +use std::collections::HashSet; + +use nodedb_types::{CollectionType, ColumnType, DatabaseId, DocumentMode}; + +use crate::control::server::shared::ddl::result::DdlError; +use crate::control::state::SharedState; + +/// The default vector field a vector index without a field name covers. +const DEFAULT_VECTOR_FIELD: &str = "embedding"; + +/// The fields of `collection` the document write indexes into a vector +/// index. Fails when the catalog cannot be read: guessing would either +/// index a field twice or not at all. +pub(super) fn indexed_vector_fields( + state: &SharedState, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, + collection_type: Option<&CollectionType>, +) -> Result, DdlError> { + if let Some(CollectionType::Document(DocumentMode::Strict(schema))) = collection_type { + let strict: HashSet = schema + .columns + .iter() + .filter(|c| matches!(c.column_type, ColumnType::Vector(_))) + .map(|c| c.name.clone()) + .collect(); + if !strict.is_empty() { + return Ok(strict); + } + } + + let params = state + .credentials + .catalog() + .list_vector_index_params_in_database(database_id.as_u64()) + .map_err(|e| { + DdlError::new( + "XX000", + format!("read vector indexes of \"{collection}\" for INSERT: {e}"), + ) + })?; + let mut named = HashSet::new(); + let mut has_default = false; + for p in params + .iter() + .filter(|p| p.tenant_id == tenant_id && p.collection == collection) + { + if p.field_name.is_empty() { + has_default = true; + } else { + named.insert(p.field_name.clone()); + } + } + if named.is_empty() && has_default { + named.insert(DEFAULT_VECTOR_FIELD.to_string()); + } + Ok(named) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs index 031b042f4..a7a00536d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs @@ -15,6 +15,7 @@ use crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate; use crate::control::server::shared::session::{DmlTxnCtx, PendingFieldInference}; use crate::control::state::SharedState; +use super::indexed_vector_fields::indexed_vector_fields; use super::parse::{ authorize_write_target, dispatch_plan, extract_vector_fields, fields_to_insert_sql, parse_write_statement, plan_and_dispatch, @@ -245,10 +246,25 @@ pub async fn insert_document( return Some(err); } - // Dispatch VectorInsert for vector fields. + // Dispatch VectorInsert for the numeric-array fields no vector index + // covers. The document write above already indexed the covered ones, and + // a second insert would append a second HNSW node for the same row. + let indexed = match indexed_vector_fields( + state, + database_id, + tenant_id.as_u64(), + &parsed.coll_name, + parsed.collection_type.as_ref(), + ) { + Ok(indexed) => indexed, + Err(e) => return Some(Err(e)), + }; let vec_vshard = crate::types::VShardId::from_collection_in_database(database_id, &parsed.coll_name); for (field_name, vector) in extract_vector_fields(&fields) { + if indexed.contains(&field_name) { + continue; + } let dim = vector.len(); { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/mod.rs index 544c1b542..d2f1a15a9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/mod.rs @@ -2,6 +2,7 @@ //! Protocol-neutral collection DML: INSERT INTO / UPSERT INTO. +mod indexed_vector_fields; mod insert; mod parse; mod triggers; diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs index cd7b2a89f..3c68e9105 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs @@ -94,6 +94,9 @@ pub async fn handle_show_vector_index( format!("{:.1}", stats.disk_bytes as f64 / (1024.0 * 1024.0)), ), ("build_in_progress", stats.build_in_progress.to_string()), + ("builds_queued", stats.builds_queued.to_string()), + ("builds_completed", stats.builds_completed.to_string()), + ("builds_failed", stats.builds_failed.to_string()), ("hnsw_m", stats.hnsw_m.to_string()), ("hnsw_m0", stats.hnsw_m0.to_string()), ( diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index 7a7e63664..117445733 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -219,6 +219,8 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, format!("function {name}() does not exist"), ), ErrorCode::DataException { detail } => ("ERROR", sqlstate::DATA_EXCEPTION, detail.clone()), + // The same SQLSTATE the Control Plane gives `crate::Error::BadRequest`. + ErrorCode::BadRequest { detail } => ("ERROR", sqlstate::SYNTAX_ERROR, detail.clone()), // Transient: the client retries after a backoff. ErrorCode::DispatchCapacity { reason } => { ("ERROR", sqlstate::SERVER_OVERLOAD, reason.clone()) @@ -306,4 +308,15 @@ mod tests { assert_eq!(state, sqlstate::DATA_EXCEPTION); assert_eq!(message, "vector dimension mismatch: expected 3, got 2"); } + + /// A request the Data Plane rejects as malformed is a syntax error, the + /// same SQLSTATE the Control Plane returns for it. + #[test] + fn a_bad_request_from_the_data_plane_is_a_syntax_error() { + let code = ErrorCode::from(crate::Error::BadRequest { + detail: "bad text query".into(), + }); + let (_, state, _) = error_code_to_sqlstate(&code); + assert_eq!(state, sqlstate::SYNTAX_ERROR); + } } diff --git a/nodedb/src/data/executor/core_loop/maintenance.rs b/nodedb/src/data/executor/core_loop/maintenance.rs index 29804b0a6..328575296 100644 --- a/nodedb/src/data/executor/core_loop/maintenance.rs +++ b/nodedb/src/data/executor/core_loop/maintenance.rs @@ -107,6 +107,12 @@ impl CoreLoop { /// captures these limits when it is CREATED and keeps them for its whole /// life, so a memtable built ahead of this call would silently keep the /// defaults. + /// Apply vector engine tuning. Must land before the checkpoint restore: + /// restored collections take its seal threshold. + pub fn set_vector_tuning(&mut self, tuning: nodedb_types::config::tuning::VectorTuning) { + self.vector_tuning = tuning; + } + pub fn set_timeseries_tuning( &mut self, tuning: nodedb_types::config::tuning::TimeseriesToning, diff --git a/nodedb/src/data/executor/core_loop/mod.rs b/nodedb/src/data/executor/core_loop/mod.rs index 6da403379..44439ca2d 100644 --- a/nodedb/src/data/executor/core_loop/mod.rs +++ b/nodedb/src/data/executor/core_loop/mod.rs @@ -27,6 +27,7 @@ mod state; mod test_governor; mod tick; mod ts_declared_schema; +pub(in crate::data::executor) mod vector_build_queue; mod vector_index_rebuild; mod vector_index_seed; pub(in crate::data::executor) mod write_index; diff --git a/nodedb/src/data/executor/core_loop/open.rs b/nodedb/src/data/executor/core_loop/open.rs index c86cd1dfd..f13e93e8e 100644 --- a/nodedb/src/data/executor/core_loop/open.rs +++ b/nodedb/src/data/executor/core_loop/open.rs @@ -110,8 +110,7 @@ impl CoreLoop { sparse, crdt_engines: HashMap::new(), vector_collections: HashMap::new(), - build_tx: None, - build_rx: None, + vector_builds: super::vector_build_queue::VectorBuildQueue::spawn(core_id), vector_params: HashMap::new(), declared_dims: HashMap::new(), edge_store, @@ -159,6 +158,7 @@ impl CoreLoop { query_tuning: nodedb_types::config::tuning::QueryTuning::default(), graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), ts_tuning: nodedb_types::config::tuning::TimeseriesToning::default(), + vector_tuning: nodedb_types::config::tuning::VectorTuning::default(), kv_engine: crate::engine::kv::KvEngine::from_tuning( crate::engine::kv::current_ms(), &nodedb_types::config::tuning::KvTuning::default(), diff --git a/nodedb/src/data/executor/core_loop/state.rs b/nodedb/src/data/executor/core_loop/state.rs index 0fd2633f2..f948d83e2 100644 --- a/nodedb/src/data/executor/core_loop/state.rs +++ b/nodedb/src/data/executor/core_loop/state.rs @@ -84,11 +84,9 @@ pub struct CoreLoop { pub(in crate::data::executor) vector_collections: HashMap<(DatabaseId, TenantId, String), VectorCollection>, - /// Background HNSW builder: send requests. - pub(in crate::data::executor) build_tx: Option, - /// Background HNSW builder: receive completed builds. - pub(in crate::data::executor) build_rx: - Option, + /// This core's HNSW builder thread, its bounded queues and the backlog + /// of builds waiting for room. + pub(in crate::data::executor) vector_builds: super::vector_build_queue::VectorBuildQueue, /// Per-collection HNSW parameters set via DDL. If a collection has no /// entry here, `HnswParams::default()` is used on first insert. @@ -347,6 +345,9 @@ pub struct CoreLoop { /// Read when a collection's `ColumnarMemtable` is created and by the ingest /// path's record-boundary admission gate. pub(in crate::data::executor) ts_tuning: nodedb_types::config::tuning::TimeseriesToning, + /// Vector engine tuning: the seal threshold new and restored collections + /// take, and the PQ / IVF defaults an index declaration leaves out. + pub(in crate::data::executor) vector_tuning: nodedb_types::config::tuning::VectorTuning, /// Per-core KV engine: hash tables + expiry wheel. `!Send`. pub(in crate::data::executor) kv_engine: crate::engine::kv::KvEngine, diff --git a/nodedb/src/data/executor/core_loop/vector_build_queue.rs b/nodedb/src/data/executor/core_loop/vector_build_queue.rs new file mode 100644 index 000000000..97c8d485e --- /dev/null +++ b/nodedb/src/data/executor/core_loop/vector_build_queue.rs @@ -0,0 +1,123 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-core HNSW build queue: the core's side of its builder thread. +//! +//! Each Data Plane core owns one builder thread (`nodedb_vector::builder`). +//! A build is CPU-heavy, so it never runs on the core's reactor, and the +//! core never blocks on the builder: +//! +//! - A job enters the `backlog` as a descriptor (index key plus segment), +//! not as vectors. The request is read from the collection when the job +//! is sent, so a job whose segment is gone is dropped then. +//! - Sending uses `try_send` on the bounded request queue. A full queue +//! leaves the job at the head of the backlog for the next tick, and the +//! segment stays searchable by brute force. +//! - Finished builds come back on a bounded completion queue the core +//! drains every tick; the builder thread waits while it is full. +//! +//! The backlog holds one small descriptor per unbuilt segment, so its size +//! is bounded by the segments the core's collections hold. + +use std::collections::{HashMap, VecDeque}; + +use crate::data::executor::handlers::vector_direct_row::VectorIndexKey; +use crate::engine::vector::builder::{BUILD_QUEUE_CAPACITY, BuildSender, CompleteReceiver}; + +/// What a queued build produces. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::data::executor) enum BuildJobKind { + /// Build the graph of building segment `segment_id`. + Seal { segment_id: u32 }, + /// Rebuild the sealed segment at `base_id` in place. + Rebuild { base_id: u32 }, +} + +/// One queued build. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(in crate::data::executor) struct BuildJob { + pub key: VectorIndexKey, + pub kind: BuildJobKind, +} + +/// The core's builder channels, backlog and in-flight count. +pub(in crate::data::executor) struct VectorBuildQueue { + /// `None` when the builder thread could not be spawned. Jobs then wait + /// in the backlog and the next dispatch spawns it again. + pub(in crate::data::executor) tx: Option, + pub(in crate::data::executor) rx: Option, + /// Jobs waiting for room in the builder queue, oldest first. + pub(in crate::data::executor) backlog: VecDeque, + /// Jobs sent and not yet finished, per index. + pub(in crate::data::executor) in_flight: HashMap, + /// Pending jobs (backlog plus in flight) last added to the shared + /// backlog gauge. + pub(in crate::data::executor) reported_pending: u64, +} + +impl VectorBuildQueue { + /// Spawn the builder thread for core `core_id`. + pub(in crate::data::executor) fn spawn(core_id: usize) -> Self { + let mut queue = Self { + tx: None, + rx: None, + backlog: VecDeque::new(), + in_flight: HashMap::new(), + reported_pending: 0, + }; + queue.respawn(core_id); + queue + } + + /// Spawn a fresh builder thread, replacing a dead one. On failure the + /// channels stay `None` and the next dispatch tries again. + pub(in crate::data::executor) fn respawn(&mut self, core_id: usize) { + match crate::engine::vector::builder::spawn_builder(core_id, BUILD_QUEUE_CAPACITY) { + Ok((tx, rx, _handle)) => { + // The thread stops on its own once `tx` drops. + self.tx = Some(tx); + self.rx = Some(rx); + // Builds the dead thread held never come back. + self.in_flight.clear(); + } + Err(e) => { + tracing::error!(core = core_id, error = %e, "HNSW builder thread spawn failed"); + crate::diag::vector_builder_spawn_failed(&e, core_id); + self.tx = None; + self.rx = None; + } + } + } + + /// Queue `job` unless the same job already waits. + pub(in crate::data::executor) fn push(&mut self, job: BuildJob) { + if !self.backlog.contains(&job) { + self.backlog.push_back(job); + } + } + + /// Builds of `key` waiting or running. + pub(in crate::data::executor) fn pending_for(&self, key: &VectorIndexKey) -> usize { + self.backlog.iter().filter(|j| &j.key == key).count() + + self.in_flight.get(key).copied().unwrap_or(0) + } + + /// Builds waiting or running on this core. + pub(in crate::data::executor) fn pending_total(&self) -> u64 { + (self.backlog.len() + self.in_flight.values().sum::()) as u64 + } + + /// Record a job sent for `key`. + pub(in crate::data::executor) fn note_sent(&mut self, key: VectorIndexKey) { + *self.in_flight.entry(key).or_insert(0) += 1; + } + + /// Record a finished job for `key`. + pub(in crate::data::executor) fn note_finished(&mut self, key: &VectorIndexKey) { + if let Some(n) = self.in_flight.get_mut(key) { + *n = n.saturating_sub(1); + if *n == 0 { + self.in_flight.remove(key); + } + } + } +} diff --git a/nodedb/src/data/executor/handlers/control/reindex.rs b/nodedb/src/data/executor/handlers/control/reindex.rs index 199553d6a..ea46935da 100644 --- a/nodedb/src/data/executor/handlers/control/reindex.rs +++ b/nodedb/src/data/executor/handlers/control/reindex.rs @@ -2,9 +2,11 @@ //! Concurrent index rebuild (REINDEX CONCURRENTLY) for HNSW, FTS LSM, and graph CSR. //! -//! Design: the Data Plane dispatches a `RebuildIndex` op. For the concurrent -//! path a background OS thread performs the heavy build work while the owning -//! core continues to serve reads from the live index. On each subsequent tick +//! Design: the Data Plane dispatches a `RebuildIndex` op. HNSW segments are +//! rebuilt on the core's HNSW builder thread, the same path every graph build +//! takes (`handlers::vector_build`). For FTS and CSR a background OS thread +//! performs the heavy build work while the owning core continues to serve +//! reads from the live index. On each subsequent tick //! the core polls `pending_reindex` for completion via `try_recv`; when the //! build succeeds the core performs an in-memory swap and returns the ACK. //! @@ -24,8 +26,7 @@ use std::sync::mpsc; use tracing::{error, info, warn}; use super::reindex_apply::{ - FtsRebuild, RebuildOutput, apply_csr, apply_fts, apply_hnsw, rebuild_csr_thread, - rebuild_fts_thread, rebuild_hnsw_thread, + FtsRebuild, RebuildOutput, apply_csr, apply_fts, rebuild_csr_thread, rebuild_fts_thread, }; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; @@ -86,9 +87,10 @@ impl CoreLoop { .map(|n| n.eq_ignore_ascii_case("csr")) .unwrap_or(true); - // Start the first applicable background rebuild (priority: HNSW > FTS > CSR). + // Start the first applicable rebuild (priority: HNSW > FTS > CSR). let start_result = if rebuild_hnsw { - self.start_hnsw_rebuild(task, tenant_id, &collection_key) + self.start_hnsw_rebuild(task, tenant_id, &collection_key); + Ok(()) } else if rebuild_fts { self.start_fts_rebuild(task, tenant_id, &collection_key) } else if rebuild_csr { @@ -97,12 +99,7 @@ impl CoreLoop { Ok(()) }; if let Err(e) = start_result { - return self.response_error( - task, - ErrorCode::Internal { - detail: e.to_string(), - }, - ); + return self.response_error(task, e); } self.response_ok(task) @@ -160,9 +157,6 @@ impl CoreLoop { output, } => { match output { - RebuildOutput::Hnsw { bytes } => { - apply_hnsw(self, &database_id, &tenant_id, &collection_key, bytes); - } RebuildOutput::Csr { bytes } => { apply_csr(self, &database_id, &tenant_id, &collection_key, bytes); } @@ -228,82 +222,44 @@ impl CoreLoop { // ── Background-thread starters ──────────────────────────────────────────── + /// Queue a rebuild of every sealed HNSW segment of the collection's + /// vector indexes on this core's builder thread. Each segment is rebuilt + /// from its own vectors with its node ids kept, quantized again under the + /// collection's config, and swapped in on this core; search reads the old + /// graph until then. The growing and building segments are left alone. fn start_hnsw_rebuild( &mut self, task: &ExecutionTask, tenant_id: TenantId, collection_key: &str, - ) -> crate::Result<()> { + ) { // Vector collections are stored under two key forms depending on how // they were inserted: // - Bare: (db, tenant, "coll") — BatchInsert / native // - Field-qualified: (db, tenant, "coll:field_name") — SQL INSERT, DirectUpsert - // - // Collect all matching keys so field-indexed collections (the common - // case from SQL DDL) are rebuilt correctly. let db = task.request.database_id; let field_prefix = format!("{collection_key}:"); - let matching_keys: Vec<(nodedb_types::DatabaseId, TenantId, String)> = self .vector_collections - .keys() - .filter(|(d, t, k)| { + .iter() + .filter(|((d, t, k), coll)| { *d == db && *t == tenant_id && (k.as_str() == collection_key || k.starts_with(&field_prefix)) + // An IVF-PQ collection keeps no HNSW segments to rebuild. + && !coll.is_ivf() }) - .cloned() + .map(|(key, _)| key.clone()) .collect(); - - if matching_keys.is_empty() { - return Ok(()); // no vector index for this collection; nothing to rebuild - } - for key in matching_keys { - let coll = match self.vector_collections.get(&key) { - Some(c) => c, - None => continue, - }; - // An IVF-PQ collection keeps no HNSW segments to rebuild. - if coll.is_ivf() { - continue; - } - - let dim = coll.dim(); - let params = coll.hnsw_params(); - - // Extract all live vectors from sealed segments. - let mut vectors: Vec> = Vec::new(); - for sealed in coll.sealed_segments() { - for id in 0..sealed.index.len() as u32 { - if !sealed.index.is_deleted(id) - && let Some(v) = sealed.index.get_vector(id) - { - vectors.push(v.to_vec()); - } - } - } - // Also include live vectors from the growing flat index. - let growing = coll.growing_flat(); - for id in 0..growing.len() as u32 { - if let Some(v) = growing.get_vector(id) { - vectors.push(v.to_vec()); - } - } - - let (tx, rx) = mpsc::sync_channel::>(1); - std::thread::spawn(move || { - let _ = tx.send(rebuild_hnsw_thread(vectors, dim, params)); - }); - - self.maintenance.pending_reindex.push(PendingReindex { - database_id: db, - tenant_id, - collection_key: key.2, - rx, - }); + let queued = self.queue_vector_rebuild(&key); + info!( + core = self.core_id, + collection = %key.2, + queued, + "HNSW rebuild queued" + ); } - Ok(()) } fn start_fts_rebuild( diff --git a/nodedb/src/data/executor/handlers/control/reindex_apply.rs b/nodedb/src/data/executor/handlers/control/reindex_apply.rs index b1cb0b78b..87a487c1f 100644 --- a/nodedb/src/data/executor/handlers/control/reindex_apply.rs +++ b/nodedb/src/data/executor/handlers/control/reindex_apply.rs @@ -1,7 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 //! Background-thread rebuild functions and Data-Plane cutover appliers for -//! concurrent index rebuild. See `reindex.rs` for the dispatch/poll surface. +//! concurrent FTS and CSR rebuild. See `reindex.rs` for the dispatch/poll +//! surface. HNSW rebuilds go through the core's HNSW builder thread. //! //! All functions in this module either run on a plain OS thread (operating on //! pure `Send` data) or run on the owning Data Plane core during cutover. They @@ -15,8 +16,6 @@ use crate::types::TenantId; // ── Background-thread output types (all Send) ──────────────────────────────── pub(super) enum RebuildOutput { - /// Serialized rebuilt HNSW index bytes. - Hnsw { bytes: Vec }, /// Serialized rebuilt CSR bytes. Csr { bytes: Vec }, /// Compacted FTS data ready for write-back. @@ -35,28 +34,6 @@ pub(super) struct FtsRebuild { // ── Background thread rebuild functions (pure Send, no !Send types) ────────── -pub(super) fn rebuild_hnsw_thread( - vectors: Vec>, - dim: usize, - params: nodedb_vector::HnswParams, -) -> crate::Result { - let mut index = nodedb_vector::HnswIndex::new(dim, params); - for v in vectors { - index.insert(v).map_err(|e| crate::Error::Storage { - engine: "vector".to_string(), - detail: format!("HNSW insert: {e}"), - })?; - } - Ok(RebuildOutput::Hnsw { - bytes: index - .checkpoint_to_bytes() - .map_err(|e| crate::Error::Storage { - engine: "vector".to_string(), - detail: format!("HNSW checkpoint encode: {e}"), - })?, - }) -} - pub(super) fn rebuild_fts_thread(input: FtsRebuild) -> crate::Result { // Compact: deduplicate posting entries by surrogate, keeping highest TF. let mut compacted: Vec<(String, Vec)> = @@ -111,58 +88,6 @@ pub(super) fn rebuild_csr_thread( // ── Cutover: apply rebuilt state to Data Plane in-memory structures ─────────── -pub(super) fn apply_hnsw( - core: &mut CoreLoop, - database_id: &nodedb_types::DatabaseId, - tenant_id: &TenantId, - collection_key: &str, - bytes: Vec, -) { - use nodedb_vector::HnswIndex; - use nodedb_vector::collection::segment::SealedSegment; - use nodedb_vector::collection::tier::StorageTier; - - let index = match HnswIndex::from_checkpoint(&bytes) { - Ok(Some(idx)) => idx, - Ok(None) => { - warn!( - core = core.core_id, - collection = %collection_key, - "HNSW rebuild: checkpoint had no magic; skipping cutover" - ); - return; - } - Err(e) => { - error!( - core = core.core_id, - collection = %collection_key, - error = %e, - "HNSW rebuild: restore failed; live index unchanged" - ); - return; - } - }; - - let key = (*database_id, *tenant_id, collection_key.to_string()); - if let Some(coll) = core.vector_collections.get_mut(&key) { - let new_seg = SealedSegment { - index, - base_id: 0, - sq8: None, - pq: None, - tier: StorageTier::L0Ram, - mmap_vectors: None, - }; - coll.replace_sealed(vec![new_seg]); - info!( - target: "nodedb::reindex", - core = core.core_id, - collection = %collection_key, - "atomic_cutover", - ); - } -} - pub(super) fn apply_fts( core: &mut CoreLoop, database_id: &nodedb_types::DatabaseId, diff --git a/nodedb/src/data/executor/handlers/mod.rs b/nodedb/src/data/executor/handlers/mod.rs index 62bd550cb..c4fcfb8fe 100644 --- a/nodedb/src/data/executor/handlers/mod.rs +++ b/nodedb/src/data/executor/handlers/mod.rs @@ -82,6 +82,7 @@ pub(super) mod update_from_join_types; pub(super) mod update_from_join_write; pub mod upsert; pub mod vector; +pub mod vector_build; pub mod vector_direct_delete; pub mod vector_direct_resolve; pub mod vector_direct_row; diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs index c81f4107b..72dc9ccfc 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/put.rs @@ -178,11 +178,13 @@ impl CoreLoop { } } - // A committed-redo install trains once the whole record landed, so a - // rollback finds its inserts in the growing segment. + // A full growing segment seals and queues its HNSW build, and an IVF-PQ + // collection at its threshold trains. A committed-redo install settles + // once the whole record landed, so a rollback finds its inserts in the + // growing segment. if !self.recording_redo_undo() { for delta in &inserts { - self.train_ivf_if_ready(&delta.index_key); + self.settle_vector_collection(&delta.index_key); } } Ok(inserts) @@ -472,6 +474,45 @@ mod tests { ); } + /// A vector write over an indexed row replaces the node the document put + /// recorded. Deleting the row must remove the node that replaced it, or + /// the deleted row's vector keeps scoring in searches. + #[test] + fn deleting_a_row_removes_the_node_a_vector_write_bound_to_it() { + let mut harness = make_core(); + let core = &mut harness.core; + let (db_id, tid, collection) = (0u64, 1u64, "docs"); + let surrogate = Surrogate::new(1); + let storage_key = crate::engine::document::store::StorageKey::for_surrogate(surrogate); + register_bare_field(core, db_id, tid, collection); + + let doc = doc_with_vectors(&[("embedding", &[1.0, 0.0, 0.0])]); + core.apply_point_put_vector_indexes(VectorIndexPutParams { + database_id: db_id, + tid, + collection, + storage_key, + value: &doc, + wal_lsn: 0, + }) + .expect("vector indexing must accept this fixture"); + let key = CoreLoop::vector_index_key(db_id, tid, collection, "embedding"); + core.vector_collections + .get_mut(&key) + .expect("collection") + .insert_with_surrogate(vec![0.0, 1.0, 0.0], surrogate) + .expect("vector write"); + assert_eq!(live_count(core, db_id, tid, collection, "embedding"), 1); + + let removed = core.remove_document_vector_indexes(db_id, tid, collection, storage_key); + assert_eq!(removed.len(), 1); + assert_eq!( + live_count(core, db_id, tid, collection, "embedding"), + 0, + "the node bound to the deleted row must be gone" + ); + } + /// A schemaless vector arriving as an SQL string literal must be parsed /// and indexed like an `ARRAY[...]` literal. /// diff --git a/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs b/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs index 57be0fdce..e9c3f9b5c 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/vector/remove.rs @@ -33,9 +33,17 @@ impl CoreLoop { field.to_string(), storage_key, ); - let vector_id = self.vector_doc_map.remove(&doc_key)?; + let recorded = self.vector_doc_map.remove(&doc_key)?; let index_key = Self::vector_index_key(database_id, tid, collection, field); + // The row's live node is the one bound to its surrogate. A later + // insert under the same surrogate (a vector write over this row) + // replaces the recorded node, so the recorded id can name a node that + // is already soft-deleted while the live one keeps scoring. + let mut vector_id = recorded; if let Some(coll) = self.vector_collections.get_mut(&index_key) { + if let Some(bound) = coll.local_for_surrogate(storage_key.surrogate()) { + vector_id = bound; + } coll.delete(vector_id); } Some(VectorIndexDelta { diff --git a/nodedb/src/data/executor/handlers/vector.rs b/nodedb/src/data/executor/handlers/vector.rs index dd560013d..15ef53ba9 100644 --- a/nodedb/src/data/executor/handlers/vector.rs +++ b/nodedb/src/data/executor/handlers/vector.rs @@ -95,6 +95,7 @@ impl CoreLoop { } let core_id = self.core_id; + let seal_threshold = self.vector_tuning.seal_threshold.max(1); match self.vector_collections.entry(index_key) { Entry::Occupied(entry) => { let existing = entry.into_mut(); @@ -108,7 +109,13 @@ impl CoreLoop { vector_index_config_for(&self.index_configs, &self.vector_params, entry.key()); check_ivf_dim(&config, dim).map_err(ErrorCode::from)?; debug!(core = core_id, dim, index_type = ?config.index_type, "creating vector collection"); - Ok(entry.insert(VectorCollection::with_index_config(dim, config))) + Ok( + entry.insert(VectorCollection::with_seal_threshold_and_config( + dim, + config, + seal_threshold, + )), + ) } } } diff --git a/nodedb/src/data/executor/handlers/vector_build.rs b/nodedb/src/data/executor/handlers/vector_build.rs new file mode 100644 index 000000000..235c0523d --- /dev/null +++ b/nodedb/src/data/executor/handlers/vector_build.rs @@ -0,0 +1,305 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sending HNSW builds to this core's builder thread. +//! +//! Every graph build — a sealed segment's first build, a boot re-queue of a +//! segment sealed before a restart, and a REINDEX or `ALTER VECTOR INDEX` +//! rebuild — goes through [`CoreLoop::dispatch_vector_builds`]. See +//! `core_loop::vector_build_queue` for the queue and its backpressure. +//! Finished builds are installed by `poll_build_completions`. + +use std::sync::mpsc::TrySendError; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::core_loop::vector_build_queue::{BuildJob, BuildJobKind}; +use crate::data::executor::handlers::vector_direct_row::VectorIndexKey; +use crate::engine::vector::collection::BuildRequest; + +impl CoreLoop { + /// Send the build request `seal` produced for `key`. It goes straight to + /// the builder when nothing waits ahead of it and the queue has room; + /// otherwise it waits in the backlog as a descriptor. + pub(in crate::data::executor) fn queue_sealed_build( + &mut self, + key: &VectorIndexKey, + req: BuildRequest, + ) { + let segment_id = req.segment_id; + if self.vector_builds.backlog.is_empty() + && let Some(tx) = &self.vector_builds.tx + { + match tx.try_send(req) { + Ok(()) => { + self.vector_builds.note_sent(key.clone()); + if let Some(m) = &self.metrics { + m.record_vector_build_started(); + } + self.sync_build_pending_metric(); + return; + } + Err(TrySendError::Full(_)) => { + if let Some(m) = &self.metrics { + m.record_vector_build_deferred(); + } + } + Err(TrySendError::Disconnected(_)) => {} + } + } + self.vector_builds.push(BuildJob { + key: key.clone(), + kind: BuildJobKind::Seal { segment_id }, + }); + self.dispatch_vector_builds(); + } + + /// Queue a rebuild of every non-empty sealed segment of `key`, under the + /// collection's current params. Returns the number of segments queued. + pub(in crate::data::executor) fn queue_vector_rebuild( + &mut self, + key: &VectorIndexKey, + ) -> usize { + let Some(coll) = self.vector_collections.get(key) else { + return 0; + }; + let base_ids = coll.sealed_base_ids(); + for &base_id in &base_ids { + self.vector_builds.push(BuildJob { + key: key.clone(), + kind: BuildJobKind::Rebuild { base_id }, + }); + } + self.dispatch_vector_builds(); + base_ids.len() + } + + /// Queue a build for every building segment of every collection. Boot + /// runs it after replay, so a segment sealed before a restart gets its + /// graph; a builder respawn runs it for the builds the dead thread held. + pub fn queue_unbuilt_segments(&mut self) { + for job in self.unbuilt_segment_jobs() { + self.vector_builds.push(job); + } + self.dispatch_vector_builds(); + } + + /// A seal job for every building segment of every collection. + fn unbuilt_segment_jobs(&self) -> Vec { + self.vector_collections + .iter() + .flat_map(|(key, coll)| { + coll.building_segment_ids() + .into_iter() + .map(|segment_id| BuildJob { + key: key.clone(), + kind: BuildJobKind::Seal { segment_id }, + }) + }) + .collect() + } + + /// Send backlog jobs, oldest first, until the builder queue is full. + /// + /// Each job's request is read from its collection now. A job whose + /// segment is gone (truncated, dropped, already rebuilt) is dropped. A + /// rebuild whose vectors cannot be read counts as a failed build. A + /// dead builder thread is replaced and every unbuilt segment re-queued. + pub(in crate::data::executor) fn dispatch_vector_builds(&mut self) { + let mut respawned = false; + while let Some(job) = self.vector_builds.backlog.front().cloned() { + if self.vector_builds.tx.is_none() { + if respawned { + break; + } + self.vector_builds.respawn(self.core_id); + respawned = true; + continue; + } + let Some(req) = self.read_build_request(&job) else { + self.vector_builds.backlog.pop_front(); + continue; + }; + let Some(tx) = &self.vector_builds.tx else { + break; + }; + match tx.try_send(req) { + Ok(()) => { + self.vector_builds.backlog.pop_front(); + self.vector_builds.note_sent(job.key); + if let Some(m) = &self.metrics { + m.record_vector_build_started(); + } + } + Err(TrySendError::Full(_)) => { + // The segment stays on brute force; the next tick retries. + if let Some(m) = &self.metrics { + m.record_vector_build_deferred(); + } + break; + } + Err(TrySendError::Disconnected(_)) => { + tracing::error!( + core = self.core_id, + "HNSW builder thread gone; spawning a new one" + ); + crate::diag::vector_builder_disconnected(self.core_id); + if respawned { + break; + } + self.vector_builds.respawn(self.core_id); + respawned = true; + for job in self.unbuilt_segment_jobs() { + self.vector_builds.push(job); + } + } + } + } + self.sync_build_pending_metric(); + } + + /// Read the build request `job` names, or `None` when its segment is + /// gone or a rebuild's vectors cannot be read. + fn read_build_request(&mut self, job: &BuildJob) -> Option { + let build_key = CoreLoop::vector_build_key(&job.key); + let coll = self.vector_collections.get_mut(&job.key)?; + match job.kind { + BuildJobKind::Seal { segment_id } => coll.build_request_for(&build_key, segment_id), + BuildJobKind::Rebuild { base_id } => { + match coll.rebuild_request_for(&build_key, base_id) { + Ok(req) => req, + Err(e) => { + coll.note_build_failed(); + crate::diag::vector_rebuild_unreadable( + &e, + &crate::diag::VectorBuildTarget { + kind: "rebuild", + database_id: job.key.0.as_u64(), + tenant_id: job.key.1.as_u64(), + index: &job.key.2, + segment: base_id, + }, + ); + tracing::error!( + core = self.core_id, + key = %job.key.2, + base_id, + error = %e, + "HNSW rebuild cannot read its segment; the segment stays as it is" + ); + if let Some(m) = &self.metrics { + m.record_vector_build_failed(); + } + None + } + } + } + } + } + + /// Bring the cross-core pending-build gauge up to this core's count. + pub(in crate::data::executor) fn sync_build_pending_metric(&mut self) { + let now = self.vector_builds.pending_total(); + if let Some(m) = &self.metrics { + m.move_vector_build_pending(self.vector_builds.reported_pending, now); + self.vector_builds.reported_pending = now; + } + } +} + +#[cfg(test)] +mod tests { + use std::time::{Duration, Instant}; + + use super::*; + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::engine::vector::collection::VectorCollection; + use crate::engine::vector::distance::DistanceMetric; + use crate::engine::vector::hnsw::HnswParams; + use crate::types::{DatabaseId, TenantId}; + + const SEGMENTS: usize = 6; + const PER_SEGMENT: usize = 8; + + /// A collection holding `SEGMENTS` sealed-but-unbuilt segments, as a + /// restart leaves one whose builds had not finished. + fn unbuilt_collection() -> VectorCollection { + let params = HnswParams { + metric: DistanceMetric::L2, + ..HnswParams::default() + }; + let mut coll = VectorCollection::with_seal_threshold(2, params, PER_SEGMENT); + for s in 0..SEGMENTS { + for i in 0..PER_SEGMENT { + let n = (s * PER_SEGMENT + i) as f32; + coll.insert(vec![n, (i % 3) as f32]).unwrap(); + } + // The request is dropped: the build never reached a builder. + coll.seal("lost").unwrap(); + } + coll + } + + #[test] + fn unbuilt_segments_queue_past_the_bounded_queue_and_all_install() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let key: VectorIndexKey = (DatabaseId::DEFAULT, TenantId::new(1), "docs:emb".into()); + core.vector_collections + .insert(key.clone(), unbuilt_collection()); + + core.queue_unbuilt_segments(); + assert_eq!(core.vector_builds.pending_for(&key), SEGMENTS); + + let deadline = Instant::now() + Duration::from_secs(20); + while core.vector_builds.pending_total() > 0 { + assert!(Instant::now() < deadline, "builds did not finish"); + std::thread::sleep(Duration::from_millis(10)); + core.poll_build_completions(); + } + + let coll = core.vector_collections.get(&key).expect("collection"); + let stats = coll.stats(); + assert_eq!(stats.sealed_count, SEGMENTS); + assert_eq!(stats.building_count, 0); + assert_eq!(stats.builds_completed, SEGMENTS as u64); + // Every node kept its id through the build. + for id in [0u32, 7, 8, 47] { + let hit = &coll + .search(&[id as f32, (id as usize % PER_SEGMENT % 3) as f32], 1, 64) + .unwrap()[0]; + assert_eq!(hit.id, id); + } + } + + #[test] + fn a_rebuild_goes_through_the_builder_and_keeps_ids() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _tx, _rx) = make_core_with_dir(dir.path()); + let key: VectorIndexKey = (DatabaseId::DEFAULT, TenantId::new(1), "docs:emb".into()); + core.vector_collections + .insert(key.clone(), unbuilt_collection()); + core.queue_unbuilt_segments(); + let deadline = Instant::now() + Duration::from_secs(20); + while core.vector_builds.pending_total() > 0 { + assert!(Instant::now() < deadline, "builds did not finish"); + std::thread::sleep(Duration::from_millis(10)); + core.poll_build_completions(); + } + if let Some(coll) = core.vector_collections.get_mut(&key) { + coll.delete(9); + } + + assert_eq!(core.queue_vector_rebuild(&key), SEGMENTS); + while core.vector_builds.pending_total() > 0 { + assert!(Instant::now() < deadline, "rebuilds did not finish"); + std::thread::sleep(Duration::from_millis(10)); + core.poll_build_completions(); + } + + let coll = core.vector_collections.get(&key).expect("collection"); + assert_eq!(coll.stats().builds_completed, 2 * SEGMENTS as u64); + assert!(!coll.is_live(9), "a tombstone carries over"); + assert_eq!(coll.len(), SEGMENTS * PER_SEGMENT, "no node added or lost"); + let hit = &coll.search(&[40.0, 0.0], 1, 64).unwrap()[0]; + assert_eq!(hit.id, 40); + } +} diff --git a/nodedb/src/data/executor/handlers/vector_lifecycle.rs b/nodedb/src/data/executor/handlers/vector_lifecycle.rs index 2bbe2f50f..a94858853 100644 --- a/nodedb/src/data/executor/handlers/vector_lifecycle.rs +++ b/nodedb/src/data/executor/handlers/vector_lifecycle.rs @@ -5,7 +5,7 @@ //! Separated from `vector.rs` (write handlers) by concern: //! write handlers deal with inserts/deletes, these deal with index management. -use tracing::{debug, info, warn}; +use tracing::{debug, info}; use crate::bridge::envelope::{ErrorCode, Response}; use crate::data::executor::core_loop::CoreLoop; @@ -46,6 +46,7 @@ impl CoreLoop { }; let mut stats = coll.stats(); + stats.builds_queued = self.vector_builds.pending_for(&index_key); // Populate arena_bytes from the per-collection arena registry. // Only set when the collection was assigned a dedicated arena // (vector-primary collections) and the registry is wired. @@ -66,7 +67,7 @@ impl CoreLoop { } } - /// Force-seal the growing segment, triggering background HNSW build. + /// Force-seal the growing segment and queue its HNSW build. pub(in crate::data::executor) fn execute_vector_seal( &mut self, task: &ExecutionTask, @@ -94,12 +95,8 @@ impl CoreLoop { let seal_key = CoreLoop::vector_build_key(&index_key); match coll.seal(&seal_key) { Some(req) => { - if let Some(tx) = &self.build_tx - && let Err(e) = tx.send(req) - { - warn!(core = self.core_id, error = %e, "failed to send HNSW build request after seal"); - } - info!(core = self.core_id, key = %seal_key, "growing segment sealed, HNSW build dispatched"); + self.queue_sealed_build(&index_key, req); + info!(core = self.core_id, key = %seal_key, "growing segment sealed, HNSW build queued"); self.checkpoint_coordinator.mark_dirty("vector", 1); self.response_ok(task) } @@ -158,7 +155,7 @@ impl CoreLoop { &mut self, params: VectorRebuildParams<'_>, ) -> Response { - use crate::engine::vector::hnsw::{HnswIndex, HnswParams}; + use crate::engine::vector::hnsw::HnswParams; let VectorRebuildParams { task, @@ -198,80 +195,22 @@ impl CoreLoop { metric: current.metric, dtype: current.dtype, }; - coll.set_params(new_params.clone()); + self.vector_params + .insert(index_key.clone(), new_params.clone()); - // Rebuild each sealed segment in-place with the new params. - // Each segment is rebuilt atomically: new index is constructed fully - // before swapping, so a failure leaves the old segment intact. - let mut rebuilt_count = 0usize; - for seg in coll.sealed_segments_mut() { - let vectors = match seg.index.export_vectors() { - Ok(v) => v, - Err(e) => { - // Leave the segment on the old params rather than rebuild - // it from vectors we could not read. - warn!( - core = self.core_id, - key = &index_key.2, - error = %e, - "rebuild: vector export failed, skipping segment" - ); - continue; - } - }; - if vectors.is_empty() { - continue; - } - let dim = seg.index.dim(); - let expected_count = vectors.len(); - let mut new_index = HnswIndex::new(dim, new_params.clone()); - for v in &vectors { - new_index - .insert(v.clone()) - .unwrap_or_else(|e| tracing::error!(error = %e, "HNSW rebuild insert failed")); - } - // Verify all vectors were inserted before swapping. - if new_index.len() != expected_count { - warn!( - core = self.core_id, - key = &index_key.2, - expected = expected_count, - actual = new_index.len(), - "rebuild: vector count mismatch, skipping segment" - ); - continue; - } - seg.index = new_index; - // Rebuild SQ8 for the new index. - seg.sq8 = super::super::handlers::vector_lifecycle::rebuild_sq8(&seg.index); - rebuilt_count += 1; - } - + // Each sealed segment is rebuilt on the builder thread with its ids + // kept, then swapped in on this core; search reads the old graph until + // then. Segments sealed later build under the new params too. + let queued = self.queue_vector_rebuild(&index_key); info!( core = self.core_id, key = &index_key.2, - rebuilt_count, + queued, m = new_params.m, ef = new_params.ef_construction, - "vector index rebuild complete" + "vector index rebuild queued" ); - - self.checkpoint_coordinator - .mark_dirty("vector", rebuilt_count); - - // Also update the stored params for this key. - self.vector_params.insert(index_key, new_params); - self.response_ok(task) } } - -/// Rebuild SQ8 quantized data for an HNSW index. -/// -/// Public within the handler module so `execute_vector_rebuild` can call it. -pub(in crate::data::executor) fn rebuild_sq8( - index: &crate::engine::vector::hnsw::HnswIndex, -) -> Option<(crate::engine::vector::quantize::sq8::Sq8Codec, Vec)> { - crate::engine::vector::collection::VectorCollection::build_sq8_for_index(index) -} diff --git a/nodedb/src/data/executor/handlers/vector_settle.rs b/nodedb/src/data/executor/handlers/vector_settle.rs index e32dd490d..2fb54832d 100644 --- a/nodedb/src/data/executor/handlers/vector_settle.rs +++ b/nodedb/src/data/executor/handlers/vector_settle.rs @@ -4,14 +4,14 @@ //! //! Every vector index, whatever its type, is one `VectorCollection` built //! from the index configuration `CREATE VECTOR INDEX` set. After a write the -//! collection settles: a full growing segment seals and its HNSW build request -//! goes to `build_tx`, and an IVF-PQ collection that holds its training +//! collection settles: a full growing segment seals and its HNSW build goes +//! to the core's builder queue, and an IVF-PQ collection that holds its training //! threshold trains and moves its buffered vectors into the IVF-PQ index. use std::collections::HashMap; use std::collections::hash_map::Entry; -use tracing::{error, info, warn}; +use tracing::{error, info}; use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::handlers::vector_direct_row::VectorIndexKey; @@ -72,13 +72,19 @@ impl CoreLoop { let config = vector_index_config_for(&self.index_configs, &self.vector_params, config_key); check_ivf_dim(&config, dim)?; - Ok(entry.insert(VectorCollection::with_index_config(dim, config))) + Ok( + entry.insert(VectorCollection::with_seal_threshold_and_config( + dim, + config, + self.vector_tuning.seal_threshold.max(1), + )), + ) } } } /// Settle the collection under `key` after a write: seal a full growing - /// segment and send its HNSW build, or train an IVF-PQ collection that + /// segment and queue its HNSW build, or train an IVF-PQ collection that /// holds its training threshold. pub(in crate::data::executor) fn settle_vector_collection(&mut self, key: &VectorIndexKey) { let seal_key = CoreLoop::vector_build_key(key); @@ -87,10 +93,8 @@ impl CoreLoop { }; if coll.needs_seal() && let Some(req) = coll.seal(&seal_key) - && let Some(tx) = &self.build_tx - && let Err(e) = tx.send(req) { - warn!(core = self.core_id, error = %e, "failed to send HNSW build request"); + self.queue_sealed_build(key, req); } self.train_ivf_if_ready(key); } @@ -107,20 +111,18 @@ impl CoreLoop { } } - /// Train every IVF-PQ collection that holds its training threshold. Boot - /// runs it once WAL replay and the store rebuild have restored the - /// buffers, so a collection that crossed its threshold before a restart - /// searches through IVF-PQ again without waiting for a write. - pub fn train_ready_ivf_collections(&mut self) { - let ready: Vec = self - .vector_collections - .iter() - .filter(|(_, coll)| coll.needs_ivf_training()) - .map(|(key, _)| key.clone()) - .collect(); - for key in ready { - self.train_vector_collection_ivf(&key); + /// Settle every collection once boot has restored it: seal a growing + /// segment replay filled, train an IVF-PQ collection that holds its + /// threshold, and queue a build for every segment sealed but not yet + /// built — including segments a checkpoint restored as building. Boot + /// runs it after WAL replay and the store rebuild, so search uses the + /// built graphs and IVF-PQ again without waiting for a write. + pub fn settle_vector_collections_after_boot(&mut self) { + let keys: Vec = self.vector_collections.keys().cloned().collect(); + for key in &keys { + self.settle_vector_collection(key); } + self.queue_unbuilt_segments(); } /// Train the IVF-PQ index of the collection under `key`. diff --git a/nodedb/src/data/executor/vector_checkpoint/build_completions.rs b/nodedb/src/data/executor/vector_checkpoint/build_completions.rs index d124a4a2a..241d759b7 100644 --- a/nodedb/src/data/executor/vector_checkpoint/build_completions.rs +++ b/nodedb/src/data/executor/vector_checkpoint/build_completions.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Draining completed background HNSW builds into the live collections. +//! Installing finished HNSW builds into the live collections. //! //! Lives beside the checkpoint because it shares the checkpoint's key //! encoding: a `BuildComplete.key` is the same `"{db}:{tid}:{coll}"` string a @@ -9,43 +9,130 @@ use super::paths::parse_build_key; use crate::data::executor::core_loop::CoreLoop; +use crate::engine::vector::collection::{BuildComplete, BuildKind}; impl CoreLoop { - /// Drain completed HNSW builds from the background builder thread and - /// promote the corresponding building segments to sealed segments. + /// Drain finished builds from this core's builder thread, install each + /// one on its collection, then send the next backlog jobs. /// - /// Called at the top of `tick()` before draining new requests. - /// - /// `BuildComplete.key` is the `"{db}:{tid}:{coll}"` string produced by - /// `VectorCollection::seal` (fed the `vector_build_key` of the - /// index key). Parse it back to the tuple key to look up the map. + /// Called at the top of `tick()`. An install happens on this core, in one + /// call, so a search sees either the brute-force segment or the built + /// graph, never a mix. pub fn poll_build_completions(&mut self) { - let Some(rx) = &self.build_rx else { return }; - while let Ok(complete) = rx.try_recv() { - // Parse the string key `"{db}:{tid}:{coll_key}"` back into the tuple. - let Some(tuple_key) = parse_build_key(&complete.key) else { - tracing::warn!( - core = self.core_id, - key = %complete.key, - "HNSW build completion has unparseable key; dropping" + let mut drained = Vec::new(); + if let Some(rx) = &self.vector_builds.rx { + while let Ok(complete) = rx.try_recv() { + drained.push(complete); + } + } + for complete in drained { + self.install_build(complete); + } + if !self.vector_builds.backlog.is_empty() { + self.dispatch_vector_builds(); + } else { + self.sync_build_pending_metric(); + } + } + + /// Install one finished build. A failed build, or one whose segment was + /// truncated, dropped or renumbered meanwhile, leaves the collection as + /// it is; the segment keeps answering by brute force or by its old graph. + fn install_build(&mut self, complete: BuildComplete) { + let BuildComplete { + key, + segment_id, + kind, + result, + } = complete; + let Some(tuple_key) = parse_build_key(&key) else { + tracing::error!( + core = self.core_id, + key = %key, + "HNSW build completion has unparseable key; dropping" + ); + return; + }; + self.vector_builds.note_finished(&tuple_key); + let memory = nodedb_mem::ScopedMemory::new( + self.governor.clone(), + tuple_key.0, + tuple_key.1, + nodedb_mem::EngineId::Vector, + ); + let Some(coll) = self.vector_collections.get_mut(&tuple_key) else { + return; + }; + let index = match result { + Ok(index) => index, + Err(e) => { + coll.note_build_failed(); + let (kind_label, segment) = match kind { + BuildKind::Seal => ("seal", segment_id), + BuildKind::Rebuild { base_id, .. } => ("rebuild", base_id), + }; + crate::diag::vector_build_failed( + &e, + &crate::diag::VectorBuildTarget { + kind: kind_label, + database_id: tuple_key.0.as_u64(), + tenant_id: tuple_key.1.as_u64(), + index: &tuple_key.2, + segment, + }, ); - continue; - }; - if let Some(coll) = self.vector_collections.get_mut(&tuple_key) { - let memory = nodedb_mem::ScopedMemory::new( - self.governor.clone(), - tuple_key.0, - tuple_key.1, - nodedb_mem::EngineId::Vector, + tracing::error!( + core = self.core_id, + key = %key, + segment_id, + error = %e, + "HNSW build failed; the segment stays as it is" ); - coll.complete_build(complete.segment_id, complete.index, memory); + if let Some(m) = &self.metrics { + m.record_vector_build_failed(); + } + return; + } + }; + let installed = match kind { + BuildKind::Seal => coll.complete_build(segment_id, index, memory), + BuildKind::Rebuild { base_id, len } => { + coll.complete_rebuild(segment_id, base_id, len, index, memory) + } + }; + if installed { + tracing::info!( + core = self.core_id, + key = %key, + segment_id, + ?kind, + "HNSW build installed" + ); + // A rebuilt segment replaces the old one in this single call on the + // owning core, so search reads the old graph or the new one, never + // a mix. REINDEX observers count this event, one per segment. + if let BuildKind::Rebuild { base_id, len } = kind { tracing::info!( + target: "nodedb::reindex", core = self.core_id, - key = %complete.key, - segment_id = complete.segment_id, - "HNSW build completed, segment promoted to sealed" + key = %key, + base_id, + len, + "atomic_cutover" ); } + self.checkpoint_coordinator.mark_dirty("vector", 1); + if let Some(m) = &self.metrics { + m.record_vector_build_completed(); + } + } else { + tracing::debug!( + core = self.core_id, + key = %key, + segment_id, + ?kind, + "HNSW build no longer matches its segment; discarded" + ); } } } diff --git a/nodedb/src/data/executor/vector_checkpoint/load.rs b/nodedb/src/data/executor/vector_checkpoint/load.rs index d43a75ea6..daeaa5167 100644 --- a/nodedb/src/data/executor/vector_checkpoint/load.rs +++ b/nodedb/src/data/executor/vector_checkpoint/load.rs @@ -59,7 +59,8 @@ impl CoreLoop { let loaded = decoded.len(); let mut vectors = 0usize; - for (key, collection) in decoded { + for (key, mut collection) in decoded { + collection.set_seal_threshold(self.vector_tuning.seal_threshold); vectors += collection.len(); self.vector_collections.insert(key, collection); } diff --git a/nodedb/src/data/runtime/boot_replay.rs b/nodedb/src/data/runtime/boot_replay.rs index a01b54511..ed8b8d32d 100644 --- a/nodedb/src/data/runtime/boot_replay.rs +++ b/nodedb/src/data/runtime/boot_replay.rs @@ -47,10 +47,11 @@ pub(super) fn replay_wal_and_rebuild_indexes( // checkpoint + WAL replay above already restored. core.rebuild_vector_indexes_from_store(vector_index_param_seed); - // Replay applies inserts without settling. An IVF-PQ collection whose - // restored buffer holds its training threshold trains here, so search - // uses IVF-PQ again from the first request. - core.train_ready_ivf_collections(); + // Replay applies inserts without settling. Full growing segments seal, + // IVF-PQ buffers at their threshold train, and every segment sealed but + // not yet built queues on the builder thread, so search uses the built + // indexes again without waiting for a write. + core.settle_vector_collections_after_boot(); // The in-memory R-tree spatial index needs no separate backstop here. // A document collection's geometry is indexed by the same diff --git a/nodedb/src/data/runtime/config.rs b/nodedb/src/data/runtime/config.rs index 623a44316..2e9a06511 100644 --- a/nodedb/src/data/runtime/config.rs +++ b/nodedb/src/data/runtime/config.rs @@ -16,6 +16,8 @@ pub struct CoreCompactionConfig { /// Timeseries engine tuning (memtable soft/hard budgets, tag cardinality /// ceiling). Drives the record-boundary admission gate on the ingest path. pub timeseries: nodedb_types::config::tuning::TimeseriesToning, + /// Vector engine tuning: seal threshold and PQ / IVF defaults. + pub vector: nodedb_types::config::tuning::VectorTuning, /// How often this core's event loop flushes vector + sparse-vector /// indexes to disk (the per-core backstop checkpoint, distinct from but /// sourced from the same `[checkpoint].interval_secs` as the Control @@ -31,6 +33,7 @@ impl Default for CoreCompactionConfig { query: nodedb_types::config::tuning::QueryTuning::default(), graph: nodedb_types::config::tuning::GraphTuning::default(), timeseries: nodedb_types::config::tuning::TimeseriesToning::default(), + vector: nodedb_types::config::tuning::VectorTuning::default(), checkpoint_interval: std::time::Duration::from_secs(300), } } diff --git a/nodedb/src/data/runtime/spawn.rs b/nodedb/src/data/runtime/spawn.rs index ca27e89b7..003ed4b82 100644 --- a/nodedb/src/data/runtime/spawn.rs +++ b/nodedb/src/data/runtime/spawn.rs @@ -113,6 +113,10 @@ pub fn spawn_core( // whole life regardless of what the operator configured. core.set_timeseries_tuning(compaction_config.timeseries); + // 2f. Apply vector tuning, also before the checkpoint restore: + // restored collections take its seal threshold. + core.set_vector_tuning(compaction_config.vector); + // 3 → 3b → 4. Boot recovery runs in exactly this order and no other: // restore the checkpoints, THEN seed the catalog state, THEN replay // the WAL. Each checkpoint restores state as of the LSN it was diff --git a/nodedb/src/diag/context/mod.rs b/nodedb/src/diag/context/mod.rs index 85b8ef86b..57c585e03 100644 --- a/nodedb/src/diag/context/mod.rs +++ b/nodedb/src/diag/context/mod.rs @@ -16,6 +16,7 @@ mod raft_apply; mod recovery; mod retention; mod vector; +mod vector_build; mod write_path; pub(in crate::diag) use catalog::{ @@ -39,6 +40,7 @@ pub(in crate::diag) use raft_apply::{RaftEntryReapplied, ReplicatedWriteParked}; pub(in crate::diag) use recovery::{ReplayRecordUnapplied, WalArchivalFailedTruncationHeld}; pub(in crate::diag) use retention::RetentionAutowireOrphaned; pub(in crate::diag) use vector::VectorIndexNotApplied; +pub(in crate::diag) use vector_build::{VectorBuildNotInstalled, VectorBuilderUnavailable}; pub(in crate::diag) use write_path::{ BatchInsertWithoutSurrogates, FtsIndexUpdateFailed, OrphanedIndexEntryAfterDelete, StrictRowUndecodable, WriteAckedWithoutDurability, diff --git a/nodedb/src/diag/context/vector_build.rs b/nodedb/src/diag/context/vector_build.rs new file mode 100644 index 000000000..ca42dfc15 --- /dev/null +++ b/nodedb/src/diag/context/vector_build.rs @@ -0,0 +1,153 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Forensic payloads for HNSW build capture sites. +//! +//! A build that never installs leaves its segment on the path it had before: +//! brute force for a newly sealed segment, the old graph for a rebuild. +//! Search stays correct, but it runs slower than the index promises, and +//! nothing but a counter shows it. + +use faultbox::DomainContext; +use faultbox::serde_json::{Value, json}; + +/// A build of one segment of one index that did not install. +pub(in crate::diag) struct VectorBuildNotInstalled<'a> { + /// Stage that failed: `build` (the builder refused the vectors) or + /// `rebuild_read` (the core could not read a sealed segment to rebuild). + pub stage: &'static str, + /// `seal` for a first build, `rebuild` for a REINDEX or ALTER rebuild. + pub kind: &'static str, + pub database_id: u64, + pub tenant_id: u64, + /// Index key: `collection` or `collection:field`. + pub index: &'a str, + /// Segment id for a first build, base id for a rebuild. + pub segment: u32, + /// What failed, without the per-occurrence detail. + pub error_class: &'a str, +} + +impl DomainContext for VectorBuildNotInstalled<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb.vector_build_not_installed" + } + + fn grouping_key(&self) -> String { + // Stage, kind and error class name the bug. The index and segment + // are the occurrence, so a boot re-queue of the same bad segment + // files one report. + format!( + "stage={};kind={};cause={}", + self.stage, self.kind, self.error_class + ) + } + + fn to_json(&self) -> Value { + json!({ + "stage": self.stage, + "kind": self.kind, + "database_id": self.database_id, + "tenant_id": self.tenant_id, + "index": self.index, + "segment": self.segment, + "error_class": self.error_class, + "why_reported": "the segment keeps answering search without the graph this \ + build was meant to install: by brute force for a first build, \ + by its old graph for a rebuild. Results stay correct. Latency \ + grows with the segment, and a rebuild's new params or \ + quantization never take effect", + "operator_action": "run SHOW VECTOR INDEX on the index: 'building_segments' \ + stays above zero and 'builds_failed' rises. A restart \ + re-queues every unbuilt segment. If the same stage fails \ + again, the segment's vectors or the index params are bad: \ + check the dimension and metric against the data", + }) + } +} + +/// The core's builder thread could not be started, or died. +pub(in crate::diag) struct VectorBuilderUnavailable<'a> { + /// `spawn_failed` or `disconnected`. + pub cause: &'static str, + pub core_id: usize, + /// What failed, without the per-occurrence detail. Empty for a + /// disconnect, which carries no error. + pub error_class: &'a str, +} + +impl DomainContext for VectorBuilderUnavailable<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb.vector_builder_unavailable" + } + + fn grouping_key(&self) -> String { + // The core id is the occurrence: every core fails the same way. + format!("cause={};class={}", self.cause, self.error_class) + } + + fn to_json(&self) -> Value { + json!({ + "cause": self.cause, + "core_id": self.core_id, + "error_class": self.error_class, + "why_reported": "no HNSW graph gets built on this core until a builder thread \ + runs. Sealed segments answer search by brute force meanwhile, \ + and the core retries the spawn on every tick that has builds \ + waiting", + "operator_action": "'spawn_failed' means the OS refused a thread: check the \ + process thread and memory limits. 'disconnected' means \ + the builder thread panicked: the panic report filed next \ + to this one names the cause", + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn sample() -> VectorBuildNotInstalled<'static> { + VectorBuildNotInstalled { + stage: "build", + kind: "seal", + database_id: 1, + tenant_id: 2, + index: "docs:emb", + segment: 3, + error_class: "dimension mismatch", + } + } + + #[test] + fn grouping_ignores_the_index_and_segment() { + let first = sample(); + let second = VectorBuildNotInstalled { + database_id: 9, + tenant_id: 8, + index: "other:emb", + segment: 70, + ..sample() + }; + assert_eq!(first.grouping_key(), second.grouping_key()); + } + + #[test] + fn grouping_separates_first_builds_from_rebuilds() { + let rebuild = VectorBuildNotInstalled { + kind: "rebuild", + ..sample() + }; + assert_ne!(sample().grouping_key(), rebuild.grouping_key()); + } + + #[test] + fn grouping_ignores_the_core() { + let a = VectorBuilderUnavailable { + cause: "disconnected", + core_id: 0, + error_class: "", + }; + let b = VectorBuilderUnavailable { core_id: 7, ..a }; + assert_eq!(a.grouping_key(), b.grouping_key()); + } +} diff --git a/nodedb/src/diag/mod.rs b/nodedb/src/diag/mod.rs index fc67e601d..9dc9bc612 100644 --- a/nodedb/src/diag/mod.rs +++ b/nodedb/src/diag/mod.rs @@ -10,15 +10,17 @@ mod recording; pub use context::{DATABASE_SCOPE, IlpFlushOutcome, LostResponseWrite, TENANT_SCOPE}; pub use recording::{ - batch_insert_without_surrogates, calvin_apply_halted, calvin_completion_timeout, - catalog_apply_orphan_row, collection_purge_row_missing, consumer_group_offsets_retained, - data_plane_core_fail_stopped, data_plane_response_lost, data_plane_responses_lost, entry_kind, - fts_index_update_failed, history_compaction_not_applied, ilp_invalid_utf8_drop, - ilp_line_read_drop, metadata_apply_wedged, orphaned_index_entry_after_delete, - quota_row_invalid, quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, - quota_scope_replay_aborted, raft_entries_reapplied, raft_entry_reapplied, - replay_record_unapplied, replicated_write_parked, replicated_writes_parked, - retention_autowire_orphaned, scope_quota_not_installed, strict_row_undecodable, - synonym_group_not_applied, vector_index_not_applied, wal_archival_failed_truncation_held, - write_acked_without_durability, write_window_held, write_window_leaked, + VectorBuildTarget, batch_insert_without_surrogates, calvin_apply_halted, + calvin_completion_timeout, catalog_apply_orphan_row, collection_purge_row_missing, + consumer_group_offsets_retained, data_plane_core_fail_stopped, data_plane_response_lost, + data_plane_responses_lost, entry_kind, fts_index_update_failed, history_compaction_not_applied, + ilp_invalid_utf8_drop, ilp_line_read_drop, metadata_apply_wedged, + orphaned_index_entry_after_delete, quota_row_invalid, quota_row_undecodable, + quota_row_write_failed, quota_scope_purge_incomplete, quota_scope_replay_aborted, + raft_entries_reapplied, raft_entry_reapplied, replay_record_unapplied, replicated_write_parked, + replicated_writes_parked, retention_autowire_orphaned, scope_quota_not_installed, + strict_row_undecodable, synonym_group_not_applied, vector_build_failed, + vector_builder_disconnected, vector_builder_spawn_failed, vector_index_not_applied, + vector_rebuild_unreadable, wal_archival_failed_truncation_held, write_acked_without_durability, + write_window_held, write_window_leaked, }; diff --git a/nodedb/src/diag/recording/mod.rs b/nodedb/src/diag/recording/mod.rs index 8ebe9a52a..250bbdcd4 100644 --- a/nodedb/src/diag/recording/mod.rs +++ b/nodedb/src/diag/recording/mod.rs @@ -19,6 +19,7 @@ mod recovery; mod retention; mod shared; mod vector; +mod vector_build; pub use catalog::{ catalog_apply_orphan_row, collection_purge_row_missing, consumer_group_offsets_retained, @@ -46,3 +47,7 @@ pub use recovery::{ pub use retention::retention_autowire_orphaned; pub use shared::entry_kind; pub use vector::vector_index_not_applied; +pub use vector_build::{ + VectorBuildTarget, vector_build_failed, vector_builder_disconnected, + vector_builder_spawn_failed, vector_rebuild_unreadable, +}; diff --git a/nodedb/src/diag/recording/vector_build.rs b/nodedb/src/diag/recording/vector_build.rs new file mode 100644 index 000000000..af059d0c0 --- /dev/null +++ b/nodedb/src/diag/recording/vector_build.rs @@ -0,0 +1,94 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capture sites for HNSW builds that never installed, and for a core that +//! lost its builder thread. + +use faultbox::{Capture, EventKind, error_chain_of}; + +use super::shared::error_class; +use crate::diag::context; + +/// Where a failed build belongs: the index and the segment it covered. +pub struct VectorBuildTarget<'a> { + /// `seal` or `rebuild`. + pub kind: &'static str, + pub database_id: u64, + pub tenant_id: u64, + pub index: &'a str, + pub segment: u32, +} + +/// Report a build the builder thread refused. Called from the completion +/// arm that receives the builder's error. +pub fn vector_build_failed(err: &nodedb_vector::VectorError, target: &VectorBuildTarget<'_>) { + record_not_installed("build", err, target); +} + +/// Report a rebuild whose sealed segment the core could not read. Called +/// from the dispatch arm that reads the segment's vectors. +pub fn vector_rebuild_unreadable(err: &nodedb_vector::VectorError, target: &VectorBuildTarget<'_>) { + record_not_installed("rebuild_read", err, target); +} + +/// Shared emit for the two not-installed causes. Private so the only entry +/// points are the one-per-cause functions above. +fn record_not_installed( + stage: &'static str, + err: &nodedb_vector::VectorError, + target: &VectorBuildTarget<'_>, +) { + let class = error_class(err); + let ctx = context::VectorBuildNotInstalled { + stage, + kind: target.kind, + database_id: target.database_id, + tenant_id: target.tenant_id, + index: target.index, + segment: target.segment, + error_class: &class, + }; + let _ = Capture::new( + EventKind::Error, + "HNSW build did not install; the segment keeps its previous search path", + ) + .error_chain(error_chain_of(err)) + .domain(&ctx) + .with_backtrace() + .emit(); +} + +/// Report a core whose builder thread could not be spawned. Called from the +/// spawn arm of the core's build queue. +pub fn vector_builder_spawn_failed(err: &std::io::Error, core_id: usize) { + let class = error_class(err); + let ctx = context::VectorBuilderUnavailable { + cause: "spawn_failed", + core_id, + error_class: &class, + }; + let _ = Capture::new( + EventKind::Error, + "HNSW builder thread could not be spawned; no graph builds on this core", + ) + .error_chain(error_chain_of(err)) + .domain(&ctx) + .with_backtrace() + .emit(); +} + +/// Report a core whose builder thread died. Called from the dispatch arm +/// that finds the request queue disconnected. +pub fn vector_builder_disconnected(core_id: usize) { + let ctx = context::VectorBuilderUnavailable { + cause: "disconnected", + core_id, + error_class: "", + }; + let _ = Capture::new( + EventKind::InvariantViolation, + "HNSW builder thread died with builds in flight", + ) + .domain(&ctx) + .with_backtrace() + .emit(); +} diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index 71f3019da..7da49c36e 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -148,6 +148,7 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { ErrorCode::DivisionByZero => NodeDbError::division_by_zero(), ErrorCode::UndefinedFunction { name } => NodeDbError::undefined_function(name), ErrorCode::DataException { detail } => NodeDbError::data_exception(detail), + ErrorCode::BadRequest { detail } => NodeDbError::bad_request(detail), // Nothing was enqueued, and the same request succeeds once capacity // frees: the retryable overload class. ErrorCode::DispatchCapacity { reason } => NodeDbError::server_overload(reason), diff --git a/nodedb/tests/inproc/cases/reindex_vector_concurrent.rs b/nodedb/tests/inproc/cases/reindex_vector_concurrent.rs index 4bf9aeb6f..90870abb9 100644 --- a/nodedb/tests/inproc/cases/reindex_vector_concurrent.rs +++ b/nodedb/tests/inproc/cases/reindex_vector_concurrent.rs @@ -13,7 +13,9 @@ //! signature of a rebuild that took an exclusive lock instead of //! running concurrently //! 5. exactly one `atomic_cutover` tracing event was emitted by the -//! `nodedb::reindex` target during the rebuild phase +//! `nodedb::reindex` target during the rebuild phase. REINDEX rebuilds +//! each sealed segment and swaps it in on its own, one event per +//! segment, so the test force-seals its rows into one segment first. //! //! Why no p99 ratio: this test asserted `rebuild_p99 <= 2.0 * baseline_p99`, //! and the dataset had been shrunk to the point where the rebuild finished @@ -127,6 +129,21 @@ async fn nn_query(server: &TestServer, query_vec: &[f32]) -> Duration { t.elapsed() } +/// One numeric property of `SHOW VECTOR INDEX status ON vecs10k.emb`. +async fn vector_index_status(server: &TestServer, property: &str) -> u64 { + let rows = server + .query_rows("SHOW VECTOR INDEX status ON vecs10k.emb") + .await + .expect("SHOW VECTOR INDEX failed"); + let row = rows + .iter() + .find(|r| r[0] == property) + .unwrap_or_else(|| panic!("SHOW VECTOR INDEX must report {property}: {rows:?}")); + row[1] + .parse() + .unwrap_or_else(|e| panic!("{property} must be a number, got {:?}: {e}", row[1])) +} + /// Compute the p99 of a slice of `Duration` values (must be non-empty). fn p99(mut samples: Vec) -> Duration { assert!(!samples.is_empty(), "p99: empty sample set"); @@ -232,6 +249,28 @@ async fn reindex_vector_concurrent_p99() { server.exec(&sql).await.unwrap(); } + // ── Seal: REINDEX rebuilds sealed segments only ───────────────────────── + // The rows sit in the growing segment, far below the default seal + // threshold. Force-seal them into one segment and wait for its first + // build, so REINDEX has exactly one sealed segment to rebuild. + server + .exec("ALTER VECTOR INDEX ON vecs10k.emb SEAL") + .await + .unwrap(); + let seal_deadline = Instant::now() + Duration::from_secs(60); + loop { + let sealed = vector_index_status(&server, "sealed_segments").await; + let building = vector_index_status(&server, "building_segments").await; + if sealed == 1 && building == 0 { + break; + } + assert!( + Instant::now() < seal_deadline, + "the sealed segment did not build: sealed={sealed} building={building}" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + // ── Baseline phase: 200 sequential queries, record latencies ──────────── let mut baseline_latencies: Vec = Vec::with_capacity(BASELINE_QUERIES); let mut qseed: u64 = 0xCAFE_F00D_ABCD_EF01; @@ -320,8 +359,9 @@ async fn reindex_vector_concurrent_p99() { }); // Issue REINDEX CONCURRENTLY on the main client. - // This returns as soon as the background thread is started; the atomic - // cutover is applied on a later tick() — so we must wait for it. + // This returns once the segment's rebuild is queued on the core's builder + // thread. The core swaps the rebuilt segment in on a later tick, so the + // test waits for the cutover event. let rebuild_started = Instant::now(); server.exec("REINDEX CONCURRENTLY vecs10k").await.unwrap(); diff --git a/nodedb/tests/wire/cases/mod.rs b/nodedb/tests/wire/cases/mod.rs index 95800e476..3fd161b40 100644 --- a/nodedb/tests/wire/cases/mod.rs +++ b/nodedb/tests/wire/cases/mod.rs @@ -323,6 +323,7 @@ mod truncate_engine_conformance_columnar_family; mod txn_ddl_commit_registry_sync; mod user_transaction; mod vector_dimension_errors; +mod vector_hnsw_build; mod vector_index_bulk_delete_reindex; mod vector_index_bulk_update_reindex; mod vector_index_merge_reindex; diff --git a/nodedb/tests/wire/cases/sql_hybrid_search.rs b/nodedb/tests/wire/cases/sql_hybrid_search.rs index 33cf8bab3..0d0f70431 100644 --- a/nodedb/tests/wire/cases/sql_hybrid_search.rs +++ b/nodedb/tests/wire/cases/sql_hybrid_search.rs @@ -295,7 +295,24 @@ async fn hybrid_search_returns_the_text_leg_error() { .await .expect_err("a hybrid search whose text leg fails must fail"); assert!( - err.contains("at least one positive term"), - "the error must be the text leg's own error; got: {err}" + err.contains("42601") && err.contains("at least one positive term"), + "the error must be the text leg's own error, as syntax_error; got: {err}" + ); +} + +/// A NOT-only text query on a plain full-text search is the caller's syntax +/// error, SQLSTATE `42601`, as the Control Plane reports it; never `XX000`. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn invalid_fts_query_is_a_syntax_error() { + let server = TestServer::start().await; + create_hybrid_collection(&server, "hs_fts_err").await; + + let err = server + .query_rows("SELECT id FROM hs_fts_err WHERE text_match(content, 'NOT consensus')") + .await + .expect_err("a NOT-only text query must fail"); + assert!( + err.contains("42601") && err.contains("at least one positive term"), + "an invalid FTS query must be syntax_error, not XX000; got: {err}" ); } diff --git a/nodedb/tests/wire/cases/sql_three_source_rrf.rs b/nodedb/tests/wire/cases/sql_three_source_rrf.rs index 849de1e8f..574a44e2e 100644 --- a/nodedb/tests/wire/cases/sql_three_source_rrf.rs +++ b/nodedb/tests/wire/cases/sql_three_source_rrf.rs @@ -388,7 +388,7 @@ async fn rrf_score_triple_returns_the_text_leg_error() { .await .expect_err("a three-source search whose text leg fails must fail"); assert!( - err.contains("at least one positive term"), - "the error must be the text leg's own error; got: {err}" + err.contains("42601") && err.contains("at least one positive term"), + "the error must be the text leg's own error, as syntax_error; got: {err}" ); } diff --git a/nodedb/tests/wire/cases/vector_hnsw_build.rs b/nodedb/tests/wire/cases/vector_hnsw_build.rs new file mode 100644 index 000000000..ac79f6665 --- /dev/null +++ b/nodedb/tests/wire/cases/vector_hnsw_build.rs @@ -0,0 +1,242 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sealed vector segments get their HNSW graph from the core's builder +//! thread, and REINDEX rebuilds them through the same path. +//! +//! - Past the seal threshold, every sealed segment reaches built: +//! `SHOW VECTOR INDEX` reports no building segment, no queued build, and +//! completed builds. Search finds every vector once. +//! - A restart right after the inserts, with builds still queued or +//! running, recovers: the segments build again and search is unchanged. +//! - REINDEX keeps every row's identity: each row is still found under its +//! own id, deleted rows stay gone, no row appears twice, and the PQ codes +//! are present afterwards. + +use std::time::{Duration, Instant}; + +use crate::harness::TestServer; + +const COLL: &str = "hnsw_build"; +/// Vectors per sealed segment. +const SEAL: usize = 64; +/// 3 sealed segments of 64 plus 8 vectors in the growing segment. +const ROWS: usize = 200; +const SEALED_SEGMENTS: usize = ROWS / SEAL; + +/// Distinct per `i`: the first component is `i + 1`. +fn vector(i: usize) -> String { + let parts: Vec = [1usize, 7, 11, 13] + .iter() + .map(|&m| format!("{:.1}", (if m == 1 { i + 1 } else { i % m + 1 }) as f32)) + .collect(); + format!("[{}]", parts.join(", ")) +} + +async fn create(srv: &TestServer) { + srv.exec(&format!( + "CREATE COLLECTION {COLL} WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + srv.exec(&format!( + "CREATE VECTOR INDEX idx_{COLL} ON {COLL} (embedding) METRIC l2 DIM 4 \ + INDEX_TYPE hnsw_pq PQ_M 2" + )) + .await + .unwrap(); +} + +async fn insert_all(srv: &TestServer) { + for i in 0..ROWS { + srv.exec(&format!( + "INSERT INTO {COLL} {{ id: 'v{i}', embedding: {} }}", + vector(i) + )) + .await + .unwrap_or_else(|e| panic!("insert of v{i} must succeed: {e}")); + } +} + +async fn status(srv: &TestServer, property: &str) -> String { + let rows = srv + .query_rows(&format!("SHOW VECTOR INDEX status ON {COLL}.embedding")) + .await + .unwrap(); + rows.iter() + .find(|r| r[0] == property) + .map(|r| r[1].clone()) + .unwrap_or_else(|| panic!("SHOW VECTOR INDEX must report {property}: {rows:?}")) +} + +async fn status_num(srv: &TestServer, property: &str) -> u64 { + status(srv, property) + .await + .parse() + .unwrap_or_else(|e| panic!("{property} must be a number: {e}")) +} + +/// Wait until at least the expected sealed segments exist, none is still +/// building, no build is queued, and at least `completed` builds have been +/// installed. +async fn wait_built(srv: &TestServer, completed: u64) { + let deadline = Instant::now() + Duration::from_secs(30); + loop { + let sealed = status_num(srv, "sealed_segments").await; + let building = status_num(srv, "building_segments").await; + let queued = status_num(srv, "builds_queued").await; + let done = status_num(srv, "builds_completed").await; + // A restart re-indexes every document from the store, so it can hold + // more sealed segments than the first run did. + if sealed >= SEALED_SEGMENTS as u64 && building == 0 && queued == 0 && done >= completed { + return; + } + assert!( + Instant::now() < deadline, + "builds did not finish: sealed={sealed} building={building} queued={queued} \ + completed={done}" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } +} + +/// Every id a wide search returns, sorted. +async fn all_ids(srv: &TestServer) -> Vec { + let mut ids: Vec = srv + .query_rows(&format!( + "SELECT id FROM {COLL} ORDER BY vector_distance(embedding, ARRAY{}) LIMIT 1000", + vector(0) + )) + .await + .unwrap() + .into_iter() + .map(|r| r[0].clone()) + .collect(); + ids.sort(); + ids +} + +fn expected_ids(skip: &[usize]) -> Vec { + let mut ids: Vec = (0..ROWS) + .filter(|i| !skip.contains(i)) + .map(|i| format!("v{i}")) + .collect(); + ids.sort(); + ids +} + +/// Assert that a search returned every expected row exactly once. On a +/// mismatch, name the missing, extra and repeated ids instead of printing +/// two long lists. +fn assert_same_ids(actual: Vec, skip: &[usize], context: &str) { + let expected = expected_ids(skip); + if actual == expected { + return; + } + let missing: Vec<&String> = expected.iter().filter(|id| !actual.contains(id)).collect(); + let extra: Vec<&String> = actual.iter().filter(|id| !expected.contains(id)).collect(); + let repeated: Vec<&String> = actual + .iter() + .enumerate() + .filter(|(i, id)| actual[..*i].contains(id)) + .map(|(_, id)| id) + .collect(); + panic!( + "{context}: {} ids returned, {} expected; missing {missing:?}, \ + not expected {extra:?}, repeated {repeated:?}", + actual.len(), + expected.len() + ); +} + +async fn nearest(srv: &TestServer, i: usize) -> String { + let rows = srv + .query_rows(&format!( + "SELECT id FROM {COLL} ORDER BY vector_distance(embedding, ARRAY{}) LIMIT 1", + vector(i) + )) + .await + .unwrap(); + rows[0][0].clone() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn sealed_segments_build_and_search_uses_them() { + let srv = TestServer::start_with_vector_seal_threshold(SEAL).await; + create(&srv).await; + insert_all(&srv).await; + + wait_built(&srv, SEALED_SEGMENTS as u64).await; + // One HNSW node per row: an insert indexed once by the document write + // and again by a separate vector insert leaves a tombstoned twin per row. + assert_eq!(status_num(&srv, "live_count").await, ROWS as u64); + assert_eq!(status_num(&srv, "tombstone_count").await, 0); + assert_eq!( + status_num(&srv, "sealed_segments").await, + SEALED_SEGMENTS as u64 + ); + assert_eq!( + status_num(&srv, "growing_vectors").await, + (ROWS % SEAL) as u64 + ); + assert_eq!(status(&srv, "quantization").await, "pq"); + assert_same_ids(all_ids(&srv).await, &[], "after the builds"); + for i in [0, 63, 64, 150, 199] { + assert_eq!(nearest(&srv, i).await, format!("v{i}")); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_restart_during_the_builds_recovers() { + let srv = TestServer::start_with_vector_seal_threshold(SEAL).await; + create(&srv).await; + insert_all(&srv).await; + + // Stop at once: builds are still queued or running. + let (srv, dir) = srv.take_dir(); + srv.graceful_shutdown().await; + let (srv2, _dir) = TestServer::open_on_path_with_vector_seal_threshold(dir, SEAL).await; + + wait_built(&srv2, 0).await; + assert_same_ids(all_ids(&srv2).await, &[], "after the restart"); + for i in [0, 64, 199] { + assert_eq!(nearest(&srv2, i).await, format!("v{i}")); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn reindex_keeps_identities_and_codes() { + let srv = TestServer::start_with_vector_seal_threshold(SEAL).await; + create(&srv).await; + insert_all(&srv).await; + wait_built(&srv, SEALED_SEGMENTS as u64).await; + + // Tombstones in two sealed segments. + for i in [5, 70] { + srv.exec(&format!("DELETE FROM {COLL} WHERE id = 'v{i}'")) + .await + .unwrap(); + } + assert_same_ids(all_ids(&srv).await, &[5, 70], "before REINDEX"); + let done = status_num(&srv, "builds_completed").await; + + srv.exec(&format!("REINDEX CONCURRENTLY {COLL}")) + .await + .unwrap(); + wait_built(&srv, done + SEALED_SEGMENTS as u64).await; + + assert_same_ids( + all_ids(&srv).await, + &[5, 70], + "after REINDEX: every row once, deleted rows still gone", + ); + // Each row is still found at distance 0 under its own id: the rebuilt + // graphs kept every node id, so every surrogate binding still holds. + for i in [0, 4, 6, 63, 64, 69, 71, 127, 128, 191, 192, 199] { + assert_eq!(nearest(&srv, i).await, format!("v{i}")); + } + assert_eq!(status(&srv, "quantization").await, "pq"); + assert_eq!( + status_num(&srv, "growing_vectors").await, + (ROWS % SEAL) as u64 + ); +} diff --git a/nodedb/tests/wire/harness/config_toml.rs b/nodedb/tests/wire/harness/config_toml.rs index 68bffa8e0..c6f959e05 100644 --- a/nodedb/tests/wire/harness/config_toml.rs +++ b/nodedb/tests/wire/harness/config_toml.rs @@ -46,6 +46,9 @@ pub(super) struct TuningOverrides { /// Overrides `[tuning.timeseries] memtable_budget_bytes` so a test can /// observe timeseries partition flushes on a handful of rows. pub(super) timeseries_memtable_budget_bytes: Option, + /// Overrides `[tuning.vector] seal_threshold` so a test can observe HNSW + /// segment builds on a few hundred vectors. + pub(super) vector_seal_threshold: Option, /// Sets `[server] single_node_calvin = false` so the server boots with no /// cluster topology. The planner then emits the single-node plan forms /// (`ArrayOp::{Put, Delete, Slice, ...}`) instead of the `ClusterArrayOp` @@ -91,6 +94,14 @@ impl TuningOverrides { } } + /// Boot with a lowered vector seal threshold. + pub(super) fn vector_seal_threshold(vectors: usize) -> Self { + Self { + vector_seal_threshold: Some(vectors), + ..Self::default() + } + } + /// Boot without the single-node Calvin stack (no cluster topology). pub(super) fn standalone() -> Self { Self { @@ -143,6 +154,9 @@ pub(super) fn write_config(dir: &Path, auth_mode: AuthMode, tuning: TuningOverri "\n[tuning.timeseries]\nmemtable_budget_bytes = {bytes}\n" )); } + if let Some(vectors) = tuning.vector_seal_threshold { + toml.push_str(&format!("\n[tuning.vector]\nseal_threshold = {vectors}\n")); + } toml.push_str(&format!( "\n[backup_encryption]\nkey_path = {}\n", toml_quote(&write_backup_kek(dir)) diff --git a/nodedb/tests/wire/harness/lifecycle.rs b/nodedb/tests/wire/harness/lifecycle.rs index aad6330d7..b255f4929 100644 --- a/nodedb/tests/wire/harness/lifecycle.rs +++ b/nodedb/tests/wire/harness/lifecycle.rs @@ -115,6 +115,36 @@ impl TestServer { Self::connect_and_build(spawned, dir, AuthMode::Trust).await } + /// Spawn a single-core NodeDB server with a lowered vector seal threshold, + /// so a few hundred inserts seal segments and queue HNSW builds. + pub async fn start_with_vector_seal_threshold(vectors: usize) -> Self { + let dir = tempfile::tempdir().expect("tempdir"); + let spawned = process::spawn( + dir.path(), + AuthMode::Trust, + TuningOverrides::vector_seal_threshold(vectors), + 1, + ); + Self::connect_and_build(spawned, dir, AuthMode::Trust).await + } + + /// Reopen `dir` with the lowered vector seal threshold a server started + /// by [`Self::start_with_vector_seal_threshold`] used. + pub async fn open_on_path_with_vector_seal_threshold( + dir: TestDataDir, + vectors: usize, + ) -> (Self, TestDataDir) { + let spawned = process::spawn( + dir.path(), + AuthMode::Trust, + TuningOverrides::vector_seal_threshold(vectors), + 1, + ); + let placeholder = tempfile::tempdir().expect("placeholder tempdir"); + let server = Self::connect_and_build(spawned, placeholder, AuthMode::Trust).await; + (server, dir) + } + /// Spawn a single-core NodeDB server with `single_node_calvin = false`: /// no cluster topology, so the planner emits the single-node plan forms /// (`ArrayOp::{Put, Delete, Slice, ...}`) rather than the `ClusterArrayOp` From ad0ec5d97c77ce4182c34d41de8e959560ee22bc Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 10:19:23 +0800 Subject: [PATCH 48/64] feat(reindex): rebuild graph CSR and FTS indexes without blocking writes REINDEX and the vector settle/seal path used to rebuild the graph CSR adjacency index and the FTS inverted index inline, blocking the core for the duration of the build. Both now build a shadow copy off the core while it keeps serving reads and writes, journal writes made during the build, replay the journal onto the shadow copy at cutover, and swap it in atomically. A write that diverges between the live and shadow index, or that overflows the journal's byte bound, discards the shadow copy and leaves the live index untouched so REINDEX can be retried. - nodedb-graph gains csr::rebuild (seed, journal, install) and new GraphError variants for a rebuild already running, a superseded rebuild, a journal overflow, a replay divergence, and an invalid snapshot. - nodedb::engine::sparse::inverted gains the matching rebuild_snapshot/rebuild_journal/rebuild_install modules for the FTS posting index. - The executor's reindex handler is split into a control/reindex/ module (dispatch, csr, fts, pending holds, waiters) replacing the old single-file reindex/reindex_apply handlers, and MetaOp::RebuildIndex's contract is updated to describe the non-blocking build, deadline-based wait, and cutover semantics. - Point and bulk-DML updates, snapshot restore, and CONVERT now route text-index maintenance through the same rebuild-aware path (update_reindex_text, snapshot restore/text.rs) instead of updating the FTS index inline, and CONVERT reports a schema mismatch as a data-exception naming the row and column instead of a generic internal error. - Diagnostics record index-rebuild lifecycle events (diag/context/recording index_rebuild) alongside the existing vector-build diagnostics. --- nodedb-graph/src/csr/index/interning.rs | 42 ++ nodedb-graph/src/csr/index/lookup.rs | 11 +- nodedb-graph/src/csr/index/mutation.rs | 58 ++ nodedb-graph/src/csr/index/restore.rs | 65 ++- nodedb-graph/src/csr/index/types.rs | 5 + nodedb-graph/src/csr/mod.rs | 1 + nodedb-graph/src/csr/persist.rs | 2 + nodedb-graph/src/csr/rebuild/install.rs | 185 +++++++ nodedb-graph/src/csr/rebuild/journal.rs | 301 +++++++++++ nodedb-graph/src/csr/rebuild/mod.rs | 15 + nodedb-graph/src/csr/rebuild/seed.rs | 126 +++++ nodedb-graph/src/error.rs | 29 + nodedb-physical/src/physical_plan/meta.rs | 33 +- .../shared/ddl/neutral/convert/driver.rs | 29 +- .../shared/ddl/neutral/maintenance/reindex.rs | 65 ++- .../executor/core_loop/maintenance_state.rs | 17 +- nodedb/src/data/executor/core_loop/tick.rs | 6 +- nodedb/src/data/executor/dispatch/meta.rs | 13 +- .../data/executor/handlers/bulk_dml/update.rs | 13 +- .../handlers/bulk_dml/update_persist.rs | 67 +-- .../src/data/executor/handlers/control/mod.rs | 1 - .../data/executor/handlers/control/reindex.rs | 386 -------------- .../executor/handlers/control/reindex/csr.rs | 147 ++++++ .../handlers/control/reindex/dispatch.rs | 182 +++++++ .../executor/handlers/control/reindex/fts.rs | 117 +++++ .../executor/handlers/control/reindex/hold.rs | 45 ++ .../executor/handlers/control/reindex/mod.rs | 20 + .../handlers/control/reindex/pending.rs | 146 +++++ .../handlers/control/reindex/waiter.rs | 208 ++++++++ .../handlers/control/reindex_apply.rs | 215 -------- nodedb/src/data/executor/handlers/convert.rs | 157 +++++- .../src/data/executor/handlers/point/mod.rs | 1 + .../executor/handlers/point/update/persist.rs | 24 +- .../handlers/point/update_reindex_text.rs | 60 +++ .../executor/handlers/snapshot/restore/mod.rs | 4 +- .../snapshot/restore/tenant_snapshot.rs | 9 + .../handlers/snapshot/restore/text.rs | 244 +++++++++ .../handlers/update_from_join_write.rs | 39 +- .../vector_checkpoint/build_completions.rs | 1 + nodedb/src/diag/context/index_rebuild.rs | 56 ++ nodedb/src/diag/context/mod.rs | 2 + nodedb/src/diag/mod.rs | 4 +- nodedb/src/diag/recording/index_rebuild.rs | 40 ++ nodedb/src/diag/recording/mod.rs | 2 + nodedb/src/engine/sparse/inverted/core.rs | 7 + .../src/engine/sparse/inverted/doc_image.rs | 9 +- nodedb/src/engine/sparse/inverted/indexing.rs | 1 + nodedb/src/engine/sparse/inverted/mod.rs | 6 + .../engine/sparse/inverted/rebuild_install.rs | 399 ++++++++++++++ .../engine/sparse/inverted/rebuild_journal.rs | 248 +++++++++ .../sparse/inverted/rebuild_snapshot.rs | 195 +++++++ nodedb/src/engine/sparse/inverted/removal.rs | 1 + .../tests/inproc/cases/fts_update_reindex.rs | 213 ++++++++ nodedb/tests/inproc/cases/mod.rs | 2 + .../inproc/cases/reindex_concurrent_writes.rs | 497 ++++++++++++++++++ .../inproc/cases/reindex_vector_concurrent.rs | 45 +- 56 files changed, 4042 insertions(+), 774 deletions(-) create mode 100644 nodedb-graph/src/csr/rebuild/install.rs create mode 100644 nodedb-graph/src/csr/rebuild/journal.rs create mode 100644 nodedb-graph/src/csr/rebuild/mod.rs create mode 100644 nodedb-graph/src/csr/rebuild/seed.rs delete mode 100644 nodedb/src/data/executor/handlers/control/reindex.rs create mode 100644 nodedb/src/data/executor/handlers/control/reindex/csr.rs create mode 100644 nodedb/src/data/executor/handlers/control/reindex/dispatch.rs create mode 100644 nodedb/src/data/executor/handlers/control/reindex/fts.rs create mode 100644 nodedb/src/data/executor/handlers/control/reindex/hold.rs create mode 100644 nodedb/src/data/executor/handlers/control/reindex/mod.rs create mode 100644 nodedb/src/data/executor/handlers/control/reindex/pending.rs create mode 100644 nodedb/src/data/executor/handlers/control/reindex/waiter.rs delete mode 100644 nodedb/src/data/executor/handlers/control/reindex_apply.rs create mode 100644 nodedb/src/data/executor/handlers/point/update_reindex_text.rs create mode 100644 nodedb/src/data/executor/handlers/snapshot/restore/text.rs create mode 100644 nodedb/src/diag/context/index_rebuild.rs create mode 100644 nodedb/src/diag/recording/index_rebuild.rs create mode 100644 nodedb/src/engine/sparse/inverted/rebuild_install.rs create mode 100644 nodedb/src/engine/sparse/inverted/rebuild_journal.rs create mode 100644 nodedb/src/engine/sparse/inverted/rebuild_snapshot.rs create mode 100644 nodedb/tests/inproc/cases/fts_update_reindex.rs create mode 100644 nodedb/tests/inproc/cases/reindex_concurrent_writes.rs diff --git a/nodedb-graph/src/csr/index/interning.rs b/nodedb-graph/src/csr/index/interning.rs index df9f702fb..48f76b829 100644 --- a/nodedb-graph/src/csr/index/interning.rs +++ b/nodedb-graph/src/csr/index/interning.rs @@ -5,6 +5,7 @@ use std::collections::hash_map::Entry; use super::types::CsrIndex; +use crate::csr::rebuild::journal::{CsrWriteOp, OpOutcome}; impl CsrIndex { /// Get or create a dense ID for a node. @@ -114,6 +115,18 @@ impl CsrIndex { /// is a no-op — the zero sentinel is the initial state and has no meaning. pub fn set_node_surrogate(&mut self, node: &str, surrogate: nodedb_types::Surrogate) { let raw = surrogate.as_u32(); + self.apply_set_node_surrogate(node, raw); + self.journal_record( + || CsrWriteOp::SetNodeSurrogate { + node: node.to_string(), + surrogate: raw, + }, + OpOutcome::Applied, + ); + } + + /// The surrogate bind itself, unjournaled. `raw == 0` is a no-op. + pub(crate) fn apply_set_node_surrogate(&mut self, node: &str, raw: u32) { if raw == 0 { return; } @@ -215,6 +228,23 @@ impl CsrIndex { /// label is silently ignored). Returns `Err(GraphError::NodeOverflow)` if /// the node is new and the partition's node-id space is exhausted. pub fn add_node_label(&mut self, node: &str, label: &str) -> Result { + let result = self.apply_add_node_label(node, label); + self.journal_record( + || CsrWriteOp::AddNodeLabel { + node: node.to_string(), + label: label.to_string(), + }, + OpOutcome::of_label(&result), + ); + result + } + + /// The label add itself, unjournaled. + pub(crate) fn apply_add_node_label( + &mut self, + node: &str, + label: &str, + ) -> Result { let node_id = self.ensure_node(node)?; let Some(label_id) = self.ensure_node_label(label) else { return Ok(false); @@ -225,6 +255,18 @@ impl CsrIndex { /// Remove a label from a node. pub fn remove_node_label(&mut self, node: &str, label: &str) { + self.apply_remove_node_label(node, label); + self.journal_record( + || CsrWriteOp::RemoveNodeLabel { + node: node.to_string(), + label: label.to_string(), + }, + OpOutcome::Applied, + ); + } + + /// The label removal itself, unjournaled. + pub(crate) fn apply_remove_node_label(&mut self, node: &str, label: &str) { let Some(&node_id) = self.node_to_id.get(node) else { return; }; diff --git a/nodedb-graph/src/csr/index/lookup.rs b/nodedb-graph/src/csr/index/lookup.rs index f39af8171..80cf8de13 100644 --- a/nodedb-graph/src/csr/index/lookup.rs +++ b/nodedb-graph/src/csr/index/lookup.rs @@ -14,6 +14,7 @@ use nodedb_mem::ScopedMemory; use super::types::{CsrIndex, Direction}; use crate::GraphError; use crate::csr::LocalNodeId; +use crate::csr::rebuild::journal::{CsrWriteOp, OpOutcome}; /// Contiguous CSR adjacency arrays produced by [`CsrIndex::build_dense`]. pub(crate) struct DenseAdjacency { @@ -127,8 +128,14 @@ impl CsrIndex { /// Returns `Err(GraphError::NodeOverflow)` when the partition's node-id /// space is exhausted (more than `MAX_NODES_PER_CSR` distinct nodes). pub fn add_node(&mut self, name: &str) -> Result { - let raw = self.ensure_node(name)?; - Ok(LocalNodeId::new(raw, self.partition_tag)) + let result = self.ensure_node(name); + self.journal_record( + || CsrWriteOp::AddNode { + name: name.to_string(), + }, + OpOutcome::of(&result), + ); + Ok(LocalNodeId::new(result?, self.partition_tag)) } pub fn node_count(&self) -> usize { diff --git a/nodedb-graph/src/csr/index/mutation.rs b/nodedb-graph/src/csr/index/mutation.rs index 5b2b3853c..6a1599036 100644 --- a/nodedb-graph/src/csr/index/mutation.rs +++ b/nodedb-graph/src/csr/index/mutation.rs @@ -3,6 +3,7 @@ //! Edge insert / remove paths and node-edge cleanup. use super::types::CsrIndex; +use crate::csr::rebuild::journal::{CsrWriteOp, OpOutcome}; impl CsrIndex { /// Incrementally add an unweighted edge (goes into mutable buffer). @@ -69,6 +70,31 @@ impl CsrIndex { collection: &str, weight: f64, force_weights: bool, + ) -> Result<(), crate::GraphError> { + let result = self.apply_add_edge(src, label, dst, collection, weight, force_weights); + self.journal_record( + || CsrWriteOp::AddEdge { + src: src.to_string(), + label: label.to_string(), + dst: dst.to_string(), + collection: collection.to_string(), + weight, + force_weights, + }, + OpOutcome::of(&result), + ); + result + } + + /// The edge insert itself, unjournaled. + pub(crate) fn apply_add_edge( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, + weight: f64, + force_weights: bool, ) -> Result<(), crate::GraphError> { let src_id = self.ensure_node(src)?; let dst_id = self.ensure_node(dst)?; @@ -136,6 +162,26 @@ impl CsrIndex { label: &str, dst: &str, collection: &str, + ) { + self.apply_remove_edge(src, label, dst, collection); + self.journal_record( + || CsrWriteOp::RemoveEdge { + src: src.to_string(), + label: label.to_string(), + dst: dst.to_string(), + collection: collection.to_string(), + }, + OpOutcome::Applied, + ); + } + + /// The edge removal itself, unjournaled. + pub(crate) fn apply_remove_edge( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, ) { let (Some(&src_id), Some(&dst_id)) = (self.node_to_id.get(src), self.node_to_id.get(dst)) else { @@ -185,6 +231,18 @@ impl CsrIndex { /// Remove ALL edges touching a node. Returns the number of edges removed. pub fn remove_node_edges(&mut self, node: &str) -> usize { + let removed = self.apply_remove_node_edges(node); + self.journal_record( + || CsrWriteOp::RemoveNodeEdges { + node: node.to_string(), + }, + OpOutcome::Applied, + ); + removed + } + + /// The node-edge removal itself, unjournaled. + pub(crate) fn apply_remove_node_edges(&mut self, node: &str) -> usize { let Some(&node_id) = self.node_to_id.get(node) else { return 0; }; diff --git a/nodedb-graph/src/csr/index/restore.rs b/nodedb-graph/src/csr/index/restore.rs index 3173a871d..7b5b0d4b6 100644 --- a/nodedb-graph/src/csr/index/restore.rs +++ b/nodedb-graph/src/csr/index/restore.rs @@ -10,6 +10,7 @@ use super::types::CsrIndex; use crate::GraphError; +use crate::csr::rebuild::journal::{CsrWriteOp, OpOutcome}; impl CsrIndex { /// Weight of the live `(src, label, dst)` edge in `collection`, `None` @@ -79,11 +80,34 @@ impl CsrIndex { dst: &str, collection: &str, weight: f64, + ) -> Result, GraphError> { + let result = self.apply_put_edge(src, label, dst, collection, weight); + self.journal_record( + || CsrWriteOp::PutEdge { + src: src.to_string(), + label: label.to_string(), + dst: dst.to_string(), + collection: collection.to_string(), + weight, + }, + OpOutcome::of(&result), + ); + result + } + + /// The edge put itself, unjournaled. + pub(crate) fn apply_put_edge( + &mut self, + src: &str, + label: &str, + dst: &str, + collection: &str, + weight: f64, ) -> Result, GraphError> { let prior = self.edge_weight_in_collection(src, label, dst, collection); match prior { Some(current) if current == weight => return Ok(prior), - Some(_) => self.remove_edge_in_collection(src, label, dst, collection), + Some(_) => self.apply_remove_edge(src, label, dst, collection), None => {} } let src_id = self.ensure_node(src)?; @@ -133,6 +157,18 @@ impl CsrIndex { /// Put `node`'s surrogate back to `prior`, `0` for none. pub fn restore_node_surrogate(&mut self, node: &str, prior: u32) { + self.apply_restore_node_surrogate(node, prior); + self.journal_record( + || CsrWriteOp::RestoreNodeSurrogate { + node: node.to_string(), + prior, + }, + OpOutcome::Applied, + ); + } + + /// The surrogate restore itself, unjournaled. + pub(crate) fn apply_restore_node_surrogate(&mut self, node: &str, prior: u32) { let Some(&id) = self.node_to_id.get(node) else { return; }; @@ -163,6 +199,18 @@ impl CsrIndex { /// edge and no label: withdrawing any other would renumber or orphan live /// state, so that is refused. pub fn withdraw_newest_node(&mut self, node: &str) -> Result<(), GraphError> { + let result = self.apply_withdraw_newest_node(node); + self.journal_record( + || CsrWriteOp::WithdrawNewestNode { + node: node.to_string(), + }, + OpOutcome::of(&result), + ); + result + } + + /// The node withdraw itself, unjournaled. + pub(crate) fn apply_withdraw_newest_node(&mut self, node: &str) -> Result<(), GraphError> { let Some(&id) = self.node_to_id.get(node) else { return Ok(()); }; @@ -214,6 +262,21 @@ impl CsrIndex { /// An absent label is a no-op. The label must be the newest one and no /// node may carry it, or the withdraw is refused. pub fn withdraw_newest_node_label(&mut self, label: &str) -> Result<(), GraphError> { + let result = self.apply_withdraw_newest_node_label(label); + self.journal_record( + || CsrWriteOp::WithdrawNewestNodeLabel { + label: label.to_string(), + }, + OpOutcome::of(&result), + ); + result + } + + /// The node-label withdraw itself, unjournaled. + pub(crate) fn apply_withdraw_newest_node_label( + &mut self, + label: &str, + ) -> Result<(), GraphError> { let Some(&id) = self.node_label_to_id.get(label) else { return Ok(()); }; diff --git a/nodedb-graph/src/csr/index/types.rs b/nodedb-graph/src/csr/index/types.rs index 4d99e3fa2..fb6303f11 100644 --- a/nodedb-graph/src/csr/index/types.rs +++ b/nodedb-graph/src/csr/index/types.rs @@ -147,6 +147,10 @@ pub struct CsrIndex { /// reserve bytes against the bound database, tenant, and `EngineId::Graph` /// before allocating and release them on drop via `ReservationToken`. pub(crate) memory: ScopedMemory, + + /// Mutations recorded while a rebuild of this index runs. `None` when + /// no rebuild runs. See [`crate::csr::rebuild`]. + pub(crate) rebuild_journal: Option, } impl CsrIndex { @@ -191,6 +195,7 @@ impl CsrIndex { query_epoch: 0, partition_tag: crate::csr::local_node_id::next_partition_tag(), memory, + rebuild_journal: None, } } diff --git a/nodedb-graph/src/csr/mod.rs b/nodedb-graph/src/csr/mod.rs index ceccab3ac..bf9aa654b 100644 --- a/nodedb-graph/src/csr/mod.rs +++ b/nodedb-graph/src/csr/mod.rs @@ -6,6 +6,7 @@ pub mod index; pub mod local_node_id; pub mod memory; pub mod persist; +pub mod rebuild; pub mod slice_accessors; pub mod statistics; pub mod weights; diff --git a/nodedb-graph/src/csr/persist.rs b/nodedb-graph/src/csr/persist.rs index 896bbd3ec..7c39b345a 100644 --- a/nodedb-graph/src/csr/persist.rs +++ b/nodedb-graph/src/csr/persist.rs @@ -291,6 +291,7 @@ impl CsrIndex { query_epoch: 0, partition_tag: crate::csr::local_node_id::next_partition_tag(), memory, + rebuild_journal: None, }) } @@ -394,6 +395,7 @@ impl CsrIndex { query_epoch: 0, partition_tag: crate::csr::local_node_id::next_partition_tag(), memory, + rebuild_journal: None, } } } diff --git a/nodedb-graph/src/csr/rebuild/install.rs b/nodedb-graph/src/csr/rebuild/install.rs new file mode 100644 index 000000000..740034bc0 --- /dev/null +++ b/nodedb-graph/src/csr/rebuild/install.rs @@ -0,0 +1,185 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Finishing a rebuild on the owning thread: restore the compacted copy +//! and replay the journal onto it. + +use nodedb_mem::ScopedMemory; + +use super::seed::{CsrRebuilt, NodeLabelState, restore_checkpoint}; +use crate::GraphError; +use crate::csr::index::CsrIndex; + +/// Most node labels one index interns. The label bitset is a `u64`. +const MAX_NODE_LABELS: usize = 64; + +impl CsrIndex { + /// Close the journal of `rebuilt` and return the copy that replaces + /// this index. + /// + /// The copy holds the snapshot, compacted, plus every mutation this + /// index took since `begin_rebuild`, in order. It keeps this index's + /// partition tag, so node ids handed out before the swap stay valid. + /// The caller installs it in one step. On error the journal is closed, + /// this index stays as it is, and the copy is dropped: + /// + /// - [`GraphError::RebuildSuperseded`]: no journal of this rebuild is open. + /// - [`GraphError::RebuildJournalOverflow`]: the journal hit its bound. + /// - [`GraphError::RebuildReplayDiverged`]: a replayed mutation returned + /// another outcome than it did here. + /// - [`GraphError::RebuildSnapshotInvalid`]: the copy does not decode. + pub fn finish_rebuild( + &mut self, + rebuilt: CsrRebuilt, + memory: ScopedMemory, + ) -> Result { + let journal = match self.rebuild_journal.take() { + Some(journal) if journal.token == rebuilt.token => journal, + other => { + self.rebuild_journal = other; + return Err(GraphError::RebuildSuperseded); + } + }; + let ops = journal.into_ops()?; + let mut copy = restore_checkpoint(&rebuilt.checkpoint, memory)?; + copy.install_node_labels(rebuilt.node_labels)?; + for (op, live_outcome) in &ops { + if copy.replay_op(op) != *live_outcome { + return Err(GraphError::RebuildReplayDiverged { op: op.kind() }); + } + } + copy.partition_tag = self.partition_tag; + Ok(copy) + } + + /// Put the node labels a rebuild carried onto a freshly restored copy. + fn install_node_labels(&mut self, labels: NodeLabelState) -> Result<(), GraphError> { + if labels.bits.len() != self.id_to_node.len() { + return Err(GraphError::RebuildSnapshotInvalid { + detail: format!( + "node label bitsets cover {} nodes, the snapshot holds {}", + labels.bits.len(), + self.id_to_node.len() + ), + }); + } + if labels.names.len() > MAX_NODE_LABELS { + return Err(GraphError::RebuildSnapshotInvalid { + detail: format!( + "{} node labels exceed the {MAX_NODE_LABELS}-label bitset", + labels.names.len() + ), + }); + } + self.node_label_to_id = labels + .names + .iter() + .enumerate() + .map(|(id, name)| (name.clone(), id as u8)) + .collect(); + self.node_label_names = labels.names; + self.node_label_bits = labels.bits; + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::Surrogate; + + use crate::GraphError; + use crate::csr::index::{CsrIndex, Direction}; + use crate::test_support::test_memory; + + const JOURNAL_BYTES: usize = 1 << 20; + + fn seeded() -> CsrIndex { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge_in_collection("a", "L", "b", "c").unwrap(); + csr.add_edge_in_collection("b", "L", "c", "c").unwrap(); + csr.add_node_label("a", "Person").unwrap(); + csr.set_node_surrogate("a", Surrogate::new(7)); + csr + } + + fn out_of(csr: &CsrIndex, node: &str) -> Vec { + let mut dsts: Vec = csr + .neighbors(node, None, Direction::Out) + .into_iter() + .map(|(_, d)| d) + .collect(); + dsts.sort(); + dsts + } + + #[test] + fn writes_during_the_build_reach_the_installed_copy() { + let mut live = seeded(); + let seed = live.begin_rebuild(JOURNAL_BYTES).unwrap(); + + // Writes that land after the snapshot. + live.add_edge_in_collection("a", "L", "d", "c").unwrap(); + live.remove_edge_in_collection("b", "L", "c", "c"); + live.put_edge_in_collection("x", "L", "y", "c", 2.5) + .unwrap(); + live.add_node_label("x", "Person").unwrap(); + live.set_node_surrogate("x", Surrogate::new(9)); + + let rebuilt = seed.build(test_memory()).unwrap(); + let tag_before = live.partition_tag; + let copy = live.finish_rebuild(rebuilt, test_memory()).unwrap(); + + assert_eq!(out_of(©, "a"), vec!["b".to_string(), "d".to_string()]); + assert!( + out_of(©, "b").is_empty(), + "a delete during the build carries over" + ); + assert_eq!( + copy.edge_weight_in_collection("x", "L", "y", "c"), + Some(2.5) + ); + let x = copy.node_id_raw("x").unwrap(); + assert!(copy.node_has_label(x, "Person")); + let a = copy.node_id_raw("a").unwrap(); + assert!(copy.node_has_label(a, "Person"), "snapshot labels survive"); + assert_eq!(copy.node_id_for_surrogate(Surrogate::new(9)), Some("x")); + assert_eq!(copy.partition_tag, tag_before); + assert!(!live.rebuild_in_progress(), "finishing closes the journal"); + } + + #[test] + fn a_result_from_another_rebuild_is_refused() { + let mut live = seeded(); + let seed = live.begin_rebuild(JOURNAL_BYTES).unwrap(); + live.abort_rebuild(seed.token()); + let _second = live.begin_rebuild(JOURNAL_BYTES).unwrap(); + let rebuilt = seed.build(test_memory()).unwrap(); + assert!(matches!( + live.finish_rebuild(rebuilt, test_memory()), + Err(GraphError::RebuildSuperseded) + )); + assert!(live.rebuild_in_progress(), "the newer journal stays open"); + } + + #[test] + fn a_second_rebuild_is_refused_while_one_runs() { + let mut live = seeded(); + let _seed = live.begin_rebuild(JOURNAL_BYTES).unwrap(); + assert!(matches!( + live.begin_rebuild(JOURNAL_BYTES), + Err(GraphError::RebuildInProgress) + )); + } + + #[test] + fn journal_overflow_refuses_the_copy_and_keeps_the_live_writes() { + let mut live = seeded(); + let seed = live.begin_rebuild(1).unwrap(); + live.add_edge_in_collection("a", "L", "z", "c").unwrap(); + let rebuilt = seed.build(test_memory()).unwrap(); + assert!(matches!( + live.finish_rebuild(rebuilt, test_memory()), + Err(GraphError::RebuildJournalOverflow { cap_bytes: 1 }) + )); + assert!(out_of(&live, "a").contains(&"z".to_string())); + } +} diff --git a/nodedb-graph/src/csr/rebuild/journal.rs b/nodedb-graph/src/csr/rebuild/journal.rs new file mode 100644 index 000000000..86edf9052 --- /dev/null +++ b/nodedb-graph/src/csr/rebuild/journal.rs @@ -0,0 +1,301 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The write journal a `CsrIndex` keeps while a rebuild of it runs. +//! +//! Every public mutation records itself here with the outcome it had on +//! the live index. At cutover the journal replays onto the rebuilt copy, +//! so the copy holds every write the live index took after the snapshot. +//! +//! The journal is bounded in bytes. Past the bound it drops its entries, +//! frees their memory and marks itself overflowed. The cutover then +//! refuses the rebuilt copy with [`GraphError::RebuildJournalOverflow`]. +//! The live index keeps every write either way. + +use std::mem::size_of; + +use crate::GraphError; +use crate::csr::index::CsrIndex; + +/// One mutation of the live index, recorded for replay. +#[derive(Debug, Clone, PartialEq)] +pub(crate) enum CsrWriteOp { + AddEdge { + src: String, + label: String, + dst: String, + collection: String, + weight: f64, + force_weights: bool, + }, + PutEdge { + src: String, + label: String, + dst: String, + collection: String, + weight: f64, + }, + RemoveEdge { + src: String, + label: String, + dst: String, + collection: String, + }, + RemoveNodeEdges { + node: String, + }, + AddNode { + name: String, + }, + SetNodeSurrogate { + node: String, + surrogate: u32, + }, + RestoreNodeSurrogate { + node: String, + prior: u32, + }, + AddNodeLabel { + node: String, + label: String, + }, + RemoveNodeLabel { + node: String, + label: String, + }, + WithdrawNewestNode { + node: String, + }, + WithdrawNewestNodeLabel { + label: String, + }, +} + +impl CsrWriteOp { + /// Name of the operation, for errors. + pub(crate) fn kind(&self) -> &'static str { + match self { + Self::AddEdge { .. } => "add_edge", + Self::PutEdge { .. } => "put_edge", + Self::RemoveEdge { .. } => "remove_edge", + Self::RemoveNodeEdges { .. } => "remove_node_edges", + Self::AddNode { .. } => "add_node", + Self::SetNodeSurrogate { .. } => "set_node_surrogate", + Self::RestoreNodeSurrogate { .. } => "restore_node_surrogate", + Self::AddNodeLabel { .. } => "add_node_label", + Self::RemoveNodeLabel { .. } => "remove_node_label", + Self::WithdrawNewestNode { .. } => "withdraw_newest_node", + Self::WithdrawNewestNodeLabel { .. } => "withdraw_newest_node_label", + } + } + + /// Bytes the entry holds: the enum itself plus its string contents. + fn byte_cost(&self) -> usize { + let strings = match self { + Self::AddEdge { + src, + label, + dst, + collection, + .. + } + | Self::PutEdge { + src, + label, + dst, + collection, + .. + } + | Self::RemoveEdge { + src, + label, + dst, + collection, + } => src.len() + label.len() + dst.len() + collection.len(), + Self::RemoveNodeEdges { node } + | Self::SetNodeSurrogate { node, .. } + | Self::RestoreNodeSurrogate { node, .. } + | Self::WithdrawNewestNode { node } => node.len(), + Self::AddNode { name } => name.len(), + Self::AddNodeLabel { node, label } | Self::RemoveNodeLabel { node, label } => { + node.len() + label.len() + } + Self::WithdrawNewestNodeLabel { label } => label.len(), + }; + size_of::<(Self, OpOutcome)>() + strings + } +} + +/// What a mutation returned on the index it ran against. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum OpOutcome { + /// The call returned `Ok`, or returns nothing. + Applied, + /// `add_node_label` returned `Ok(false)`: the label limit ignored it. + Ignored, + /// The call returned an error. + Failed, +} + +impl OpOutcome { + pub(crate) fn of(result: &Result) -> Self { + if result.is_ok() { + Self::Applied + } else { + Self::Failed + } + } + + pub(crate) fn of_label(result: &Result) -> Self { + match result { + Ok(true) => Self::Applied, + Ok(false) => Self::Ignored, + Err(_) => Self::Failed, + } + } +} + +/// Mutations recorded since a rebuild's snapshot, oldest first. +#[derive(Debug)] +pub struct CsrJournal { + pub(crate) token: u64, + cap_bytes: usize, + used_bytes: usize, + ops: Vec<(CsrWriteOp, OpOutcome)>, + overflowed: bool, +} + +impl CsrJournal { + pub(crate) fn new(token: u64, cap_bytes: usize) -> Self { + Self { + token, + cap_bytes, + used_bytes: 0, + ops: Vec::new(), + overflowed: false, + } + } + + fn record(&mut self, op: CsrWriteOp, outcome: OpOutcome) { + if self.overflowed { + return; + } + let cost = op.byte_cost(); + if self.used_bytes.saturating_add(cost) > self.cap_bytes { + self.overflowed = true; + self.ops = Vec::new(); + self.used_bytes = 0; + return; + } + self.used_bytes += cost; + self.ops.push((op, outcome)); + } + + /// The recorded entries, or the overflow error when the bound was hit. + pub(crate) fn into_ops(self) -> Result, GraphError> { + if self.overflowed { + return Err(GraphError::RebuildJournalOverflow { + cap_bytes: self.cap_bytes, + }); + } + Ok(self.ops) + } +} + +impl CsrIndex { + /// Record a mutation when a rebuild journal is open. `op` runs only + /// then, so an index with no rebuild allocates nothing here. + pub(crate) fn journal_record(&mut self, op: impl FnOnce() -> CsrWriteOp, outcome: OpOutcome) { + if let Some(journal) = self.rebuild_journal.as_mut() { + journal.record(op(), outcome); + } + } + + /// Run one recorded mutation against this index and return its outcome. + pub(crate) fn replay_op(&mut self, op: &CsrWriteOp) -> OpOutcome { + match op { + CsrWriteOp::AddEdge { + src, + label, + dst, + collection, + weight, + force_weights, + } => OpOutcome::of(&self.apply_add_edge( + src, + label, + dst, + collection, + *weight, + *force_weights, + )), + CsrWriteOp::PutEdge { + src, + label, + dst, + collection, + weight, + } => OpOutcome::of(&self.apply_put_edge(src, label, dst, collection, *weight)), + CsrWriteOp::RemoveEdge { + src, + label, + dst, + collection, + } => { + self.apply_remove_edge(src, label, dst, collection); + OpOutcome::Applied + } + CsrWriteOp::RemoveNodeEdges { node } => { + self.apply_remove_node_edges(node); + OpOutcome::Applied + } + CsrWriteOp::AddNode { name } => OpOutcome::of(&self.ensure_node(name)), + CsrWriteOp::SetNodeSurrogate { node, surrogate } => { + self.apply_set_node_surrogate(node, *surrogate); + OpOutcome::Applied + } + CsrWriteOp::RestoreNodeSurrogate { node, prior } => { + self.apply_restore_node_surrogate(node, *prior); + OpOutcome::Applied + } + CsrWriteOp::AddNodeLabel { node, label } => { + OpOutcome::of_label(&self.apply_add_node_label(node, label)) + } + CsrWriteOp::RemoveNodeLabel { node, label } => { + self.apply_remove_node_label(node, label); + OpOutcome::Applied + } + CsrWriteOp::WithdrawNewestNode { node } => { + OpOutcome::of(&self.apply_withdraw_newest_node(node)) + } + CsrWriteOp::WithdrawNewestNodeLabel { label } => { + OpOutcome::of(&self.apply_withdraw_newest_node_label(label)) + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn overflow_drops_entries_and_reports_the_bound() { + let op = CsrWriteOp::AddNode { + name: "n".to_string(), + }; + let cap = op.byte_cost() * 2; + let mut journal = CsrJournal::new(1, cap); + journal.record(op.clone(), OpOutcome::Applied); + journal.record(op.clone(), OpOutcome::Applied); + assert_eq!(journal.ops.len(), 2); + journal.record(op, OpOutcome::Applied); + assert!( + journal.ops.is_empty(), + "an overflowed journal frees its entries" + ); + assert!(matches!( + journal.into_ops(), + Err(GraphError::RebuildJournalOverflow { cap_bytes }) if cap_bytes == cap + )); + } +} diff --git a/nodedb-graph/src/csr/rebuild/mod.rs b/nodedb-graph/src/csr/rebuild/mod.rs new file mode 100644 index 000000000..c707e3ea8 --- /dev/null +++ b/nodedb-graph/src/csr/rebuild/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Rebuild of a live `CsrIndex` off its owning thread, with no lost writes. +//! +//! - `begin_rebuild` snapshots the index and opens a write journal. +//! - `CsrRebuildSeed::build` compacts the snapshot on any thread. +//! - `finish_rebuild` restores the compacted copy, replays the journal onto +//! it and returns it for the caller to install in one step. + +pub mod install; +pub mod journal; +pub mod seed; + +pub use journal::CsrJournal; +pub use seed::{CsrRebuildSeed, CsrRebuilt}; diff --git a/nodedb-graph/src/csr/rebuild/seed.rs b/nodedb-graph/src/csr/rebuild/seed.rs new file mode 100644 index 000000000..f8bc28129 --- /dev/null +++ b/nodedb-graph/src/csr/rebuild/seed.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Starting a rebuild on the owning thread, and the build that runs off it. + +use std::sync::atomic::{AtomicU64, Ordering}; + +use nodedb_mem::ScopedMemory; + +use super::journal::CsrJournal; +use crate::GraphError; +use crate::csr::index::CsrIndex; + +static NEXT_REBUILD_TOKEN: AtomicU64 = AtomicU64::new(1); + +/// Node labels as the index holds them. The checkpoint format omits them, +/// so a rebuild carries them beside it. +#[derive(Debug, Clone)] +pub(crate) struct NodeLabelState { + /// Label names in id order. + pub(crate) names: Vec, + /// Label bitset per node id. + pub(crate) bits: Vec, +} + +/// A rebuild's input, taken on the owning thread. Every field is `Send`. +#[derive(Debug)] +pub struct CsrRebuildSeed { + token: u64, + checkpoint: Vec, + node_labels: NodeLabelState, +} + +/// A compacted copy on its way back to the owning thread. Every field is +/// `Send`. +#[derive(Debug)] +pub struct CsrRebuilt { + pub(crate) token: u64, + pub(crate) checkpoint: Vec, + pub(crate) node_labels: NodeLabelState, +} + +impl CsrRebuildSeed { + /// The token that ties this rebuild to the journal on the live index. + pub fn token(&self) -> u64 { + self.token + } + + /// Restore the snapshot, compact it and serialize the result. Runs on + /// any thread. `memory` bounds the copy's allocations. + pub fn build(self, memory: ScopedMemory) -> Result { + let mut copy = restore_checkpoint(&self.checkpoint, memory)?; + copy.compact()?; + let checkpoint = copy.checkpoint_to_bytes()?; + Ok(CsrRebuilt { + token: self.token, + checkpoint, + node_labels: self.node_labels, + }) + } +} + +impl CsrRebuilt { + /// The token of the rebuild that produced this copy. + pub fn token(&self) -> u64 { + self.token + } +} + +impl CsrIndex { + /// Snapshot this index and open its write journal. + /// + /// Every later mutation records itself until `finish_rebuild` or + /// `abort_rebuild` closes the journal. `max_journal_bytes` bounds the + /// journal. Returns [`GraphError::RebuildInProgress`] when a journal is + /// already open. + pub fn begin_rebuild( + &mut self, + max_journal_bytes: usize, + ) -> Result { + if self.rebuild_journal.is_some() { + return Err(GraphError::RebuildInProgress); + } + let checkpoint = self.checkpoint_to_bytes()?; + let token = NEXT_REBUILD_TOKEN.fetch_add(1, Ordering::Relaxed); + self.rebuild_journal = Some(CsrJournal::new(token, max_journal_bytes)); + Ok(CsrRebuildSeed { + token, + checkpoint, + node_labels: NodeLabelState { + names: self.node_label_names.clone(), + bits: self.node_label_bits.clone(), + }, + }) + } + + /// Whether a rebuild journal is open on this index. + pub fn rebuild_in_progress(&self) -> bool { + self.rebuild_journal.is_some() + } + + /// Close the journal of rebuild `token`. The index itself is unchanged. + /// A journal of another rebuild stays open. + pub fn abort_rebuild(&mut self, token: u64) { + if self + .rebuild_journal + .as_ref() + .is_some_and(|journal| journal.token == token) + { + self.rebuild_journal = None; + } + } +} + +/// Decode a checkpoint a rebuild produced. +pub(crate) fn restore_checkpoint( + bytes: &[u8], + memory: ScopedMemory, +) -> Result { + CsrIndex::from_checkpoint(bytes, memory) + .map_err(|e| GraphError::RebuildSnapshotInvalid { + detail: e.to_string(), + })? + .ok_or_else(|| GraphError::RebuildSnapshotInvalid { + detail: "checkpoint bytes carry no CSR header".to_string(), + }) +} diff --git a/nodedb-graph/src/error.rs b/nodedb-graph/src/error.rs index d37f9b1f6..962cea9d7 100644 --- a/nodedb-graph/src/error.rs +++ b/nodedb-graph/src/error.rs @@ -61,4 +61,33 @@ pub enum GraphError { /// would renumber or orphan live state. #[error("cannot withdraw {kind} '{name}': it is not the newest {kind} or it is still in use")] WithdrawRefused { kind: &'static str, name: String }, + + /// A rebuild of this index is already running. The partition holds one + /// write journal at a time. + #[error("a rebuild of this CSR index is already running")] + RebuildInProgress, + + /// The index a rebuild started from is no longer the live one: it was + /// dropped, replaced, or its rebuild was aborted. The rebuilt copy is + /// discarded and the live index stays as it is. + #[error("the CSR index this rebuild started from is no longer live; rebuild discarded")] + RebuildSuperseded, + + /// The writes made during a rebuild exceeded the journal bound. The + /// rebuilt copy is discarded, the live index keeps every write, and a + /// new rebuild can run. + #[error( + "CSR writes during the rebuild exceeded the {cap_bytes}-byte journal bound; \ + rebuild discarded, live index unchanged; run REINDEX again" + )] + RebuildJournalOverflow { cap_bytes: usize }, + + /// A journaled write returned a different outcome on the rebuilt index + /// than on the live one. The rebuilt copy is discarded. + #[error("CSR rebuild replay of '{op}' diverged from the live index; rebuild discarded")] + RebuildReplayDiverged { op: &'static str }, + + /// The snapshot a rebuild carries does not decode into an index. + #[error("CSR rebuild snapshot is invalid: {detail}")] + RebuildSnapshotInvalid { detail: String }, } diff --git a/nodedb-physical/src/physical_plan/meta.rs b/nodedb-physical/src/physical_plan/meta.rs index da6a48196..1ef3f802c 100644 --- a/nodedb-physical/src/physical_plan/meta.rs +++ b/nodedb-physical/src/physical_plan/meta.rs @@ -396,22 +396,23 @@ pub enum MetaOp { is_group_leader: bool, }, - /// Rebuild all indexes (HNSW, FTS LSM, graph CSR) for a collection - /// on this core in a shadow-build + atomic-swap manner. - /// - /// When `concurrent = true`, the build proceeds without blocking query - /// handling: a background OS thread performs the rebuild and the Data - /// Plane polls for completion on subsequent ticks, only swapping the - /// live index in at cutover. When `concurrent = false` the rebuild - /// runs inline (same semantics as the legacy Checkpoint path). - /// - /// `index_name` narrows the rebuild to a single named index when set; - /// `None` rebuilds all index types for the collection. - /// - /// Returns `Response::Ok` on successful cutover, or a typed error if: - /// - another rebuild is already in progress for this collection - /// (`ErrorCode::Conflict`), or - /// - the shadow build fails (`ErrorCode::Internal`). + /// Rebuild a collection's indexes (HNSW, full-text, graph CSR) on + /// this core. + /// + /// `index_name` narrows the rebuild to one kind: `hnsw`, `fts` or + /// `csr`. `None` rebuilds every kind the collection has. + /// + /// Each rebuild runs off the core while the core keeps serving reads + /// and writes. Writes made during the build are replayed onto the + /// rebuilt index at cutover, and the core swaps it in on a later tick. + /// With `concurrent = true` the core answers once the rebuilds have + /// started. With `concurrent = false` it answers once they have cut + /// over, or `DeadlineExceeded` at the request deadline; the rebuilds + /// still complete. + /// + /// Errors when a rebuild of the collection already runs + /// (`ObjectNotInPrerequisiteState`), when a rebuild cannot start, or, + /// with `concurrent = false`, when a rebuild is discarded. RebuildIndex { collection: QualifiedCollection, index_name: Option, diff --git a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs index 66fb8f94d..2077ef341 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs @@ -11,11 +11,12 @@ use std::time::Duration; use sonic_rs; -use crate::bridge::envelope::PhysicalPlan; +use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::catalog_entry::persist_collection_replicated; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::pgwire::types::error_to_sqlstate; use crate::control::server::shared::ddl::sync_dispatch::{ - SystemReason, SystemTask, dispatch_system, + SystemReason, SystemTask, dispatch_system_response_with_source, }; use crate::control::state::SharedState; use nodedb_physical::physical_plan::MetaOp; @@ -96,7 +97,8 @@ pub async fn convert_collection( source_storage_mode, }); - dispatch_system( + let event_source = SystemReason::DdlApply.event_source(); + let resp = dispatch_system_response_with_source( state, SystemTask::new( SystemReason::DdlApply, @@ -106,9 +108,28 @@ pub async fn convert_collection( plan, ), Duration::from_secs(60), + event_source, ) .await - .map_err(|e| err("XX000", format!("conversion failed: {e}")))?; + .map_err(|e| { + let (_, code, message) = error_to_sqlstate(&e); + err(code, format!("conversion failed: {message}")) + })?; + + // A Data-Plane verdict (a row breaking the target schema, say) keeps its + // own typed SQLSTATE instead of collapsing to a generic internal error. + if resp.status != Status::Ok { + let verdict = match resp.error_code { + Some(code) => crate::Error::DataPlane(*code), + None => crate::Error::Internal { + detail: "conversion failed: data plane returned an error status with no error \ + code" + .into(), + }, + }; + let (_, code, message) = error_to_sqlstate(&verdict); + return Err(err(code, message)); + } // Update catalog collection type. let new_type = match target_type.as_str() { diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs index 9e6b98614..b3afd3f67 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs @@ -5,9 +5,17 @@ //! Grammar: //! REINDEX [INDEX ] [CONCURRENTLY] //! -//! Non-concurrent path: dispatches `MetaOp::Checkpoint` (existing semantics). -//! Concurrent path: dispatches `MetaOp::RebuildIndex { concurrent: true }` to -//! every core and awaits the cross-core ACK barrier before returning. +//! Both forms dispatch `MetaOp::RebuildIndex` to every core and await the +//! cross-core ACK barrier. Without `INDEX ` every index the +//! collection has is rebuilt: HNSW, full-text and CSR. +//! +//! Both forms rebuild the same way: off the core, with the core serving +//! reads and writes meanwhile and swapping each rebuilt index in on a later +//! tick. They differ only in when a core answers: +//! +//! - Non-concurrent: once its cutovers are done. Past the statement +//! deadline it answers `DeadlineExceeded`; the rebuilds still complete. +//! - Concurrent: once its rebuilds have started. //! //! The grammar is parsed once by `nodedb_sql::ddl_ast::parse` into //! `NodedbStatement::Reindex { .. }`; this handler receives the already-parsed @@ -53,39 +61,26 @@ pub async fn handle_reindex( )); } - if concurrent { - // Concurrent path: broadcast to all cores and await per-core ACK. - let plan = crate::bridge::envelope::PhysicalPlan::Meta(MetaOp::RebuildIndex { - collection: nodedb_types::QualifiedCollection::new(database_id, &collection), - index_name, - concurrent: true, - }); - let trace_id = TraceId::generate(); - crate::control::server::broadcast::broadcast_register_to_all_cores( - state, - tenant_id, - database_id, - plan, - trace_id, - ) - .await - .map_err(|e| ddl_err("XX000", format!("REINDEX CONCURRENTLY failed: {e}")))?; + // Every core rebuilds the indexes it holds for the collection. A core + // answers the plain form after its cutovers, the concurrent form after + // its rebuilds start. + let plan = crate::bridge::envelope::PhysicalPlan::Meta(MetaOp::RebuildIndex { + collection: nodedb_types::QualifiedCollection::new(database_id, &collection), + index_name, + concurrent, + }); + let trace_id = TraceId::generate(); + crate::control::server::broadcast::broadcast_register_to_all_cores( + state, + tenant_id, + database_id, + plan, + trace_id, + ) + .await + .map_err(|e| ddl_err("XX000", format!("REINDEX failed: {e}")))?; - tracing::info!( - %collection, - concurrent = true, - "REINDEX CONCURRENTLY dispatched and acknowledged by all cores" - ); - } else { - // Non-concurrent path: fire-and-forget (same as legacy Checkpoint). - super::distributed::dispatch_maintenance_to_all_cores( - state, - tenant_id, - database_id, - MetaOp::Checkpoint, - ); - tracing::info!(%collection, concurrent = false, "REINDEX dispatched"); - } + tracing::info!(%collection, concurrent, "REINDEX acknowledged by all cores"); Ok(vec![DdlResult::Status { command: "REINDEX".to_string(), diff --git a/nodedb/src/data/executor/core_loop/maintenance_state.rs b/nodedb/src/data/executor/core_loop/maintenance_state.rs index c5c151411..b46d44d7c 100644 --- a/nodedb/src/data/executor/core_loop/maintenance_state.rs +++ b/nodedb/src/data/executor/core_loop/maintenance_state.rs @@ -4,7 +4,7 @@ use std::sync::Arc; -use crate::data::executor::handlers::control::reindex::PendingReindex; +use crate::data::executor::handlers::control::reindex::{PendingReindex, ReindexWaiter}; /// Compaction pacing, the maintenance CPU budget, and in-flight index /// rebuilds. @@ -28,13 +28,17 @@ pub(in crate::data::executor) struct MaintenanceState { pub(in crate::data::executor) maintenance_budget: Option>, - /// In-flight concurrent index rebuilds, polled each tick. + /// In-flight full-text and CSR rebuilds, polled each tick. /// - /// Each entry yields a `RebuildResult` once its background OS thread - /// finishes the shadow build. Only one rebuild per collection runs at a - /// time. `execute_rebuild_index` returns `ErrorCode::Conflict` for a - /// second one. + /// Each entry yields its rebuilt index once its OS thread finishes the + /// build; the poll cuts it over on this core. One collection runs one + /// concurrent REINDEX at a time; `execute_rebuild_index` refuses a + /// second one with `ObjectNotInPrerequisiteState`. pub(in crate::data::executor) pending_reindex: Vec, + + /// Plain REINDEX requests waiting for their rebuilds to cut over. The + /// tick answers each one when they have, or at its deadline. + pub(in crate::data::executor) reindex_waiters: Vec, } impl MaintenanceState { @@ -46,6 +50,7 @@ impl MaintenanceState { segment_compaction_config: crate::storage::compaction::CompactionConfig::default(), maintenance_budget: None, pending_reindex: Vec::new(), + reindex_waiters: Vec::new(), } } } diff --git a/nodedb/src/data/executor/core_loop/tick.rs b/nodedb/src/data/executor/core_loop/tick.rs index aea4f4a43..92731cbfa 100644 --- a/nodedb/src/data/executor/core_loop/tick.rs +++ b/nodedb/src/data/executor/core_loop/tick.rs @@ -58,7 +58,10 @@ impl CoreLoop { self.io_metrics.record_wait(tier, wait_ns); // A write to a row a staged Calvin transaction owns waits for it. - if let Some(task) = self.park_if_calvin_owned(qt.task) { + // A plain REINDEX starts its rebuilds and waits for their cutovers. + if let Some(task) = self.park_if_calvin_owned(qt.task) + && let Some(task) = self.hold_plain_reindex(task) + { self.run_task(task); } self.release_resolved_calvin_owners(); @@ -164,6 +167,7 @@ impl CoreLoop { pub fn tick(&mut self) -> usize { self.poll_build_completions(); self.poll_pending_reindex(); + self.answer_reindex_waiters(); // Adjust SPSC read depth based on current memory pressure. self.apply_spsc_pressure(); self.drain_requests(); diff --git a/nodedb/src/data/executor/dispatch/meta.rs b/nodedb/src/data/executor/dispatch/meta.rs index 25ceeb0b1..b1aeeb334 100644 --- a/nodedb/src/data/executor/dispatch/meta.rs +++ b/nodedb/src/data/executor/dispatch/meta.rs @@ -218,17 +218,14 @@ impl CoreLoop { injected_reads, ), + // Both forms start the same rebuilds here. A plain REINDEX is + // held by the core loop's reindex waiter before it reaches this + // arm, and answered at cutover. MetaOp::RebuildIndex { collection, index_name, - concurrent, - } => self.execute_rebuild_index( - task, - tid, - collection.as_str(), - index_name.as_deref(), - *concurrent, - ), + concurrent: _, + } => self.execute_rebuild_index(task, tid, collection.as_str(), index_name.as_deref()), MetaOp::PutSynonymGroup { tenant_id, diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update.rs b/nodedb/src/data/executor/handlers/bulk_dml/update.rs index a86f0182c..2b64056d3 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update.rs @@ -270,15 +270,10 @@ impl CoreLoop { }, ); let (touched, target_writes) = match persisted { - // The row's own reindex failed and it is skipped, as it - // always was — the helper logged which row and why. - Ok(None) => continue, - Ok(Some(persisted)) => (persisted.touched, persisted.target_writes), - // A rejected materialized sum is NOT a skippable row: - // skipping it would report a smaller affected count as - // the truth while the rest of the predicate's matches - // were rewritten, and leave the stored total short of the - // `SUM(...)` over the rows that did land. The row's own + Ok(persisted) => (persisted.touched, persisted.target_writes), + // A row that fails is NOT skipped: skipping it would report + // a smaller affected count as the truth while the rest of + // the predicate's matches were rewritten. The row's own // transaction did not commit, but earlier rows did. Err(e) if affected > 0 => { return self diff --git a/nodedb/src/data/executor/handlers/bulk_dml/update_persist.rs b/nodedb/src/data/executor/handlers/bulk_dml/update_persist.rs index 3b58e16cf..90ab249ba 100644 --- a/nodedb/src/data/executor/handlers/bulk_dml/update_persist.rs +++ b/nodedb/src/data/executor/handlers/bulk_dml/update_persist.rs @@ -1,21 +1,21 @@ // SPDX-License-Identifier: BUSL-1.1 //! Landing one bulk-UPDATE row: the write transaction the row's body, its -//! secondary-index diff, and its materialized-sum deltas share. +//! secondary-index diff, its full-text postings and its materialized-sum +//! deltas share. //! //! Its own file because the transaction boundary is the concern — the bulk //! handler decides WHICH rows change and what they become, and this decides //! when that becomes durable. Every sparse-database write a row produces is //! staged into the transaction opened here and lands on its commit, so a row -//! that fails at any step drops the transaction un-committed and is skipped -//! whole rather than left with a body the index no longer describes. +//! that fails at any step drops the transaction un-committed and fails the +//! statement, rather than being left with a body the index no longer +//! describes or dropped from the affected count. //! //! The materialized-sum fold runs inside that same transaction, one level above //! the row's own write, so a credited target row can never survive a row whose //! commit did not happen. -use tracing::warn; - use crate::data::executor::core_loop::CoreLoop; use crate::data::executor::enforcement::images::RowImages; use crate::data::executor::enforcement::materialized_sum::apply::TargetWrite; @@ -39,16 +39,16 @@ impl CoreLoop { /// entries, and fold its materialized-sum deltas — committing all three /// together. /// - /// `Ok(None)` is the row's own write failing: it is skipped, exactly as it - /// always was, and the transaction is dropped un-committed so it leaves - /// nothing behind. `Err` is an enforcement REJECTION, which is not - /// skippable — a rejected constraint must fail the statement, not quietly - /// shrink its affected count. + /// Any failure is an `Err` and fails the statement: the row's write, its + /// index diff, its full-text postings, an enforcement rejection, or the + /// commit. The transaction is dropped un-committed, so the row leaves + /// nothing behind. A row is never skipped, because a skipped row would + /// report a smaller affected count than the predicate matched. pub(super) fn persist_bulk_update_row( &mut self, p: NonbitemporalUpdateReindex<'_>, hook: &HookCtx<'_>, - ) -> crate::Result> { + ) -> crate::Result { // Hoisted before the transaction opens: the images below borrow them, // and a collection that declares no image-folding enforcement must not // pay for the fold at all. @@ -60,20 +60,17 @@ impl CoreLoop { let new_doc: &serde_json::Value = p.new_doc; let folds = write_hook::folds_images(self, hook); - let txn = match self.sparse.begin_write() { - Ok(txn) => txn, - Err(e) => { - warn!(%doc_id, error = %e, "bulk update: write txn failed, skipping document"); - return Ok(None); - } - }; - let touched = match self.nonbitemporal_update_reindex(&txn, p) { - Ok(touched) => touched, - Err(e) => { - warn!(%doc_id, error = %e, "update reindex failed, skipping document"); - return Ok(None); - } + let txn = self.sparse.begin_write()?; + let text = crate::data::executor::handlers::point::update_reindex_text::UpdateTextReindex { + database_id: p.database_id, + tid: p.tid, + collection: p.collection, + surrogate: doc_id.surrogate(), + new_doc, }; + let touched = self.nonbitemporal_update_reindex(&txn, p)?; + // The row's postings follow its new text in the same transaction. + self.update_reindex_text(&txn, text)?; // Both images were materialized by the caller for the index diff, so the // fold re-reads and re-decodes nothing. `RowImages::Update` is the only @@ -100,19 +97,13 @@ impl CoreLoop { Vec::new() }; - match txn.commit() { - Ok(()) => Ok(Some(PersistedBulkUpdateRow { - touched, - target_writes, - })), - Err(e) => { - warn!( - %doc_id, - error = %e, - "bulk update commit failed, skipping document" - ); - Ok(None) - } - } + txn.commit().map_err(|e| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("bulk update commit of row {doc_id}: {e}"), + })?; + Ok(PersistedBulkUpdateRow { + touched, + target_writes, + }) } } diff --git a/nodedb/src/data/executor/handlers/control/mod.rs b/nodedb/src/data/executor/handlers/control/mod.rs index e036991e7..946975bf8 100644 --- a/nodedb/src/data/executor/handlers/control/mod.rs +++ b/nodedb/src/data/executor/handlers/control/mod.rs @@ -23,7 +23,6 @@ pub mod crdt_preview; pub mod move_tenant; mod range_scan_versioned; pub mod reindex; -mod reindex_apply; pub mod snapshot; pub mod synonym_group; diff --git a/nodedb/src/data/executor/handlers/control/reindex.rs b/nodedb/src/data/executor/handlers/control/reindex.rs deleted file mode 100644 index ea46935da..000000000 --- a/nodedb/src/data/executor/handlers/control/reindex.rs +++ /dev/null @@ -1,386 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Concurrent index rebuild (REINDEX CONCURRENTLY) for HNSW, FTS LSM, and graph CSR. -//! -//! Design: the Data Plane dispatches a `RebuildIndex` op. HNSW segments are -//! rebuilt on the core's HNSW builder thread, the same path every graph build -//! takes (`handlers::vector_build`). For FTS and CSR a background OS thread -//! performs the heavy build work while the owning core continues to serve -//! reads from the live index. On each subsequent tick -//! the core polls `pending_reindex` for completion via `try_recv`; when the -//! build succeeds the core performs an in-memory swap and returns the ACK. -//! -//! The non-concurrent path runs the rebuild inline, compacting tombstones and -//! write buffers without moving data off-core. -//! -//! Plane rules: the background thread is a plain OS thread (not a tokio task). -//! It receives plain `Send` data, builds in isolation, and sends serialized -//! bytes back. The `!Send` engine state is touched only on the Data Plane -//! thread. -//! -//! Background-thread rebuild functions and Data-Plane cutover appliers live in -//! the sibling `reindex_apply` module. - -use std::sync::mpsc; - -use tracing::{error, info, warn}; - -use super::reindex_apply::{ - FtsRebuild, RebuildOutput, apply_csr, apply_fts, rebuild_csr_thread, rebuild_fts_thread, -}; -use crate::bridge::envelope::{ErrorCode, Response}; -use crate::data::executor::core_loop::CoreLoop; -use crate::data::executor::task::ExecutionTask; -use crate::types::TenantId; - -// ── PendingReindex ──────────────────────────────────────────────────────────── - -/// An in-flight concurrent rebuild tracked on the `CoreLoop`. -pub struct PendingReindex { - pub database_id: nodedb_types::DatabaseId, - pub tenant_id: TenantId, - pub collection_key: String, - rx: mpsc::Receiver>, -} - -// ── CoreLoop integration ────────────────────────────────────────────────────── - -impl CoreLoop { - /// Handle a `MetaOp::RebuildIndex` dispatch. - pub(in crate::data::executor) fn execute_rebuild_index( - &mut self, - task: &ExecutionTask, - tid: u64, - collection: &str, - index_name: Option<&str>, - concurrent: bool, - ) -> Response { - let tenant_id = TenantId::new(tid); - let collection_key = collection.to_string(); - - if !concurrent { - return self.rebuild_index_inline(task, tenant_id, &collection_key); - } - - // Reject duplicate concurrent rebuild for same collection. - if self - .maintenance - .pending_reindex - .iter() - .any(|p| p.tenant_id == tenant_id && p.collection_key == collection_key) - { - return self.response_error( - task, - ErrorCode::Internal { - detail: format!("rebuild already in progress for collection \"{collection}\""), - }, - ); - } - - let rebuild_hnsw = index_name - .map(|n| n.eq_ignore_ascii_case("hnsw")) - .unwrap_or(true); - let rebuild_fts = index_name - .map(|n| n.eq_ignore_ascii_case("fts")) - .unwrap_or(true); - let rebuild_csr = index_name - .map(|n| n.eq_ignore_ascii_case("csr")) - .unwrap_or(true); - - // Start the first applicable rebuild (priority: HNSW > FTS > CSR). - let start_result = if rebuild_hnsw { - self.start_hnsw_rebuild(task, tenant_id, &collection_key); - Ok(()) - } else if rebuild_fts { - self.start_fts_rebuild(task, tenant_id, &collection_key) - } else if rebuild_csr { - self.start_csr_rebuild(task, tenant_id, &collection_key) - } else { - Ok(()) - }; - if let Err(e) = start_result { - return self.response_error(task, e); - } - - self.response_ok(task) - } - - /// Poll all in-flight concurrent rebuilds. Called from `tick()`. - pub fn poll_pending_reindex(&mut self) { - // Collect completed and failed entries, leaving only still-running ones. - // We must separate the poll loop from the apply loop to satisfy the borrow checker: - // apply_* functions take &mut self, which conflicts with holding a reference into - // self.maintenance.pending_reindex at the same time. - enum Outcome { - Done { - database_id: nodedb_types::DatabaseId, - tenant_id: nodedb_types::TenantId, - collection_key: String, - output: RebuildOutput, - }, - Failed { - collection_key: String, - error: String, - }, - } - - let mut outcomes: Vec = Vec::new(); - let mut still_running: Vec = Vec::new(); - - for pending in self.maintenance.pending_reindex.drain(..) { - match pending.rx.try_recv() { - Ok(Ok(output)) => outcomes.push(Outcome::Done { - database_id: pending.database_id, - tenant_id: pending.tenant_id, - collection_key: pending.collection_key, - output, - }), - Ok(Err(e)) => outcomes.push(Outcome::Failed { - collection_key: pending.collection_key, - error: e.to_string(), - }), - Err(mpsc::TryRecvError::Disconnected) => outcomes.push(Outcome::Failed { - collection_key: pending.collection_key, - error: "rebuild thread disconnected".to_owned(), - }), - Err(mpsc::TryRecvError::Empty) => still_running.push(pending), - } - } - self.maintenance.pending_reindex = still_running; - - for outcome in outcomes { - match outcome { - Outcome::Done { - database_id, - tenant_id, - collection_key, - output, - } => { - match output { - RebuildOutput::Csr { bytes } => { - apply_csr(self, &database_id, &tenant_id, &collection_key, bytes); - } - RebuildOutput::Fts(rebuild) => { - apply_fts(self, &database_id, &tenant_id, &collection_key, rebuild); - } - } - info!( - core = self.core_id, - collection = %collection_key, - "concurrent index rebuild cutover complete" - ); - } - Outcome::Failed { - collection_key, - error, - } => { - error!( - core = self.core_id, - collection = %collection_key, - error = %error, - "concurrent index rebuild failed; live index unchanged" - ); - } - } - } - } - - // ── Inline (non-concurrent) rebuild ────────────────────────────────────── - - fn rebuild_index_inline( - &mut self, - task: &ExecutionTask, - tenant_id: TenantId, - collection_key: &str, - ) -> Response { - // HNSW: compact tombstones from sealed segments. - let db = task.request.database_id; - if let Some(coll) = - self.vector_collections - .get_mut(&(db, tenant_id, collection_key.to_string())) - { - let removed = coll.compact_tombstones(); - info!( - core = self.core_id, - collection = %collection_key, - removed, - "inline HNSW tombstone compaction" - ); - } - - // CSR: compact write buffers into dense arrays. - if let Err(e) = self.csr.compact_all() { - warn!( - core = self.core_id, - error = %e, - "inline CSR compact failed (budget); continuing" - ); - } - - self.response_ok(task) - } - - // ── Background-thread starters ──────────────────────────────────────────── - - /// Queue a rebuild of every sealed HNSW segment of the collection's - /// vector indexes on this core's builder thread. Each segment is rebuilt - /// from its own vectors with its node ids kept, quantized again under the - /// collection's config, and swapped in on this core; search reads the old - /// graph until then. The growing and building segments are left alone. - fn start_hnsw_rebuild( - &mut self, - task: &ExecutionTask, - tenant_id: TenantId, - collection_key: &str, - ) { - // Vector collections are stored under two key forms depending on how - // they were inserted: - // - Bare: (db, tenant, "coll") — BatchInsert / native - // - Field-qualified: (db, tenant, "coll:field_name") — SQL INSERT, DirectUpsert - let db = task.request.database_id; - let field_prefix = format!("{collection_key}:"); - let matching_keys: Vec<(nodedb_types::DatabaseId, TenantId, String)> = self - .vector_collections - .iter() - .filter(|((d, t, k), coll)| { - *d == db - && *t == tenant_id - && (k.as_str() == collection_key || k.starts_with(&field_prefix)) - // An IVF-PQ collection keeps no HNSW segments to rebuild. - && !coll.is_ivf() - }) - .map(|(key, _)| key.clone()) - .collect(); - for key in matching_keys { - let queued = self.queue_vector_rebuild(&key); - info!( - core = self.core_id, - collection = %key.2, - queued, - "HNSW rebuild queued" - ); - } - } - - fn start_fts_rebuild( - &mut self, - task: &ExecutionTask, - tenant_id: TenantId, - collection_key: &str, - ) -> crate::Result<()> { - use nodedb_fts::backend::FtsBackend; - - let database_id = task.request.database_id; - let db_u64 = database_id.as_u64(); - let tid = tenant_id.as_u64(); - let backend = self.inverted.backend(); - - let terms = backend - .collection_terms(db_u64, tid, collection_key) - .map_err(|e| crate::Error::Storage { - engine: "fts".to_string(), - detail: format!("FTS terms: {e}"), - })?; - - let mut postings: Vec<(String, Vec)> = - Vec::with_capacity(terms.len()); - for term in &terms { - let ps = backend - .read_postings(db_u64, tid, collection_key, term) - .map_err(|e| crate::Error::Storage { - engine: "fts".to_string(), - detail: format!("FTS postings '{term}': {e}"), - })?; - postings.push((term.clone(), ps)); - } - - // Collect doc lengths from posting entries. - let mut dl_map: std::collections::HashMap = std::collections::HashMap::new(); - for (_, ps) in &postings { - for p in ps { - let k = p.doc_id.as_u32(); - if let std::collections::hash_map::Entry::Vacant(slot) = dl_map.entry(k) - && let Ok(Some(dl)) = - backend.read_doc_length(db_u64, tid, collection_key, p.doc_id) - { - slot.insert(dl); - } - } - } - let doc_lengths: Vec<(nodedb_types::Surrogate, u32)> = dl_map - .iter() - .map(|(&k, &dl)| (nodedb_types::Surrogate::new(k), dl)) - .collect(); - - let (doc_count, total_tokens) = backend - .collection_stats(db_u64, tid, collection_key) - .map_err(|e| crate::Error::Storage { - engine: "fts".to_string(), - detail: format!("FTS stats: {e}"), - })?; - - let analyzer_meta = backend - .read_meta(db_u64, tid, collection_key, "analyzer") - .unwrap_or(None); - - let input = FtsRebuild { - postings, - doc_lengths, - doc_count, - total_tokens, - analyzer_meta, - }; - let (tx, rx) = mpsc::sync_channel::>(1); - std::thread::spawn(move || { - let _ = tx.send(rebuild_fts_thread(input)); - }); - - self.maintenance.pending_reindex.push(PendingReindex { - database_id, - tenant_id, - collection_key: collection_key.to_string(), - rx, - }); - Ok(()) - } - - fn start_csr_rebuild( - &mut self, - task: &ExecutionTask, - tenant_id: TenantId, - collection_key: &str, - ) -> crate::Result<()> { - let database_id = task.request.database_id; - let partition = match self.csr.partition(database_id, tenant_id) { - Some(p) => p, - None => return Ok(()), // nothing to rebuild - }; - - let snapshot_bytes = - partition - .checkpoint_to_bytes() - .map_err(|e| crate::Error::Storage { - engine: "graph".to_string(), - detail: format!("CSR serialize: {e}"), - })?; - - let memory = nodedb_mem::ScopedMemory::new( - self.governor.clone(), - database_id, - tenant_id, - nodedb_mem::EngineId::Graph, - ); - let (tx, rx) = mpsc::sync_channel::>(1); - std::thread::spawn(move || { - let _ = tx.send(rebuild_csr_thread(snapshot_bytes, memory)); - }); - - self.maintenance.pending_reindex.push(PendingReindex { - database_id, - tenant_id, - collection_key: collection_key.to_string(), - rx, - }); - Ok(()) - } -} diff --git a/nodedb/src/data/executor/handlers/control/reindex/csr.rs b/nodedb/src/data/executor/handlers/control/reindex/csr.rs new file mode 100644 index 000000000..b316ee6df --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/reindex/csr.rs @@ -0,0 +1,147 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! CSR rebuild of the tenant partition that holds a collection's edges. +//! +//! A tenant's edges from every collection share one CSR partition, so the +//! rebuild covers the whole partition. It runs only when the collection +//! has edges in it. +//! +//! Start, on the owning core: snapshot the partition and open its write +//! journal in one step, then compact the snapshot on a plain OS thread. +//! The core keeps serving from the live partition; every mutation of it +//! records itself in the journal. +//! +//! Cutover, on the owning core: restore the compacted copy, replay the +//! journal onto it and swap it into the partition map in one call. A +//! traversal sees the old partition or the new one, never a mix. The +//! cutover emits `atomic_cutover` on the `nodedb::reindex` target. + +use std::sync::mpsc; + +use nodedb_graph::csr::rebuild::{CsrRebuildSeed, CsrRebuilt}; +use nodedb_mem::{EngineId, ScopedMemory}; +use tracing::info; + +use super::hold::{CSR_BUILD_HOLD, hold_build}; +use super::pending::{PendingBuild, PendingReindex, RebuildTarget}; +use crate::data::executor::core_loop::CoreLoop; + +/// Bound on the bytes of mutations one CSR rebuild journal records. Past +/// it the rebuild is discarded at cutover and the live partition stays. +pub const CSR_REBUILD_JOURNAL_MAX_BYTES: usize = 64 << 20; + +/// Map a graph-engine error into the crate error. +pub(super) fn graph_err(e: nodedb_graph::GraphError) -> crate::Error { + crate::Error::Storage { + engine: "graph".to_string(), + detail: e.to_string(), + } +} + +impl CoreLoop { + /// Start a CSR rebuild for `target` on its own thread. + pub(super) fn start_csr_rebuild(&mut self, target: &RebuildTarget) -> crate::Result<()> { + let Some(seed) = self.begin_csr_rebuild(target)? else { + return Ok(()); + }; + let token = seed.token(); + let memory = self.graph_memory(target); + let (tx, rx) = mpsc::sync_channel::>(1); + let spawned = std::thread::Builder::new() + .name(format!("reindex-csr-{}", self.core_id)) + .spawn(move || { + hold_build(CSR_BUILD_HOLD); + // The receiver is gone only when the core shut down. + let _ = tx.send(seed.build(memory)); + }); + if let Err(e) = spawned { + self.abort_csr_rebuild(target, token); + return Err(crate::Error::Io(e)); + } + self.maintenance.pending_reindex.push(PendingReindex { + target: target.clone(), + build: PendingBuild::Csr { token, rx }, + }); + Ok(()) + } + + /// Snapshot the partition and open its journal, or `None` when the + /// collection has no edges in it, or when a rebuild of the partition + /// already runs: that rebuild covers this collection too. + fn begin_csr_rebuild( + &mut self, + target: &RebuildTarget, + ) -> crate::Result> { + let core_id = self.core_id; + let Some(partition) = self.csr.partition_mut(target.database_id, target.tenant_id) else { + return Ok(None); + }; + if partition.collection_id(&target.collection).is_none() { + return Ok(None); + } + if partition.rebuild_in_progress() { + info!( + target: "nodedb::reindex", + core = core_id, + index = "csr", + collection = %target.collection, + "CSR partition rebuild already running; it covers this collection" + ); + return Ok(None); + } + let seed = partition + .begin_rebuild(CSR_REBUILD_JOURNAL_MAX_BYTES) + .map_err(graph_err)?; + info!( + target: "nodedb::reindex", + core = core_id, + index = "csr", + collection = %target.collection, + "rebuild_started" + ); + Ok(Some(seed)) + } + + /// Replay the journal onto `rebuilt` and swap it in on this core. + pub(super) fn install_csr( + &mut self, + target: &RebuildTarget, + rebuilt: CsrRebuilt, + ) -> crate::Result<()> { + let memory = self.graph_memory(target); + let Some(live) = self.csr.partition_mut(target.database_id, target.tenant_id) else { + return Err(graph_err(nodedb_graph::GraphError::RebuildSuperseded)); + }; + let copy = live.finish_rebuild(rebuilt, memory).map_err(graph_err)?; + let nodes = copy.node_count(); + let edges = copy.edge_count(); + self.csr + .install_partition(target.database_id, target.tenant_id, copy); + info!( + target: "nodedb::reindex", + core = self.core_id, + index = "csr", + collection = %target.collection, + nodes, + edges, + "atomic_cutover" + ); + Ok(()) + } + + /// Close the partition's journal of rebuild `token`. + pub(super) fn abort_csr_rebuild(&mut self, target: &RebuildTarget, token: u64) { + if let Some(partition) = self.csr.partition_mut(target.database_id, target.tenant_id) { + partition.abort_rebuild(token); + } + } + + fn graph_memory(&self, target: &RebuildTarget) -> ScopedMemory { + ScopedMemory::new( + self.governor.clone(), + target.database_id, + target.tenant_id, + EngineId::Graph, + ) + } +} diff --git a/nodedb/src/data/executor/handlers/control/reindex/dispatch.rs b/nodedb/src/data/executor/handlers/control/reindex/dispatch.rs new file mode 100644 index 000000000..c6777a6c5 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/reindex/dispatch.rs @@ -0,0 +1,182 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The `MetaOp::RebuildIndex` handler. +//! +//! `index_name` picks the index kinds to rebuild: `hnsw`, `fts` or `csr`, +//! matched without case. `None` rebuilds every kind the collection has. +//! +//! Both REINDEX forms rebuild the same way, and never on the core: HNSW +//! segments on the core's HNSW builder thread, FTS and CSR on their own +//! threads as `fts` and `csr` describe. The core keeps serving reads and +//! writes, journals the writes, and swaps each rebuilt index in on a later +//! tick. The forms differ only in when the core answers: +//! +//! - Concurrent: once every rebuild has started. This handler answers. +//! - Plain: once every rebuild has cut over. `waiter` holds the request +//! and answers it from the tick. + +use tracing::info; + +use super::pending::RebuildTarget; +use crate::bridge::envelope::Response; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::task::ExecutionTask; +use crate::types::TenantId; + +/// The index kinds one REINDEX covers. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct IndexSelection { + pub(super) hnsw: bool, + pub(super) fts: bool, + pub(super) csr: bool, +} + +impl IndexSelection { + pub(super) fn from_name(index_name: Option<&str>) -> Self { + match index_name { + None => Self { + hnsw: true, + fts: true, + csr: true, + }, + Some(name) => Self { + hnsw: name.eq_ignore_ascii_case("hnsw"), + fts: name.eq_ignore_ascii_case("fts"), + csr: name.eq_ignore_ascii_case("csr"), + }, + } + } +} + +impl CoreLoop { + /// Handle a `MetaOp::RebuildIndex` dispatch: start every selected + /// rebuild and answer. A plain REINDEX reaches the core loop's waiter + /// first, which holds its answer until the cutovers; this handler + /// answers a concurrent one. + pub(in crate::data::executor) fn execute_rebuild_index( + &mut self, + task: &ExecutionTask, + tid: u64, + collection: &str, + index_name: Option<&str>, + ) -> Response { + let target = RebuildTarget { + database_id: task.request.database_id, + tenant_id: TenantId::new(tid), + collection: collection.to_string(), + }; + match self.start_rebuilds(&target, IndexSelection::from_name(index_name)) { + Ok(()) => self.response_ok(task), + Err(e) => self.response_error(task, e), + } + } + + /// Start every selected rebuild. Each kind starts even when another + /// fails to; the first error is returned. + pub(super) fn start_rebuilds( + &mut self, + target: &RebuildTarget, + selection: IndexSelection, + ) -> crate::Result<()> { + if self + .maintenance + .pending_reindex + .iter() + .any(|p| p.target == *target) + { + return Err(crate::Error::ObjectNotInPrerequisiteState { + object: format!("collection \"{}\"", target.collection), + detail: "a rebuild of its indexes is already running".to_string(), + }); + } + if selection.hnsw { + self.start_hnsw_rebuild(target); + } + let fts = if selection.fts { + self.start_fts_rebuild(target) + } else { + Ok(()) + }; + let csr = if selection.csr { + self.start_csr_rebuild(target) + } else { + Ok(()) + }; + fts.and(csr) + } + + /// Queue a rebuild of every sealed HNSW segment of the collection's + /// vector indexes on this core's builder thread. Each segment is rebuilt + /// from its own vectors with its node ids kept, quantized again under the + /// collection's config, and swapped in on this core; search reads the old + /// graph until then. The growing and building segments are left alone. + fn start_hnsw_rebuild(&mut self, target: &RebuildTarget) { + for key in self.vector_keys_of(target) { + let queued = self.queue_vector_rebuild(&key); + info!( + core = self.core_id, + collection = %key.2, + queued, + "HNSW rebuild queued" + ); + } + } + + /// Keys of the collection's HNSW indexes on this core. A collection + /// keys its indexes as `coll` (batch and native inserts) or as + /// `coll:field` (SQL inserts). An IVF-PQ index keeps no HNSW segments, + /// so it is left out. + pub(super) fn vector_keys_of( + &self, + target: &RebuildTarget, + ) -> Vec<(nodedb_types::DatabaseId, TenantId, String)> { + let field_prefix = format!("{}:", target.collection); + self.vector_collections + .iter() + .filter(|((d, t, k), coll)| { + *d == target.database_id + && *t == target.tenant_id + && (k.as_str() == target.collection || k.starts_with(&field_prefix)) + && !coll.is_ivf() + }) + .map(|(key, _)| key.clone()) + .collect() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn no_name_selects_every_kind() { + assert_eq!( + IndexSelection::from_name(None), + IndexSelection { + hnsw: true, + fts: true, + csr: true, + } + ); + } + + #[test] + fn a_name_selects_its_kind_only() { + assert_eq!( + IndexSelection::from_name(Some("FTS")), + IndexSelection { + hnsw: false, + fts: true, + csr: false, + } + ); + assert_eq!( + IndexSelection::from_name(Some("tag_idx")), + IndexSelection { + hnsw: false, + fts: false, + csr: false, + } + ); + } +} diff --git a/nodedb/src/data/executor/handlers/control/reindex/fts.rs b/nodedb/src/data/executor/handlers/control/reindex/fts.rs new file mode 100644 index 000000000..4d7e39a1a --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/reindex/fts.rs @@ -0,0 +1,117 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Full-text rebuild of one collection. +//! +//! Start, on the owning core: pin a redb read snapshot of the collection and +//! open its write journal in one step, then hand the snapshot to a plain OS +//! thread. The thread reads the snapshot and derives the canonical rows. +//! The core keeps serving reads and writes from the live index; every write +//! to the collection notes its document in the journal. +//! +//! Cutover, on the owning core: one redb write transaction replaces the +//! collection's rows with the rebuilt ones and writes the live footprint of +//! every noted document over them. A reader sees the old rows or the new +//! ones, never a mix. The cutover emits `atomic_cutover` on the +//! `nodedb::reindex` target. + +use std::sync::mpsc; + +use tracing::info; + +use super::hold::{FTS_BUILD_HOLD, hold_build}; +use super::pending::{PendingBuild, PendingReindex, RebuildTarget}; +use crate::data::executor::core_loop::CoreLoop; +use crate::engine::sparse::inverted::{ + FTS_REBUILD_JOURNAL_MAX_DOCS, FtsInstallOutcome, FtsRebuildTicket, FtsRebuilt, FtsSnapshot, +}; + +impl CoreLoop { + /// Start a full-text rebuild of `target` on its own thread. A collection + /// with no full-text rows has nothing to rebuild. + pub(super) fn start_fts_rebuild(&mut self, target: &RebuildTarget) -> crate::Result<()> { + let Some(ticket) = self.begin_fts_rebuild(target)? else { + return Ok(()); + }; + let token = ticket.token(); + let (tx, rx) = mpsc::sync_channel::>(1); + let spawned = std::thread::Builder::new() + .name(format!("reindex-fts-{}", self.core_id)) + .spawn(move || { + hold_build(FTS_BUILD_HOLD); + // The receiver is gone only when the core shut down. + let _ = tx.send(ticket.read().map(FtsSnapshot::compact)); + }); + if let Err(e) = spawned { + self.inverted.abort_rebuild(token); + return Err(crate::Error::Io(e)); + } + self.maintenance.pending_reindex.push(PendingReindex { + target: target.clone(), + build: PendingBuild::Fts { token, rx }, + }); + Ok(()) + } + + /// Pin the snapshot and open the journal, or `None` when the + /// collection has no full-text rows. + fn begin_fts_rebuild( + &mut self, + target: &RebuildTarget, + ) -> crate::Result> { + let db = target.database_id.as_u64(); + if !self + .inverted + .has_collection_rows(db, target.tenant_id, &target.collection)? + { + return Ok(None); + } + let ticket = self.inverted.begin_rebuild( + db, + target.tenant_id, + &target.collection, + FTS_REBUILD_JOURNAL_MAX_DOCS, + )?; + info!( + target: "nodedb::reindex", + core = self.core_id, + index = "fts", + collection = %target.collection, + "rebuild_started" + ); + Ok(Some(ticket)) + } + + /// Cut `rebuilt` over on this core. A refusal is an error: the live + /// index stays as it is. + pub(super) fn install_fts( + &mut self, + target: &RebuildTarget, + rebuilt: FtsRebuilt, + ) -> crate::Result<()> { + match self.inverted.install_rebuild(rebuilt)? { + FtsInstallOutcome::Installed { + terms, + docs, + replayed, + } => { + info!( + target: "nodedb::reindex", + core = self.core_id, + index = "fts", + collection = %target.collection, + terms, + docs, + replayed, + "atomic_cutover" + ); + Ok(()) + } + FtsInstallOutcome::Refused(refusal) => { + Err(crate::Error::ObjectNotInPrerequisiteState { + object: format!("full-text index of collection \"{}\"", target.collection), + detail: format!("rebuild discarded, live index unchanged: {refusal}"), + }) + } + } + } +} diff --git a/nodedb/src/data/executor/handlers/control/reindex/hold.rs b/nodedb/src/data/executor/handlers/control/reindex/hold.rs new file mode 100644 index 000000000..4d11035e9 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/reindex/hold.rs @@ -0,0 +1,45 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Failpoints that hold a rebuild thread, so a test can write while a +//! rebuild runs. +//! +//! The hold runs on the rebuild's own OS thread, never on a core. Without +//! the `failpoints` feature it compiles to nothing. + +/// Holds a full-text rebuild thread before it reads its snapshot. +pub const FTS_BUILD_HOLD: &str = "reindex::fts_build_hold"; + +/// Holds a CSR rebuild thread before it compacts its snapshot. +pub const CSR_BUILD_HOLD: &str = "reindex::csr_build_hold"; + +/// Hold the calling rebuild thread at failpoint `name`. +/// +/// `WaitForFile(path)` parks the thread until `path` exists, for at most +/// two minutes. Every other action runs as at any failpoint. +pub(super) fn hold_build(name: &str) { + #[cfg(feature = "failpoints")] + hold_armed(name); + #[cfg(not(feature = "failpoints"))] + let _ = name; +} + +#[cfg(feature = "failpoints")] +fn hold_armed(name: &str) { + use std::time::{Duration, Instant}; + + use crate::fail_point::{FailAction, eval, lookup}; + + const HOLD_LIMIT: Duration = Duration::from_secs(120); + const POLL: Duration = Duration::from_millis(5); + + match lookup(name) { + Some(FailAction::WaitForFile(path)) => { + let deadline = Instant::now() + HOLD_LIMIT; + while !path.exists() && Instant::now() < deadline { + std::thread::sleep(POLL); + } + } + Some(_) => eval(name), + None => {} + } +} diff --git a/nodedb/src/data/executor/handlers/control/reindex/mod.rs b/nodedb/src/data/executor/handlers/control/reindex/mod.rs new file mode 100644 index 000000000..688973a77 --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/reindex/mod.rs @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! REINDEX of a collection's HNSW, full-text and CSR indexes. +//! +//! - `dispatch`: the `MetaOp::RebuildIndex` handler: starts the rebuilds. +//! - `fts`, `csr`: start a rebuild on its own OS thread and cut it over on +//! the owning core with every write made during the build replayed. +//! - `pending`: in-flight rebuilds, polled every tick. +//! - `hold`: the failpoint that holds a rebuild thread in tests. +//! - `waiter`: holds a plain REINDEX's answer until its cutovers. + +pub mod csr; +pub mod dispatch; +pub mod fts; +pub mod hold; +pub mod pending; +pub mod waiter; + +pub use pending::{PendingReindex, RebuildTarget}; +pub use waiter::ReindexWaiter; diff --git a/nodedb/src/data/executor/handlers/control/reindex/pending.rs b/nodedb/src/data/executor/handlers/control/reindex/pending.rs new file mode 100644 index 000000000..c470e60bb --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/reindex/pending.rs @@ -0,0 +1,146 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! In-flight rebuilds a core tracks, and the per-tick poll that cuts each +//! finished one over. + +use std::sync::mpsc::{self, TryRecvError}; + +use nodedb_graph::csr::rebuild::CsrRebuilt; +use nodedb_types::DatabaseId; +use tracing::error; + +use crate::data::executor::core_loop::CoreLoop; +use crate::engine::sparse::inverted::FtsRebuilt; +use crate::types::TenantId; + +/// The collection one REINDEX covers. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RebuildTarget { + pub database_id: DatabaseId, + pub tenant_id: TenantId, + pub collection: String, +} + +/// A build running on its own OS thread. `token` ties it to the write +/// journal on the live index. +pub(in crate::data::executor) enum PendingBuild { + Fts { + token: u64, + rx: mpsc::Receiver>, + }, + Csr { + token: u64, + rx: mpsc::Receiver>, + }, +} + +/// One in-flight rebuild of one index of one collection. +pub struct PendingReindex { + pub target: RebuildTarget, + pub(in crate::data::executor) build: PendingBuild, +} + +/// What one poll of a build found. +enum BuildPoll { + Running(PendingBuild), + Fts { + token: u64, + result: crate::Result, + }, + Csr { + token: u64, + result: crate::Result, + }, +} + +impl PendingBuild { + fn poll(self) -> BuildPoll { + match self { + Self::Fts { token, rx } => match rx.try_recv() { + Ok(result) => BuildPoll::Fts { token, result }, + Err(TryRecvError::Empty) => BuildPoll::Running(Self::Fts { token, rx }), + Err(TryRecvError::Disconnected) => BuildPoll::Fts { + token, + result: Err(thread_lost("fts")), + }, + }, + Self::Csr { token, rx } => match rx.try_recv() { + Ok(result) => BuildPoll::Csr { + token, + result: result.map_err(super::csr::graph_err), + }, + Err(TryRecvError::Empty) => BuildPoll::Running(Self::Csr { token, rx }), + Err(TryRecvError::Disconnected) => BuildPoll::Csr { + token, + result: Err(thread_lost("csr")), + }, + }, + } + } +} + +fn thread_lost(index: &str) -> crate::Error { + crate::Error::Internal { + detail: format!("{index} rebuild thread exited without a result"), + } +} + +impl CoreLoop { + /// Cut over every finished rebuild. Called from `tick()`. + pub fn poll_pending_reindex(&mut self) { + if self.maintenance.pending_reindex.is_empty() { + return; + } + let entries = std::mem::take(&mut self.maintenance.pending_reindex); + let mut running = Vec::with_capacity(entries.len()); + for PendingReindex { target, build } in entries { + match build.poll() { + BuildPoll::Running(build) => running.push(PendingReindex { target, build }), + BuildPoll::Fts { token, result } => { + let installed = result.and_then(|rebuilt| self.install_fts(&target, rebuilt)); + if let Err(e) = installed { + self.inverted.abort_rebuild(token); + self.report_rebuild_refused("fts", &target, &e); + self.note_reindex_refused(&target, &e); + } + } + BuildPoll::Csr { token, result } => { + let installed = result.and_then(|rebuilt| self.install_csr(&target, rebuilt)); + if let Err(e) = installed { + self.abort_csr_rebuild(&target, token); + self.report_rebuild_refused("csr", &target, &e); + self.note_reindex_refused(&target, &e); + } + } + } + } + self.maintenance.pending_reindex.extend(running); + } + + /// Log and record a rebuild the core discarded. The live index keeps + /// every write; only the rebuild is lost. + pub(super) fn report_rebuild_refused( + &self, + index: &'static str, + target: &RebuildTarget, + err: &crate::Error, + ) { + error!( + target: "nodedb::reindex", + core = self.core_id, + index, + collection = %target.collection, + error = %err, + "rebuild_refused" + ); + crate::diag::index_rebuild_not_installed( + err, + &crate::diag::IndexRebuildTarget { + index, + database_id: target.database_id.as_u64(), + tenant_id: target.tenant_id.as_u64(), + collection: &target.collection, + }, + ); + } +} diff --git a/nodedb/src/data/executor/handlers/control/reindex/waiter.rs b/nodedb/src/data/executor/handlers/control/reindex/waiter.rs new file mode 100644 index 000000000..0e43d6aae --- /dev/null +++ b/nodedb/src/data/executor/handlers/control/reindex/waiter.rs @@ -0,0 +1,208 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A plain REINDEX answers once its rebuilds have cut over. +//! +//! The rebuilds run exactly as for REINDEX CONCURRENTLY: off the core, with +//! the core serving every other request meanwhile. Only the answer waits. +//! The core holds the request's task here and answers it from the tick, +//! the way it holds a write parked behind a staged Calvin transaction +//! (`core_loop::calvin_fence`): the held task builds the response, the +//! response ring carries it, and the task's deadline bounds the wait. +//! +//! - Every rebuild cut over: `Ok`. +//! - A rebuild was discarded, or an HNSW segment rebuild failed: the first +//! error. +//! - The deadline passed first: `DeadlineExceeded`. The rebuilds are not +//! cancelled; each one still cuts over on its own tick, or is refused +//! cleanly with the live index unchanged, as for REINDEX CONCURRENTLY. + +use std::collections::HashMap; + +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; + +use super::dispatch::IndexSelection; +use super::pending::RebuildTarget; +use crate::bridge::dispatch::BridgeResponse; +use crate::bridge::envelope::{ErrorCode, Response}; +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::vector_direct_row::VectorIndexKey; +use crate::data::executor::task::ExecutionTask; + +/// A plain REINDEX waiting for its rebuilds. +pub struct ReindexWaiter { + task: ExecutionTask, + target: RebuildTarget, + /// HNSW indexes with queued segment rebuilds, and each index's failed + /// build count when the rebuilds were queued. + vector_failed_before: HashMap, + /// The first FTS or CSR rebuild the core discarded. + first_error: Option, +} + +impl CoreLoop { + /// Start the rebuilds of a plain REINDEX and hold its task until they + /// cut over. Returns the task unchanged when it is anything else, or + /// when it expired before it started. + pub(in crate::data::executor) fn hold_plain_reindex( + &mut self, + task: ExecutionTask, + ) -> Option { + let plain = if let PhysicalPlan::Meta(MetaOp::RebuildIndex { + collection, + index_name, + concurrent: false, + }) = task.plan() + { + Some((collection.as_str().to_string(), index_name.clone())) + } else { + None + }; + let Some((collection, index_name)) = plain else { + return Some(task); + }; + if past_deadline(&task) { + // Nothing started, so nothing is left running. + let response = self.response_error(&task, ErrorCode::DeadlineExceeded); + self.send_reindex_answer(response); + return None; + } + let target = RebuildTarget { + database_id: task.request.database_id, + tenant_id: task.request.tenant_id, + collection, + }; + let selection = IndexSelection::from_name(index_name.as_deref()); + + if let Err(e) = self.start_rebuilds(&target, selection) { + let response = self.response_error(&task, e); + self.send_reindex_answer(response); + return None; + } + let vector_failed_before = if selection.hnsw { + self.vector_keys_of(&target) + .into_iter() + .filter(|key| self.vector_builds.pending_for(key) > 0) + .map(|key| { + let failed = self.vector_builds_failed(&key); + (key, failed) + }) + .collect() + } else { + HashMap::new() + }; + self.maintenance.reindex_waiters.push(ReindexWaiter { + task, + target, + vector_failed_before, + first_error: None, + }); + // A collection with nothing to rebuild answers on this tick. + self.answer_reindex_waiters(); + None + } + + /// Record a discarded FTS or CSR rebuild against the plain REINDEX that + /// waits for it. + pub(super) fn note_reindex_refused(&mut self, target: &RebuildTarget, err: &crate::Error) { + for waiter in self + .maintenance + .reindex_waiters + .iter_mut() + .filter(|w| w.target == *target && w.first_error.is_none()) + { + waiter.first_error = Some(crate::Error::ObjectNotInPrerequisiteState { + object: format!("indexes of collection \"{}\"", target.collection), + detail: format!("a rebuild was discarded: {err}"), + }); + } + } + + /// Answer every plain REINDEX whose rebuilds finished or whose deadline + /// passed. Called from `tick()` after the cutovers. + pub fn answer_reindex_waiters(&mut self) { + if self.maintenance.reindex_waiters.is_empty() { + return; + } + let waiters = std::mem::take(&mut self.maintenance.reindex_waiters); + for waiter in waiters { + if past_deadline(&waiter.task) { + let response = self.response_error(&waiter.task, ErrorCode::DeadlineExceeded); + self.send_reindex_answer(response); + continue; + } + let fts_csr_running = self + .maintenance + .pending_reindex + .iter() + .any(|p| p.target == waiter.target); + let hnsw_running = waiter + .vector_failed_before + .keys() + .any(|key| self.vector_builds.pending_for(key) > 0); + if fts_csr_running || hnsw_running { + self.maintenance.reindex_waiters.push(waiter); + continue; + } + let ReindexWaiter { + task, + vector_failed_before, + first_error, + .. + } = waiter; + let error = first_error.or_else(|| self.failed_vector_rebuilds(&vector_failed_before)); + let response = match error { + None => self.response_ok(&task), + Some(e) => self.response_error(&task, e), + }; + self.send_reindex_answer(response); + } + } + + /// An error naming each HNSW index whose failed build count rose + /// since its rebuilds were queued, or `None` when none did. + fn failed_vector_rebuilds( + &self, + failed_before: &HashMap, + ) -> Option { + let failed: Vec<&str> = failed_before + .iter() + .filter(|(key, before)| self.vector_builds_failed(key) > **before) + .map(|(key, _)| key.2.as_str()) + .collect(); + if failed.is_empty() { + return None; + } + Some(crate::Error::ObjectNotInPrerequisiteState { + object: format!("HNSW indexes {failed:?}"), + detail: "a segment rebuild failed; the segment keeps its previous graph".to_string(), + }) + } + + /// Failed HNSW builds of `key` so far. + fn vector_builds_failed(&self, key: &VectorIndexKey) -> u64 { + self.vector_collections + .get(key) + .map_or(0, |coll| coll.stats().builds_failed) + } + + fn send_reindex_answer(&mut self, response: Response) { + if let Err(e) = self + .response_tx + .try_push(BridgeResponse { inner: response }) + { + tracing::warn!( + core = self.core_id, + error = %e, + "failed to send a REINDEX answer: response queue full" + ); + } + } +} + +/// Whether `task`'s request deadline passed. The wait is bounded by the +/// request deadline whatever the request's admission class. +fn past_deadline(task: &ExecutionTask) -> bool { + // no-determinism: the deadline bounds only when the answer is sent; the + // rebuilds and their cutovers do not depend on it. + std::time::Instant::now() > task.request.deadline +} diff --git a/nodedb/src/data/executor/handlers/control/reindex_apply.rs b/nodedb/src/data/executor/handlers/control/reindex_apply.rs deleted file mode 100644 index 87a487c1f..000000000 --- a/nodedb/src/data/executor/handlers/control/reindex_apply.rs +++ /dev/null @@ -1,215 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Background-thread rebuild functions and Data-Plane cutover appliers for -//! concurrent FTS and CSR rebuild. See `reindex.rs` for the dispatch/poll -//! surface. HNSW rebuilds go through the core's HNSW builder thread. -//! -//! All functions in this module either run on a plain OS thread (operating on -//! pure `Send` data) or run on the owning Data Plane core during cutover. They -//! never spawn tokio tasks and never share `!Send` state across threads. - -use tracing::{error, info, warn}; - -use crate::data::executor::core_loop::CoreLoop; -use crate::types::TenantId; - -// ── Background-thread output types (all Send) ──────────────────────────────── - -pub(super) enum RebuildOutput { - /// Serialized rebuilt CSR bytes. - Csr { bytes: Vec }, - /// Compacted FTS data ready for write-back. - Fts(FtsRebuild), -} - -/// Compacted FTS state produced by a rebuild thread, ready to apply to the live -/// backend on the owning Data Plane core. -pub(super) struct FtsRebuild { - pub postings: Vec<(String, Vec)>, - pub doc_lengths: Vec<(nodedb_types::Surrogate, u32)>, - pub doc_count: u32, - pub total_tokens: u64, - pub analyzer_meta: Option>, -} - -// ── Background thread rebuild functions (pure Send, no !Send types) ────────── - -pub(super) fn rebuild_fts_thread(input: FtsRebuild) -> crate::Result { - // Compact: deduplicate posting entries by surrogate, keeping highest TF. - let mut compacted: Vec<(String, Vec)> = - Vec::with_capacity(input.postings.len()); - for (term, mut ps) in input.postings { - ps.sort_unstable_by(|a, b| { - a.doc_id - .as_u32() - .cmp(&b.doc_id.as_u32()) - .then(b.term_freq.cmp(&a.term_freq)) - }); - ps.dedup_by_key(|p| p.doc_id.as_u32()); - if !ps.is_empty() { - compacted.push((term, ps)); - } - } - Ok(RebuildOutput::Fts(FtsRebuild { - postings: compacted, - ..input - })) -} - -pub(super) fn rebuild_csr_thread( - snapshot_bytes: Vec, - memory: nodedb_mem::ScopedMemory, -) -> crate::Result { - let restored = nodedb_graph::CsrIndex::from_checkpoint(&snapshot_bytes, memory) - .map_err(|e| crate::Error::Storage { - engine: "graph".to_string(), - detail: format!("CSR restore: {e}"), - })? - .ok_or_else(|| crate::Error::Storage { - engine: "graph".to_string(), - detail: "CSR checkpoint bytes missing magic header".to_string(), - })?; - - let mut rebuilt = restored; - rebuilt.compact().map_err(|e| crate::Error::Storage { - engine: "graph".to_string(), - detail: format!("CSR compact: {e}"), - })?; - - let bytes = rebuilt - .checkpoint_to_bytes() - .map_err(|e| crate::Error::Storage { - engine: "graph".to_string(), - detail: format!("CSR re-serialize: {e}"), - })?; - - Ok(RebuildOutput::Csr { bytes }) -} - -// ── Cutover: apply rebuilt state to Data Plane in-memory structures ─────────── - -pub(super) fn apply_fts( - core: &mut CoreLoop, - database_id: &nodedb_types::DatabaseId, - tenant_id: &TenantId, - collection_key: &str, - rebuild: FtsRebuild, -) { - use nodedb_fts::backend::FtsBackend; - - let FtsRebuild { - postings, - doc_lengths, - doc_count, - total_tokens, - analyzer_meta, - } = rebuild; - - let db_u64 = database_id.as_u64(); - let tid = tenant_id.as_u64(); - let backend = core.inverted.backend(); - - // Purge existing postings for the collection, then write the compacted set. - if let Err(e) = backend.purge_collection(db_u64, tid, collection_key) { - error!( - core = core.core_id, - collection = %collection_key, - error = %e, - "FTS rebuild purge failed; index may be inconsistent" - ); - return; - } - - for (term, ps) in &postings { - if let Err(e) = backend.write_postings(db_u64, tid, collection_key, term, ps) { - error!( - core = core.core_id, - term, - error = %e, - "FTS rebuild write_postings failed" - ); - return; - } - } - for (surrogate, len) in &doc_lengths { - if let Err(e) = backend.write_doc_length(db_u64, tid, collection_key, *surrogate, *len) { - error!( - core = core.core_id, - error = %e, - "FTS rebuild write_doc_length failed" - ); - return; - } - } - // Restore collection-level stats via synthetic increment. - for _ in 0..doc_count { - // increment_stats with doc_len=0 just bumps the doc counter. - let _ = backend.increment_stats(db_u64, tid, collection_key, 0); - } - // Restore total_tokens by writing a single doc with all tokens. - if total_tokens > 0 && doc_count > 0 { - let tokens_per_doc = (total_tokens / doc_count as u64) as u32; - let _ = backend.increment_stats(db_u64, tid, collection_key, tokens_per_doc); - } - - if let Some(meta_bytes) = analyzer_meta - && let Err(e) = backend.write_meta(db_u64, tid, collection_key, "analyzer", &meta_bytes) - { - warn!( - core = core.core_id, - error = %e, - "FTS rebuild: analyzer meta write failed" - ); - } - - info!( - core = core.core_id, - collection = %collection_key, - terms = postings.len(), - docs = doc_lengths.len(), - "FTS cutover: postings replaced with compacted snapshot" - ); -} - -pub(super) fn apply_csr( - core: &mut CoreLoop, - database_id: &nodedb_types::DatabaseId, - tenant_id: &TenantId, - collection_key: &str, - bytes: Vec, -) { - let memory = nodedb_mem::ScopedMemory::new( - core.governor.clone(), - *database_id, - *tenant_id, - nodedb_mem::EngineId::Graph, - ); - let rebuilt = match nodedb_graph::CsrIndex::from_checkpoint(&bytes, memory) { - Ok(Some(r)) => r, - Ok(None) => { - warn!( - core = core.core_id, - collection = %collection_key, - "CSR rebuild: empty checkpoint; skipping cutover" - ); - return; - } - Err(e) => { - error!( - core = core.core_id, - collection = %collection_key, - error = %e, - "CSR rebuild: restore failed; live index unchanged" - ); - return; - } - }; - - core.csr - .install_partition(*database_id, *tenant_id, rebuilt); - info!( - core = core.core_id, - collection = %collection_key, - "CSR cutover: partition replaced with rebuilt index" - ); -} diff --git a/nodedb/src/data/executor/handlers/convert.rs b/nodedb/src/data/executor/handlers/convert.rs index c9abc2171..1e1a561ea 100644 --- a/nodedb/src/data/executor/handlers/convert.rs +++ b/nodedb/src/data/executor/handlers/convert.rs @@ -17,6 +17,7 @@ use sonic_rs; use nodedb_physical::physical_plan::StorageMode; use nodedb_query::msgpack_scan; +use nodedb_types::RowIdentity; use nodedb_types::columnar::{ColumnDef, StrictSchema}; use crate::bridge::envelope::{ErrorCode, Response}; @@ -125,6 +126,11 @@ impl CoreLoop { dropped_columns: Vec::new(), bitemporal: false, }; + let declared_primary_key = schema + .columns + .iter() + .find(|c| c.primary_key) + .map(|c| c.name.as_str()); // Scan all existing documents. let database_id = task.request.database_id.as_u64(); @@ -147,29 +153,72 @@ impl CoreLoop { for (doc_id, doc_bytes) in &docs { let normalized = sparse_body_to_msgpack(doc_bytes, source_format.as_format_ref()); - let identity = doc_id.to_identity(); - let with_id = msgpack_scan::inject_str_field(&normalized, "id", identity.as_str()); + // A row with no declared primary key has no client-visible identity + // yet: inject the surrogate's decimal string as `id` so the target + // schema's NOT NULL primary key column has something to validate. + let synth_id = doc_id.to_identity(); + let with_id = msgpack_scan::inject_str_field(&normalized, "id", synth_id.as_str()); + // The identity a user recognizes: the declared primary key's value, + // or `id`, read from the row itself — never the internal surrogate. + let identity = RowIdentity::of_stored_row(&normalized, declared_primary_key, *doc_id); let tuple_bytes = match super::super::strict_format::bytes_to_binary_tuple( &with_id, &schema, collection, ) { Ok(bytes) => bytes, - Err(e) => { + // A row carrying a field the target schema does not declare is + // a schema mismatch the client caused, not an internal fault: + // name the offending row and column as SQLSTATE 22000 + // (data_exception) instead of collapsing into a generic + // internal error. `ErrorCode::UndefinedColumn`, the Data + // Plane's other 42703-classified code, carries only a column + // name, with no room for the collection or the row that + // failed, so it would drop both here. + Err(crate::Error::UnknownStrictField { + collection, column, .. + }) => { return self.response_error( - task, - ErrorCode::Internal { - detail: format!( - "collection '{collection}': row '{identity}' failed to convert to document_strict: {e}" - ), - }, - ); + task, + ErrorCode::DataException { + detail: format!( + "column \"{column}\" of collection \"{collection}\" does not \ + exist (row \"{identity}\")" + ), + }, + ); } + Err(e) => return self.response_error(task, e), }; - if let Err(e) = self - .sparse - .put(database_id, tid, collection, doc_id, &tuple_bytes) - { + // The text index follows what the tuple stores: fields the schema + // does not declare leave the row, and their words leave the index. + let Some(stored_msgpack) = + super::super::strict_format::binary_tuple_to_msgpack(&tuple_bytes, &schema) + else { + let e = super::super::strict_format::undecodable_strict_row( + collection, + identity.as_str(), + ); + return self.response_error( + task, + ErrorCode::Internal { + detail: format!( + "collection '{collection}': row '{identity}' converted to a tuple \ + that does not decode: {e}" + ), + }, + ); + }; + if let Err(e) = self.put_converted_row( + ConvertedRow { + database_id, + tid, + collection, + doc_id, + }, + &tuple_bytes, + &stored_msgpack, + ) { return self.response_error( task, ErrorCode::Internal { @@ -181,7 +230,7 @@ impl CoreLoop { } // Write-through: a point-get after this statement must see the // re-encoded bytes, not a stale cache entry from before the - // conversion. `sparse.put` alone never touches this cache. + // conversion. The row write alone never touches this cache. self.doc_cache .put(database_id, tid, collection, doc_id, &tuple_bytes); converted += 1; @@ -237,27 +286,47 @@ impl CoreLoop { let converted = match source_format { SparseBodyFormat::Strict(schema) => { + let declared_primary_key = schema + .columns + .iter() + .find(|c| c.primary_key) + .map(|c| c.name.as_str()); let mut converted = 0u64; for (doc_id, doc_bytes) in &docs { - let identity = doc_id.to_identity(); + // Before decode, the row's own identity is unreadable — + // name it by the internal surrogate, the only handle a + // tuple that fails to decode has left. + let undecoded_identity = doc_id.to_identity(); let Some(mp) = super::super::strict_format::binary_tuple_to_msgpack(doc_bytes, &schema) else { let e = super::super::strict_format::undecodable_strict_row( collection, - identity.as_str(), + undecoded_identity.as_str(), ); return self.response_error( task, ErrorCode::Internal { detail: format!( - "collection '{collection}': row '{identity}' failed to convert to {target_type}: {e}" + "collection '{collection}': row '{undecoded_identity}' failed to convert to {target_type}: {e}" ), }, ); }; + // The identity a user recognizes: the declared primary + // key's value, or `id`, read from the decoded row. + let identity = RowIdentity::of_stored_row(&mp, declared_primary_key, *doc_id); - if let Err(e) = self.sparse.put(database_id, tid, collection, doc_id, &mp) { + if let Err(e) = self.put_converted_row( + ConvertedRow { + database_id, + tid, + collection, + doc_id, + }, + &mp, + &mp, + ) { return self.response_error( task, ErrorCode::Internal { @@ -269,7 +338,7 @@ impl CoreLoop { } // Write-through: a point-get after this statement must see // the re-encoded bytes, not a stale cache entry from before - // the conversion. `sparse.put` alone never touches this cache. + // the conversion. The row write alone never touches this cache. self.doc_cache .put(database_id, tid, collection, doc_id, &mp); converted += 1; @@ -295,3 +364,51 @@ impl CoreLoop { } } } + +/// Where one converted row lands. +struct ConvertedRow<'a> { + database_id: u64, + tid: u64, + collection: &'a str, + doc_id: &'a nodedb_types::StorageKey, +} + +impl CoreLoop { + /// Write one converted row and re-index its text from the converted + /// content, in one transaction. A conversion that drops fields also + /// drops their words from the full-text index. + /// + /// `body` is the stored form; `msgpack` is the same row as MessagePack, + /// which the text is extracted from. + fn put_converted_row( + &mut self, + row: ConvertedRow<'_>, + body: &[u8], + msgpack: &[u8], + ) -> crate::Result<()> { + let new_doc = crate::data::executor::doc_format::decode_document(msgpack)?; + let txn = self.sparse.begin_write()?; + self.sparse.put_in_txn( + &txn, + row.database_id, + row.tid, + row.collection, + row.doc_id, + body, + )?; + self.update_reindex_text( + &txn, + super::point::update_reindex_text::UpdateTextReindex { + database_id: row.database_id, + tid: row.tid, + collection: row.collection, + surrogate: row.doc_id.surrogate(), + new_doc: &new_doc, + }, + )?; + txn.commit().map_err(|e| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("convert commit: {e}"), + }) + } +} diff --git a/nodedb/src/data/executor/handlers/point/mod.rs b/nodedb/src/data/executor/handlers/point/mod.rs index a41e4dd98..5d1fe6785 100644 --- a/nodedb/src/data/executor/handlers/point/mod.rs +++ b/nodedb/src/data/executor/handlers/point/mod.rs @@ -18,4 +18,5 @@ pub mod update; pub mod update_reindex; pub mod update_reindex_secondary; pub mod update_reindex_sparse; +pub mod update_reindex_text; pub mod update_reindex_vector; diff --git a/nodedb/src/data/executor/handlers/point/update/persist.rs b/nodedb/src/data/executor/handlers/point/update/persist.rs index 690cecb2d..1f2b18562 100644 --- a/nodedb/src/data/executor/handlers/point/update/persist.rs +++ b/nodedb/src/data/executor/handlers/point/update/persist.rs @@ -11,7 +11,8 @@ //! collection with neither writes the body alone. Keeping the three side by //! side in one file is what makes it visible that only the last one is allowed //! to skip the diff, and that a body whose index diff cannot be computed must -//! fail rather than write alone. +//! fail rather than write alone. Every shape re-indexes the row's full-text +//! postings from the new body in the same transaction. //! //! All three run inside ONE transaction this function owns, and image-folding //! enforcement runs inside that same transaction before it commits. That is why @@ -223,6 +224,27 @@ impl CoreLoop { }; let touched = write_result?; + // The row's postings follow its new text in the same transaction. + // A registered collection's stored image must decode; an + // unregistered one indexes what `decode_document` reads, and a body + // it cannot read has no fields to index, as on the insert path. + let new_doc = match self.doc_configs.get(config_key) { + Some(cfg) => Some(self.decode_stored_document(cfg, updated_bytes)?), + None => crate::data::executor::doc_format::decode_document(updated_bytes).ok(), + }; + if let Some(new_doc) = &new_doc { + self.update_reindex_text( + &txn, + super::super::update_reindex_text::UpdateTextReindex { + database_id, + tid, + collection, + surrogate: storage_key.surrogate(), + new_doc, + }, + )?; + } + // Image-folding enforcement, inside the transaction the body just landed // in. Both images are STORED bytes: `current_bytes` came off the store // and `updated_bytes` was re-encoded in the collection's own mode, so a diff --git a/nodedb/src/data/executor/handlers/point/update_reindex_text.rs b/nodedb/src/data/executor/handlers/point/update_reindex_text.rs new file mode 100644 index 000000000..82a2c052e --- /dev/null +++ b/nodedb/src/data/executor/handlers/point/update_reindex_text.rs @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Full-text maintenance for UPDATE. +//! +//! An UPDATE rewrites a row's body in place, so the row keeps its surrogate +//! while its text changes. The inverted index must follow in the update's +//! own transaction: without it, the row keeps matching the words its old +//! body held and misses the words its new body holds. +//! +//! The text is extracted exactly as the insert path extracts it +//! (`fts_text::extract_fts_text`), so an updated row indexes the same way a +//! freshly inserted row with the same body does. `index_document_in_txn` +//! retracts the terms the new text no longer contains, and removes the row +//! from the index when the new text has no indexable word. + +use redb::WriteTransaction; + +use nodedb_types::Surrogate; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::fts_text::extract_fts_text; +use crate::engine::sparse::inverted::IndexDocScope; +use crate::types::TenantId; + +/// Inputs to [`CoreLoop::update_reindex_text`]. +pub(in crate::data::executor) struct UpdateTextReindex<'a> { + pub database_id: u64, + pub tid: u64, + pub collection: &'a str, + pub surrogate: Surrogate, + /// The row's post-update document. + pub new_doc: &'a serde_json::Value, +} + +impl CoreLoop { + /// Re-index the updated row's text inside `txn`. An error rejects the + /// update: `txn` must then be dropped un-committed, so the body and its + /// postings never disagree. + pub(in crate::data::executor) fn update_reindex_text( + &self, + txn: &WriteTransaction, + p: UpdateTextReindex<'_>, + ) -> crate::Result<()> { + let text = extract_fts_text(p.new_doc); + self.inverted + .index_document_in_txn( + txn, + IndexDocScope { + database_id: p.database_id, + tid: TenantId::new(p.tid), + collection: p.collection, + surrogate: p.surrogate, + }, + &text, + ) + .inspect_err(|e| { + crate::diag::fts_index_update_failed(e, p.collection, p.surrogate.as_u32()); + }) + } +} diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/mod.rs b/nodedb/src/data/executor/handlers/snapshot/restore/mod.rs index 437e45d6f..b702c373a 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore/mod.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore/mod.rs @@ -6,11 +6,13 @@ //! a full-tenant restore across every engine. `engines` holds the per-engine //! install helpers it calls (sparse/document, vector, KV, CRDT, timeseries). //! `keys` holds the snapshot-key parsing helpers shared across engines (and, -//! for the timeseries key parser, by `restore_segments.rs`). +//! for the timeseries key parser, by `restore_segments.rs`). `text` indexes +//! the restored rows' full-text postings. mod engines; mod keys; mod tenant_snapshot; +mod text; pub(in crate::data::executor) use keys::database_id_from_qualified; pub(in crate::data::executor::handlers::snapshot) use keys::parse_timeseries_snapshot_key; diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs b/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs index 270db83cd..f27ad9c6d 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs @@ -83,6 +83,15 @@ impl CoreLoop { ); } }; + // The snapshot carries no postings: index the restored rows' text. + if let Err(e) = self.restore_text_index(&snap) { + return self.response_error( + task, + ErrorCode::Internal { + detail: format!("restore: full-text reindex failed: {e}"), + }, + ); + } let mut edges_written = 0u64; let mut vectors_written = 0u64; diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/text.rs b/nodedb/src/data/executor/handlers/snapshot/restore/text.rs new file mode 100644 index 000000000..f1097f3d4 --- /dev/null +++ b/nodedb/src/data/executor/handlers/snapshot/restore/text.rs @@ -0,0 +1,244 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Full-text postings for restored documents. +//! +//! A snapshot carries document rows but no full-text postings, and the raw +//! row install bypasses the write path that indexes text. After the rows +//! land, every restored collection's current rows are re-indexed here from +//! their stored bodies, so a restored row is findable by its words and a +//! row the restore replaced no longer matches its old words. +//! +//! Each collection re-indexes in one write transaction, through the same +//! per-row helper an UPDATE uses. + +use std::collections::BTreeMap; + +use nodedb_types::StorageKey; + +use crate::data::executor::core_loop::CoreLoop; +use crate::data::executor::handlers::point::update_reindex_text::UpdateTextReindex; +use crate::engine::sparse::btree_versioned::VersionedScanParams; +use crate::types::{DatabaseId, TenantId}; + +/// `(database_id, tenant_id, collection)` of a restored document key. +/// +/// Plain keys are `{db}:{tenant}:{collection}:{doc_id}`; versioned keys append +/// `\x00{sys_from}`. A document id never holds `:`, so the collection is +/// everything between the tenant and the last `:`. +fn collection_of(key: &str) -> Option<(u64, u64, String)> { + let mut parts = key.splitn(3, ':'); + let db = parts.next()?.parse().ok()?; + let tid = parts.next()?.parse().ok()?; + let rest = parts.next()?.split('\x00').next()?; + let (collection, _doc_id) = rest.rsplit_once(':')?; + Some((db, tid, collection.to_string())) +} + +impl CoreLoop { + /// Re-index the text of every current row of every collection the + /// snapshot restored rows into. Returns the rows re-indexed. The first + /// failure fails the restore, as a failed row install does. + pub(super) fn restore_text_index( + &mut self, + snap: &crate::types::TenantDataSnapshot, + ) -> crate::Result { + // A collection is versioned when the snapshot carries versioned rows + // for it: its current rows live in the versioned table. + let mut collections: BTreeMap<(u64, u64, String), bool> = BTreeMap::new(); + let keys = snap + .documents + .iter() + .map(|(key, _)| (key.as_str(), false)) + .chain( + snap.documents_versioned + .iter() + .map(|(key, _)| (key.as_str(), true)), + ); + for (key, versioned) in keys { + let parsed = collection_of(key).ok_or_else(|| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("restore: document key {key:?} names no collection"), + })?; + *collections.entry(parsed).or_insert(false) |= versioned; + } + let mut reindexed = 0u64; + for ((db, tid, collection), versioned) in collections { + reindexed += self.reindex_collection_text(db, tid, &collection, versioned)?; + } + Ok(reindexed) + } + + /// Re-index the text of every current row of one collection in one + /// transaction. `versioned` reads the current version of each row from + /// the versioned table. + fn reindex_collection_text( + &mut self, + database_id: u64, + tid: u64, + collection: &str, + versioned: bool, + ) -> crate::Result { + let rows: Vec<(StorageKey, Vec)> = if versioned { + self.sparse.versioned_scan_as_of( + VersionedScanParams { + database_id, + tenant: tid, + coll: collection, + sys_cutoff_ms: None, + valid_at_ms: None, + limit: usize::MAX, + }, + &|_, _| true, + &crate::engine::sparse::scan_stop::never_stop, + )? + } else { + self.sparse + .scan_documents(database_id, tid, collection, usize::MAX)? + }; + let config_key = ( + DatabaseId::new(database_id), + TenantId::new(tid), + collection.to_string(), + ); + let txn = self.sparse.begin_write()?; + let mut reindexed = 0u64; + for (key, body) in &rows { + // A registered collection's stored rows must decode. An + // unregistered one indexes what `decode_document` reads, and a + // body it cannot read has no fields to index, as on the insert + // path. + let doc = match self.doc_configs.get(&config_key) { + Some(cfg) => Some(self.decode_stored_document(cfg, body)?), + None => crate::data::executor::doc_format::decode_document(body).ok(), + }; + let Some(doc) = doc else { + continue; + }; + self.update_reindex_text( + &txn, + UpdateTextReindex { + database_id, + tid, + collection, + surrogate: key.surrogate(), + new_doc: &doc, + }, + )?; + reindexed += 1; + } + txn.commit().map_err(|e| crate::Error::Storage { + engine: "sparse".into(), + detail: format!("restore text reindex commit ({collection}): {e}"), + })?; + Ok(reindexed) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + use nodedb_fts::FtsSearchParams; + use nodedb_fts::posting::QueryMode; + + use crate::data::executor::core_loop::tests::make_core_with_dir; + use crate::data::executor::doc_format; + use crate::engine::sparse::inverted::IndexDocScope; + + const DB: u64 = 0; + const TID: u64 = 1; + const COLL: &str = "restore_fts"; + + #[test] + fn plain_and_versioned_keys_name_their_collection() { + assert_eq!( + collection_of("0:7:docs:0000002a"), + Some((0, 7, "docs".to_string())) + ); + assert_eq!( + collection_of("3:7:docs:0000002a\x0000000000000000001234"), + Some((3, 7, "docs".to_string())) + ); + assert_eq!(collection_of("not-a-key"), None); + } + + fn searchable(core: &CoreLoop, term: &str) -> bool { + !core + .inverted + .search( + DB, + TenantId::new(TID), + COLL, + FtsSearchParams { + query: term, + top_k: 10, + fuzzy_enabled: false, + mode: QueryMode::And, + prefilter: None, + }, + ) + .expect("search") + .is_empty() + } + + /// A snapshot install writes a row's body straight into the sparse + /// store, bypassing the write path that indexes text. This proves + /// `restore_text_index` catches the row up: the text an old index entry + /// named is gone, and the row's restored text is findable with no + /// manual `REINDEX`. + #[test] + fn restore_text_index_replaces_postings_a_direct_row_install_left_stale() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut core, _req, _resp) = make_core_with_dir(dir.path()); + let surrogate = nodedb_types::Surrogate::new(9); + let key = StorageKey::for_surrogate(surrogate); + + // Index the row's OLD text through the normal indexing path, as a + // live write would have before the snapshot install below replaced + // the row underneath that index. + let txn = core.sparse.begin_write().expect("begin write"); + core.inverted + .index_document_in_txn( + &txn, + IndexDocScope { + database_id: DB, + tid: TenantId::new(TID), + collection: COLL, + surrogate, + }, + "alpha original", + ) + .expect("index old text"); + txn.commit().expect("commit old index"); + assert!(searchable(&core, "alpha"), "the old text must be indexed"); + + // The snapshot install writes the restored body straight into the + // store, exactly as `restore_sparse` does, with no FTS side effect. + let restored = serde_json::json!({"body": "beta replacement"}); + core.sparse + .put( + DB, + TID, + COLL, + &key, + &doc_format::encode_to_msgpack(&restored), + ) + .expect("install the restored row"); + + let snap = crate::types::TenantDataSnapshot { + documents: vec![(format!("{DB}:{TID}:{COLL}:{key}"), Vec::new())], + ..Default::default() + }; + let reindexed = core.restore_text_index(&snap).expect("restore text index"); + assert_eq!(reindexed, 1, "the one restored row must be reindexed"); + + assert!( + !searchable(&core, "alpha"), + "the pre-restore text must no longer match" + ); + assert!( + searchable(&core, "beta"), + "the restored row's text must be findable with no manual REINDEX" + ); + } +} diff --git a/nodedb/src/data/executor/handlers/update_from_join_write.rs b/nodedb/src/data/executor/handlers/update_from_join_write.rs index fb2e3af71..c647558cc 100644 --- a/nodedb/src/data/executor/handlers/update_from_join_write.rs +++ b/nodedb/src/data/executor/handlers/update_from_join_write.rs @@ -13,6 +13,7 @@ use crate::data::executor::enforcement::write_hook; use crate::data::executor::handlers::partial_refusal::{ refusal_after_partial_apply, refusal_after_rows, }; +use crate::data::executor::handlers::point::update_reindex_text::UpdateTextReindex; use crate::data::executor::handlers::point::update_reindex_vector::UpdateVectorReindex; use crate::data::executor::handlers::returning_doc; use crate::data::executor::handlers::transaction::stage_write::stored_row_identity; @@ -126,15 +127,35 @@ impl CoreLoop { Ok(txn) => txn, Err(e) => return Err(self.response_error(task, refusal_after_rows(affected, e))), }; - let stored = self.sparse.put_in_txn( - &row_txn, - database_id, - tid, - target_collection, - &storage_key, - &updated_bytes, - ); - if stored.is_ok() { + // The body and the row's full-text postings land together; a row + // whose write fails refuses the statement rather than drop out of + // its affected count. + let stored = self + .sparse + .put_in_txn( + &row_txn, + database_id, + tid, + target_collection, + &storage_key, + &updated_bytes, + ) + .and_then(|_prior| { + self.update_reindex_text( + &row_txn, + UpdateTextReindex { + database_id, + tid, + collection: target_collection, + surrogate: storage_key.surrogate(), + new_doc: &doc, + }, + ) + }); + if let Err(e) = stored { + return Err(self.response_error(task, refusal_after_rows(affected, e))); + } + { let enforcement = write_hook::run( self, &row_txn, diff --git a/nodedb/src/data/executor/vector_checkpoint/build_completions.rs b/nodedb/src/data/executor/vector_checkpoint/build_completions.rs index 241d759b7..9ffdd79cb 100644 --- a/nodedb/src/data/executor/vector_checkpoint/build_completions.rs +++ b/nodedb/src/data/executor/vector_checkpoint/build_completions.rs @@ -115,6 +115,7 @@ impl CoreLoop { tracing::info!( target: "nodedb::reindex", core = self.core_id, + index = "hnsw", key = %key, base_id, len, diff --git a/nodedb/src/diag/context/index_rebuild.rs b/nodedb/src/diag/context/index_rebuild.rs new file mode 100644 index 000000000..0a9980830 --- /dev/null +++ b/nodedb/src/diag/context/index_rebuild.rs @@ -0,0 +1,56 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Forensic payload for a REINDEX of a full-text or CSR index that did not +//! install. +//! +//! REINDEX CONCURRENTLY acknowledges once the rebuild starts. A rebuild the +//! core refuses at cutover, or one whose build fails, leaves the live index +//! in place with every write it took. Nothing but this report and a log +//! line shows that the requested rebuild never happened. + +use faultbox::DomainContext; +use faultbox::serde_json::{Value, json}; + +/// One rebuild of one index that did not install. +pub(in crate::diag) struct IndexRebuildNotInstalled<'a> { + /// `fts` or `csr`. + pub index: &'static str, + pub database_id: u64, + pub tenant_id: u64, + pub collection: &'a str, + /// What stopped the install, without the per-occurrence detail. + pub cause_class: &'a str, +} + +impl DomainContext for IndexRebuildNotInstalled<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb.index_rebuild_not_installed" + } + + fn grouping_key(&self) -> String { + // The index kind and the cause name the bug. The collection is the + // occurrence, so repeated REINDEX runs that hit one cause file one + // report. + format!("index={};cause={}", self.index, self.cause_class) + } + + fn to_json(&self) -> Value { + json!({ + "index": self.index, + "database_id": self.database_id, + "tenant_id": self.tenant_id, + "collection": self.collection, + "cause_class": self.cause_class, + "why_reported": "REINDEX acknowledged this rebuild when it started. The \ + rebuilt index was discarded, so the live index stays as it \ + was, with every write it took. Results stay correct; the \ + compaction the rebuild was meant to install never happened", + "operator_action": "read the cause: a journal overflow means more writes landed \ + during the rebuild than its journal holds, so run REINDEX \ + again at a quieter time. A purge or supersede means the \ + collection was dropped or rebuilt again, and needs no action. \ + Any other cause is a storage or snapshot error: check the \ + core's log for the same collection", + }) + } +} diff --git a/nodedb/src/diag/context/mod.rs b/nodedb/src/diag/context/mod.rs index 57c585e03..a33246f6e 100644 --- a/nodedb/src/diag/context/mod.rs +++ b/nodedb/src/diag/context/mod.rs @@ -9,6 +9,7 @@ mod catalog; mod crdt; mod data_plane; +mod index_rebuild; mod ingest; mod outcome_floor; mod quota; @@ -28,6 +29,7 @@ pub use data_plane::LostResponseWrite; pub(in crate::diag) use data_plane::{ CalvinApplyHalted, CalvinCompletionTimeout, CoreFailStopped, DataPlaneResponseLost, }; +pub(in crate::diag) use index_rebuild::IndexRebuildNotInstalled; pub(in crate::diag) use ingest::IlpAcceptedLinesDropped; pub use ingest::IlpFlushOutcome; pub(in crate::diag) use outcome_floor::{WriteWindowHeld, WriteWindowLeaked}; diff --git a/nodedb/src/diag/mod.rs b/nodedb/src/diag/mod.rs index 9dc9bc612..918afd60d 100644 --- a/nodedb/src/diag/mod.rs +++ b/nodedb/src/diag/mod.rs @@ -10,11 +10,11 @@ mod recording; pub use context::{DATABASE_SCOPE, IlpFlushOutcome, LostResponseWrite, TENANT_SCOPE}; pub use recording::{ - VectorBuildTarget, batch_insert_without_surrogates, calvin_apply_halted, + IndexRebuildTarget, VectorBuildTarget, batch_insert_without_surrogates, calvin_apply_halted, calvin_completion_timeout, catalog_apply_orphan_row, collection_purge_row_missing, consumer_group_offsets_retained, data_plane_core_fail_stopped, data_plane_response_lost, data_plane_responses_lost, entry_kind, fts_index_update_failed, history_compaction_not_applied, - ilp_invalid_utf8_drop, ilp_line_read_drop, metadata_apply_wedged, + ilp_invalid_utf8_drop, ilp_line_read_drop, index_rebuild_not_installed, metadata_apply_wedged, orphaned_index_entry_after_delete, quota_row_invalid, quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, quota_scope_replay_aborted, raft_entries_reapplied, raft_entry_reapplied, replay_record_unapplied, replicated_write_parked, diff --git a/nodedb/src/diag/recording/index_rebuild.rs b/nodedb/src/diag/recording/index_rebuild.rs new file mode 100644 index 000000000..774c5ffe5 --- /dev/null +++ b/nodedb/src/diag/recording/index_rebuild.rs @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capture site for a full-text or CSR rebuild that did not install. + +use faultbox::{Capture, EventKind, error_chain_of}; + +use super::shared::error_class; +use crate::diag::context; + +/// Where a rebuild belongs: the index kind and the collection it covered. +pub struct IndexRebuildTarget<'a> { + /// `fts` or `csr`. + pub index: &'static str, + pub database_id: u64, + pub tenant_id: u64, + pub collection: &'a str, +} + +/// Report a rebuild the core discarded. Called from the cutover arm that +/// receives the refusal or the build error. +pub fn index_rebuild_not_installed( + err: &(dyn std::error::Error + 'static), + target: &IndexRebuildTarget<'_>, +) { + let class = error_class(err); + let ctx = context::IndexRebuildNotInstalled { + index: target.index, + database_id: target.database_id, + tenant_id: target.tenant_id, + collection: target.collection, + cause_class: &class, + }; + let _ = Capture::new( + EventKind::Error, + "index rebuild did not install; the live index stays in place", + ) + .error_chain(error_chain_of(err)) + .domain(&ctx) + .emit(); +} diff --git a/nodedb/src/diag/recording/mod.rs b/nodedb/src/diag/recording/mod.rs index 250bbdcd4..8a14ff74f 100644 --- a/nodedb/src/diag/recording/mod.rs +++ b/nodedb/src/diag/recording/mod.rs @@ -11,6 +11,7 @@ mod catalog; mod crdt; mod data_plane; +mod index_rebuild; mod ingest; mod outcome_floor; mod quota; @@ -30,6 +31,7 @@ pub use data_plane::{ calvin_apply_halted, calvin_completion_timeout, data_plane_core_fail_stopped, data_plane_response_lost, data_plane_responses_lost, }; +pub use index_rebuild::{IndexRebuildTarget, index_rebuild_not_installed}; pub use ingest::{ilp_invalid_utf8_drop, ilp_line_read_drop}; pub use outcome_floor::{write_window_held, write_window_leaked}; pub use quota::{ diff --git a/nodedb/src/engine/sparse/inverted/core.rs b/nodedb/src/engine/sparse/inverted/core.rs index a401584e6..84d5fab4e 100644 --- a/nodedb/src/engine/sparse/inverted/core.rs +++ b/nodedb/src/engine/sparse/inverted/core.rs @@ -4,6 +4,7 @@ //! tenant/collection purge. All other concerns (indexing, search, //! synonyms, compaction) live in sibling modules. +use std::cell::RefCell; use std::sync::Arc; use redb::Database; @@ -12,12 +13,15 @@ use nodedb_mem::MemoryGovernor; use nodedb_types::TenantId; use super::errors::into_result_err; +use super::rebuild_journal::FtsJournals; use crate::engine::sparse::fts_redb::RedbFtsBackend; use crate::storage::quarantine::QuarantineRegistry; /// Full-text inverted index backed by redb via `nodedb-fts`. pub struct InvertedIndex { pub(super) inner: nodedb_fts::index::FtsIndex, + /// Write journals of the collection rebuilds running on this index. + pub(super) journals: RefCell, } impl InvertedIndex { @@ -27,6 +31,7 @@ impl InvertedIndex { let backend = RedbFtsBackend::open(db)?; Ok(Self { inner: nodedb_fts::index::FtsIndex::new(backend, governor), + journals: RefCell::new(FtsJournals::default()), }) } @@ -53,6 +58,7 @@ impl InvertedIndex { /// Purge all inverted index entries for a `(database, tenant)`. Structural /// drop via tuple ranges on every FTS table. pub fn purge_tenant(&self, database_id: u64, tid: TenantId) -> crate::Result { + self.note_purge(database_id, tid.as_u64(), None); self.inner .purge_tenant(database_id, tid.as_u64()) .map_err(into_result_err) @@ -67,6 +73,7 @@ impl InvertedIndex { tid: TenantId, collection: &str, ) -> crate::Result { + self.note_purge(database_id, tid.as_u64(), Some(collection)); self.inner .purge_collection(database_id, tid.as_u64(), collection) .map_err(into_result_err) diff --git a/nodedb/src/engine/sparse/inverted/doc_image.rs b/nodedb/src/engine/sparse/inverted/doc_image.rs index 245cff996..98e22951d 100644 --- a/nodedb/src/engine/sparse/inverted/doc_image.rs +++ b/nodedb/src/engine/sparse/inverted/doc_image.rs @@ -26,6 +26,13 @@ pub struct FtsDocImage { tokens: Vec, } +impl FtsDocImage { + /// The analyzed token stream, in document order. + pub(super) fn tokens(&self) -> &[String] { + &self.tokens + } +} + impl InvertedIndex { /// The index footprint of `surrogate` in `collection`, or `None` when the /// document is not indexed. Reads only: the write transaction it opens to @@ -78,7 +85,7 @@ impl InvertedIndex { Ok(()) } - fn read_document_image( + pub(super) fn read_document_image( txn: &redb::WriteTransaction, scope: IndexDocScope<'_>, ) -> crate::Result> { diff --git a/nodedb/src/engine/sparse/inverted/indexing.rs b/nodedb/src/engine/sparse/inverted/indexing.rs index ebf1a5701..e8a853f25 100644 --- a/nodedb/src/engine/sparse/inverted/indexing.rs +++ b/nodedb/src/engine/sparse/inverted/indexing.rs @@ -153,6 +153,7 @@ impl InvertedIndex { surrogate, } = scope; let t = tid.as_u64(); + self.note_doc_write(scope); let mut term_postings: HashMap<&str, (u32, Vec)> = HashMap::new(); for (pos, token) in tokens.iter().enumerate() { diff --git a/nodedb/src/engine/sparse/inverted/mod.rs b/nodedb/src/engine/sparse/inverted/mod.rs index aa47293cb..c1d719fc4 100644 --- a/nodedb/src/engine/sparse/inverted/mod.rs +++ b/nodedb/src/engine/sparse/inverted/mod.rs @@ -20,6 +20,9 @@ mod doc_image; mod doc_terms; mod errors; mod indexing; +mod rebuild_install; +mod rebuild_journal; +mod rebuild_snapshot; mod removal; mod search; mod synonyms; @@ -29,4 +32,7 @@ pub use doc_image::FtsDocImage; pub use indexing::IndexDocScope; pub use nodedb_fts::FtsSearchParams; pub use nodedb_fts::posting::{MatchOffset, Posting, QueryMode, TextSearchResult}; +pub use rebuild_install::{FtsInstallOutcome, FtsRebuildRefusal}; +pub use rebuild_journal::FTS_REBUILD_JOURNAL_MAX_DOCS; +pub use rebuild_snapshot::{FtsRebuildTicket, FtsRebuilt, FtsSnapshot}; pub use search::PhraseSearchParams; diff --git a/nodedb/src/engine/sparse/inverted/rebuild_install.rs b/nodedb/src/engine/sparse/inverted/rebuild_install.rs new file mode 100644 index 000000000..d80d01e15 --- /dev/null +++ b/nodedb/src/engine/sparse/inverted/rebuild_install.rs @@ -0,0 +1,399 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The on-core half of a collection rebuild: the atomic cutover. +//! +//! One redb write transaction: +//! +//! 1. reads the live footprint of every document the journal noted, +//! 2. replaces the collection's `POSTINGS`, `DOC_LENGTHS`, `DOC_TERMS` and +//! `STATS` rows with the rebuilt ones, +//! 3. writes each noted document's live footprint over the rebuilt rows, +//! or removes the document when it is no longer indexed. +//! +//! Readers see the transaction's state before or after the commit, never a +//! mix. `INDEX_META` (analyzer, fuzzy flag, synonyms) and `SEGMENTS` are +//! left alone. + +use redb::{ReadableTable as _, WriteTransaction}; + +use nodedb_types::{Surrogate, TenantId}; + +use super::core::InvertedIndex; +use super::doc_image::FtsDocImage; +use super::errors::inverted_err; +use super::indexing::IndexDocScope; +use super::rebuild_journal::JournalState; +use super::rebuild_snapshot::FtsRebuilt; +use crate::engine::sparse::fts_redb::tables::{DOC_LENGTHS, DOC_TERMS, POSTINGS, STATS}; + +/// Upper bound for the `term` component of a posting range scan. +const MAX_TERM: &str = "\u{10ffff}"; + +/// Why a cutover left the live index as it is. +#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] +pub enum FtsRebuildRefusal { + /// The journal of this rebuild is gone: it was aborted, or a newer + /// rebuild of the collection replaced it. + #[error("the rebuild's journal is closed; a newer rebuild or an abort replaced it")] + Superseded, + /// The collection or its tenant was purged during the rebuild. + #[error("the collection was purged during the rebuild")] + Purged, + /// More distinct documents were written during the rebuild than the + /// journal holds. + #[error( + "more than {max_docs} documents were written during the rebuild; \ + run REINDEX again" + )] + JournalOverflow { max_docs: usize }, +} + +/// What a cutover did. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum FtsInstallOutcome { + /// The rebuilt rows are live. + Installed { + /// Terms with a posting list. + terms: usize, + /// Documents in the rebuilt rows before replay. + docs: usize, + /// Documents written during the rebuild and replayed. + replayed: usize, + }, + /// The live index is unchanged. + Refused(FtsRebuildRefusal), +} + +impl InvertedIndex { + /// Install `rebuilt` over the live rows of its collection, with every + /// write the journal noted replayed on top, in one transaction. + /// + /// Closes the journal in every case. A refusal or an error leaves the + /// live index as it is. + pub fn install_rebuild(&self, rebuilt: FtsRebuilt) -> crate::Result { + let Some(journal) = self.take_journal(rebuilt.token) else { + return Ok(FtsInstallOutcome::Refused(FtsRebuildRefusal::Superseded)); + }; + match journal.state { + JournalState::Recording => {} + JournalState::Purged => { + return Ok(FtsInstallOutcome::Refused(FtsRebuildRefusal::Purged)); + } + JournalState::Overflowed => { + return Ok(FtsInstallOutcome::Refused( + FtsRebuildRefusal::JournalOverflow { + max_docs: journal.max_docs, + }, + )); + } + } + let mut touched: Vec = journal.touched.into_iter().collect(); + touched.sort_unstable(); + + let tid = TenantId::new(rebuilt.tid); + let scope_of = |surrogate: u32| IndexDocScope { + database_id: rebuilt.database_id, + tid, + collection: &rebuilt.collection, + surrogate: Surrogate::new(surrogate), + }; + + let db = self.inner.backend().db(); + let txn = db + .begin_write() + .map_err(|e| inverted_err("rebuild cutover txn", e))?; + + let mut live: Vec<(u32, Option)> = Vec::with_capacity(touched.len()); + for &surrogate in &touched { + live.push(( + surrogate, + Self::read_document_image(&txn, scope_of(surrogate))?, + )); + } + + clear_collection_rows(&txn, rebuilt.database_id, rebuilt.tid, &rebuilt.collection)?; + write_rebuilt_rows(&txn, &rebuilt)?; + + for (surrogate, image) in &live { + match image { + Some(image) => self.write_index_data(&txn, scope_of(*surrogate), image.tokens())?, + None => self.remove_document_in_txn(&txn, scope_of(*surrogate))?, + } + } + + txn.commit() + .map_err(|e| inverted_err("rebuild cutover commit", e))?; + Ok(FtsInstallOutcome::Installed { + terms: rebuilt.postings.len(), + docs: rebuilt.doc_lengths.len(), + replayed: live.len(), + }) + } +} + +/// Remove the collection's rows from every table the rebuild owns. +fn clear_collection_rows( + txn: &WriteTransaction, + database_id: u64, + tid: u64, + collection: &str, +) -> crate::Result<()> { + { + let mut table = txn + .open_table(POSTINGS) + .map_err(|e| inverted_err("rebuild open postings", e))?; + let terms: Vec = table + .range((database_id, tid, collection, "")..=(database_id, tid, collection, MAX_TERM)) + .map_err(|e| inverted_err("rebuild postings range", e))? + .map(|entry| entry.map(|(k, _)| k.value().3.to_string())) + .collect::>() + .map_err(|e| inverted_err("rebuild postings entry", e))?; + for term in &terms { + table + .remove((database_id, tid, collection, term.as_str())) + .map_err(|e| inverted_err("rebuild remove postings", e))?; + } + } + for (def, name) in [(DOC_LENGTHS, "doc_lengths"), (DOC_TERMS, "doc_terms")] { + let mut table = txn + .open_table(def) + .map_err(|e| inverted_err(&format!("rebuild open {name}"), e))?; + let docs: Vec = table + .range((database_id, tid, collection, 0u32)..=(database_id, tid, collection, u32::MAX)) + .map_err(|e| inverted_err(&format!("rebuild {name} range"), e))? + .map(|entry| entry.map(|(k, _)| k.value().3)) + .collect::>() + .map_err(|e| inverted_err(&format!("rebuild {name} entry"), e))?; + for doc in docs { + table + .remove((database_id, tid, collection, doc)) + .map_err(|e| inverted_err(&format!("rebuild remove {name}"), e))?; + } + } + let mut stats = txn + .open_table(STATS) + .map_err(|e| inverted_err("rebuild open stats", e))?; + stats + .remove((database_id, tid, collection)) + .map_err(|e| inverted_err("rebuild remove stats", e))?; + Ok(()) +} + +/// Write the rebuilt rows of the collection. +fn write_rebuilt_rows(txn: &WriteTransaction, rebuilt: &FtsRebuilt) -> crate::Result<()> { + let db = rebuilt.database_id; + let t = rebuilt.tid; + let coll = rebuilt.collection.as_str(); + { + let mut table = txn + .open_table(POSTINGS) + .map_err(|e| inverted_err("rebuild open postings", e))?; + for (term, list) in &rebuilt.postings { + let bytes = zerompk::to_msgpack_vec(list) + .map_err(|e| inverted_err("rebuild serialize postings", e))?; + table + .insert((db, t, coll, term.as_str()), bytes.as_slice()) + .map_err(|e| inverted_err("rebuild insert postings", e))?; + } + } + { + let mut table = txn + .open_table(DOC_LENGTHS) + .map_err(|e| inverted_err("rebuild open doc_lengths", e))?; + for &(doc, len) in &rebuilt.doc_lengths { + let bytes = zerompk::to_msgpack_vec(&len) + .map_err(|e| inverted_err("rebuild serialize doc_length", e))?; + table + .insert((db, t, coll, doc), bytes.as_slice()) + .map_err(|e| inverted_err("rebuild insert doc_length", e))?; + } + } + { + let mut table = txn + .open_table(DOC_TERMS) + .map_err(|e| inverted_err("rebuild open doc_terms", e))?; + for (doc, terms) in &rebuilt.doc_terms { + let bytes = zerompk::to_msgpack_vec(terms) + .map_err(|e| inverted_err("rebuild serialize doc_terms", e))?; + table + .insert((db, t, coll, *doc), bytes.as_slice()) + .map_err(|e| inverted_err("rebuild insert doc_terms", e))?; + } + } + let mut stats = txn + .open_table(STATS) + .map_err(|e| inverted_err("rebuild open stats", e))?; + let bytes = zerompk::to_msgpack_vec(&(rebuilt.doc_count, rebuilt.total_tokens)) + .map_err(|e| inverted_err("rebuild serialize stats", e))?; + stats + .insert((db, t, coll), bytes.as_slice()) + .map_err(|e| inverted_err("rebuild insert stats", e))?; + Ok(()) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use redb::{Database, ReadableDatabase as _}; + + use nodedb_fts::FtsSearchParams; + use nodedb_fts::posting::QueryMode; + + use super::*; + use crate::engine::sparse::inverted::FTS_REBUILD_JOURNAL_MAX_DOCS; + + const DB: u64 = 0; + const T: TenantId = TenantId::new(1); + + fn open_temp() -> (InvertedIndex, tempfile::TempDir) { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("test-inverted.redb"); + let db = Arc::new(Database::create(&path).unwrap()); + let idx = + InvertedIndex::open(db, crate::data::executor::core_loop::test_governor()).unwrap(); + (idx, dir) + } + + fn hits(idx: &InvertedIndex, query: &str) -> Vec { + let mut ids: Vec = idx + .search( + DB, + T, + "docs", + FtsSearchParams { + query, + top_k: 100, + fuzzy_enabled: false, + mode: QueryMode::And, + prefilter: None, + }, + ) + .unwrap() + .into_iter() + .map(|r| r.doc_id.as_u32()) + .collect(); + ids.sort_unstable(); + ids + } + + fn rebuild_with(idx: &InvertedIndex, during: impl FnOnce(&InvertedIndex)) -> FtsInstallOutcome { + let ticket = idx + .begin_rebuild(DB, T, "docs", FTS_REBUILD_JOURNAL_MAX_DOCS) + .unwrap(); + during(idx); + let rebuilt = ticket.read().unwrap().compact(); + idx.install_rebuild(rebuilt).unwrap() + } + + #[test] + fn writes_during_the_rebuild_survive_the_cutover() { + let (idx, _dir) = open_temp(); + for i in 1..=3u32 { + idx.index_document(DB, T, "docs", Surrogate::new(i), "alpha") + .unwrap(); + } + let outcome = rebuild_with(&idx, |idx| { + idx.index_document(DB, T, "docs", Surrogate::new(4), "alpha") + .unwrap(); + idx.index_document(DB, T, "docs", Surrogate::new(1), "beta") + .unwrap(); + idx.remove_document(DB, T, "docs", Surrogate::new(2)) + .unwrap(); + }); + assert!(matches!( + outcome, + FtsInstallOutcome::Installed { replayed: 3, .. } + )); + assert_eq!(hits(&idx, "alpha"), vec![3, 4]); + assert_eq!(hits(&idx, "beta"), vec![1]); + let (count, avg_len) = idx.corpus_stats(DB, T, "docs").unwrap(); + assert_eq!(count, 3); + assert_eq!(avg_len, 1.0); + } + + #[test] + fn a_term_dropped_during_the_rebuild_leaves_its_posting_list() { + let (idx, _dir) = open_temp(); + idx.index_document(DB, T, "docs", Surrogate::new(1), "alpha bravo") + .unwrap(); + idx.index_document(DB, T, "docs", Surrogate::new(2), "alpha") + .unwrap(); + let outcome = rebuild_with(&idx, |idx| { + // The snapshot holds document 1 under `alpha`; the update drops it. + idx.index_document(DB, T, "docs", Surrogate::new(1), "bravo charlie") + .unwrap(); + }); + assert!(matches!( + outcome, + FtsInstallOutcome::Installed { replayed: 1, .. } + )); + let alpha = idx + .backend() + .db() + .begin_read() + .unwrap() + .open_table(POSTINGS) + .unwrap() + .get((DB, T.as_u64(), "docs", "alpha")) + .unwrap() + .map(|v| zerompk::from_msgpack::>(v.value()).unwrap()) + .unwrap_or_default(); + assert_eq!( + alpha.iter().map(|p| p.doc_id.as_u32()).collect::>(), + vec![2], + "the rebuilt `alpha` list must not keep the updated document" + ); + assert_eq!(idx.term_df(DB, T, "docs", "alpha").unwrap(), 1); + assert_eq!(hits(&idx, "alpha"), vec![2]); + assert_eq!(hits(&idx, "charlie"), vec![1]); + let (count, avg_len) = idx.corpus_stats(DB, T, "docs").unwrap(); + assert_eq!(count, 2, "stats count each document once"); + assert_eq!(avg_len, 1.5, "(2 + 1) tokens over 2 documents"); + } + + #[test] + fn a_purge_during_the_rebuild_refuses_the_cutover() { + let (idx, _dir) = open_temp(); + idx.index_document(DB, T, "docs", Surrogate::new(1), "alpha") + .unwrap(); + let outcome = rebuild_with(&idx, |idx| { + idx.purge_collection(DB, T, "docs").unwrap(); + }); + assert_eq!( + outcome, + FtsInstallOutcome::Refused(FtsRebuildRefusal::Purged) + ); + assert!(hits(&idx, "alpha").is_empty(), "the purge is not undone"); + } + + #[test] + fn journal_overflow_refuses_the_cutover_and_keeps_live_writes() { + let (idx, _dir) = open_temp(); + let ticket = idx.begin_rebuild(DB, T, "docs", 1).unwrap(); + idx.index_document(DB, T, "docs", Surrogate::new(1), "alpha") + .unwrap(); + idx.index_document(DB, T, "docs", Surrogate::new(2), "alpha") + .unwrap(); + let outcome = idx + .install_rebuild(ticket.read().unwrap().compact()) + .unwrap(); + assert_eq!( + outcome, + FtsInstallOutcome::Refused(FtsRebuildRefusal::JournalOverflow { max_docs: 1 }) + ); + assert_eq!(hits(&idx, "alpha"), vec![1, 2]); + } + + #[test] + fn a_second_rebuild_of_the_collection_is_refused_while_one_runs() { + let (idx, _dir) = open_temp(); + let _ticket = idx + .begin_rebuild(DB, T, "docs", FTS_REBUILD_JOURNAL_MAX_DOCS) + .unwrap(); + assert!( + idx.begin_rebuild(DB, T, "docs", FTS_REBUILD_JOURNAL_MAX_DOCS) + .is_err() + ); + } +} diff --git a/nodedb/src/engine/sparse/inverted/rebuild_journal.rs b/nodedb/src/engine/sparse/inverted/rebuild_journal.rs new file mode 100644 index 000000000..d0024a104 --- /dev/null +++ b/nodedb/src/engine/sparse/inverted/rebuild_journal.rs @@ -0,0 +1,248 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Write journals for collection rebuilds of the inverted index. +//! +//! A rebuild reads a collection's index from a redb read snapshot on +//! another thread. Every document write to the collection after that +//! snapshot notes the document's surrogate here. At cutover the owning +//! core reads each noted document's live footprint and writes it over the +//! rebuilt state in the same transaction, so the rebuilt index holds every +//! write the live index took. +//! +//! Noting a surrogate whose write later aborts is harmless: the cutover +//! reads the live footprint, which the abort left unchanged. +//! +//! A journal holds at most `max_docs` distinct surrogates. Past the bound +//! it drops them, frees their memory and marks itself overflowed. The +//! cutover then refuses the rebuilt state. The live index keeps every write. + +use std::collections::HashSet; + +use redb::{ReadableDatabase as _, ReadableTable as _}; + +use nodedb_types::TenantId; + +use super::core::InvertedIndex; +use super::errors::inverted_err; +use super::indexing::IndexDocScope; +use super::rebuild_snapshot::FtsRebuildTicket; +use crate::engine::sparse::fts_redb::tables::{DOC_LENGTHS, POSTINGS}; + +/// Upper bound for the `term` component of a posting range scan. +const MAX_TERM: &str = "\u{10ffff}"; + +/// Default bound on the distinct documents one rebuild journal records. +pub const FTS_REBUILD_JOURNAL_MAX_DOCS: usize = 1 << 20; + +/// Where a journal stands. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum JournalState { + Recording, + /// More than `max_docs` distinct documents were written. + Overflowed, + /// The collection or its tenant was purged. + Purged, +} + +/// Documents written to one collection since its rebuild snapshot. +#[derive(Debug)] +pub(super) struct FtsJournal { + pub(super) token: u64, + database_id: u64, + tid: u64, + collection: String, + pub(super) max_docs: usize, + pub(super) touched: HashSet, + pub(super) state: JournalState, +} + +impl FtsJournal { + fn is_for(&self, database_id: u64, tid: u64, collection: &str) -> bool { + self.database_id == database_id && self.tid == tid && self.collection == collection + } + + fn note(&mut self, surrogate: u32) { + if self.state != JournalState::Recording { + return; + } + if self.touched.len() >= self.max_docs && !self.touched.contains(&surrogate) { + self.state = JournalState::Overflowed; + self.touched = HashSet::new(); + return; + } + self.touched.insert(surrogate); + } +} + +/// The open journals of one inverted index. +#[derive(Debug, Default)] +pub(super) struct FtsJournals { + next_token: u64, + active: Vec, +} + +impl InvertedIndex { + /// Open a rebuild of `collection`: pin a read snapshot and start its + /// write journal in the same step. + /// + /// The returned ticket reads the snapshot on any thread. Errors when a + /// rebuild of the collection already runs, or when redb cannot open the + /// snapshot. + pub fn begin_rebuild( + &self, + database_id: u64, + tid: TenantId, + collection: &str, + max_docs: usize, + ) -> crate::Result { + let t = tid.as_u64(); + let mut journals = self.journals.borrow_mut(); + if journals + .active + .iter() + .any(|j| j.is_for(database_id, t, collection)) + { + return Err(crate::Error::ObjectNotInPrerequisiteState { + object: format!("full-text index of collection \"{collection}\""), + detail: "a rebuild of it is already running".to_string(), + }); + } + let txn = self + .inner + .backend() + .db() + .begin_read() + .map_err(|e| inverted_err("rebuild snapshot", e))?; + journals.next_token += 1; + let token = journals.next_token; + journals.active.push(FtsJournal { + token, + database_id, + tid: t, + collection: collection.to_string(), + max_docs, + touched: HashSet::new(), + state: JournalState::Recording, + }); + Ok(FtsRebuildTicket::new( + token, + database_id, + t, + collection.to_string(), + txn, + )) + } + + /// Whether `collection` holds any posting or document-length row, i.e. + /// whether it has a full-text index to rebuild. + pub fn has_collection_rows( + &self, + database_id: u64, + tid: TenantId, + collection: &str, + ) -> crate::Result { + let t = tid.as_u64(); + let txn = self + .inner + .backend() + .db() + .begin_read() + .map_err(|e| inverted_err("rebuild probe txn", e))?; + let postings = txn + .open_table(POSTINGS) + .map_err(|e| inverted_err("rebuild probe postings", e))?; + if postings + .range((database_id, t, collection, "")..=(database_id, t, collection, MAX_TERM)) + .map_err(|e| inverted_err("rebuild probe postings range", e))? + .next() + .is_some() + { + return Ok(true); + } + let lengths = txn + .open_table(DOC_LENGTHS) + .map_err(|e| inverted_err("rebuild probe doc_lengths", e))?; + let any = lengths + .range((database_id, t, collection, 0u32)..=(database_id, t, collection, u32::MAX)) + .map_err(|e| inverted_err("rebuild probe doc_lengths range", e))? + .next() + .is_some(); + Ok(any) + } + + /// Close the journal of rebuild `token`. The index is unchanged. + pub fn abort_rebuild(&self, token: u64) { + self.take_journal(token); + } + + /// Remove and return the journal of rebuild `token`. + pub(super) fn take_journal(&self, token: u64) -> Option { + let mut journals = self.journals.borrow_mut(); + let pos = journals.active.iter().position(|j| j.token == token)?; + Some(journals.active.swap_remove(pos)) + } + + /// Note a write of the document `scope` names. + pub(super) fn note_doc_write(&self, scope: IndexDocScope<'_>) { + let mut journals = self.journals.borrow_mut(); + let t = scope.tid.as_u64(); + for journal in journals + .active + .iter_mut() + .filter(|j| j.is_for(scope.database_id, t, scope.collection)) + { + journal.note(scope.surrogate.as_u32()); + } + } + + /// Mark the journals a purge invalidates: one collection, or every + /// collection of the tenant when `collection` is `None`. + pub(super) fn note_purge(&self, database_id: u64, tid: u64, collection: Option<&str>) { + let mut journals = self.journals.borrow_mut(); + for journal in journals.active.iter_mut().filter(|j| { + j.database_id == database_id + && j.tid == tid + && collection.is_none_or(|c| j.collection == c) + }) { + journal.state = JournalState::Purged; + journal.touched = HashSet::new(); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn journal(max_docs: usize) -> FtsJournal { + FtsJournal { + token: 1, + database_id: 0, + tid: 1, + collection: "docs".to_string(), + max_docs, + touched: HashSet::new(), + state: JournalState::Recording, + } + } + + #[test] + fn a_repeat_write_does_not_count_twice() { + let mut j = journal(2); + j.note(1); + j.note(1); + j.note(2); + assert_eq!(j.state, JournalState::Recording); + assert_eq!(j.touched.len(), 2); + } + + #[test] + fn a_write_past_the_bound_overflows_and_frees_the_set() { + let mut j = journal(2); + j.note(1); + j.note(2); + j.note(3); + assert_eq!(j.state, JournalState::Overflowed); + assert!(j.touched.is_empty()); + } +} diff --git a/nodedb/src/engine/sparse/inverted/rebuild_snapshot.rs b/nodedb/src/engine/sparse/inverted/rebuild_snapshot.rs new file mode 100644 index 000000000..226389819 --- /dev/null +++ b/nodedb/src/engine/sparse/inverted/rebuild_snapshot.rs @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The off-core half of a collection rebuild: read the pinned snapshot and +//! derive the collection's canonical index rows from it. +//! +//! Every type here is `Send` and touches no `!Send` engine state. The redb +//! read transaction was opened on the owning core by +//! `InvertedIndex::begin_rebuild`, in the same step that opened the write +//! journal, so the snapshot and the journal meet with no gap. + +use std::collections::{BTreeMap, BTreeSet}; + +use redb::{ReadTransaction, ReadableTable as _}; + +use nodedb_fts::posting::Posting; + +use super::errors::inverted_err; +use crate::engine::sparse::fts_redb::tables::{DOC_LENGTHS, POSTINGS}; + +/// Upper bound for the `term` component of a posting range scan. +const MAX_TERM: &str = "\u{10ffff}"; + +/// A pinned snapshot of one collection's index, ready to read. +pub struct FtsRebuildTicket { + token: u64, + database_id: u64, + tid: u64, + collection: String, + txn: ReadTransaction, +} + +/// One collection's postings and document lengths as the snapshot holds them. +pub struct FtsSnapshot { + token: u64, + database_id: u64, + tid: u64, + collection: String, + postings: Vec<(String, Vec)>, + doc_lengths: Vec<(u32, u32)>, +} + +/// The canonical index rows of one collection, derived from a snapshot. +pub struct FtsRebuilt { + pub(super) token: u64, + pub(super) database_id: u64, + pub(super) tid: u64, + pub(super) collection: String, + /// One list per term, one posting per document, ordered by surrogate. + pub(super) postings: Vec<(String, Vec)>, + /// `(surrogate, token count)` per indexed document. + pub(super) doc_lengths: Vec<(u32, u32)>, + /// `(surrogate, distinct terms)` per document with a posting. + pub(super) doc_terms: Vec<(u32, Vec)>, + /// Number of indexed documents. + pub(super) doc_count: u32, + /// Sum of the indexed documents' token counts. + pub(super) total_tokens: u64, +} + +impl FtsRebuildTicket { + pub(super) fn new( + token: u64, + database_id: u64, + tid: u64, + collection: String, + txn: ReadTransaction, + ) -> Self { + Self { + token, + database_id, + tid, + collection, + txn, + } + } + + /// The token that ties this rebuild to its journal. + pub fn token(&self) -> u64 { + self.token + } + + /// Read the collection's postings and document lengths from the pinned + /// snapshot. Runs on any thread. The snapshot is released on return. + pub fn read(self) -> crate::Result { + let db = self.database_id; + let t = self.tid; + let coll = self.collection.as_str(); + + let postings_table = self + .txn + .open_table(POSTINGS) + .map_err(|e| inverted_err("rebuild open postings", e))?; + let mut postings = Vec::new(); + for entry in postings_table + .range((db, t, coll, "")..=(db, t, coll, MAX_TERM)) + .map_err(|e| inverted_err("rebuild postings range", e))? + { + let (key, value) = entry.map_err(|e| inverted_err("rebuild postings entry", e))?; + let list: Vec = zerompk::from_msgpack(value.value()) + .map_err(|e| inverted_err("rebuild decode postings", e))?; + postings.push((key.value().3.to_string(), list)); + } + + let lengths_table = self + .txn + .open_table(DOC_LENGTHS) + .map_err(|e| inverted_err("rebuild open doc_lengths", e))?; + let mut doc_lengths = Vec::new(); + for entry in lengths_table + .range((db, t, coll, 0u32)..=(db, t, coll, u32::MAX)) + .map_err(|e| inverted_err("rebuild doc_lengths range", e))? + { + let (key, value) = entry.map_err(|e| inverted_err("rebuild doc_length entry", e))?; + let len: u32 = zerompk::from_msgpack(value.value()) + .map_err(|e| inverted_err("rebuild decode doc_length", e))?; + doc_lengths.push((key.value().3, len)); + } + + Ok(FtsSnapshot { + token: self.token, + database_id: self.database_id, + tid: self.tid, + collection: self.collection, + postings, + doc_lengths, + }) + } +} + +impl FtsSnapshot { + /// The token that ties this rebuild to its journal. + pub fn token(&self) -> u64 { + self.token + } + + /// Derive the canonical rows: one posting per document per term, the + /// term set of each document, and corpus stats counted from the + /// document lengths. Runs on any thread. + pub fn compact(self) -> FtsRebuilt { + let mut postings = Vec::with_capacity(self.postings.len()); + let mut terms_by_doc: BTreeMap> = BTreeMap::new(); + for (term, mut list) in self.postings { + list.sort_unstable_by(|a, b| { + a.doc_id + .as_u32() + .cmp(&b.doc_id.as_u32()) + .then(b.term_freq.cmp(&a.term_freq)) + }); + list.dedup_by_key(|p| p.doc_id.as_u32()); + if list.is_empty() { + continue; + } + for posting in &list { + terms_by_doc + .entry(posting.doc_id.as_u32()) + .or_default() + .insert(term.clone()); + } + postings.push((term, list)); + } + let doc_terms = terms_by_doc + .into_iter() + .map(|(doc, terms)| (doc, terms.into_iter().collect())) + .collect(); + let doc_count = u32::try_from(self.doc_lengths.len()).unwrap_or(u32::MAX); + let total_tokens = self + .doc_lengths + .iter() + .map(|&(_, len)| u64::from(len)) + .sum(); + FtsRebuilt { + token: self.token, + database_id: self.database_id, + tid: self.tid, + collection: self.collection, + postings, + doc_lengths: self.doc_lengths, + doc_terms, + doc_count, + total_tokens, + } + } +} + +impl FtsRebuilt { + /// The token that ties this rebuild to its journal. + pub fn token(&self) -> u64 { + self.token + } + + /// The collection this rebuild covers. + pub fn collection(&self) -> &str { + &self.collection + } +} diff --git a/nodedb/src/engine/sparse/inverted/removal.rs b/nodedb/src/engine/sparse/inverted/removal.rs index eca372ecc..306d2fe0c 100644 --- a/nodedb/src/engine/sparse/inverted/removal.rs +++ b/nodedb/src/engine/sparse/inverted/removal.rs @@ -57,6 +57,7 @@ impl InvertedIndex { txn: &WriteTransaction, scope: IndexDocScope<'_>, ) -> crate::Result<()> { + self.note_doc_write(scope); let Some(old_len) = prior_doc_length(txn, scope)? else { return Ok(()); }; diff --git a/nodedb/tests/inproc/cases/fts_update_reindex.rs b/nodedb/tests/inproc/cases/fts_update_reindex.rs new file mode 100644 index 000000000..1e624a926 --- /dev/null +++ b/nodedb/tests/inproc/cases/fts_update_reindex.rs @@ -0,0 +1,213 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! UPDATE must keep the full-text index in step with the row. +//! +//! An updated row keeps its surrogate while its text changes. After the +//! update it must match the words its new body holds and must stop +//! matching the words only its old body held, for a point update by id and +//! for a bulk update by predicate alike. + +use nodedb_test_support::pgwire_harness::TestServer; + +/// Ids a full-text match on `body` returns, sorted. +async fn text_ids(server: &TestServer, collection: &str, term: &str) -> Vec { + text_ids_on(server, collection, "body", term).await +} + +/// Ids a full-text match on `field` returns, sorted. +async fn text_ids_on( + server: &TestServer, + collection: &str, + field: &str, + term: &str, +) -> Vec { + let mut ids = server + .query_text(&format!( + "SELECT id FROM {collection} WHERE text_match({field}, '{term}')" + )) + .await + .unwrap(); + ids.sort(); + ids +} + +async fn seeded(server: &TestServer, collection: &str) { + server + .exec(&format!( + "CREATE COLLECTION {collection} WITH (engine='document_schemaless')" + )) + .await + .unwrap(); + for (id, body) in [("u1", "alpha first"), ("u2", "alpha second")] { + server + .exec(&format!( + "INSERT INTO {collection} {{ id: '{id}', body: '{body}', kind: 'k' }}" + )) + .await + .unwrap(); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn point_update_moves_the_row_to_its_new_terms() { + let server = TestServer::start().await; + seeded(&server, "fu_point").await; + + server + .exec("UPDATE fu_point SET body = 'beta moved' WHERE id = 'u1'") + .await + .unwrap(); + + assert_eq!(text_ids(&server, "fu_point", "alpha").await, ["u2"]); + assert_eq!(text_ids(&server, "fu_point", "beta").await, ["u1"]); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn bulk_update_moves_every_row_to_its_new_terms() { + let server = TestServer::start().await; + seeded(&server, "fu_bulk").await; + + server + .exec("UPDATE fu_bulk SET body = 'gamma bulk' WHERE kind = 'k'") + .await + .unwrap(); + + assert!(text_ids(&server, "fu_bulk", "alpha").await.is_empty()); + assert_eq!(text_ids(&server, "fu_bulk", "gamma").await, ["u1", "u2"]); +} + +/// `UPDATE ... FROM` writes through its own row-by-row transaction, not the +/// point/bulk UPDATE paths above, so it must reindex text on its own. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn update_from_join_moves_the_row_to_its_new_terms() { + let server = TestServer::start().await; + server + .exec( + "CREATE COLLECTION fu_join_target (\ + id TEXT PRIMARY KEY, sku TEXT, body TEXT) \ + WITH (engine='document_strict')", + ) + .await + .unwrap(); + server + .exec( + "CREATE COLLECTION fu_join_source (\ + id TEXT PRIMARY KEY, sku TEXT, new_body TEXT) \ + WITH (engine='document_strict')", + ) + .await + .unwrap(); + server + .exec("INSERT INTO fu_join_target (id, sku, body) VALUES ('t1', 'k1', 'alpha original')") + .await + .unwrap(); + server + .exec("INSERT INTO fu_join_source (id, sku, new_body) VALUES ('s1', 'k1', 'beta updated')") + .await + .unwrap(); + + server + .exec( + "UPDATE fu_join_target SET body = s.new_body \ + FROM fu_join_source s WHERE fu_join_target.sku = s.sku", + ) + .await + .unwrap(); + + assert!( + text_ids(&server, "fu_join_target", "alpha") + .await + .is_empty() + ); + assert_eq!(text_ids(&server, "fu_join_target", "beta").await, ["t1"]); +} + +/// `CONVERT COLLECTION` re-encodes every row against the target schema — a +/// schema that declares every field the source rows carry converts cleanly, +/// and every field's words stay findable afterward. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn convert_collection_to_a_full_schema_keeps_every_fields_words() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION fu_convert_full WITH (engine='document_schemaless')") + .await + .unwrap(); + server + .exec("INSERT INTO fu_convert_full { id: 'c1', title: 'alpha kept', notes: 'beta kept' }") + .await + .unwrap(); + + server + .exec( + "CONVERT COLLECTION fu_convert_full TO document_strict \ + (id TEXT PRIMARY KEY, title TEXT, notes TEXT)", + ) + .await + .unwrap(); + + assert_eq!( + text_ids_on(&server, "fu_convert_full", "title", "alpha").await, + ["c1"] + ); + assert_eq!( + text_ids_on(&server, "fu_convert_full", "notes", "beta").await, + ["c1"] + ); +} + +/// `CONVERT COLLECTION` to a schema that omits a field a source row carries +/// is refused rather than silently dropping data: the row's own primary key +/// names it, the missing column and collection are named, the error is a +/// typed SQLSTATE 22000 (data_exception) rather than an internal fault, and +/// the collection — and its full-text index — are left exactly as they were. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn convert_collection_to_a_partial_schema_is_refused_and_leaves_the_collection_unchanged() { + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION fu_convert_partial WITH (engine='document_schemaless')") + .await + .unwrap(); + server + .exec( + "INSERT INTO fu_convert_partial \ + { id: 'c1', title: 'alpha kept', notes: 'beta dropped' }", + ) + .await + .unwrap(); + + let err = server + .exec( + "CONVERT COLLECTION fu_convert_partial TO document_strict \ + (id TEXT PRIMARY KEY, title TEXT)", + ) + .await + .expect_err("CONVERT must refuse a schema that drops a populated field"); + + assert!( + err.contains("(SQLSTATE 22000)"), + "expected SQLSTATE 22000 (data_exception), got: {err}" + ); + assert!( + err.contains("column \"notes\""), + "error must name the missing column: {err}" + ); + assert!( + err.contains("collection \"fu_convert_partial\""), + "error must name the collection: {err}" + ); + assert!( + err.contains("row \"c1\""), + "error must name the row by its primary key, not an internal surrogate: {err}" + ); + assert!( + !err.contains("Internal"), + "a schema mismatch is a client error, not an internal fault: {err}" + ); + + // Nothing converted: the collection's full-text index still holds the + // row's original body, on its original schemaless engine. + assert_eq!( + text_ids_on(&server, "fu_convert_partial", "notes", "beta").await, + ["c1"] + ); +} diff --git a/nodedb/tests/inproc/cases/mod.rs b/nodedb/tests/inproc/cases/mod.rs index 70e3a5861..34c2c6225 100644 --- a/nodedb/tests/inproc/cases/mod.rs +++ b/nodedb/tests/inproc/cases/mod.rs @@ -92,6 +92,7 @@ mod event_trigger; mod event_trigger_descriptor_fence; mod event_wal_replay_source; mod fts_compaction_budget; +mod fts_update_reindex; mod gateway_local_descriptor_fence; mod graph_collection_isolation; mod graph_cross_core_bfs; @@ -147,6 +148,7 @@ mod quota_drop_cleanup; mod quota_live_enforcement_apply; mod quota_three_level_denial; mod redaction_policy_ddl; +mod reindex_concurrent_writes; mod reindex_vector_concurrent; mod request_tracker_backpressure; mod resp_row_level_security; diff --git a/nodedb/tests/inproc/cases/reindex_concurrent_writes.rs b/nodedb/tests/inproc/cases/reindex_concurrent_writes.rs new file mode 100644 index 000000000..c3023ac4a --- /dev/null +++ b/nodedb/tests/inproc/cases/reindex_concurrent_writes.rs @@ -0,0 +1,497 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! REINDEX must keep every write, and must rebuild every index it names. +//! +//! - A full-text or CSR rebuild snapshots its index, builds off the core, +//! then swaps the result in. Rows written, updated or deleted between the +//! snapshot and the swap must show in the swapped-in index exactly as +//! they do in the rows. A failpoint holds the rebuild thread so the +//! writes land inside that window. +//! - REINDEX with no index name rebuilds every index kind the collection +//! has: HNSW, full-text and CSR. Both the concurrent and the plain form +//! are covered. +//! - Plain REINDEX rebuilds off the core like the concurrent form and +//! answers only after the cutover. While its rebuild is held, the same +//! core still answers queries on other collections. +//! +//! Each rebuild reports on the `nodedb::reindex` tracing target: +//! `rebuild_started` when it starts, then `atomic_cutover` when it is +//! swapped in or `rebuild_refused` when it is discarded. The tests count +//! those events. + +use std::sync::{Arc, Mutex, OnceLock}; +use std::time::{Duration, Instant}; + +use nodedb_test_support::pgwire_harness::TestServer; +use tracing_subscriber::layer::SubscriberExt; +use tracing_subscriber::util::SubscriberInitExt; + +// ── Event recorder ──────────────────────────────────────────────────────────── + +/// `(message, index, scope)` of every `nodedb::reindex` event. `scope` is +/// the event's `collection` field, or its `key` field for HNSW. +#[derive(Default)] +struct EventLog(Mutex>); + +impl EventLog { + fn count(&self, message: &str, index: &str, collection: &str) -> usize { + self.0 + .lock() + .unwrap() + .iter() + .filter(|(m, i, s)| m == message && i == index && s.contains(collection)) + .count() + } +} + +struct Recorder(Arc); + +#[derive(Default)] +struct Fields { + message: String, + index: String, + scope: String, +} + +impl Fields { + fn record_text(&mut self, field: &tracing::field::Field, text: &str) { + match field.name() { + "message" => self.message = text.to_string(), + "index" => self.index = text.to_string(), + "collection" | "key" => self.scope = text.to_string(), + _ => {} + } + } +} + +impl tracing::field::Visit for Fields { + fn record_debug(&mut self, field: &tracing::field::Field, value: &dyn std::fmt::Debug) { + let text = format!("{value:?}"); + self.record_text(field, text.trim_matches('"')); + } + + fn record_str(&mut self, field: &tracing::field::Field, value: &str) { + self.record_text(field, value); + } +} + +impl tracing_subscriber::Layer for Recorder { + fn on_event( + &self, + event: &tracing::Event<'_>, + _ctx: tracing_subscriber::layer::Context<'_, S>, + ) { + if event.metadata().target() != "nodedb::reindex" { + return; + } + let mut fields = Fields::default(); + event.record(&mut fields); + self.0 + .0 + .lock() + .unwrap() + .push((fields.message, fields.index, fields.scope)); + } +} + +/// The process-wide event log. Installed before the first server starts. +fn events() -> Arc { + static LOG: OnceLock> = OnceLock::new(); + Arc::clone(LOG.get_or_init(|| { + let log = Arc::new(EventLog::default()); + tracing_subscriber::registry() + .with(Recorder(Arc::clone(&log))) + .try_init() + .expect("the reindex event recorder must be the process's tracing subscriber"); + log + })) +} + +/// Wait until every started `index` rebuild of `collection` has been swapped +/// in, and fail on any rebuild that was discarded instead. +async fn wait_for_cutovers(log: &EventLog, index: &str, collection: &str) { + let started = log.count("rebuild_started", index, collection); + assert!( + started > 0, + "REINDEX must start a {index} rebuild of {collection}" + ); + let deadline = Instant::now() + Duration::from_secs(60); + loop { + let refused = log.count("rebuild_refused", index, collection); + assert_eq!( + refused, 0, + "a {index} rebuild of {collection} was discarded instead of swapped in" + ); + let cutovers = log.count("atomic_cutover", index, collection); + if cutovers >= started { + assert_eq!(cutovers, started, "one cutover per started {index} rebuild"); + return; + } + assert!( + Instant::now() < deadline, + "{index} rebuild of {collection} did not cut over: {cutovers} of {started}" + ); + tokio::time::sleep(Duration::from_millis(10)).await; + } +} + +/// Ids a full-text match on `body` returns, sorted. +async fn text_ids(server: &TestServer, collection: &str, term: &str) -> Vec { + let mut ids = server + .query_text(&format!( + "SELECT id FROM {collection} WHERE text_match(body, '{term}')" + )) + .await + .unwrap(); + ids.sort(); + ids +} + +/// Nodes one hop out of `root` along label `l`, as one text blob. +async fn out_of_root(server: &TestServer, collection: &str) -> String { + server + .query_text_joined(&format!( + "GRAPH TRAVERSE IN '{collection}' FROM 'root' DEPTH 1 LABEL 'l' DIRECTION out" + )) + .await + .unwrap() + .join("\n") +} + +async fn insert_edge(server: &TestServer, collection: &str, dst: &str) { + server + .exec(&format!( + "GRAPH INSERT EDGE IN '{collection}' FROM 'root' TO '{dst}' TYPE 'l'" + )) + .await + .unwrap(); +} + +// ── Writes during a held rebuild ────────────────────────────────────────────── + +#[cfg(feature = "failpoints")] +mod held { + use nodedb::fail_point::{FailAction, FailGuard}; + + use super::*; + + /// Arm `failpoint` to hold its rebuild thread until the returned path + /// exists. + fn hold(failpoint: &str) -> (FailGuard, tempfile::TempDir, std::path::PathBuf) { + let dir = tempfile::tempdir().unwrap(); + let release = dir.path().join("release"); + let guard = FailGuard::install(failpoint, FailAction::WaitForFile(release.clone())); + (guard, dir, release) + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn fts_rebuild_keeps_writes_made_during_the_rebuild() { + let log = events(); + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION rx_fts WITH (engine='document_schemaless')") + .await + .unwrap(); + for i in 1..=4 { + server + .exec(&format!( + "INSERT INTO rx_fts {{ id: 'd{i}', body: 'alpha seed' }}" + )) + .await + .unwrap(); + } + + let (_guard, _dir, release) = hold("reindex::fts_build_hold"); + server + .exec("REINDEX INDEX fts CONCURRENTLY rx_fts") + .await + .unwrap(); + assert!(log.count("rebuild_started", "fts", "rx_fts") > 0); + + // The rebuild's snapshot is pinned; these writes land after it. + for i in 5..=7 { + server + .exec(&format!( + "INSERT INTO rx_fts {{ id: 'd{i}', body: 'alpha fresh' }}" + )) + .await + .unwrap(); + } + server + .exec("UPDATE rx_fts SET body = 'beta moved' WHERE id = 'd1'") + .await + .unwrap(); + server + .exec("DELETE FROM rx_fts WHERE id = 'd2'") + .await + .unwrap(); + assert_eq!( + log.count("atomic_cutover", "fts", "rx_fts"), + 0, + "the held rebuild must not cut over before it is released" + ); + + std::fs::write(&release, b"").unwrap(); + wait_for_cutovers(&log, "fts", "rx_fts").await; + + assert_eq!( + text_ids(&server, "rx_fts", "alpha").await, + ["d3", "d4", "d5", "d6", "d7"], + "rows inserted during the rebuild are found once, the updated row \ + lost the term, and the deleted row is gone" + ); + assert_eq!(text_ids(&server, "rx_fts", "beta").await, ["d1"]); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn csr_rebuild_keeps_writes_made_during_the_rebuild() { + let log = events(); + let server = TestServer::start().await; + server.exec("CREATE COLLECTION rx_csr").await.unwrap(); + for dst in ["keep0", "gone1", "keep2"] { + insert_edge(&server, "rx_csr", dst).await; + } + + let (_guard, _dir, release) = hold("reindex::csr_build_hold"); + server + .exec("REINDEX INDEX csr CONCURRENTLY rx_csr") + .await + .unwrap(); + assert!(log.count("rebuild_started", "csr", "rx_csr") > 0); + + // The partition snapshot is taken; these writes land after it. + insert_edge(&server, "rx_csr", "new3").await; + insert_edge(&server, "rx_csr", "new4").await; + server + .exec("GRAPH DELETE EDGE IN 'rx_csr' FROM 'root' TO 'gone1' TYPE 'l'") + .await + .unwrap(); + assert_eq!( + log.count("atomic_cutover", "csr", "rx_csr"), + 0, + "the held rebuild must not cut over before it is released" + ); + + std::fs::write(&release, b"").unwrap(); + wait_for_cutovers(&log, "csr", "rx_csr").await; + + let blob = out_of_root(&server, "rx_csr").await; + for kept in ["keep0", "keep2", "new3", "new4"] { + assert!( + blob.contains(kept), + "traversal after the cutover must reach {kept}; got: {blob}" + ); + } + assert!( + !blob.contains("gone1"), + "an edge deleted during the rebuild must stay deleted; got: {blob}" + ); + } + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn plain_reindex_leaves_the_core_serving_while_its_rebuild_is_held() { + let log = events(); + let server = TestServer::start().await; + server + .exec("CREATE COLLECTION rx_held WITH (engine='document_schemaless')") + .await + .unwrap(); + server + .exec("INSERT INTO rx_held { id: 'h1', body: 'alpha held' }") + .await + .unwrap(); + server + .exec("CREATE COLLECTION rx_other WITH (engine='document_schemaless')") + .await + .unwrap(); + server + .exec("INSERT INTO rx_other { id: 'o1', body: 'other row' }") + .await + .unwrap(); + + let (_guard, _dir, release) = hold("reindex::fts_build_hold"); + + // The plain REINDEX runs on its own connection: it answers only + // after the cutover, and the cutover waits for the release. + let conn_str = format!( + "host=127.0.0.1 port={} user=nodedb dbname=default", + server.pg_port + ); + let (client, conn) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .unwrap(); + tokio::spawn(async move { + let _ = conn.await; + }); + let reindex = tokio::spawn(async move { + client + .simple_query("REINDEX INDEX fts rx_held") + .await + .map(|_| ()) + .map_err(|e| e.to_string()) + }); + + let deadline = Instant::now() + Duration::from_secs(30); + while log.count("rebuild_started", "fts", "rx_held") == 0 { + assert!(Instant::now() < deadline, "the plain REINDEX never started"); + tokio::time::sleep(Duration::from_millis(10)).await; + } + + // The core that holds the rebuild answers other requests. + let rows = tokio::time::timeout( + Duration::from_secs(10), + server.query_text("SELECT id FROM rx_other"), + ) + .await + .expect("a query on another collection must answer while the rebuild is held") + .unwrap(); + assert_eq!(rows, ["o1"]); + assert!( + !reindex.is_finished(), + "plain REINDEX must not answer before its cutover" + ); + assert_eq!(log.count("atomic_cutover", "fts", "rx_held"), 0); + + std::fs::write(&release, b"").unwrap(); + tokio::time::timeout(Duration::from_secs(60), reindex) + .await + .expect("plain REINDEX must answer after the release") + .unwrap() + .expect("plain REINDEX must succeed"); + assert_eq!( + log.count("atomic_cutover", "fts", "rx_held"), + log.count("rebuild_started", "fts", "rx_held"), + "plain REINDEX answers only after its cutover" + ); + assert_eq!(text_ids(&server, "rx_held", "alpha").await, ["h1"]); + } +} + +// ── REINDEX with no index name ──────────────────────────────────────────────── + +/// A collection with an HNSW index, full-text rows and graph edges. +async fn collection_with_every_index(server: &TestServer, collection: &str) { + server + .exec(&format!("CREATE COLLECTION {collection} TYPE document")) + .await + .unwrap(); + server + .exec(&format!( + "CREATE VECTOR INDEX idx_{collection} ON {collection} (embedding) METRIC cosine DIM 4" + )) + .await + .unwrap(); + for (id, body, emb) in [ + ("r1", "alpha one", "1.0,0.0,0.0,0.0"), + ("r2", "alpha two", "0.0,1.0,0.0,0.0"), + ("r3", "alpha three", "0.0,0.0,1.0,0.0"), + ] { + server + .exec(&format!( + "INSERT INTO {collection} (id, body, embedding) VALUES ('{id}', '{body}', ARRAY[{emb}])" + )) + .await + .unwrap(); + } + insert_edge(server, collection, "r1").await; + insert_edge(server, collection, "r2").await; +} + +/// One numeric property of the collection's vector index status. +async fn vector_status(server: &TestServer, collection: &str, property: &str) -> u64 { + let rows = server + .query_rows(&format!( + "SHOW VECTOR INDEX status ON {collection}.embedding" + )) + .await + .unwrap(); + rows.iter() + .find(|r| r[0] == property) + .unwrap_or_else(|| panic!("SHOW VECTOR INDEX must report {property}: {rows:?}"))[1] + .parse() + .unwrap() +} + +/// Seal the vector rows into one segment and wait for its first build, so +/// an HNSW rebuild has a sealed segment to work on. +async fn seal_vectors(server: &TestServer, collection: &str) { + server + .exec(&format!( + "ALTER VECTOR INDEX ON {collection}.embedding SEAL" + )) + .await + .unwrap(); + let deadline = Instant::now() + Duration::from_secs(60); + loop { + let sealed = vector_status(server, collection, "sealed_segments").await; + let building = vector_status(server, collection, "building_segments").await; + if sealed >= 1 && building == 0 { + return; + } + assert!( + Instant::now() < deadline, + "the sealed segment did not build: sealed={sealed} building={building}" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn concurrent_reindex_without_a_name_rebuilds_every_index_kind() { + let log = events(); + let server = TestServer::start().await; + collection_with_every_index(&server, "rx_all").await; + seal_vectors(&server, "rx_all").await; + + server.exec("REINDEX CONCURRENTLY rx_all").await.unwrap(); + + wait_for_cutovers(&log, "fts", "rx_all").await; + wait_for_cutovers(&log, "csr", "rx_all").await; + let deadline = Instant::now() + Duration::from_secs(60); + while log.count("atomic_cutover", "hnsw", "rx_all") == 0 { + assert!( + Instant::now() < deadline, + "REINDEX with no index name must rebuild the HNSW index too" + ); + tokio::time::sleep(Duration::from_millis(10)).await; + } + + assert_eq!( + text_ids(&server, "rx_all", "alpha").await, + ["r1", "r2", "r3"] + ); + let blob = out_of_root(&server, "rx_all").await; + assert!(blob.contains("r1") && blob.contains("r2"), "got: {blob}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn plain_reindex_without_a_name_rebuilds_every_index_kind() { + let log = events(); + let server = TestServer::start().await; + collection_with_every_index(&server, "rx_plain").await; + seal_vectors(&server, "rx_plain").await; + + // The plain form answers after every cutover. + server.exec("REINDEX rx_plain").await.unwrap(); + + for index in ["fts", "csr"] { + let started = log.count("rebuild_started", index, "rx_plain"); + assert!(started > 0, "REINDEX must rebuild the {index} index"); + assert_eq!( + log.count("atomic_cutover", index, "rx_plain"), + started, + "the plain {index} rebuild must cut over before REINDEX returns" + ); + assert_eq!(log.count("rebuild_refused", index, "rx_plain"), 0); + } + assert!( + log.count("atomic_cutover", "hnsw", "rx_plain") > 0, + "the plain HNSW rebuild must cut over before REINDEX returns" + ); + + assert_eq!( + text_ids(&server, "rx_plain", "alpha").await, + ["r1", "r2", "r3"] + ); + let blob = out_of_root(&server, "rx_plain").await; + assert!(blob.contains("r1") && blob.contains("r2"), "got: {blob}"); +} diff --git a/nodedb/tests/inproc/cases/reindex_vector_concurrent.rs b/nodedb/tests/inproc/cases/reindex_vector_concurrent.rs index 90870abb9..b41db74e6 100644 --- a/nodedb/tests/inproc/cases/reindex_vector_concurrent.rs +++ b/nodedb/tests/inproc/cases/reindex_vector_concurrent.rs @@ -12,7 +12,7 @@ //! 4. the query p99 stayed a small share of the rebuild window — the //! signature of a rebuild that took an exclusive lock instead of //! running concurrently -//! 5. exactly one `atomic_cutover` tracing event was emitted by the +//! 5. exactly one HNSW `atomic_cutover` tracing event was emitted by the //! `nodedb::reindex` target during the rebuild phase. REINDEX rebuilds //! each sealed segment and swaps it in on its own, one event per //! segment, so the test force-seals its rows into one segment first. @@ -42,29 +42,36 @@ use nodedb_test_support::pgwire_harness::TestServer; use tracing_subscriber::layer::SubscriberExt; use tracing_subscriber::util::SubscriberInitExt; -// ── Tracing layer that counts `atomic_cutover` events ──────────────────────── +// ── Tracing layer that counts HNSW `atomic_cutover` events ───────────────── struct CutoverCounter(Arc); -/// Visitor that checks whether the `message` field equals "atomic_cutover". -struct MessageVisitor(bool); +/// Visitor that reads the `message` and `index` fields of an event. +#[derive(Default)] +struct CutoverVisitor { + is_cutover: bool, + index: Option, +} -impl tracing::field::Visit for MessageVisitor { - fn record_debug(&mut self, field: &tracing::field::Field, value: &dyn std::fmt::Debug) { - if field.name() == "message" { - let s = format!("{value:?}"); - // Debug formatting wraps strings in quotes; strip them. - let trimmed = s.trim_matches('"'); - if trimmed == "atomic_cutover" { - self.0 = true; - } +impl CutoverVisitor { + fn record_text(&mut self, field: &tracing::field::Field, text: &str) { + match field.name() { + "message" => self.is_cutover |= text == "atomic_cutover", + "index" => self.index = Some(text.to_string()), + _ => {} } } +} + +impl tracing::field::Visit for CutoverVisitor { + fn record_debug(&mut self, field: &tracing::field::Field, value: &dyn std::fmt::Debug) { + let s = format!("{value:?}"); + // Debug formatting wraps strings in quotes; strip them. + self.record_text(field, s.trim_matches('"')); + } fn record_str(&mut self, field: &tracing::field::Field, value: &str) { - if field.name() == "message" && value == "atomic_cutover" { - self.0 = true; - } + self.record_text(field, value); } } @@ -79,9 +86,11 @@ where ) { let meta = event.metadata(); if meta.target().contains("reindex") { - let mut visitor = MessageVisitor(false); + let mut visitor = CutoverVisitor::default(); event.record(&mut visitor); - if visitor.0 { + // REINDEX without an index name also rebuilds any full-text or + // CSR index the collection has; only HNSW cutovers count here. + if visitor.is_cutover && visitor.index.as_deref() == Some("hnsw") { self.0.fetch_add(1, Ordering::Relaxed); } } From 56fca1df2f4a6d15f05c8b3e7749c6e27d15522f Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 13:43:13 +0800 Subject: [PATCH 49/64] feat(errors): keep a typed refusal's SQLSTATE class across every plane Data-Plane refusals now carry a typed cause end to end: a phase error such as MOVE_TENANT_SNAPSHOT_FAILED keeps its own code while the refusal that caused it rides alongside as ErrorCausePayload, rebuilt client-side as NodeDbError::cause. DDL apply, system dispatch, and the native/pgwire/HTTP gateways now render a Data-Plane ErrorCode's own SQLSTATE and class consistently instead of folding refusals to Internal, and native error codes are referenced by their named constants instead of magic numbers. Adds the ProgramLimitExceeded error variant for statements that exceed a server size or depth limit. --- .../src/native/connection/response.rs | 9 +- .../cases/listeners_typed_not_leader.rs | 8 +- .../cases/native_gateway_migration.rs | 63 ++-- .../cases/sync_constraint_version_fence.rs | 6 +- .../cases/sync_peer_id_collision.rs | 2 +- .../cases/sync_retryable_delta_refusal.rs | 2 +- nodedb-types/src/error/code.rs | 3 + nodedb-types/src/error/code_table.rs | 2 + .../src/error/ctors/read_query_auth.rs | 13 + nodedb-types/src/error/ctors/write_path.rs | 15 +- nodedb-types/src/error/details.rs | 4 + nodedb-types/src/error/msgpack/constants.rs | 2 + .../error/msgpack/decode/from_messagepack.rs | 12 + nodedb-types/src/error/msgpack/encode.rs | 3 + nodedb-types/src/error/types.rs | 1 + nodedb-types/src/protocol/error_cause.rs | 71 +++++ nodedb-types/src/protocol/frames.rs | 30 ++ nodedb-types/src/protocol/mod.rs | 2 + nodedb/src/control/crdt_admission.rs | 20 +- .../control/gateway/error_map/class_parity.rs | 289 ++++++++++++++++++ nodedb/src/control/gateway/error_map/http.rs | 12 +- nodedb/src/control/gateway/error_map/mod.rs | 4 + .../src/control/gateway/error_map/native.rs | 92 ++---- .../control/gateway/error_map/remote_code.rs | 45 ++- .../error_map/system_dispatch_refusal.rs | 188 ++++++++++++ .../sql_plan_convert/dml/balanced_gate.rs | 9 +- .../planner/sql_plan_convert/dml/crdt_gate.rs | 10 +- .../sql_plan_convert/dml/insert/identity.rs | 6 +- .../dml/update_delete/shared.rs | 5 +- .../http/routes/query/materialized/encode.rs | 13 +- .../server/native/dispatch/conversion.rs | 60 +++- .../src/control/server/native/dispatch/mod.rs | 2 +- .../control/server/native/sqlstate_code.rs | 173 ++++------- .../src/control/server/pgwire/ddl_encode.rs | 33 ++ .../control/server/pgwire/types/error_map.rs | 60 ++++ nodedb/src/control/server/result_stream.rs | 16 +- .../control/server/shared/ddl/engine_apply.rs | 48 +-- .../ddl/neutral/continuous_agg/create.rs | 2 +- .../shared/ddl/neutral/continuous_agg/drop.rs | 2 +- .../server/shared/ddl/neutral/crdt_ops.rs | 6 +- .../shared/ddl/neutral/deferred_effects.rs | 2 - .../shared/ddl/neutral/dsl/crdt_merge.rs | 12 +- .../shared/ddl/neutral/dsl/text_index.rs | 2 - .../shared/ddl/neutral/dsl/vector_index.rs | 2 - .../ddl/neutral/graph_ops/rag_fusion.rs | 2 +- .../server/shared/ddl/neutral/last_value.rs | 4 +- .../server/shared/ddl/neutral/rate_gate.rs | 2 +- .../ddl/neutral/tenant/move_tenant/cutover.rs | 5 +- .../ddl/neutral/tenant/move_tenant/entry.rs | 4 +- .../neutral/tenant/move_tenant/recovery.rs | 4 +- .../neutral/tenant/move_tenant/snapshot.rs | 8 +- .../server/shared/ddl/neutral/tenant/purge.rs | 2 +- .../ddl/neutral/version_history/checkpoint.rs | 2 +- .../ddl/neutral/version_history/dispatch.rs | 4 +- .../ddl/neutral/version_history/restore.rs | 14 +- .../src/control/server/shared/ddl/result.rs | 23 ++ .../src/control/server/shared/ddl/sqlstate.rs | 4 +- .../shared/ddl/sync_dispatch/dispatch.rs | 47 ++- .../control/server/shared/response_payload.rs | 4 +- .../server/shared/session/ddl_effect.rs | 4 +- nodedb/src/error_classify.rs | 27 +- nodedb/src/error_from_data_plane.rs | 64 ++-- .../inproc/cases/system_task_call_sites.rs | 101 +++++- .../native/cases/native_kv_counter_faults.rs | 16 +- .../cases/crdt_write_rls_database_scope.rs | 15 +- 65 files changed, 1309 insertions(+), 408 deletions(-) create mode 100644 nodedb-types/src/protocol/error_cause.rs create mode 100644 nodedb/src/control/gateway/error_map/class_parity.rs create mode 100644 nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs diff --git a/nodedb-client/src/native/connection/response.rs b/nodedb-client/src/native/connection/response.rs index d97e66c2c..15a265a43 100644 --- a/nodedb-client/src/native/connection/response.rs +++ b/nodedb-client/src/native/connection/response.rs @@ -40,11 +40,16 @@ fn error_frame_to_typed( if payload.ndb_code == 0 { return NodeDbError::internal(payload.message.clone()); } - NodeDbError::from_wire_with_details( + let error = NodeDbError::from_wire_with_details( nodedb_types::error::ErrorCode(payload.ndb_code), payload.message.clone(), payload.details.clone(), - ) + ); + // The typed cause, when the server sent one, becomes the error's cause. + match &payload.cause { + Some(cause) => error.with_cause(cause.to_error()), + None => error, + } } pub(super) fn response_to_query_result(resp: NativeResponse) -> NodeDbResult { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs b/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs index 08dd33abc..9ab4585ae 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs @@ -463,7 +463,8 @@ async fn native_not_leader_gateway_error_mapping() { assert_eq!(node.not_leader_retry_count(), 0); - // Error-mapping proof: GatewayErrorMap::to_native maps NotLeader to code 40. + // Error-mapping proof: GatewayErrorMap::to_native maps NotLeader to the + // public NOT_LEADER code. let not_leader = Error::NotLeader { vshard_id: VShardId::new(0), leader_node: 1, @@ -471,8 +472,9 @@ async fn native_not_leader_gateway_error_mapping() { }; let (native_code, _native_msg) = GatewayErrorMap::to_native(¬_leader); assert_eq!( - native_code, 10, - "NotLeader must map to native error code 10 (CODE_NOT_LEADER)" + native_code, + nodedb::ErrorCode::NOT_LEADER, + "NotLeader must map to the public NOT_LEADER code" ); node.shutdown().await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs index fc46e4228..ecb671784 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs @@ -7,8 +7,8 @@ //! assert rows returned. //! 2. **Cross-node SELECT** — 3-node cluster, gateway on follower routes a //! KV GET to the leaseholder; asserts success. -//! 3. **Typed error → native code** — trigger `CollectionNotFound`, assert the -//! native error code matches `GatewayErrorMap::to_native` mapping (code 40). +//! 3. **Typed error → native code** — map each error variant through +//! `GatewayErrorMap::to_native` and assert its stable `nodedb_types` code. use crate::common; @@ -22,6 +22,7 @@ use nodedb::control::gateway::core::QueryContext; use nodedb::types::{RequestId, TenantId, VShardId}; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use nodedb_types::QualifiedCollection; +use nodedb_types::error::ErrorCode; use common::cluster_harness::{TestCluster, TestClusterNode}; @@ -188,21 +189,17 @@ async fn native_gateway_migration_cross_node_select() { // Test 3: Typed error → native code mapping // --------------------------------------------------------------------------- // -// `GatewayErrorMap::to_native` maps each error variant to a numeric code. -// The migrated `direct_ops.rs` and `sql_gateway.rs` call this mapper. -// These tests verify the codes align with the constants defined in error_map.rs. +// `GatewayErrorMap::to_native` returns the stable `nodedb_types` error code +// and the message the native error frame carries for each error variant. #[test] -fn native_gateway_error_collection_not_found_is_code_40() { +fn native_gateway_error_collection_not_found_code() { let err = Error::CollectionNotFound { tenant_id: TenantId::new(0), collection: "missing_native_col".into(), }; let (code, msg) = GatewayErrorMap::to_native(&err); - assert_eq!( - code, 40, - "CollectionNotFound should map to code 40, got {code}" - ); + assert_eq!(code, ErrorCode::COLLECTION_NOT_FOUND, "got {code}"); assert!( msg.contains("missing_native_col"), "error message should name the collection: {msg}" @@ -210,14 +207,14 @@ fn native_gateway_error_collection_not_found_is_code_40() { } #[test] -fn native_gateway_error_not_leader_is_code_10() { +fn native_gateway_error_not_leader_code() { let err = Error::NotLeader { vshard_id: VShardId::new(1), leader_node: 2, leader_addr: "10.0.0.1:9000".into(), }; let (code, msg) = GatewayErrorMap::to_native(&err); - assert_eq!(code, 10, "NotLeader should map to code 10, got {code}"); + assert_eq!(code, ErrorCode::NOT_LEADER, "got {code}"); assert!( msg.contains("hint:"), "not-leader message should contain hint: {msg}" @@ -225,50 +222,31 @@ fn native_gateway_error_not_leader_is_code_10() { } #[test] -fn native_gateway_error_deadline_is_code_20() { +fn native_gateway_error_deadline_code() { let err = Error::DeadlineExceeded { request_id: RequestId::new(1), }; let (code, _msg) = GatewayErrorMap::to_native(&err); - assert_eq!( - code, 20, - "DeadlineExceeded should map to code 20, got {code}" - ); + assert_eq!(code, ErrorCode::DEADLINE_EXCEEDED, "got {code}"); } #[test] -fn native_gateway_error_schema_changed_is_code_30() { - let err = Error::RetryableSchemaChanged { - descriptor: "users".into(), - }; - let (code, msg) = GatewayErrorMap::to_native(&err); - assert_eq!( - code, 30, - "RetryableSchemaChanged should map to code 30, got {code}" - ); - assert!( - msg.contains("users"), - "message should name descriptor: {msg}" - ); -} - -#[test] -fn native_gateway_error_authz_is_code_50() { +fn native_gateway_error_authz_code() { let err = Error::RejectedAuthz { tenant_id: TenantId::new(0), resource: "secret".into(), }; let (code, _msg) = GatewayErrorMap::to_native(&err); - assert_eq!(code, 50, "RejectedAuthz should map to code 50, got {code}"); + assert_eq!(code, ErrorCode::AUTHORIZATION_DENIED, "got {code}"); } #[test] -fn native_gateway_error_bad_request_is_code_60() { +fn native_gateway_error_bad_request_code() { let err = Error::BadRequest { detail: "invalid plan".into(), }; let (code, msg) = GatewayErrorMap::to_native(&err); - assert_eq!(code, 60, "BadRequest should map to code 60, got {code}"); + assert_eq!(code, ErrorCode::BAD_REQUEST, "got {code}"); assert!( msg.contains("invalid plan"), "message should contain detail: {msg}" @@ -276,24 +254,21 @@ fn native_gateway_error_bad_request_is_code_60() { } #[test] -fn native_gateway_error_constraint_is_code_70() { +fn native_gateway_error_constraint_code() { let err = Error::RejectedConstraint { detail: "unique violation".into(), constraint: "pk".into(), collection: "orders".into(), }; let (code, _msg) = GatewayErrorMap::to_native(&err); - assert_eq!( - code, 70, - "RejectedConstraint should map to code 70, got {code}" - ); + assert_eq!(code, ErrorCode::CONSTRAINT_VIOLATION, "got {code}"); } #[test] -fn native_gateway_error_internal_is_code_99() { +fn native_gateway_error_internal_code() { let err = Error::Internal { detail: "unexpected state".into(), }; let (code, _msg) = GatewayErrorMap::to_native(&err); - assert_eq!(code, 99, "Internal should map to code 99, got {code}"); + assert_eq!(code, ErrorCode::INTERNAL, "got {code}"); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_constraint_version_fence.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_constraint_version_fence.rs index 1a1be059d..b1d3c7df6 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_constraint_version_fence.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_constraint_version_fence.rs @@ -99,11 +99,11 @@ async fn read_crdt_doc( } // A missing document is terminal, not transient: the CRDT read path // answers `NotFound`, which the `crdt_state` scalar function surfaces - // as an internal error whose message carries "NotFound". That is - // exactly the "not imported" state the fence assertion expects. + // as SQLSTATE `02000` (no_data). That is exactly the "not imported" + // state the fence assertion expects. Err(e) if e.as_db_error() - .is_some_and(|d| d.message().contains("NotFound")) => + .is_some_and(|d| d.code() == &tokio_postgres::error::SqlState::NO_DATA) => { return Ok(None); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_peer_id_collision.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_peer_id_collision.rs index 0eeca7bb6..21269ed21 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_peer_id_collision.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_peer_id_collision.rs @@ -113,7 +113,7 @@ async fn row_is_readable( } Err(e) if e.as_db_error() - .is_some_and(|d| d.message().contains("NotFound")) => + .is_some_and(|d| d.code() == &tokio_postgres::error::SqlState::NO_DATA) => { return Ok(false); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_retryable_delta_refusal.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_retryable_delta_refusal.rs index 61f781b44..b918206c7 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_retryable_delta_refusal.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_retryable_delta_refusal.rs @@ -106,7 +106,7 @@ async fn read_crdt_doc( } Err(e) if e.as_db_error() - .is_some_and(|d| d.message().contains("NotFound")) => + .is_some_and(|d| d.code() == &tokio_postgres::error::SqlState::NO_DATA) => { return Ok(None); } diff --git a/nodedb-types/src/error/code.rs b/nodedb-types/src/error/code.rs index 6a462952a..c9e3d8b4c 100644 --- a/nodedb-types/src/error/code.rs +++ b/nodedb-types/src/error/code.rs @@ -70,6 +70,9 @@ impl ErrorCode { /// A function received a value it cannot compute on: a vector of the /// wrong dimension, an argument of the wrong shape, a malformed path. pub const DATA_EXCEPTION: Self = Self(1208); + /// A statement exceeded a server limit on its own size or depth: a + /// recursion depth, a per-transaction staging budget. + pub const PROGRAM_LIMIT_EXCEEDED: Self = Self(1209); // Engine ops (1300–1399) pub const ARRAY: Self = Self(1300); diff --git a/nodedb-types/src/error/code_table.rs b/nodedb-types/src/error/code_table.rs index 11524d5de..eb0a2f8df 100644 --- a/nodedb-types/src/error/code_table.rs +++ b/nodedb-types/src/error/code_table.rs @@ -96,6 +96,7 @@ error_code_table! { AMBIGUOUS_COLUMN => AmbiguousColumn { column: String::new() }, DIVISION_BY_ZERO => DivisionByZero, DATA_EXCEPTION => DataException { detail: message.to_owned() }, + PROGRAM_LIMIT_EXCEEDED => ProgramLimitExceeded { detail: message.to_owned() }, INVALID_LIMIT_VALUE => InvalidLimitValue { clause: "remote".into(), value: message.to_owned() }, // Auth / tenant quota. @@ -238,6 +239,7 @@ mod tests { ErrorCode::CANNOT_DROP_DEFAULT_DATABASE, ErrorCode::COLLECTION_DEACTIVATED, ErrorCode::ARRAY, + ErrorCode::PROGRAM_LIMIT_EXCEEDED, ErrorCode::QUOTA_OVERCOMMIT, ErrorCode::CLONE_DEPTH_EXCEEDED, ErrorCode::CLONE_WRITE_REQUIRES_MATERIALIZE, diff --git a/nodedb-types/src/error/ctors/read_query_auth.rs b/nodedb-types/src/error/ctors/read_query_auth.rs index 9bf562b46..b564d0740 100644 --- a/nodedb-types/src/error/ctors/read_query_auth.rs +++ b/nodedb-types/src/error/ctors/read_query_auth.rs @@ -212,6 +212,19 @@ impl NodeDbError { } } + /// A statement exceeded a server limit on its own size or depth: a + /// recursion depth, a per-transaction staging budget. SQLSTATE `54000` + /// (`program_limit_exceeded`). `detail` is the full message. + pub fn program_limit_exceeded(detail: impl Into) -> Self { + let detail = detail.into(); + Self { + code: ErrorCode::PROGRAM_LIMIT_EXCEEDED, + message: detail.clone(), + details: ErrorDetails::ProgramLimitExceeded { detail }, + cause: None, + } + } + /// A LIMIT/OFFSET/FETCH bound did not resolve to `[0, usize::MAX]`. /// Distinct from `plan_error` so clients match the code, SQLSTATE /// `2201W`, instead of parsing the message. diff --git a/nodedb-types/src/error/ctors/write_path.rs b/nodedb-types/src/error/ctors/write_path.rs index 309b1e3cd..156c5c4ce 100644 --- a/nodedb-types/src/error/ctors/write_path.rs +++ b/nodedb-types/src/error/ctors/write_path.rs @@ -207,8 +207,13 @@ impl NodeDbError { /// /// `fault` is the client text, e.g. `value is not an integer or out of /// range`. The message is `"{fault} on {collection}"`, the text the SQL - /// surfaces send. `out_of_range` picks the class: `OVERFLOW` for a result - /// out of range, `TYPE_MISMATCH` for a stored value that does not parse. + /// surfaces send. `out_of_range` picks the class, and both classes are + /// the data-exception class (`22`) the SQL surfaces send: + /// + /// - `OVERFLOW` for a result out of range (SQLSTATE `22003`). + /// - `DATA_EXCEPTION` for a stored value that does not parse (SQLSTATE + /// `22P02`). `TYPE_MISMATCH` is the class for a key that holds the + /// wrong kind of value (SQLSTATE `42846`), a different condition. pub fn kv_counter_fault( collection: impl Into, fault: impl fmt::Display, @@ -225,9 +230,9 @@ impl NodeDbError { } } else { Self { - code: ErrorCode::TYPE_MISMATCH, - message, - details: ErrorDetails::TypeMismatch { collection }, + code: ErrorCode::DATA_EXCEPTION, + message: message.clone(), + details: ErrorDetails::DataException { detail: message }, cause: None, } } diff --git a/nodedb-types/src/error/details.rs b/nodedb-types/src/error/details.rs index 3286eee9d..314d997b5 100644 --- a/nodedb-types/src/error/details.rs +++ b/nodedb-types/src/error/details.rs @@ -127,6 +127,10 @@ pub enum ErrorDetails { /// the function and the offending value. #[serde(rename = "data_exception")] DataException { detail: String }, + /// A statement exceeded a server limit on its own size or depth. + /// `detail` names the limit. + #[serde(rename = "program_limit_exceeded")] + ProgramLimitExceeded { detail: String }, /// A LIMIT/OFFSET/FETCH bound resolved outside `[0, usize::MAX]`. #[serde(rename = "invalid_limit_value")] InvalidLimitValue { clause: String, value: String }, diff --git a/nodedb-types/src/error/msgpack/constants.rs b/nodedb-types/src/error/msgpack/constants.rs index e6625c049..c2fef997a 100644 --- a/nodedb-types/src/error/msgpack/constants.rs +++ b/nodedb-types/src/error/msgpack/constants.rs @@ -86,6 +86,7 @@ // | 80 | AmbiguousColumn | // | 81 | PeriodLockMisconfigured | // | 82 | DataException | +// | 83 | ProgramLimitExceeded | pub(super) const TAG_CONSTRAINT_VIOLATION: u16 = 1; pub(super) const TAG_WRITE_CONFLICT: u16 = 2; @@ -169,3 +170,4 @@ pub(super) const TAG_UNDEFINED_COLUMN: u16 = 79; pub(super) const TAG_AMBIGUOUS_COLUMN: u16 = 80; pub(super) const TAG_PERIOD_LOCK_MISCONFIGURED: u16 = 81; pub(super) const TAG_DATA_EXCEPTION: u16 = 82; +pub(super) const TAG_PROGRAM_LIMIT_EXCEEDED: u16 = 83; diff --git a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs index 0dc316edd..1789a6ddf 100644 --- a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs +++ b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs @@ -159,6 +159,10 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { let (detail,) = read1_str(reader, field_count)?; Ok(ErrorDetails::DataException { detail }) } + TAG_PROGRAM_LIMIT_EXCEEDED => { + let (detail,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::ProgramLimitExceeded { detail }) + } TAG_INVALID_LIMIT_VALUE => { let (clause, value) = read2_str(reader, field_count)?; Ok(ErrorDetails::InvalidLimitValue { clause, value }) @@ -512,6 +516,14 @@ mod tests { assert_eq!(roundtrip(&v), v); } + #[test] + fn program_limit_exceeded_roundtrip() { + let v = ErrorDetails::ProgramLimitExceeded { + detail: "WITH RECURSIVE CTE 'walk' exceeded max recursion depth 100".into(), + }; + assert_eq!(roundtrip(&v), v); + } + #[test] fn bridge_enriched_roundtrip() { let v = ErrorDetails::Bridge { diff --git a/nodedb-types/src/error/msgpack/encode.rs b/nodedb-types/src/error/msgpack/encode.rs index 7c36827b1..d43c5f64e 100644 --- a/nodedb-types/src/error/msgpack/encode.rs +++ b/nodedb-types/src/error/msgpack/encode.rs @@ -179,6 +179,9 @@ impl ToMessagePack for ErrorDetails { } ErrorDetails::DivisionByZero => write_unit(writer, TAG_DIVISION_BY_ZERO), ErrorDetails::DataException { detail } => write1(writer, TAG_DATA_EXCEPTION, detail), + ErrorDetails::ProgramLimitExceeded { detail } => { + write1(writer, TAG_PROGRAM_LIMIT_EXCEEDED, detail) + } ErrorDetails::InvalidLimitValue { clause, value } => { write2(writer, TAG_INVALID_LIMIT_VALUE, clause, value) } diff --git a/nodedb-types/src/error/types.rs b/nodedb-types/src/error/types.rs index b967ae3df..9837849c9 100644 --- a/nodedb-types/src/error/types.rs +++ b/nodedb-types/src/error/types.rs @@ -103,6 +103,7 @@ impl NodeDbError { | ErrorDetails::AmbiguousColumn { .. } | ErrorDetails::DivisionByZero | ErrorDetails::DataException { .. } + | ErrorDetails::ProgramLimitExceeded { .. } | ErrorDetails::InvalidLimitValue { .. } | ErrorDetails::BackupTenantMismatch { .. } | ErrorDetails::BackupKeyMismatch diff --git a/nodedb-types/src/protocol/error_cause.rs b/nodedb-types/src/protocol/error_cause.rs new file mode 100644 index 000000000..838c8b2c3 --- /dev/null +++ b/nodedb-types/src/protocol/error_cause.rs @@ -0,0 +1,71 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The typed cause an error frame carries alongside its own classification. + +use serde::{Deserialize, Serialize}; + +use crate::error::{ErrorCode, ErrorDetails, NodeDbError}; + +/// The typed error that caused the one an error frame reports. +/// +/// A phase error such as `MOVE_TENANT_SNAPSHOT_FAILED` keeps its own code, +/// and the Data-Plane refusal that failed the phase rides here with its own +/// code and details. A client rebuilds it as the typed error's +/// [`NodeDbError::cause`]. +#[derive( + Debug, + Clone, + PartialEq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct ErrorCausePayload { + /// Stable numeric NodeDB code of the cause. + pub ndb_code: u16, + /// Human-readable message of the cause. + pub message: String, + /// Structured details of the cause, when it had any. + #[serde(default, skip_serializing_if = "Option::is_none")] + #[msgpack(default)] + pub details: Option, +} + +impl From<&NodeDbError> for ErrorCausePayload { + fn from(error: &NodeDbError) -> Self { + Self { + ndb_code: error.code().0, + message: error.message().to_owned(), + details: Some(error.details().clone()), + } + } +} + +impl ErrorCausePayload { + /// Rebuild the typed cause. + pub fn to_error(&self) -> NodeDbError { + NodeDbError::from_wire_with_details( + ErrorCode(self.ndb_code), + self.message.clone(), + self.details.clone(), + ) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn cause_round_trips_its_class() { + let cause = NodeDbError::division_by_zero(); + let payload = ErrorCausePayload::from(&cause); + let bytes = zerompk::to_msgpack_vec(&payload).expect("encode"); + let decoded: ErrorCausePayload = zerompk::from_msgpack(&bytes).expect("decode"); + let rebuilt = decoded.to_error(); + assert_eq!(rebuilt.code(), ErrorCode::DIVISION_BY_ZERO); + assert_eq!(rebuilt.details(), cause.details()); + } +} diff --git a/nodedb-types/src/protocol/frames.rs b/nodedb-types/src/protocol/frames.rs index 2c89de52d..827ee1db9 100644 --- a/nodedb-types/src/protocol/frames.rs +++ b/nodedb-types/src/protocol/frames.rs @@ -130,6 +130,12 @@ pub struct ErrorPayload { #[serde(default, skip_serializing_if = "Option::is_none")] #[msgpack(default)] pub details: Option, + /// The typed error that caused this one, when the server held one: the + /// Data-Plane refusal behind a phase failure, say. `None` when there is + /// no cause to report. + #[serde(default, skip_serializing_if = "Option::is_none")] + #[msgpack(default)] + pub cause: Option, } /// `skip_serializing_if` predicate for [`ErrorPayload::ndb_code`]: zero is the @@ -202,6 +208,7 @@ impl NativeResponse { message: message.into(), ndb_code, details: None, + cause: None, }), auth: None, warnings: Vec::new(), @@ -217,6 +224,15 @@ impl NativeResponse { self } + /// Attach the typed cause to an error response. A response with no error + /// payload is returned unchanged. + pub fn with_error_cause(mut self, cause: super::error_cause::ErrorCausePayload) -> Self { + if let Some(payload) = self.error.as_mut() { + payload.cause = Some(cause); + } + self + } + /// Create an auth success response. pub fn auth_ok(seq: u64, username: String, tenant_id: u64) -> Self { Self { @@ -359,6 +375,20 @@ mod tests { assert_eq!(payload.details, Some(details)); } + #[test] + fn error_payload_round_trips_the_cause() { + let cause = super::super::error_cause::ErrorCausePayload::from( + &crate::error::NodeDbError::division_by_zero(), + ); + let frame = NativeResponse::error_with_code(7, "XX000", "snapshot failed", 1602) + .with_error_cause(cause.clone()); + let bytes = zerompk::to_msgpack_vec(&frame).expect("encode"); + let decoded: NativeResponse = zerompk::from_msgpack(&bytes).expect("decode"); + let payload = decoded.error.expect("error payload survives the wire"); + assert_eq!(payload.ndb_code, 1602); + assert_eq!(payload.cause, Some(cause)); + } + #[test] fn error_payload_without_numeric_code_still_decodes() { // Hand-rolled 2-key map: exactly what a peer built before the numeric diff --git a/nodedb-types/src/protocol/mod.rs b/nodedb-types/src/protocol/mod.rs index 1034d15d3..8b16fcb9a 100644 --- a/nodedb-types/src/protocol/mod.rs +++ b/nodedb-types/src/protocol/mod.rs @@ -2,6 +2,7 @@ pub mod auth; pub mod batch; +pub mod error_cause; pub mod frames; pub mod handshake; pub mod opcodes; @@ -10,6 +11,7 @@ pub mod text_fields; pub use auth::{AuthMethod, AuthResponse}; pub use batch::{BatchDocument, BatchVector}; +pub use error_cause::ErrorCausePayload; pub use frames::{ErrorPayload, NativeRequest, NativeResponse}; pub use handshake::{ CAP_COLUMNAR, CAP_CRDT, CAP_FTS, CAP_GRAPHRAG, CAP_MSGPACK, CAP_SPATIAL, CAP_STREAMING, diff --git a/nodedb/src/control/crdt_admission.rs b/nodedb/src/control/crdt_admission.rs index 4914410a6..ce8d0f407 100644 --- a/nodedb/src/control/crdt_admission.rs +++ b/nodedb/src/control/crdt_admission.rs @@ -31,6 +31,9 @@ const FRONTIER_RETRY_LIMIT: usize = 8; pub struct AuthorizedCrdtApplyAdmissionRequest<'a> { pub authorized: AuthorizedTask, + /// The db-qualified collection (`QualifiedCollection::as_str`), the same + /// string the plan's `CrdtOp::Apply` carries. The preview and the apply + /// address the Data Plane by it, and the catalog lookup de-qualifies it. pub collection: &'a str, pub timeout: Duration, pub event_source: EventSource, @@ -40,6 +43,8 @@ pub struct AuthorizedCrdtApplyAdmissionRequest<'a> { pub struct CrdtApplyAdmissionRequest<'a> { pub tenant_id: TenantId, pub database_id: DatabaseId, + /// The db-qualified collection, equal to the plan's `CrdtOp::Apply` + /// collection. pub collection: &'a str, pub plan: PhysicalPlan, pub timeout: Duration, @@ -70,6 +75,7 @@ impl CrdtAdmissionOutcome { pub struct CrdtRestoreAdmissionRequest<'a> { pub tenant_id: TenantId, pub database_id: DatabaseId, + /// The db-qualified collection the generated restore ops address. pub collection: &'a str, pub document_id: &'a str, pub target_version_json: &'a str, @@ -147,31 +153,35 @@ pub async fn dispatch_authorized_crdt_apply_admitted_outcome( .await } +/// `collection` is db-qualified. The catalog keys collections by the bare +/// name, so the lookup de-qualifies it first. fn enforce_external_signing_policy( state: &SharedState, authorized: &AuthorizedTask, collection: &str, ) -> crate::Result<()> { + let bare = crate::control::target_identity::naming::bare_collection_name( + authorized.database_id(), + collection, + ); let stored = state .credentials .catalog() .get_collection( authorized.database_id(), authorized.tenant_id().as_u64(), - collection, + &bare, )? .ok_or_else(|| crate::Error::CollectionNotFound { tenant_id: authorized.tenant_id(), - collection: collection.to_owned(), + collection: bare.clone(), })?; if stored.crdt_signing_required && matches!(authorized.plan(), PhysicalPlan::Crdt(CrdtOp::Apply { .. })) { return Err(crate::Error::RejectedAuthz { tenant_id: authorized.tenant_id(), - resource: format!( - "collection:{collection}:unsigned_crdt_delta_requires_authenticated_sync" - ), + resource: format!("collection:{bare}:unsigned_crdt_delta_requires_authenticated_sync"), }); } Ok(()) diff --git a/nodedb/src/control/gateway/error_map/class_parity.rs b/nodedb/src/control/gateway/error_map/class_parity.rs new file mode 100644 index 000000000..8adc049cb --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity.rs @@ -0,0 +1,289 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every Data-Plane `ErrorCode` answers one class on native, pgwire and HTTP. +//! +//! pgwire renders a Data-Plane verdict as its SQLSTATE. A native client reads +//! the numeric `nodedb_types` code on the frame. The two agree when the +//! numeric code renders, through the numeric-code SQLSTATE table, in the same +//! SQLSTATE class (the first two characters) as the pgwire SQLSTATE. That same +//! table renders a code that crossed a node as a bare number, so agreement +//! also keeps a verdict's class across nodes. + +use nodedb_types::error::sqlstate; +use nodedb_types::sync::violation::ViolationType; +use nodedb_types::sync::wire::SyncProvenance; + +use crate::bridge::envelope::{CounterFault, ErrorCode, SyncHold}; +use crate::control::server::native::dispatch::{error_code_to_native, native_error_fields}; +use crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate; +use crate::control::server::pgwire::types::error_to_sqlstate; + +use super::gateway_map::GatewayErrorMap; + +/// The number of `ErrorCode` variants [`variant_index`] numbers. +const VARIANT_COUNT: usize = 41; + +/// A dense index per variant. Exhaustive, so a new variant fails to compile +/// here until it gets an index, and [`every_variant_has_a_sample`] then fails +/// until [`samples`] carries it. +fn variant_index(code: &ErrorCode) -> usize { + match code { + ErrorCode::DeadlineExceeded => 0, + ErrorCode::RejectedConstraint { .. } => 1, + ErrorCode::RejectedPrevalidation { .. } => 2, + ErrorCode::RetryableRefusal { .. } => 3, + ErrorCode::SyncRejected { .. } => 4, + ErrorCode::SyncNotApplied { .. } => 5, + ErrorCode::NotFound => 6, + ErrorCode::RejectedAuthz { .. } => 7, + ErrorCode::ConflictRetry => 8, + ErrorCode::CrdtFrontierMismatch { .. } => 9, + ErrorCode::FanOutExceeded => 10, + ErrorCode::ResourcesExhausted => 11, + ErrorCode::RejectedDanglingEdge { .. } => 12, + ErrorCode::DuplicateWrite => 13, + ErrorCode::AppendOnlyViolation { .. } => 14, + ErrorCode::BalanceViolation { .. } => 15, + ErrorCode::PeriodLocked { .. } => 16, + ErrorCode::PeriodLockMisconfigured { .. } => 17, + ErrorCode::RetentionViolation { .. } => 18, + ErrorCode::LegalHoldActive { .. } => 19, + ErrorCode::StateTransitionViolation { .. } => 20, + ErrorCode::TransitionCheckViolation { .. } => 21, + ErrorCode::TypeGuardViolation { .. } => 22, + ErrorCode::TypeMismatch { .. } => 23, + ErrorCode::CounterFault { .. } => 24, + ErrorCode::InsufficientBalance { .. } => 25, + ErrorCode::RateExceeded { .. } => 26, + ErrorCode::CollectionDraining { .. } => 27, + ErrorCode::RecursionDepthExceeded { .. } => 28, + ErrorCode::UndefinedColumn { .. } => 29, + ErrorCode::Internal { .. } => 30, + ErrorCode::Unsupported { .. } => 31, + ErrorCode::RollbackFailed { .. } => 32, + ErrorCode::OllpRetryRequired => 33, + ErrorCode::TxnOverlayMemoryExceeded { .. } => 34, + ErrorCode::DivisionByZero => 35, + ErrorCode::UndefinedFunction { .. } => 36, + ErrorCode::DataException { .. } => 37, + ErrorCode::DispatchCapacity { .. } => 38, + ErrorCode::ExpiredBeforeExecution => 39, + ErrorCode::BadRequest { .. } => 40, + } +} + +fn provenance() -> SyncProvenance { + SyncProvenance { + producer_id: 1, + epoch: 1, + stream_id: 1, + seq: 1, + } +} + +/// One sample per variant, plus one per value that picks a different +/// SQLSTATE: each constraint kind and each counter fault. +fn samples() -> Vec { + let text = || "detail".to_owned(); + let collection = || "c".to_owned(); + let mut samples = vec![ + ErrorCode::DeadlineExceeded, + ErrorCode::RejectedPrevalidation { reason: text() }, + ErrorCode::RetryableRefusal { reason: text() }, + ErrorCode::SyncRejected { + violation: ViolationType::PermissionDenied, + applied_seq: 1, + provenance: provenance(), + }, + ErrorCode::SyncRejected { + violation: ViolationType::RateLimited, + applied_seq: 1, + provenance: provenance(), + }, + ErrorCode::SyncNotApplied { + hold: SyncHold::Gap { expected: 2 }, + applied_seq: 1, + }, + ErrorCode::NotFound, + ErrorCode::RejectedAuthz { resource: text() }, + ErrorCode::ConflictRetry, + ErrorCode::CrdtFrontierMismatch { + expected: [0; 32], + actual: [1; 32], + }, + ErrorCode::FanOutExceeded, + ErrorCode::ResourcesExhausted, + ErrorCode::RejectedDanglingEdge { + missing_node: text(), + }, + ErrorCode::DuplicateWrite, + ErrorCode::AppendOnlyViolation { + collection: collection(), + }, + ErrorCode::BalanceViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::PeriodLocked { + collection: collection(), + }, + ErrorCode::PeriodLockMisconfigured { + collection: collection(), + ref_table: "periods".into(), + status_column: "status".into(), + row_identity: "p1".into(), + }, + ErrorCode::RetentionViolation { + collection: collection(), + }, + ErrorCode::LegalHoldActive { + collection: collection(), + }, + ErrorCode::StateTransitionViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TransitionCheckViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TypeGuardViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TypeMismatch { + collection: collection(), + detail: text(), + }, + ErrorCode::InsufficientBalance { + collection: collection(), + detail: text(), + }, + ErrorCode::RateExceeded { + gate: "g".into(), + retry_after_ms: 10, + }, + ErrorCode::CollectionDraining { + collection: collection(), + }, + ErrorCode::RecursionDepthExceeded { + cte_name: "walk".into(), + max_depth: 100, + }, + ErrorCode::UndefinedColumn { column: "x".into() }, + ErrorCode::Internal { detail: text() }, + ErrorCode::Unsupported { detail: text() }, + ErrorCode::RollbackFailed { + entry_index: 0, + detail: text(), + }, + ErrorCode::OllpRetryRequired, + ErrorCode::TxnOverlayMemoryExceeded { limit: 1 << 20 }, + ErrorCode::DivisionByZero, + ErrorCode::UndefinedFunction { name: "f".into() }, + ErrorCode::DataException { detail: text() }, + ErrorCode::DispatchCapacity { reason: text() }, + ErrorCode::ExpiredBeforeExecution, + ErrorCode::BadRequest { detail: text() }, + ]; + for constraint in [ + "not_null", + "unique", + "generated_always", + "fk_missing", + "rls_policy", + "permission_denied", + "check", + ] { + samples.push(ErrorCode::RejectedConstraint { + constraint: constraint.into(), + detail: text(), + }); + } + for fault in [ + CounterFault::NotAnInteger, + CounterFault::NotAFloat, + CounterFault::IntegerOverflow, + CounterFault::NonFinite, + ] { + samples.push(ErrorCode::CounterFault { + collection: collection(), + fault, + }); + } + samples +} + +fn class(state: &str) -> &str { + state.get(..2).unwrap_or(state) +} + +#[test] +fn every_variant_has_a_sample() { + let mut seen = [false; VARIANT_COUNT]; + for code in samples() { + seen[variant_index(&code)] = true; + } + let missing: Vec = (0..VARIANT_COUNT).filter(|i| !seen[*i]).collect(); + assert!(missing.is_empty(), "variants with no sample: {missing:?}"); +} + +/// The native frame carries pgwire's SQLSTATE, and its numeric code has the +/// same SQLSTATE class, on both native renderings: the typed `Err` and the +/// raw response frame. +#[test] +fn every_data_plane_code_has_one_class_on_native_and_pgwire() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, pg_state, "native SQLSTATE for {code:?}"); + let native_state = numeric_code_to_sqlstate(native.code); + assert_eq!( + class(native_state), + class(pg_state), + "{code:?}: pgwire sends {pg_state}, native code {} renders {native_state}", + native.code + ); + + let frame = error_code_to_native(1, Some(&code)); + let payload = frame.error.expect("error frames carry a payload"); + assert_eq!( + payload.code, pg_state, + "response-frame SQLSTATE for {code:?}" + ); + assert_eq!( + payload.ndb_code, native.code.0, + "response-frame code for {code:?}" + ); + } +} + +/// A classified Data-Plane verdict never reads as a server fault over HTTP. +#[test] +fn classified_data_plane_codes_are_not_http_500() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + if pg_state == sqlstate::INTERNAL_ERROR { + assert_eq!(status, 500, "{code:?}"); + } else { + assert_ne!(status, 500, "{code:?} is {pg_state} on pgwire"); + } + } +} + +/// `Unsupported` is feature-not-supported on every surface. +#[test] +fn unsupported_is_feature_not_supported_everywhere() { + let err = crate::Error::DataPlane(ErrorCode::Unsupported { + detail: "not on this engine".into(), + }); + let (_, pg_state, _) = error_to_sqlstate(&err); + assert_eq!(pg_state, sqlstate::FEATURE_NOT_SUPPORTED); + let native = native_error_fields(&err); + assert_eq!(native.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); + assert_eq!(GatewayErrorMap::to_http(&err).0, 501); +} diff --git a/nodedb/src/control/gateway/error_map/http.rs b/nodedb/src/control/gateway/error_map/http.rs index 231f3375a..b194b72c2 100644 --- a/nodedb/src/control/gateway/error_map/http.rs +++ b/nodedb/src/control/gateway/error_map/http.rs @@ -13,6 +13,8 @@ impl GatewayErrorMap { /// - 400 Bad Request for client-side errors (bad SQL, not found) /// - 403 Forbidden for authz errors /// - 409 Conflict for write-conflict / constraint violations + /// - 429 Too Many Requests for a rate-gate refusal + /// - 501 Not Implemented for an unsupported feature /// - 503 Service Unavailable for routing/leader errors and dispatch overload /// - 504 Gateway Timeout for deadline exceeded /// - 500 Internal Server Error as the default fallback @@ -31,13 +33,17 @@ impl GatewayErrorMap { Error::BadRequest { detail } => (400, detail.clone()), Error::PlanError { detail } => (400, detail.clone()), Error::RejectedConstraint { detail, .. } => (409, detail.clone()), + Error::RateExceeded { .. } => (429, err.to_string()), Error::NoLeader { .. } => (503, err.to_string()), Error::DispatchCapacity { .. } => (503, err.to_string()), Error::Serialization { .. } | Error::Codec { .. } => (500, err.to_string()), Error::Internal { .. } => (500, err.to_string()), - // 501 Not Implemented: a valid op refused because cross-core - // source-shipping is not yet supported (fail-closed safety floor). - Error::CrossCollectionNotColocated { .. } => (501, err.to_string()), + // 501 Not Implemented: a valid op this server does not support, + // such as a cross-collection write whose collections are not + // co-resident (fail-closed safety floor). + Error::CrossCollectionNotColocated { .. } | Error::FeatureNotSupported { .. } => { + (501, err.to_string()) + } Error::RemoteTyped { code, message } => { (remote_code_to_http_status(*code), message.clone()) } diff --git a/nodedb/src/control/gateway/error_map/mod.rs b/nodedb/src/control/gateway/error_map/mod.rs index b3f51476c..095a7de80 100644 --- a/nodedb/src/control/gateway/error_map/mod.rs +++ b/nodedb/src/control/gateway/error_map/mod.rs @@ -6,6 +6,8 @@ //! One module per protocol surface owns that surface's mapping, so a change //! to its SQLSTATE / HTTP / RESP / native codes is a one-file edit. +#[cfg(test)] +mod class_parity; mod gateway_map; mod http; mod native; @@ -13,6 +15,8 @@ mod pgwire; mod remote_code; mod resp; #[cfg(test)] +mod system_dispatch_refusal; +#[cfg(test)] mod test_fixtures; pub use gateway_map::GatewayErrorMap; diff --git a/nodedb/src/control/gateway/error_map/native.rs b/nodedb/src/control/gateway/error_map/native.rs index f9686cb90..b3eff1b2b 100644 --- a/nodedb/src/control/gateway/error_map/native.rs +++ b/nodedb/src/control/gateway/error_map/native.rs @@ -2,99 +2,71 @@ //! Native-protocol error shape: `(numeric code, message)`. +use nodedb_types::error::ErrorCode; + use super::gateway_map::GatewayErrorMap; use crate::Error; -/// Error code constants (subset matching `nodedb_types` numeric codes). -const CODE_NOT_LEADER: u32 = 10; -const CODE_DEADLINE: u32 = 20; -const CODE_SCHEMA_CHANGED: u32 = 30; -const CODE_NOT_FOUND: u32 = 40; -const CODE_AUTHZ: u32 = 50; -const CODE_BAD_REQUEST: u32 = 60; -const CODE_CONSTRAINT: u32 = 70; -const CODE_INTERNAL: u32 = 99; - impl GatewayErrorMap { /// Map a gateway error into `(code, message)` for the native protocol. /// - /// Error codes are aligned with `nodedb_types::error::ErrorCode` numeric - /// values so native clients can switch on the code without string matching. - pub fn to_native(err: &Error) -> (u32, String) { - match err { - Error::NotLeader { leader_addr, .. } => { - (CODE_NOT_LEADER, format!("not leader; hint: {leader_addr}")) - } - Error::DeadlineExceeded { .. } => (CODE_DEADLINE, err.to_string()), - Error::RetryableSchemaChanged { .. } => (CODE_SCHEMA_CHANGED, err.to_string()), - Error::CollectionNotFound { collection, .. } => ( - CODE_NOT_FOUND, - format!("collection \"{collection}\" not found"), - ), - Error::RejectedAuthz { .. } => (CODE_AUTHZ, err.to_string()), - Error::BadRequest { detail } | Error::PlanError { detail } => { - (CODE_BAD_REQUEST, detail.clone()) - } - Error::RejectedConstraint { detail, .. } => (CODE_CONSTRAINT, detail.clone()), - Error::CrossCollectionNotColocated { .. } => (CODE_BAD_REQUEST, err.to_string()), - Error::RemoteTyped { code, message } => { - use nodedb_types::error::ErrorCode as Ec; - let native_code = match *code { - Ec::DEADLINE_EXCEEDED => CODE_DEADLINE, - Ec::COLLECTION_NOT_FOUND => CODE_NOT_FOUND, - Ec::AUTHORIZATION_DENIED => CODE_AUTHZ, - Ec::BAD_REQUEST | Ec::PLAN_ERROR => CODE_BAD_REQUEST, - Ec::CONSTRAINT_VIOLATION => CODE_CONSTRAINT, - _ => CODE_INTERNAL, - }; - (native_code, message.clone()) - } - _ => (CODE_INTERNAL, err.to_string()), - } + /// The code and message are the ones the native error frame carries, + /// from the one native mapping the listener uses. The code is the stable + /// `nodedb_types::error::ErrorCode`, so a native client switches on it + /// without string matching. + pub fn to_native(err: &Error) -> (ErrorCode, String) { + let fields = crate::control::server::native::dispatch::native_error_fields(err); + (fields.code, fields.message) } } #[cfg(test)] mod tests { - use super::super::test_fixtures::{ - authz, deadline, internal, not_found, not_leader, schema_changed, - }; + use super::super::test_fixtures::{authz, deadline, internal, not_found, not_leader}; use super::*; #[test] fn native_not_leader() { - let (code, msg) = GatewayErrorMap::to_native(¬_leader()); - assert_eq!(code, 10); - assert!(msg.contains("hint:")); + let (code, _) = GatewayErrorMap::to_native(¬_leader()); + assert_eq!(code, ErrorCode::NOT_LEADER); } #[test] fn native_deadline() { let (code, _) = GatewayErrorMap::to_native(&deadline()); - assert_eq!(code, 20); - } - - #[test] - fn native_schema_changed() { - let (code, _) = GatewayErrorMap::to_native(&schema_changed()); - assert_eq!(code, 30); + assert_eq!(code, ErrorCode::DEADLINE_EXCEEDED); } #[test] fn native_not_found() { - let (code, _) = GatewayErrorMap::to_native(¬_found()); - assert_eq!(code, 40); + let (code, msg) = GatewayErrorMap::to_native(¬_found()); + assert_eq!(code, ErrorCode::COLLECTION_NOT_FOUND); + assert!(msg.contains("missing_col")); } #[test] fn native_authz() { let (code, _) = GatewayErrorMap::to_native(&authz()); - assert_eq!(code, 50); + assert_eq!(code, ErrorCode::AUTHORIZATION_DENIED); } #[test] fn native_internal() { let (code, _) = GatewayErrorMap::to_native(&internal()); - assert_eq!(code, 99); + assert_eq!(code, ErrorCode::INTERNAL); + } + + /// The gateway map and the native listener read one mapping, so the code + /// a gateway caller sees is the code the wire frame carries. + #[test] + fn gateway_map_matches_the_wire_frame() { + let err = Error::DataPlane(crate::bridge::envelope::ErrorCode::Unsupported { + detail: "not here".into(), + }); + let frame = crate::control::server::native::dispatch::error_to_native(1, &err); + let payload = frame.error.expect("error frames carry a payload"); + let (code, message) = GatewayErrorMap::to_native(&err); + assert_eq!(code.0, payload.ndb_code); + assert_eq!(message, payload.message); } } diff --git a/nodedb/src/control/gateway/error_map/remote_code.rs b/nodedb/src/control/gateway/error_map/remote_code.rs index 6deb5287b..6dc2f4d64 100644 --- a/nodedb/src/control/gateway/error_map/remote_code.rs +++ b/nodedb/src/control/gateway/error_map/remote_code.rs @@ -5,18 +5,49 @@ //! A remote peer on a newer build can mint a code this build does not know, //! so both helpers degrade to the generic shape rather than misclassifying. -/// Map a numeric `ErrorCode` from a `RemoteTyped` error to an HTTP status, -/// mirroring the local variant arms in `to_http` for the same condition -/// (e.g. `CONSTRAINT_VIOLATION` mirrors `RejectedConstraint`'s 409). +/// Map a numeric `ErrorCode` to an HTTP status, mirroring the local variant +/// arms in `to_http` for the same condition (e.g. `CONSTRAINT_VIOLATION` +/// mirrors `RejectedConstraint`'s 409). +/// +/// Serves a `RemoteTyped` error and every Data-Plane verdict. Each code a +/// Data-Plane code classifies to has its own status here, so a typed +/// refusal never reads as a server fault. pub(super) fn remote_code_to_http_status(code: nodedb_types::error::ErrorCode) -> u16 { use nodedb_types::error::ErrorCode as Ec; match code { - Ec::NOT_LEADER | Ec::NO_LEADER | Ec::SERVER_OVERLOAD => 503, + Ec::NOT_LEADER + | Ec::NO_LEADER + | Ec::SERVER_OVERLOAD + | Ec::MEMORY_EXHAUSTED + | Ec::COLLECTION_DRAINING => 503, Ec::DEADLINE_EXCEEDED => 504, - Ec::COLLECTION_NOT_FOUND => 404, + Ec::COLLECTION_NOT_FOUND | Ec::DOCUMENT_NOT_FOUND => 404, Ec::AUTHORIZATION_DENIED => 403, - Ec::BAD_REQUEST | Ec::PLAN_ERROR => 400, - Ec::CONSTRAINT_VIOLATION | Ec::WRITE_CONFLICT => 409, + Ec::BAD_REQUEST + | Ec::PLAN_ERROR + | Ec::TYPE_MISMATCH + | Ec::OVERFLOW + | Ec::DATA_EXCEPTION + | Ec::DIVISION_BY_ZERO + | Ec::UNDEFINED_COLUMN + | Ec::UNDEFINED_FUNCTION + | Ec::FAN_OUT_EXCEEDED + | Ec::PROGRAM_LIMIT_EXCEEDED => 400, + Ec::CONSTRAINT_VIOLATION + | Ec::WRITE_CONFLICT + | Ec::PREVALIDATION_REJECTED + | Ec::APPEND_ONLY_VIOLATION + | Ec::BALANCE_VIOLATION + | Ec::PERIOD_LOCKED + | Ec::PERIOD_LOCK_MISCONFIGURED + | Ec::STATE_TRANSITION_VIOLATION + | Ec::TRANSITION_CHECK_VIOLATION + | Ec::TYPE_GUARD_VIOLATION + | Ec::RETENTION_VIOLATION + | Ec::LEGAL_HOLD_ACTIVE + | Ec::INSUFFICIENT_BALANCE => 409, + Ec::RATE_EXCEEDED => 429, + Ec::SQL_NOT_ENABLED => 501, _ => 500, } } diff --git a/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs b/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs new file mode 100644 index 000000000..71b4e2ac4 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs @@ -0,0 +1,188 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A typed Data-Plane refusal keeps its SQLSTATE through the system door. +//! +//! `CREATE VECTOR INDEX` registers its parameters through `apply_in_engine`, +//! which dispatches `VectorOp::SetParams` through `dispatch_system`. A core +//! whose index already holds vectors refuses that plan with +//! `ErrorCode::Unsupported`. A fake core gives that exact refusal here, so the +//! test runs the real dispatch, response routing and DDL error mapping with no +//! real Data-Plane core. + +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use nodedb_physical::physical_plan::VectorOp; +use nodedb_types::error::sqlstate; + +use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; +use crate::bridge::envelope::{ErrorCode, Payload, PhysicalPlan, Response, Status}; +use crate::control::server::shared::ddl::engine_apply::apply_in_engine; +use crate::control::server::shared::ddl::sync_dispatch::{ + SystemReason, SystemTask, dispatch_system, +}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, TenantId}; +use crate::wal::WalManager; + +const COLLECTION: &str = "vectors"; + +/// The refusal a core gives `SetParams` on an index that holds vectors. +fn materialized_refusal() -> ErrorCode { + ErrorCode::Unsupported { + detail: "changing vector index params after the index holds vectors is not \ + supported; drop and recreate the collection" + .into(), + } +} + +fn set_params_plan() -> PhysicalPlan { + PhysicalPlan::Vector(VectorOp::SetParams { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLLECTION), + field_name: "emb".into(), + dim: 3, + m: 16, + ef_construction: 200, + metric: "cosine".into(), + index_type: "hnsw".into(), + pq_m: 0, + ivf_cells: 0, + ivf_nprobe: 0, + }) +} + +/// State plus the fake core's data side. The caller keeps the `TempDir` +/// alive for as long as `state` is in use. +fn fixture() -> (Arc, CoreChannelDataSide, tempfile::TempDir) { + let dir = tempfile::tempdir().expect("create test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("test.wal")).expect("open test WAL"), + ); + let (dispatcher, mut sides) = Dispatcher::new(1, 64); + let side = sides.pop().expect("one data side"); + let state = SharedState::new(dispatcher, wal).expect("construct shared state"); + (state, side, dir) +} + +/// Fake core: pops the one dispatched request and answers it with an error +/// status carrying `code`. +async fn refuse_once( + mut side: CoreChannelDataSide, + state: Arc, + code: Option, +) { + let deadline = Instant::now() + Duration::from_secs(5); + let mut handled = false; + while !handled && Instant::now() < deadline { + if let Ok(request) = side.request_rx.try_pop() { + side.response_tx + .try_push(BridgeResponse { + inner: Response { + request_id: request.inner.request_id, + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.clone().map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .expect("fake core response queue has capacity"); + handled = true; + } + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + assert!(handled, "fake core received the dispatched request"); + state.poll_and_route_responses(); +} + +/// The system door returns the refusal's own code, never `Internal`. +#[tokio::test] +async fn dispatch_system_keeps_the_refusal_code() { + let (state, side, _dir) = fixture(); + let responder = tokio::spawn(refuse_once( + side, + Arc::clone(&state), + Some(materialized_refusal()), + )); + let result = dispatch_system( + &state, + SystemTask::new( + SystemReason::DdlApply, + TenantId::new(1), + DatabaseId::DEFAULT, + COLLECTION, + set_params_plan(), + ), + Duration::from_secs(5), + ) + .await; + responder.await.expect("responder completes"); + + match result { + Err(crate::Error::DataPlane(code)) => assert_eq!(code, materialized_refusal()), + other => panic!("expected the typed Data-Plane refusal, got {other:?}"), + } +} + +/// The DDL statement answers the refusal's SQLSTATE (`0A000`) and class, with +/// the statement context before the message. +#[tokio::test] +async fn create_vector_index_refusal_keeps_its_sqlstate() { + let (state, side, _dir) = fixture(); + let responder = tokio::spawn(refuse_once( + side, + Arc::clone(&state), + Some(materialized_refusal()), + )); + let result = apply_in_engine( + &state, + TenantId::new(1), + DatabaseId::DEFAULT, + COLLECTION, + set_params_plan(), + "CREATE VECTOR INDEX", + ) + .await; + responder.await.expect("responder completes"); + + let err = result.expect_err("the refused SetParams must fail the statement"); + assert_eq!(err.sqlstate, sqlstate::FEATURE_NOT_SUPPORTED, "{err:?}"); + assert_ne!(err.sqlstate, sqlstate::INTERNAL_ERROR); + assert_eq!(err.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); + assert!( + err.message.starts_with("CREATE VECTOR INDEX: "), + "context leads the message: {}", + err.message + ); + assert!( + err.message.contains("holds vectors"), + "the refusal detail survives: {}", + err.message + ); +} + +/// A refusal with no code has no class of its own, so it is the one case +/// that stays internal. +#[tokio::test] +async fn refusal_without_a_code_is_internal() { + let (state, side, _dir) = fixture(); + let responder = tokio::spawn(refuse_once(side, Arc::clone(&state), None)); + let result = apply_in_engine( + &state, + TenantId::new(1), + DatabaseId::DEFAULT, + COLLECTION, + set_params_plan(), + "CREATE VECTOR INDEX", + ) + .await; + responder.await.expect("responder completes"); + + let err = result.expect_err("an uncoded refusal must fail the statement"); + assert_eq!(err.sqlstate, sqlstate::INTERNAL_ERROR, "{err:?}"); +} diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/balanced_gate.rs b/nodedb/src/control/planner/sql_plan_convert/dml/balanced_gate.rs index c0284b647..b1957fe90 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/balanced_gate.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/balanced_gate.rs @@ -62,6 +62,11 @@ pub(in crate::control::planner::sql_plan_convert::dml) struct WriteGates { /// split a balanced statement's boundary and refuse journals the constraint /// permits. An absent credential store or an absent collection row declares /// neither gate. +/// +/// `collection` may be bare or db-qualified: the catalog keys collections by +/// the bare name, so the lookup de-qualifies it. A qualified name looked up +/// as-is finds no row outside the default database, and the INSERT would +/// silently skip the CRDT and BALANCED routing. pub(in crate::control::planner::sql_plan_convert::dml) fn document_collection_write_gates( ctx: &ConvertContext, collection: &str, @@ -70,8 +75,10 @@ pub(in crate::control::planner::sql_plan_convert::dml) fn document_collection_wr return Ok(WriteGates::default()); }; let catalog = credentials.catalog(); + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); Ok(catalog - .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), collection)? + .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), &bare)? .map(|c| WriteGates { crdt: c.crdt, balanced: c.balanced.is_some(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/crdt_gate.rs b/nodedb/src/control/planner/sql_plan_convert/dml/crdt_gate.rs index 9a6287856..24e632b96 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/crdt_gate.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/crdt_gate.rs @@ -20,8 +20,10 @@ use nodedb_sql::types::{SqlExpr, SqlValue}; use crate::control::planner::sql_plan_convert::convert::ConvertContext; use crate::control::planner::sql_plan_convert::value::row_to_msgpack; -/// `true` when `collection` (already db-qualified by the caller) is a CRDT -/// document collection. +/// `true` when `collection` is a CRDT document collection. +/// +/// `collection` may be bare or db-qualified: the catalog keys collections by +/// the bare name, so the lookup de-qualifies it. /// /// A genuine catalog READ error propagates: misrouting a write to the non-CRDT /// path would silently bypass CRDT convergence. An ABSENT credential store or @@ -36,8 +38,10 @@ pub(in crate::control::planner::sql_plan_convert::dml) fn document_collection_is return Ok(false); }; let catalog = credentials.catalog(); + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); Ok(catalog - .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), collection)? + .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), &bare)? .map(|c| c.crdt) .unwrap_or(false)) } diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs index 3d89fdbe0..a94261e07 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs @@ -38,9 +38,13 @@ pub(in super::super::super) fn declared_primary_key_name( let Some(credentials) = ctx.credentials.as_ref() else { return Ok(None); }; + // `collection` may be bare or db-qualified. The catalog keys collections + // by the bare name. + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); credentials .catalog() - .declared_primary_key(ctx.database_id, ctx.tenant_id.as_u64(), collection) + .declared_primary_key(ctx.database_id, ctx.tenant_id.as_u64(), &bare) } /// Resolve a row's document id and surrogate, refusing a NULL or omitted diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs index d077830a8..f71aa744b 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs @@ -25,9 +25,12 @@ pub(super) fn document_collection_is_edge_bearing( let Some(credentials) = ctx.credentials.as_ref() else { return Ok(false); }; + // The catalog keys collections by the bare name. let catalog = credentials.catalog(); + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); Ok(catalog - .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), collection)? + .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), &bare)? .map(|c| c.has_implicit_edges) .unwrap_or(false)) } diff --git a/nodedb/src/control/server/http/routes/query/materialized/encode.rs b/nodedb/src/control/server/http/routes/query/materialized/encode.rs index f6d4be59c..e8effeb3d 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/encode.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/encode.rs @@ -22,13 +22,14 @@ pub(super) fn gateway_error(error: crate::Error) -> ApiError { ApiError::HttpStatus(status, msg) } +/// Map a Data-Plane refusal to the HTTP error the client reads. A typed +/// refusal takes the status its code maps to. Only a refusal with no code is +/// an internal error. pub(super) fn response_error(response: &crate::bridge::envelope::Response) -> ApiError { - let detail = response - .error_code - .as_ref() - .map(|code| format!("{code:?}")) - .unwrap_or_else(|| "unknown error".into()); - ApiError::Internal(detail) + match response.error_code.as_deref() { + Some(code) => gateway_error(crate::Error::DataPlane(code.clone())), + None => ApiError::Internal("data plane returned an error status with no error code".into()), + } } #[cfg(test)] diff --git a/nodedb/src/control/server/native/dispatch/conversion.rs b/nodedb/src/control/server/native/dispatch/conversion.rs index a69792b3d..5c2abe257 100644 --- a/nodedb/src/control/server/native/dispatch/conversion.rs +++ b/nodedb/src/control/server/native/dispatch/conversion.rs @@ -13,6 +13,15 @@ use crate::control::server::response_shape::types::{ use crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate; use crate::control::server::shared::ddl::{DdlError, DdlResult}; +/// The SQLSTATE, message and numeric code a native error frame carries for +/// one Control-Plane error. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct NativeErrorFields { + pub(crate) sqlstate: &'static str, + pub(crate) message: String, + pub(crate) code: nodedb_types::error::ErrorCode, +} + /// Convert a Control-Plane error into a native error frame. /// /// The stable numeric NodeDB code travels alongside the SQLSTATE, taken from @@ -24,7 +33,14 @@ use crate::control::server::shared::ddl::{DdlError, DdlResult}; /// The SQLSTATE is chosen here because it is a protocol-level rendering, while /// the numeric code is the classification itself. pub(crate) fn error_to_native(seq: u64, e: &crate::Error) -> NativeResponse { - let (code, message) = match e { + let fields = native_error_fields(e); + NativeResponse::error_with_code(seq, fields.sqlstate, fields.message, fields.code.0) +} + +/// The fields [`error_to_native`] puts on the frame. The one native mapping: +/// every native rendering of an `Error` reads it. +pub(crate) fn native_error_fields(e: &crate::Error) -> NativeErrorFields { + let (sqlstate, message) = match e { crate::Error::BadRequest { detail } => ("42601", detail.clone()), crate::Error::RejectedAuthz { resource, .. } => ("42501", resource.clone()), crate::Error::RateExceeded { .. } => ( @@ -76,8 +92,11 @@ pub(crate) fn error_to_native(seq: u64, e: &crate::Error) -> NativeResponse { (sqlstate, message) } }; - let ndb_code = crate::error_classify::classify(e).code().0; - NativeResponse::error_with_code(seq, code, message, ndb_code) + NativeErrorFields { + sqlstate, + message, + code: crate::error_classify::classify(e).code(), + } } /// Convert a Control-Plane error into a native error frame under a SQLSTATE @@ -183,11 +202,18 @@ pub(crate) fn ddl_result_to_native( code, message, details, + cause, }) => { let frame = NativeResponse::error_with_code(seq, sqlstate, message, code.0); - match details { + let frame = match details { Some(details) => frame.with_error_details(*details), None => frame, + }; + match cause { + Some(cause) => frame.with_error_cause( + nodedb_types::protocol::ErrorCausePayload::from(cause.as_ref()), + ), + None => frame, } } // Unknown pgwire response variants are dropped during translation, so @@ -473,6 +499,32 @@ mod tests { ); } + /// A phase failure keeps its own code, and the typed Data-Plane cause + /// rides beside it with its own code, so a client sees both. + #[test] + fn ddl_phase_failure_carries_its_typed_cause() { + let phase = nodedb_types::NodeDbError::move_tenant_snapshot_failed("7", "dispatch") + .with_cause(nodedb_types::NodeDbError::division_by_zero()); + let response = ddl_result_to_native( + 1, + Err(DdlError::move_tenant_snapshot_failed(phase.message()).with_cause_of(&phase)), + ); + + let bytes = zerompk::to_msgpack_vec(&response).expect("encode native response"); + let decoded: NativeResponse = + zerompk::from_msgpack(&bytes).expect("decode native response"); + let error = decoded.error.expect("error responses carry a payload"); + assert_eq!( + error.ndb_code, + nodedb_types::error::ErrorCode::MOVE_TENANT_SNAPSHOT_FAILED.0 + ); + let cause = error.cause.expect("the typed cause survives the wire"); + assert_eq!( + cause.to_error().code(), + nodedb_types::error::ErrorCode::DIVISION_BY_ZERO + ); + } + /// A count-bearing DDL-router status (the `{ ... }` document INSERT) is /// a DML answer: `(rows_affected, command)` exactly as the dispatch /// loop's folded tag reports, never a status row with no verb. diff --git a/nodedb/src/control/server/native/dispatch/mod.rs b/nodedb/src/control/server/native/dispatch/mod.rs index 273dc5815..0901ed537 100644 --- a/nodedb/src/control/server/native/dispatch/mod.rs +++ b/nodedb/src/control/server/native/dispatch/mod.rs @@ -31,7 +31,7 @@ pub(crate) use admission_op::admission_operation; pub(crate) use auth::{NativeAuthOutcome, handle_auth, handle_ping}; pub(crate) use conversion::{ apply_dml_outcome, ddl_result_to_native, dml_fold_error_to_native, error_code_to_native, - error_response_to_native, error_to_native, error_to_native_with_sqlstate, + error_response_to_native, error_to_native, error_to_native_with_sqlstate, native_error_fields, shape_error_to_native, to_native_columns_rows, }; pub(crate) use ctx::DispatchCtx; diff --git a/nodedb/src/control/server/native/sqlstate_code.rs b/nodedb/src/control/server/native/sqlstate_code.rs index 6f9f23212..4e0123cb2 100644 --- a/nodedb/src/control/server/native/sqlstate_code.rs +++ b/nodedb/src/control/server/native/sqlstate_code.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Numeric NodeDB codes for native error frames authored as a bare SQLSTATE. +//! Native error frames authored as a bare SQLSTATE. //! //! A native error frame carries both a SQLSTATE and the stable numeric NodeDB //! code, and the client rebuilds its typed error from the number: a frame that @@ -10,84 +10,25 @@ //! //! Most frames get their number from [`crate::error_classify::classify`], //! which is the one internal-`Error`-to-public mapping the crate owns. This -//! module exists for the frames that never held an `Error` to classify: a DDL -//! refusal ([`DdlError`](crate::control::server::shared::ddl::DdlError) is -//! authored as a SQLSTATE plus a message, in ~600 places, and has no numeric -//! code to carry), and the session/dispatch guards that reject a request with -//! a literal SQLSTATE and a static message. For those the SQLSTATE *is* the -//! only classification the server ever produced, so reading it back is a -//! lookup rather than a guess. +//! module serves the frames that never held an `Error` to classify: the +//! session and dispatch guards that reject a request with a literal SQLSTATE +//! and a static message. For those the SQLSTATE *is* the only classification +//! the server produced, so the number comes from the one SQLSTATE-to-code +//! table, [`code_for_sqlstate`], which DDL refusals read too. A bare `0A000` +//! therefore carries the same feature-not-supported class on native as on +//! pgwire. //! -//! This is the inverse of the client-side rule in -//! `NodeDbError::from_wire`, and deliberately so. There, every SQLSTATE the -//! server can emit arrives through one funnel, so a reverse mapping would have -//! to resolve `23505` into either a unique violation or a duplicate -//! idempotency key with no way to tell them apart. Here the lookup happens at -//! the site that chose the SQLSTATE, and the table only carries SQLSTATEs -//! whose NodeDB classification is unambiguous *whatever* site emitted them. -//! -//! Everything else maps to `0`, which is exactly the frame today's code ships, -//! so an unmapped SQLSTATE is never worse off than before this table existed. -//! Three groups stay unmapped on purpose: -//! -//! - **Overloaded SQLSTATEs.** `53400` is `QUOTA_OVERCOMMIT`, -//! `TENANT_QUOTA_EXCEEDED`, `DATABASE_QUOTA_EXCEEDED` and `SERVER_OVERLOAD`; -//! `0A000` is `SQL_NOT_ENABLED` and `CANNOT_CLONE_MIRROR` (the default- -//! database drop guard also sends `0A000` but has no numeric code at all). -//! A caller that knows which one it is passes the code explicitly instead -//! of routing through this table. -//! - **SQLSTATEs with no NodeDB variant.** `42P07` (duplicate table), `42704` -//! (undefined object), `25P02` (aborted transaction), `3B001` (no such -//! savepoint). These need new `ErrorCode`/`ErrorDetails` variants to type at -//! all, which is a public-API change tracked separately. -//! - **SQLSTATEs that are deliberately undistinguished.** Every credential -//! failure renders as `28P01` and every ILP auth failure as a single code -//! with one message, precisely so a caller cannot tell a wrong password from -//! an unknown user. Typing them would rebuild the oracle that collapsing -//! removed. -//! - **SQLSTATEs that mean different things to different emitters.** `57014` -//! is `query_canceled`, which this server sends both for a deadline and for -//! a cancellation that is not one — `COPY restore aborted` renders as `57014` -//! with no deadline anywhere near it. Mapping it to `DEADLINE_EXCEEDED` would -//! be the one entry here that fails the rule above, and it fails it in the -//! expensive direction: `DeadlineExceeded` is retriable, so a cancelled -//! operation would come back classified as worth retrying. A site that -//! cancels for a deadline holds the error and passes the code explicitly, as -//! the Calvin abort paths already do. +//! A SQLSTATE that more than one code shares (`0A000`, `55006`, `57014`, +//! `XX000`, `02000` in their special meanings) has a typed constant a `&str` +//! parameter rejects, so a site that means one of those special codes builds +//! its frame from the code, not from this table. -use nodedb_types::error::{ErrorCode, sqlstate}; use nodedb_types::protocol::NativeResponse; -/// The numeric NodeDB code a bare `sqlstate` classifies to, or `0` when it -/// carries no unambiguous classification. -pub(crate) fn ndb_code_for_sqlstate(sqlstate_str: &str) -> u16 { - let code = match sqlstate_str { - sqlstate::UNDEFINED_TABLE => ErrorCode::COLLECTION_NOT_FOUND, - sqlstate::INVALID_CATALOG_NAME => ErrorCode::DATABASE_NOT_FOUND, - sqlstate::INSUFFICIENT_PRIVILEGE => ErrorCode::AUTHORIZATION_DENIED, - sqlstate::UNDEFINED_FUNCTION => ErrorCode::UNDEFINED_FUNCTION, - sqlstate::UNDEFINED_COLUMN => ErrorCode::UNDEFINED_COLUMN, - sqlstate::AMBIGUOUS_COLUMN => ErrorCode::AMBIGUOUS_COLUMN, - sqlstate::DATA_EXCEPTION => ErrorCode::DATA_EXCEPTION, - // Both a malformed request and a plan that cannot be built render as - // `42601`, so this cannot say which. It does not have to: the two - // differ in which side wrote the bad statement, not in how a client - // must react, and both `BadRequest` and `PlanError` are client errors - // that no caller should retry. - sqlstate::SYNTAX_ERROR => ErrorCode::BAD_REQUEST, - // A cross-shard OCC abort and a retryable refusal both mean "nothing - // applied, retry the whole thing" — the same contract `WriteConflict` - // states, and the classification a retry loop reads. - sqlstate::SERIALIZATION_FAILURE => ErrorCode::WRITE_CONFLICT, - sqlstate::TOO_MANY_CONNECTIONS => ErrorCode::RATE_EXCEEDED, - sqlstate::INTERNAL_ERROR => ErrorCode::INTERNAL, - _ => return 0, - }; - code.0 -} +use crate::control::server::shared::ddl::result::code_for_sqlstate; /// Build a native error frame from a bare SQLSTATE, classifying it through -/// [`ndb_code_for_sqlstate`]. +/// [`code_for_sqlstate`]. /// /// Use this wherever a site rejects a request with a literal SQLSTATE and no /// `Error` value. A site that holds an `Error` must use @@ -99,30 +40,30 @@ pub(crate) fn sqlstate_error( message: impl Into, ) -> NativeResponse { let sqlstate_str = sqlstate_str.into(); - let ndb_code = ndb_code_for_sqlstate(&sqlstate_str); + let ndb_code = code_for_sqlstate(&sqlstate_str).0; NativeResponse::error_with_code(seq, sqlstate_str, message, ndb_code) } #[cfg(test)] mod tests { + use nodedb_types::error::ErrorCode; + use super::*; + fn frame_code(sqlstate: &str) -> u16 { + sqlstate_error(1, sqlstate, "refused") + .error + .expect("error frames carry a payload") + .ndb_code + } + #[test] fn classified_sqlstates_carry_their_code() { - assert_eq!( - ndb_code_for_sqlstate("42P01"), - ErrorCode::COLLECTION_NOT_FOUND.0 - ); - assert_eq!( - ndb_code_for_sqlstate("42501"), - ErrorCode::AUTHORIZATION_DENIED.0 - ); - assert_eq!(ndb_code_for_sqlstate("42601"), ErrorCode::BAD_REQUEST.0); - assert_eq!( - ndb_code_for_sqlstate("3D000"), - ErrorCode::DATABASE_NOT_FOUND.0 - ); - assert_eq!(ndb_code_for_sqlstate("XX000"), ErrorCode::INTERNAL.0); + assert_eq!(frame_code("42P01"), ErrorCode::COLLECTION_NOT_FOUND.0); + assert_eq!(frame_code("42501"), ErrorCode::AUTHORIZATION_DENIED.0); + assert_eq!(frame_code("42601"), ErrorCode::BAD_REQUEST.0); + assert_eq!(frame_code("3D000"), ErrorCode::DATABASE_NOT_FOUND.0); + assert_eq!(frame_code("XX000"), ErrorCode::INTERNAL.0); } /// A retry loop reads the numeric code, so the SQLSTATE the server sends @@ -138,38 +79,44 @@ mod tests { ); } - /// An overloaded or unmapped SQLSTATE must fall through to `0` rather than - /// pick a side: `0` is what the frame ships today, so an unknown SQLSTATE - /// is no worse off, while a wrong guess would misreport retriability. + /// A bare `0A000` guard ("opcode not supported") carries the same class a + /// DDL `0A000` refusal does, not the internal class. + #[test] + fn feature_not_supported_matches_the_ddl_class() { + assert_eq!(frame_code("0A000"), ErrorCode::SQL_NOT_ENABLED.0); + assert_eq!( + frame_code("0A000"), + crate::control::server::shared::ddl::DdlError::new("0A000", "x") + .code + .0 + ); + } + + /// Credential failures stay undistinguished: every one gets the same + /// code, so a caller cannot tell a wrong password from an unknown user. + #[test] + fn credential_failures_share_one_code() { + assert_eq!(frame_code("28P01"), frame_code("28000")); + assert!( + !nodedb_types::NodeDbError::from_wire(ErrorCode(frame_code("28P01")), "x") + .is_retriable() + ); + } + + /// `57014` is sent both for a deadline and for a cancellation that is not + /// one, so it must not classify as the retriable deadline class. #[test] - fn ambiguous_and_unknown_sqlstates_stay_unclassified() { - // Overloaded across several NodeDB variants. - assert_eq!(ndb_code_for_sqlstate("53400"), 0); - assert_eq!(ndb_code_for_sqlstate("0A000"), 0); - // No NodeDB variant exists to map onto. - assert_eq!(ndb_code_for_sqlstate("42P07"), 0); - assert_eq!(ndb_code_for_sqlstate("42704"), 0); - // Deliberately undistinguished so credential failures stay opaque. - assert_eq!(ndb_code_for_sqlstate("28P01"), 0); - assert_eq!(ndb_code_for_sqlstate("28000"), 0); - // Sent both for a deadline and for a cancellation that is not one, so - // it cannot be typed here. `DeadlineExceeded` is retriable, and a - // cancelled operation classified as retriable is one this table told a - // client to run again. - assert_eq!(ndb_code_for_sqlstate("57014"), 0); - // Not a SQLSTATE this server emits. - assert_eq!(ndb_code_for_sqlstate("99999"), 0); + fn query_canceled_is_not_retriable() { + assert_ne!(frame_code("57014"), ErrorCode::DEADLINE_EXCEEDED.0); } - /// An unclassified frame must still reach the client exactly as it does - /// today — same SQLSTATE, same message, `ndb_code == 0` — so adding the - /// table cannot regress a path it does not cover. + /// The frame keeps the SQLSTATE and message the site chose. #[test] - fn unclassified_frame_is_unchanged() { + fn frame_keeps_sqlstate_and_message() { let frame = sqlstate_error(7, "42P07", "table 'repro_t' already exists"); let payload = frame.error.expect("error frames carry a payload"); assert_eq!(payload.code, "42P07"); assert_eq!(payload.message, "table 'repro_t' already exists"); - assert_eq!(payload.ndb_code, 0); + assert_eq!(payload.ndb_code, ErrorCode::ALREADY_EXISTS.0); } } diff --git a/nodedb/src/control/server/pgwire/ddl_encode.rs b/nodedb/src/control/server/pgwire/ddl_encode.rs index 29dd1992c..8b717da47 100644 --- a/nodedb/src/control/server/pgwire/ddl_encode.rs +++ b/nodedb/src/control/server/pgwire/ddl_encode.rs @@ -43,10 +43,23 @@ pub fn ddl_results_to_pgwire( sqlstate, code, message, + cause, .. }) => { let mut info = ErrorInfo::new("ERROR".to_owned(), sqlstate, message); info.routine = Some(code.to_string()); + // The typed cause travels in `detail`: its SQLSTATE, its numeric + // code, and its message. + info.detail = cause.map(|cause| { + format!( + "caused by {} ({}): {}", + crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate( + cause.code() + ), + cause.code(), + cause.message() + ) + }); return Err(PgWireError::UserError(Box::new(info))); } }; @@ -176,6 +189,26 @@ mod tests { use super::*; + /// A phase failure keeps its SQLSTATE, and the typed cause travels in + /// `detail` with its own SQLSTATE and code. + #[test] + fn ddl_phase_failure_names_its_cause_in_detail() { + let phase = nodedb_types::NodeDbError::move_tenant_snapshot_failed("7", "dispatch") + .with_cause(nodedb_types::NodeDbError::division_by_zero()); + let result: Result, DdlError> = + Err(DdlError::move_tenant_snapshot_failed(phase.message()).with_cause_of(&phase)); + + let err = ddl_results_to_pgwire(result).expect_err("must map to a pgwire error"); + let PgWireError::UserError(info) = err else { + panic!("expected a UserError carrying ErrorInfo"); + }; + let info = *info; + assert_eq!(info.code, "XX000"); + let detail = info.detail.expect("the cause travels in detail"); + assert!(detail.contains("22012"), "{detail}"); + assert!(detail.contains("NDB-1204"), "{detail}"); + } + /// Round-trips through the actual PostgreSQL wire bytes `ErrorResponse` /// encodes and a client's `pgwire` codec decodes — proving the code /// reaches the wire, not just that the server set it. diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index ca0c85a26..875092e82 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -129,6 +129,47 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::TxnOverlayMemoryExceeded { .. } => { ("ERROR", sqlstate::PROGRAM_LIMIT_EXCEEDED, err.to_string()) } + // Control-Plane twins of Data-Plane codes take the SQLSTATE their + // Data-Plane code has, so one condition answers one class wherever + // it is detected. + crate::Error::RejectedPrevalidation { .. } | crate::Error::InsufficientBalance { .. } => { + ("ERROR", sqlstate::CHECK_VIOLATION, err.to_string()) + } + crate::Error::RetryableRefusal { .. } => { + ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) + } + crate::Error::AppendOnlyViolation { .. } => { + ("ERROR", sqlstate::APPEND_ONLY_VIOLATION, err.to_string()) + } + crate::Error::BalanceViolation { .. } => { + ("ERROR", sqlstate::BALANCE_VIOLATION, err.to_string()) + } + crate::Error::PeriodLocked { .. } => ("ERROR", sqlstate::PERIOD_LOCKED, err.to_string()), + crate::Error::PeriodLockMisconfigured { .. } => ( + "ERROR", + sqlstate::PERIOD_LOCK_MISCONFIGURED, + err.to_string(), + ), + crate::Error::RetentionViolation { .. } => { + ("ERROR", sqlstate::RETENTION_VIOLATION, err.to_string()) + } + crate::Error::LegalHoldActive { .. } => { + ("ERROR", sqlstate::LEGAL_HOLD_ACTIVE, err.to_string()) + } + crate::Error::StateTransitionViolation { .. } => ( + "ERROR", + sqlstate::STATE_TRANSITION_VIOLATION, + err.to_string(), + ), + crate::Error::TransitionCheckViolation { .. } => ( + "ERROR", + sqlstate::TRANSITION_CHECK_VIOLATION, + err.to_string(), + ), + crate::Error::TypeGuardViolation { .. } => { + ("ERROR", sqlstate::TYPE_GUARD_VIOLATION, err.to_string()) + } + crate::Error::TypeMismatch { .. } => ("ERROR", sqlstate::CANNOT_COERCE, err.to_string()), crate::Error::DeadlineExceeded { .. } => { ("ERROR", sqlstate::QUERY_CANCELED, err.to_string()) } @@ -299,6 +340,25 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> Ec::NOT_LEADER => sqlstate::DATABASE_DROPPED, // Mirrors the `CloneWriteRequiresMaterialize` arm. Ec::CLONE_WRITE_REQUIRES_MATERIALIZE => sqlstate::CLONE_WRITE_REQUIRES_MATERIALIZE.0, + // The codes below mirror the Data-Plane code table + // (`error_code_to_sqlstate`) for the public code each Data-Plane code + // classifies to, so a verdict that crossed a node as a numeric code + // renders in the class it has locally. + Ec::PREVALIDATION_REJECTED | Ec::INSUFFICIENT_BALANCE => sqlstate::CHECK_VIOLATION, + Ec::APPEND_ONLY_VIOLATION => sqlstate::APPEND_ONLY_VIOLATION, + Ec::BALANCE_VIOLATION => sqlstate::BALANCE_VIOLATION, + Ec::PERIOD_LOCKED => sqlstate::PERIOD_LOCKED, + Ec::PERIOD_LOCK_MISCONFIGURED => sqlstate::PERIOD_LOCK_MISCONFIGURED, + Ec::STATE_TRANSITION_VIOLATION => sqlstate::STATE_TRANSITION_VIOLATION, + Ec::TRANSITION_CHECK_VIOLATION => sqlstate::TRANSITION_CHECK_VIOLATION, + Ec::RETENTION_VIOLATION => sqlstate::RETENTION_VIOLATION, + Ec::LEGAL_HOLD_ACTIVE => sqlstate::LEGAL_HOLD_ACTIVE, + Ec::TYPE_GUARD_VIOLATION => sqlstate::TYPE_GUARD_VIOLATION, + Ec::TYPE_MISMATCH => sqlstate::CANNOT_COERCE, + Ec::OVERFLOW => sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, + Ec::COLLECTION_DRAINING => sqlstate::CANNOT_CONNECT_NOW, + Ec::SQL_NOT_ENABLED => sqlstate::FEATURE_NOT_SUPPORTED, + Ec::PROGRAM_LIMIT_EXCEEDED => sqlstate::PROGRAM_LIMIT_EXCEEDED, _ => sqlstate::INTERNAL_ERROR, } } diff --git a/nodedb/src/control/server/result_stream.rs b/nodedb/src/control/server/result_stream.rs index 6682bca5d..2d188768a 100644 --- a/nodedb/src/control/server/result_stream.rs +++ b/nodedb/src/control/server/result_stream.rs @@ -78,10 +78,11 @@ pub(crate) fn stream_response_channel( Err(error) => error, // `NotFound` is the one code that conversion reads as an // empty observation rather than an error. This stream - // declined to tolerate it, so it stops here. - Ok(()) => crate::Error::Dispatch { - detail: "data plane error: NotFound".to_string(), - }, + // declined to tolerate it, so it stops here with the + // code's own class. + Ok(()) => crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::NotFound, + ), }; Err(error)?; return; @@ -288,7 +289,7 @@ mod tests { } /// An untolerated `NotFound` still stops the stream rather than reading as - /// an empty success. + /// an empty success, and keeps its typed code. #[tokio::test] async fn untolerated_not_found_errors() { let (tx, rx) = mpsc::channel(8); @@ -299,7 +300,10 @@ mod tests { 1 << 20, false, ); - assert!(materialize(stream).await.is_err()); + match materialize(stream).await { + Err(crate::Error::DataPlane(ErrorCode::NotFound)) => {} + other => panic!("expected the typed NotFound refusal, got {other:?}"), + } } #[tokio::test] diff --git a/nodedb/src/control/server/shared/ddl/engine_apply.rs b/nodedb/src/control/server/shared/ddl/engine_apply.rs index b0728e087..14e79df1f 100644 --- a/nodedb/src/control/server/shared/ddl/engine_apply.rs +++ b/nodedb/src/control/server/shared/ddl/engine_apply.rs @@ -9,7 +9,7 @@ use std::time::Duration; -use crate::bridge::envelope::PhysicalPlan; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId}; @@ -17,14 +17,16 @@ use super::result::DdlError; use super::sync_dispatch::{SystemReason, SystemTask, dispatch_system}; /// Dispatch `plan` for `collection` and translate any Data-Plane refusal into -/// a [`DdlError`] carrying `sqlstate` and `context`. +/// a [`DdlError`] with `context` before its message. +/// +/// A typed refusal keeps its own SQLSTATE and code. A refusal with no code is +/// an internal error. pub(crate) async fn apply_in_engine( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, collection: &str, plan: PhysicalPlan, - sqlstate: &str, context: &str, ) -> Result<(), DdlError> { let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); @@ -41,24 +43,25 @@ pub(crate) async fn apply_in_engine( ) .await .map(|_| ()) - .map_err(|e| DdlError::new(sqlstate, format!("{context}: {e}"))) + .map_err(|e| DdlError::from_error_in_context(context, &e)) } /// Refuse a vector index definition the engine would refuse, without /// changing engine state. /// -/// `VectorOp::SetParams` refuses a core whose index already materialized. -/// Inside an explicit transaction the parameters install at COMMIT, so the -/// statement probes with the read-only `VectorOp::QueryStats` instead: an -/// index that answers has materialized, and `NotFound` means the parameters -/// will install. +/// `VectorOp::SetParams` refuses a core whose index already materialized with +/// `ErrorCode::Unsupported`. Inside an explicit transaction the parameters +/// install at COMMIT, so the statement probes with the read-only +/// `VectorOp::QueryStats` instead: an index that answers has materialized, and +/// `NotFound` means the parameters will install. A materialized index gets the +/// same `Unsupported` verdict the engine gives, so both paths answer one +/// SQLSTATE. pub(crate) async fn refuse_materialized_vector_index( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, collection: &str, field_name: &str, - sqlstate: &str, context: &str, ) -> Result<(), DdlError> { let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); @@ -79,19 +82,24 @@ pub(crate) async fn refuse_materialized_vector_index( crate::event::EventSource::User, ) .await - .map_err(|e| DdlError::new("XX000", format!("{context}: {e}")))?; + .map_err(|e| DdlError::from_error_in_context(context, &e))?; match (response.status, response.error_code.as_deref()) { - (crate::bridge::envelope::Status::Ok, _) => Err(DdlError::new( - sqlstate, - format!( - "{context}: cannot change index params after creation; drop and recreate \ - the collection" - ), + (crate::bridge::envelope::Status::Ok, _) => Err(DdlError::from_error_in_context( + context, + &crate::Error::DataPlane(ErrorCode::Unsupported { + detail: "changing vector index params after the index holds vectors is not \ + supported; drop and recreate the collection" + .into(), + }), + )), + (_, Some(ErrorCode::NotFound)) => Ok(()), + (_, Some(code)) => Err(DdlError::from_error_in_context( + &format!("{context}: vector index probe"), + &crate::Error::DataPlane(code.clone()), )), - (_, Some(crate::bridge::envelope::ErrorCode::NotFound)) => Ok(()), - (_, code) => Err(DdlError::new( + (_, None) => Err(DdlError::new( "XX000", - format!("{context}: vector index probe failed: {code:?}"), + format!("{context}: vector index probe failed with no error code"), )), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs index fd436a472..de45cf73c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs @@ -256,7 +256,7 @@ pub async fn create_continuous_aggregate( Duration::from_secs(5), ) .await - .map_err(|e| err("XX000", format!("dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; } tracing::info!( diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs index 599ecd20a..319a35247 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs @@ -121,7 +121,7 @@ pub async fn drop_continuous_aggregate( Duration::from_secs(5), ) .await - .map_err(|e| err("XX000", format!("dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; } tracing::info!(name, "continuous aggregate dropped"); diff --git a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs index 832adb071..f733a5a08 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs @@ -96,7 +96,7 @@ pub async fn crdt_state( }, ) .await - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let columns = vec!["crdt_state".to_string()]; @@ -219,14 +219,14 @@ pub async fn crdt_apply( state, crate::control::crdt_admission::AuthorizedCrdtApplyAdmissionRequest { authorized, - collection, + collection: &qualified_collection, timeout: Duration::from_secs(state.tuning.network.default_deadline_secs), event_source: crate::event::EventSource::User, policy: &policy, }, ) .await - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let columns = vec!["result".to_string()]; let mut row = Map::new(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs b/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs index 8b86fdf89..7f1a7abda 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs @@ -53,7 +53,6 @@ async fn run_one(state: &SharedState, effect: DeferredDdlEffect) -> Result<(), D database_id, collection, plan, - sqlstate, context, } => { crate::control::server::shared::ddl::engine_apply::apply_in_engine( @@ -62,7 +61,6 @@ async fn run_one(state: &SharedState, effect: DeferredDdlEffect) -> Result<(), D database_id, &collection, plan, - &sqlstate, &context, ) .await diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs index 6e2790798..0d6a1b2a1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs @@ -87,7 +87,7 @@ pub async fn crdt_merge( }, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if source_bytes.is_empty() { return Err(ddl_err( "02000", @@ -138,9 +138,9 @@ pub async fn crdt_merge( // // RLS write policies are stored keyed by `db_qualified(database_id, // collection)`, so the policy is handed that same key or it silently - // misses a policy on a non-default database. `collection` itself stays - // bare: it also feeds vShard routing and the admission request's equality - // check against this same (unqualified) plan. + // misses a policy on a non-default database. The admission request takes + // the same qualified name: it must equal the plan's `CrdtOp::Apply` + // collection, and the preview addresses the Data Plane by it. let qualified_collection = crate::control::planner::sql_plan_convert::convert::db_qualified(database_id, collection); let policy = ExternalCrdtPostImagePolicy::from_identity( @@ -156,14 +156,14 @@ pub async fn crdt_merge( state, crate::control::crdt_admission::AuthorizedCrdtApplyAdmissionRequest { authorized, - collection, + collection: &qualified_collection, timeout: Duration::from_secs(state.tuning.network.default_deadline_secs), event_source: crate::event::EventSource::User, policy: &policy, }, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs index 45290802f..d1d8172cb 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs @@ -192,7 +192,6 @@ async fn create_text_index( database_id, collection: collection.clone(), plan: set_config_plan.clone(), - sqlstate: "58000".to_string(), context: command.to_string(), }); if !deferred { @@ -202,7 +201,6 @@ async fn create_text_index( database_id, &collection, set_config_plan, - "58000", command, ) .await?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs index 74d67f5a0..1d2b8679e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs @@ -195,7 +195,6 @@ pub async fn create_vector_index( database_id, collection, &field_name, - "42P16", CONTEXT, ) .await?; @@ -206,7 +205,6 @@ pub async fn create_vector_index( database_id, collection, set_params_plan, - "42P16", CONTEXT, ) .await?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs index 2d5f1b1c9..df6d66902 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs @@ -134,7 +134,7 @@ pub async fn rag_fusion( admission: user_dispatch::RequestAdmission::AlreadyAdmitted, }) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let json_text = response_codec::decode_payload_to_json(&payload); let mut row = Map::new(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/last_value.rs b/nodedb/src/control/server/shared/ddl/neutral/last_value.rs index 9d12a5c17..2eee908ab 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/last_value.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/last_value.rs @@ -51,7 +51,7 @@ pub async fn query_last_values( }, ) .await - .map_err(|e| ddl_err("XX000", format!("dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; // `meta_query_last_values` encodes with `response_codec::encode`, which is // MessagePack — `decode_payload` is its counterpart. A JSON parser on those @@ -120,7 +120,7 @@ pub async fn query_last_value( }, ) .await - .map_err(|e| ddl_err("XX000", format!("dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; // MessagePack, as in `query_last_values` above — an absent series is // encoded as a null (decoding to `None`), which is a different fact from a diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index c6bd82749..c5ed236da 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -129,7 +129,7 @@ pub async fn rate_check( // Read TTL to compute retry_after_ms. let ttl_remaining = read_ttl_ms(state, tenant_id, vshard, &rate_key).await; Err(ddl_err( - "54001", + "53300", format!( "rate limit exceeded for {gate_name}:{key}, retry after {ttl_remaining}ms (current={current}, max={max_count})" ), diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/cutover.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/cutover.rs index d9f34985a..5dd704a9b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/cutover.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/cutover.rs @@ -143,11 +143,14 @@ async fn dispatch_rename_ops( RENAME_DISPATCH_TIMEOUT, ) .await + // The phase code stays the statement's verdict, and the typed + // dispatch error rides as its cause with its own class. .map_err(|e| { NodeDbError::move_tenant_cutover_failed( tenant_id.as_u64().to_string(), - format!("rename_collection dispatch ({old_collection} -> {new_collection}): {e}"), + format!("rename_collection dispatch ({old_collection} -> {new_collection})"), ) + .with_cause(crate::error_classify::classify(&e)) })?; } Ok(()) diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/entry.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/entry.rs index c2b2e76e3..4297742c9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/entry.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/entry.rs @@ -131,7 +131,7 @@ pub async fn handle_move_tenant( // Compensate: release drain, remove journal. drain::release(state, tenant_id, source_db_id); journal::delete_journal_entry_logged(catalog, tenant_id); - return Err(DdlError::move_tenant_snapshot_failed(e.message())); + return Err(DdlError::move_tenant_snapshot_failed(e.message()).with_cause_of(e)); } }; @@ -160,7 +160,7 @@ pub async fn handle_move_tenant( drain::release(state, tenant_id, source_db_id); let _ = snapshot::delete_temp(state, &temp_key).await; journal::delete_journal_entry_logged(catalog, tenant_id); - return Err(DdlError::move_tenant_cutover_failed(e.message())); + return Err(DdlError::move_tenant_cutover_failed(e.message()).with_cause_of(e)); } // ── Phase 5: Resume ─────────────────────────────────────────────────────── diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/recovery.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/recovery.rs index 6ae4d979f..63b5d3b4a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/recovery.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/recovery.rs @@ -116,7 +116,7 @@ pub async fn resume_or_compensate( Err(ref e) => { drain::release(state, tenant_id, source_db_id); journal::delete_journal_entry_logged(catalog, tenant_id); - return Err(DdlError::move_tenant_snapshot_failed(e.message())); + return Err(DdlError::move_tenant_snapshot_failed(e.message()).with_cause_of(e)); } }; @@ -136,7 +136,7 @@ pub async fn resume_or_compensate( let _ = snapshot::delete_temp(state, key).await; } journal::delete_journal_entry_logged(catalog, tenant_id); - return Err(DdlError::move_tenant_cutover_failed(e.message())); + return Err(DdlError::move_tenant_cutover_failed(e.message()).with_cause_of(e)); } if let Some(ref key) = entry.temp_snapshot_key { diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs index 4c848fcdc..5699bf45d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs @@ -52,8 +52,14 @@ pub async fn run( timeout, ) .await + // The phase code stays the statement's verdict, and the typed dispatch + // error rides as its cause with its own class. .map_err(|e| { - NodeDbError::move_tenant_snapshot_failed(tenant_id.as_u64().to_string(), format!("{e}")) + NodeDbError::move_tenant_snapshot_failed( + tenant_id.as_u64().to_string(), + "snapshot dispatch failed", + ) + .with_cause(crate::error_classify::classify(&e)) })?; Ok(Bytes::from(raw)) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/purge.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/purge.rs index 7b0cbf895..9d1b9de0e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/purge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/purge.rs @@ -96,6 +96,6 @@ pub async fn purge_tenant( ); Ok(status("PURGE TENANT")) } - Err(e) => Err(ddl_err("XX000", format!("purge failed: {e}"))), + Err(e) => Err(DdlError::from_error_in_context("purge failed", &e)), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/version_history/checkpoint.rs b/nodedb/src/control/server/shared/ddl/neutral/version_history/checkpoint.rs index 361613aad..42c7c568e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/version_history/checkpoint.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/version_history/checkpoint.rs @@ -55,7 +55,7 @@ pub async fn create_checkpoint( timeout, ) .await - .map_err(|e| err("XX000", format!("dispatch: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dispatch", &e))?; let vv_json = String::from_utf8(vv_bytes) .map_err(|e| err("XX000", format!("version vector decode: {e}")))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs index 7eadd1502..f63eb717b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs @@ -70,12 +70,12 @@ pub(super) async fn dispatch_authorized_read( // answer itself; the payload is flattened the same way the Data Plane's // own response is, so the two are indistinguishable to the caller. CloneCheckedOutcome::Handled(response) => { - payload_or_typed_error(response).map_err(|e| DdlError::new("XX000", format!("{e}"))) + payload_or_typed_error(response).map_err(|e| DdlError::from_error(&e)) } CloneCheckedOutcome::Proceed(checked) => { dispatch_authorized(state, checked, collection, timeout) .await - .map_err(|e| DdlError::new("XX000", format!("dispatch: {e}"))) + .map_err(|e| DdlError::from_error_in_context("dispatch", &e)) } } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/version_history/restore.rs b/nodedb/src/control/server/shared/ddl/neutral/version_history/restore.rs index c8ce1dc24..941e7f377 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/version_history/restore.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/version_history/restore.rs @@ -70,9 +70,8 @@ pub async fn restore_version( let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); // RLS write policies are stored keyed by `db_qualified(database_id, // collection)`, so the policy is handed that same key or it silently - // misses a policy on a non-default database. `collection` itself stays - // bare: it feeds vShard routing and the restore-op collection field the - // admission workflow builds internally, which must stay self-consistent. + // misses a policy on a non-default database. The admission workflow + // addresses the Data Plane by the same qualified name. let qualified_collection = crate::control::planner::sql_plan_convert::convert::db_qualified(database_id, &collection); let policy = ExternalCrdtPostImagePolicy::from_identity( @@ -89,7 +88,7 @@ pub async fn restore_version( crate::control::crdt_admission::CrdtRestoreAdmissionRequest { tenant_id, database_id, - collection: &collection, + collection: &qualified_collection, document_id: &doc_id, target_version_json: &vv_json, surrogate, @@ -100,7 +99,7 @@ pub async fn restore_version( }, ) .await - .map_err(|e| err("XX000", format!("restore dispatch: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("restore dispatch", &e))?; state .audit @@ -152,8 +151,9 @@ async fn persist_restore_delta( peer_id, delta, } = params; + let qualified = nodedb_types::QualifiedCollection::new(database_id, collection); let plan = PhysicalPlan::Crdt(CrdtOp::Apply { - collection: nodedb_types::QualifiedCollection::new(database_id, collection), + collection: qualified.clone(), document_id: document_id.to_string(), delta, peer_id, @@ -169,7 +169,7 @@ async fn persist_restore_delta( crate::control::crdt_admission::CrdtApplyAdmissionRequest { tenant_id, database_id, - collection, + collection: qualified.as_str(), plan, timeout: Duration::from_secs(state.tuning.network.default_deadline_secs), event_source: crate::event::EventSource::User, diff --git a/nodedb/src/control/server/shared/ddl/result.rs b/nodedb/src/control/server/shared/ddl/result.rs index 4e3102882..7c4bceab5 100644 --- a/nodedb/src/control/server/shared/ddl/result.rs +++ b/nodedb/src/control/server/shared/ddl/result.rs @@ -45,6 +45,9 @@ pub struct DdlError { /// The structured details of a typed verdict: the collection, gate, or /// document it names. `None` for an error built from a SQLSTATE alone. pub details: Option>, + /// The typed error that caused this one, such as the Data-Plane refusal + /// behind a MOVE TENANT phase failure. `None` when there is none. + pub cause: Option>, } impl DdlError { @@ -59,6 +62,7 @@ impl DdlError { code, message: message.into(), details: None, + cause: None, } } @@ -74,6 +78,7 @@ impl DdlError { code: public.code(), message: message.into(), details: Some(Box::new(public.details().clone())), + cause: None, } } @@ -86,6 +91,23 @@ impl DdlError { Self::from_public(sqlstate, message, &public) } + /// Build a `DdlError` from an internal error, with `context` before its + /// message. The SQLSTATE, code and details stay the error's own, so a + /// typed Data-Plane refusal keeps its class under the prefix. + pub fn from_error_in_context(context: &str, error: &crate::Error) -> Self { + let (_, sqlstate, message) = + crate::control::server::pgwire::types::error_to_sqlstate(error); + let public = crate::error_classify::classify(error); + Self::from_public(sqlstate, format!("{context}: {message}"), &public) + } + + /// Carry `error`'s cause, when it has one, as this error's cause. The + /// phase code stays this error's own, and the cause keeps its own class. + pub fn with_cause_of(mut self, error: &nodedb_types::NodeDbError) -> Self { + self.cause = error.cause().map(|cause| Box::new(cause.clone())); + self + } + /// Build a `DdlError` with an explicit code, bypassing derivation. /// Used by the named constructors below for ambiguous SQLSTATEs. fn with_code(sqlstate: &'static str, code: ErrorCode, message: impl Into) -> Self { @@ -94,6 +116,7 @@ impl DdlError { code, message: message.into(), details: None, + cause: None, } } diff --git a/nodedb/src/control/server/shared/ddl/sqlstate.rs b/nodedb/src/control/server/shared/ddl/sqlstate.rs index 117445733..90046cb75 100644 --- a/nodedb/src/control/server/shared/ddl/sqlstate.rs +++ b/nodedb/src/control/server/shared/ddl/sqlstate.rs @@ -175,12 +175,14 @@ pub fn error_code_to_sqlstate(code: &ErrorCode) -> (&'static str, &'static str, sqlstate::CHECK_VIOLATION, format!("insufficient balance on {collection}: {detail}"), ), + // The transient, retryable class, the same SQLSTATE the Control + // Plane gives `crate::Error::RateExceeded`. ErrorCode::RateExceeded { gate, retry_after_ms, } => ( "ERROR", - sqlstate::STATEMENT_TOO_COMPLEX, + sqlstate::TOO_MANY_CONNECTIONS, format!("rate limit exceeded for {gate}, retry after {retry_after_ms}ms"), ), ErrorCode::CollectionDraining { collection } => ( diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs index c85f42694..698f9eed9 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs @@ -5,11 +5,12 @@ use std::sync::Arc; use std::time::{Duration, Instant}; -use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Response, Status}; +use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Response}; use crate::control::server::dispatch_utils::{ Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, }; use crate::control::server::shared::clone_write::CloneCheckedTask; +use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::server::shared::session::statement_deadline; use crate::control::state::SharedState; use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; @@ -20,6 +21,11 @@ use super::system_task::SystemTask; /// /// This is async — it yields the Tokio thread while waiting, so the response /// poller can deliver the result without deadlocking. +/// +/// A Data-Plane refusal returns `crate::Error::DataPlane` with the response's +/// own [`crate::bridge::envelope::ErrorCode`]. Each protocol renders that code +/// through its SQLSTATE or status table. Only a refusal with no code is +/// `crate::Error::Internal`. pub(crate) async fn dispatch_system( state: &SharedState, task: SystemTask<'_>, @@ -28,34 +34,21 @@ pub(crate) async fn dispatch_system( let event_source = task.reason.event_source(); let resp = dispatch_system_response_with_source(state, task, timeout, event_source).await?; - if resp.status != Status::Ok { - // DDL/DSL callers receive the flattened message form. Callers that need - // to classify the Data-Plane rejection by type use - // `dispatch_system_response_with_source` and inspect `resp.error_code`. - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| String::from_utf8_lossy(&resp.payload).into_owned()); - return Err(crate::Error::Internal { detail }); - } - // A system task never advances the tenant's observed write-HLC. Its // `SystemReason` states that no client asked for the work, and RESTORE's // staleness gate counts only user data writes. - Ok(resp.payload.to_vec()) + payload_or_typed_error(resp) } /// Send system-initiated work and await the full [`Response`], preserving the /// typed [`crate::bridge::envelope::ErrorCode`] on a non-`Ok` status instead of /// flattening it to a string. /// -/// Infrastructure failures (dispatch, timeout, channel close) still surface as -/// typed `Error` variants. Callers that must classify a Data-Plane rejection by -/// type (e.g. the CRDT sync delta path) use this and inspect `resp.error_code`; -/// [`dispatch_system`] wraps this and flattens the code to a message -/// for DDL/DSL callers. This function does **not** advance the tenant write-HLC -/// — the caller does that on its own success path. +/// Infrastructure failures (dispatch, timeout, channel close) surface as typed +/// `Error` variants. Callers that inspect the whole response (the CRDT sync +/// delta path, CONVERT) use this. [`dispatch_system`] wraps it and returns the +/// payload or the typed refusal. This function does **not** advance the +/// tenant write-HLC — the caller does that on its own success path. pub(crate) async fn dispatch_system_response_with_source( state: &SharedState, task: SystemTask<'_>, @@ -86,6 +79,9 @@ pub(crate) async fn dispatch_system_response_with_source( /// Send clone-checked, already-authorized work to the Data Plane and await its /// payload. /// +/// A Data-Plane refusal returns `crate::Error::DataPlane` with its own code, +/// the same shape [`dispatch_system`] returns. +/// /// The capability is consumed here: the plan that reaches storage is the plan /// authorization approved, so a caller cannot authorize one shape and dispatch /// another. Client-reachable paths use this rather than the system door. @@ -119,16 +115,7 @@ pub(crate) async fn dispatch_authorized( ) .await?; - if resp.status != Status::Ok { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| String::from_utf8_lossy(&resp.payload).into_owned()); - return Err(crate::Error::Internal { detail }); - } - - Ok(resp.payload.to_vec()) + payload_or_typed_error(resp) } /// What [`dispatch_plan`] sends, and where. diff --git a/nodedb/src/control/server/shared/response_payload.rs b/nodedb/src/control/server/shared/response_payload.rs index dc51f0284..710681367 100644 --- a/nodedb/src/control/server/shared/response_payload.rs +++ b/nodedb/src/control/server/shared/response_payload.rs @@ -14,8 +14,8 @@ use crate::bridge::envelope::{Response, Status}; /// the detail. /// /// Shared by every door that returns bytes rather than a `Response`: the -/// replicated sync write, the user DDL/DSL dispatch, and the version-history -/// read. One copy so a response a clone hook served and a response the Data +/// replicated sync write, the user DDL/DSL dispatch, the system dispatch, and +/// the version-history read. One copy so a response a clone hook served and a response the Data /// Plane returned cannot be reported differently. pub(crate) fn payload_or_typed_error(response: Response) -> crate::Result> { if response.status != Status::Ok { diff --git a/nodedb/src/control/server/shared/session/ddl_effect.rs b/nodedb/src/control/server/shared/session/ddl_effect.rs index 6c176202d..668346f89 100644 --- a/nodedb/src/control/server/shared/session/ddl_effect.rs +++ b/nodedb/src/control/server/shared/session/ddl_effect.rs @@ -22,13 +22,13 @@ pub enum DeferredDdlEffect { /// `Ready`. SecondaryIndexBuild(SecondaryIndexBuild), /// Apply an index's engine configuration, such as a full-text analyzer - /// binding. A refusal fails the COMMIT with `sqlstate`. + /// binding. A refusal fails the COMMIT with the refusal's own SQLSTATE, + /// and `context` leads its message. EngineApply { tenant_id: TenantId, database_id: DatabaseId, collection: String, plan: PhysicalPlan, - sqlstate: String, context: String, }, /// Remove a dropped index's engine state: a secondary index's entries, diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index 1025c0dd1..8f2be245b 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -16,11 +16,19 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { collection, constraint, detail, - } => NodeDbError::constraint_violation(collection.clone(), constraint.clone(), detail), + } => crate::error_from_data_plane::rejected_constraint_to_public( + collection.clone(), + constraint.clone(), + detail.clone(), + ), Error::RejectedAuthz { resource, .. } => { NodeDbError::authorization_denied(resource.clone()) } - Error::TxnOverlayMemoryExceeded { .. } => NodeDbError::bad_request(e.to_string()), + // `54000` on the SQL surfaces, the class `TxnOverlayMemoryExceeded` + // has as a Data-Plane code. + Error::TxnOverlayMemoryExceeded { .. } => { + NodeDbError::program_limit_exceeded(e.to_string()) + } Error::BackupTenantMismatch { expected, actual } => { NodeDbError::backup_tenant_mismatch(*expected, *actual) } @@ -151,16 +159,23 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { shards_touched, limit, } => NodeDbError::fan_out_exceeded(*shards_touched, *limit), - Error::CrossCollectionNotColocated { .. } => NodeDbError::bad_request(e.to_string()), + // `0A000` on the SQL surfaces, the class `Unsupported` has as a + // Data-Plane code. + Error::CrossCollectionNotColocated { .. } => NodeDbError::from_wire( + nodedb_types::error::ErrorCode::SQL_NOT_ENABLED, + e.to_string(), + ), Error::BadRequest { detail } => NodeDbError::bad_request(detail), Error::QuotaOvercommit { field, detail } => { NodeDbError::quota_overcommit(field.clone(), detail) } Error::PlanError { detail } => NodeDbError::plan_error(detail), - // The native surface carries this as a plan error; the pgwire - // surface renders SQLSTATE `0A000`. - Error::FeatureNotSupported { detail } => NodeDbError::plan_error(detail), + // `0A000` on every surface, the class `Unsupported` has as a + // Data-Plane code. + Error::FeatureNotSupported { detail } => { + NodeDbError::from_wire(nodedb_types::error::ErrorCode::SQL_NOT_ENABLED, detail) + } Error::UndefinedFunction { name } => NodeDbError::undefined_function(name.clone()), Error::UndefinedObject { kind, name } => { NodeDbError::undefined_object(format!("{kind} \"{name}\"")) diff --git a/nodedb/src/error_from_data_plane.rs b/nodedb/src/error_from_data_plane.rs index 7da49c36e..effb964b5 100644 --- a/nodedb/src/error_from_data_plane.rs +++ b/nodedb/src/error_from_data_plane.rs @@ -9,9 +9,10 @@ //! key is indistinguishable from a crashed database — so the compiler is made //! to name every new variant here instead of a catch-all absorbing it. -use nodedb_types::error::{ErrorCode as PublicCode, NodeDbError}; +use nodedb_types::error::{ErrorCode as PublicCode, NodeDbError, sqlstate}; use crate::bridge::envelope::ErrorCode; +use crate::control::server::shared::ddl::sqlstate::constraint_sqlstate; /// Convert a deterministic Data-Plane code into the public error a client /// can classify. @@ -30,7 +31,7 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { // only the constraint kind and detail — leave collection blank // rather than misreport the kind string as the collection. ErrorCode::RejectedConstraint { constraint, detail } => { - NodeDbError::constraint_violation("", constraint, detail) + rejected_constraint_to_public(String::new(), constraint, detail) } ErrorCode::RejectedPrevalidation { reason } => { NodeDbError::prevalidation_rejected("data plane", reason) @@ -139,12 +140,16 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { ErrorCode::RecursionDepthExceeded { cte_name, max_depth, - } => NodeDbError::bad_request(format!( + } => NodeDbError::program_limit_exceeded(format!( "WITH RECURSIVE CTE '{cte_name}' exceeded max recursion depth {max_depth}; \ add a stricter termination condition or raise max_recursion_depth" )), ErrorCode::UndefinedColumn { column } => NodeDbError::undefined_column(column), - ErrorCode::Unsupported { detail } => NodeDbError::bad_request(detail), + // `0A000` (feature_not_supported). `SQL_NOT_ENABLED` is the class + // every bare `0A000` refusal carries. + ErrorCode::Unsupported { detail } => { + NodeDbError::from_wire(PublicCode::SQL_NOT_ENABLED, detail) + } ErrorCode::DivisionByZero => NodeDbError::division_by_zero(), ErrorCode::UndefinedFunction { name } => NodeDbError::undefined_function(name), ErrorCode::DataException { detail } => NodeDbError::data_exception(detail), @@ -152,13 +157,14 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { // Nothing was enqueued, and the same request succeeds once capacity // frees: the retryable overload class. ErrorCode::DispatchCapacity { reason } => NodeDbError::server_overload(reason), - ErrorCode::TxnOverlayMemoryExceeded { limit } => NodeDbError::bad_request(format!( - "transaction staging overlay exceeded its {limit}-byte per-core budget; \ - split the transaction into smaller batches" - )), - // Genuinely internal: the shard is in an unknown or faulted state, or - // a scheduler signal leaked past the layer that should have consumed - // it. These are the only codes for which NDB-9000 is the truth. + ErrorCode::TxnOverlayMemoryExceeded { limit } => { + NodeDbError::program_limit_exceeded(format!( + "transaction staging overlay exceeded its {limit}-byte per-core budget; \ + split the transaction into smaller batches" + )) + } + // Genuinely internal: the shard is in an unknown or faulted state. + // These are the only codes for which NDB-9000 is the truth. ErrorCode::Internal { detail } => NodeDbError::internal(detail), ErrorCode::RollbackFailed { entry_index, @@ -167,9 +173,33 @@ pub(crate) fn data_plane_code_to_public(code: ErrorCode) -> NodeDbError { "transaction rollback failed at undo entry {entry_index}: {detail}; \ shard state is unknown — restart required" )), - ErrorCode::OllpRetryRequired => { - NodeDbError::internal("optimistic predicate retry required") - } + // A scheduler signal that reached a client: nothing was written, and + // the retry that the signal asks for succeeds, so it takes the + // retriable class the SQL surfaces send (`40001`). + ErrorCode::OllpRetryRequired => NodeDbError::from_wire( + PublicCode::WRITE_CONFLICT, + "optimistic predicate retry required; retry the transaction", + ), + } +} + +/// The public error for a rejected constraint of kind `constraint`. +/// +/// The class follows the SQLSTATE the SQL surfaces send for the kind +/// ([`constraint_sqlstate`]): an RLS or permission refusal is an +/// authorization denial (`42501`), a write to a generated column is a bad +/// request (`428C9`), and every other kind is a constraint violation (`23`). +/// Shared by the Data-Plane code and the Control-Plane variant, so both +/// classify one kind the same way. +pub(crate) fn rejected_constraint_to_public( + collection: String, + constraint: String, + detail: String, +) -> NodeDbError { + match constraint_sqlstate(&constraint) { + sqlstate::INSUFFICIENT_PRIVILEGE => NodeDbError::authorization_denied(detail), + sqlstate::GENERATED_ALWAYS => NodeDbError::bad_request(detail), + _ => NodeDbError::constraint_violation(collection, constraint, detail), } } @@ -216,15 +246,15 @@ mod tests { collection: "counters".into(), fault: CounterFault::NotAnInteger, }); - assert_eq!(e.code(), PublicCode::TYPE_MISMATCH); + assert_eq!(e.code(), PublicCode::DATA_EXCEPTION); assert_eq!( e.message(), "value is not an integer or out of range on counters" ); assert_eq!( e.details(), - &nodedb_types::error::ErrorDetails::TypeMismatch { - collection: "counters".into() + &nodedb_types::error::ErrorDetails::DataException { + detail: "value is not an integer or out of range on counters".into() } ); diff --git a/nodedb/tests/inproc/cases/system_task_call_sites.rs b/nodedb/tests/inproc/cases/system_task_call_sites.rs index 785e467e7..8282f4e4b 100644 --- a/nodedb/tests/inproc/cases/system_task_call_sites.rs +++ b/nodedb/tests/inproc/cases/system_task_call_sites.rs @@ -83,8 +83,85 @@ fn rust_files(dir: &Path, out: &mut Vec) { } } -/// Every `SystemTask::new` in the tree is in a file that named itself as -/// system-initiated work. +/// The file that declares the module `file` holds, and that module's name. +/// +/// `a/b.rs` and `a/b/mod.rs` are both module `b`, declared in `a/mod.rs`, +/// `a.rs`, or the crate root (`lib.rs`, `main.rs`) when `a` is `src`. +fn declaring_file(root: &Path, file: &Path) -> Option<(PathBuf, String)> { + let stem = file.file_stem()?.to_str()?; + let (name, dir) = if stem == "mod" { + let dir = file.parent()?; + (dir.file_name()?.to_str()?.to_owned(), dir.parent()?) + } else { + (stem.to_owned(), file.parent()?) + }; + if dir == root { + return ["lib.rs", "main.rs"] + .iter() + .map(|f| root.join(f)) + .find(|p| std::fs::read_to_string(p).is_ok_and(|b| declared_attrs(&b, &name).is_some())) + .map(|p| (p, name)); + } + [dir.join("mod.rs"), dir.with_extension("rs")] + .into_iter() + .find(|p| p.is_file()) + .map(|p| (p, name)) +} + +/// The attribute lines directly above `mod ;` in `body`, or `None` +/// when `body` does not declare that module. +fn declared_attrs(body: &str, name: &str) -> Option> { + let lines: Vec<&str> = body.lines().collect(); + let target = format!("mod {name};"); + let index = lines.iter().position(|line| { + let line = line.trim(); + line == target || (line.starts_with("pub") && line.ends_with(&format!(" {target}"))) + })?; + let attrs = lines[..index] + .iter() + .rev() + .map(|line| line.trim()) + .take_while(|line| line.starts_with("#[") || line.starts_with("//")) + .filter(|line| line.starts_with("#[")) + .map(str::to_owned) + .collect(); + Some(attrs) +} + +/// Whether `file` compiles only under `cfg(test)`: the file opens with +/// `#![cfg(test)]`, its `mod` declaration carries `#[cfg(test)]`, or its +/// parent module does. Such a file never reaches a production build, so a +/// `SystemTask` it builds is a test fixture, not a dispatch door. +fn is_test_only(root: &Path, file: &Path) -> bool { + let Ok(body) = std::fs::read_to_string(file) else { + return false; + }; + if body.lines().any(|line| line.trim() == "#![cfg(test)]") { + return true; + } + let Some((parent, name)) = declaring_file(root, file) else { + return false; + }; + let Ok(parent_body) = std::fs::read_to_string(&parent) else { + return false; + }; + match declared_attrs(&parent_body, &name) { + Some(attrs) if attrs.iter().any(|a| a == "#[cfg(test)]") => true, + Some(_) => { + parent.as_path() != file && !is_crate_root(root, &parent) && is_test_only(root, &parent) + } + None => false, + } +} + +fn is_crate_root(root: &Path, file: &Path) -> bool { + file == root.join("lib.rs") || file == root.join("main.rs") +} + +/// Every `SystemTask::new` in a production build is in a file that named +/// itself as system-initiated work. Test-only files are skipped: they never +/// compile into the server, and a fixture that drives the system door must +/// not need a production allowlist entry. #[test] fn system_task_is_constructed_only_by_system_initiated_code() { let root = src_root(); @@ -100,7 +177,7 @@ fn system_task_is_constructed_only_by_system_initiated_code() { for file in &files { let body = std::fs::read_to_string(file) .unwrap_or_else(|e| panic!("read {}: {e}", file.display())); - if !body.contains("SystemTask::new") { + if !body.contains("SystemTask::new") || is_test_only(&root, file) { continue; } let relative = file @@ -141,3 +218,21 @@ fn every_allowlisted_file_still_constructs_a_system_task() { "the SystemTask allowlist has stale entries; remove them: {stale:#?}" ); } + +/// The test-only skip can never hide a production caller: no allowlisted +/// file reads as test-only, and a file declared under `#[cfg(test)]` does. +#[test] +fn test_only_detection_matches_the_module_tree() { + let root = src_root(); + for entry in ALLOWED { + assert!( + !is_test_only(&root, &root.join(entry)), + "{entry} is production code but reads as test-only" + ); + } + let fixture = root.join("control/gateway/error_map/test_fixtures.rs"); + assert!( + is_test_only(&root, &fixture), + "a `#[cfg(test)] mod` file must read as test-only" + ); +} diff --git a/nodedb/tests/native/cases/native_kv_counter_faults.rs b/nodedb/tests/native/cases/native_kv_counter_faults.rs index fea37ce41..295c34577 100644 --- a/nodedb/tests/native/cases/native_kv_counter_faults.rs +++ b/nodedb/tests/native/cases/native_kv_counter_faults.rs @@ -5,8 +5,8 @@ //! `KV_INCR` on a raw value that is not a decimal integer, or past the i64 //! range, is refused by the Data Plane with the collection it ran on. The //! native frame carries the data-exception SQLSTATE pgwire sends, the same -//! message text, the numeric code, and structured details that name the -//! collection. +//! message text, a numeric code of the same data-exception class, and +//! structured details that name the collection. use nodedb_test_support::native_harness::{do_handshake, send_sql}; use nodedb_test_support::pgwire_harness::TestServer; @@ -71,14 +71,10 @@ async fn native_counter_parse_fault_names_the_collection() { ) .await; assert_eq!(err.code, sqlstate::INVALID_TEXT_REPRESENTATION); - assert_eq!(err.ndb_code, ErrorCode::TYPE_MISMATCH.0); - assert_eq!( - err.message, - format!("value is not an integer or out of range on {COLLECTION}") - ); - let expected = ErrorDetails::TypeMismatch { - collection: COLLECTION.into(), - }; + assert_eq!(err.ndb_code, ErrorCode::DATA_EXCEPTION.0); + let message = format!("value is not an integer or out of range on {COLLECTION}"); + assert_eq!(err.message, message); + let expected = ErrorDetails::DataException { detail: message }; assert_eq!(err.details.as_ref(), Some(&expected)); let typed = NodeDbError::from_wire_with_details( diff --git a/nodedb/tests/wire/cases/crdt_write_rls_database_scope.rs b/nodedb/tests/wire/cases/crdt_write_rls_database_scope.rs index a9a83a481..8b755a49f 100644 --- a/nodedb/tests/wire/cases/crdt_write_rls_database_scope.rs +++ b/nodedb/tests/wire/cases/crdt_write_rls_database_scope.rs @@ -132,14 +132,11 @@ async fn crdt_merge_in_non_default_database_is_rls_enforced() { "a write policy on a non-default-database CRDT collection must reject \ a CRDT MERGE its predicate forbids", ); - // `CRDT MERGE`'s handler wraps every admission failure — RLS denial - // included — under this one SQLSTATE (`crdt_merge.rs`'s `dispatch_... - // .map_err(|e| ddl_err("XX000", ...))`); the substantive assertion is - // that an error surfaces here at all, since pre-fix the merge applied - // silently and no error, of any code, reached the client. + // The RLS denial keeps its own class through the `CRDT MERGE` handler: + // `42501` (insufficient_privilege). assert_eq!( - sqlstate, "XX000", - "expected the CRDT admission failure's SQLSTATE, got: {sqlstate}" + sqlstate, "42501", + "expected the RLS denial's SQLSTATE, got: {sqlstate}" ); assert_eq!( title_of(&server, "crdt_rls_notes", "target").await, @@ -199,8 +196,8 @@ async fn crdt_merge_in_default_database_is_still_rls_enforced() { MERGE its predicate forbids", ); assert_eq!( - sqlstate, "XX000", - "expected the CRDT admission failure's SQLSTATE, got: {sqlstate}" + sqlstate, "42501", + "expected the RLS denial's SQLSTATE, got: {sqlstate}" ); assert_eq!( title_of(&server, "crdt_rls_default_notes", "target").await, From 9d4faaf258c290175e6106074aded697d39a4754 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 16:53:49 +0800 Subject: [PATCH 50/64] refactor(routing): key the vShard hash and surrogates on a canonical collection key Introduce CollectionKey, the only (database_id, bare_name) pair the vShard hash and surrogate allocator accept. Every caller that used to hash a raw string, or DatabaseId::from_collection_in_database, now builds a CollectionKey via from_bare or from_qualified so a database-qualified name can never reach the hash and route rows to the wrong vShard or bind surrogates under the wrong key. Thread the type through nodedb-cluster routing, the Calvin sequencer and its diagnostics, physical layer surrogate assignment, control plane bind/dispatch paths, and the cluster and inproc test suites. Add a diagnostic capture for a Calvin batch whose participant set cannot be derived because a key-set collection name is missing its database qualifier. --- .../cases/calvin_3node_normal.rs | 25 +- .../cases/calvin_3node_shard_failover.rs | 17 +- .../cluster_suite/cases/calvin_e2e_ollp.rs | 18 +- .../cluster_suite/cases/calvin_e2e_pgwire.rs | 17 +- .../cases/calvin_sequencer_failover.rs | 9 +- .../cases/assign_surrogate_cross_node.rs | 6 +- .../cluster_array_cell_raft_replication.rs | 6 +- .../cases/cluster_crdt_replication.rs | 4 +- .../cases/cluster_surrogate_replication.rs | 3 +- .../cases/cross_node_read_occ_abort.rs | 6 +- .../cases/gather_join_in_txn_occ.rs | 6 +- .../cases/hilo_surrogate_uniqueness.rs | 5 +- .../cases/kv_atomic_autocommit_replicates.rs | 4 +- .../cases/materialized_sum_cross_core.rs | 6 +- .../cases/materialized_sum_cross_shard.rs | 6 +- .../cases/materialized_sum_replication.rs | 10 +- .../cases/multi_replica_data_groups.rs | 4 +- .../cases/native_gather_join_in_txn_occ.rs | 6 +- .../proposal_committed_twice_applies_once.rs | 10 +- .../cases/shuffle_aggregate_in_txn_occ.rs | 6 +- .../cases/shuffle_join_cost_model.rs | 10 +- .../cases/shuffle_join_end_to_end.rs | 20 +- .../cases/shuffle_join_in_txn_occ.rs | 6 +- .../cases/single_node_calvin_graph_txn.rs | 6 +- .../single_node_calvin_hot_key_reservation.rs | 10 +- .../tests/common_suite/cases/sync_failover.rs | 2 +- .../tests/common_suite/cases/vshard_names.rs | 11 +- .../cases/calvin_sequencer_starvation.rs | 9 +- .../join_cross_node.rs | 20 +- nodedb-cluster/src/array_routing.rs | 2 +- nodedb-cluster/src/calvin/sequencer/entry.rs | 6 +- nodedb-cluster/src/calvin/sequencer/inbox.rs | 6 +- nodedb-cluster/src/calvin/sequencer/replay.rs | 34 +- .../src/calvin/sequencer/service/core.rs | 6 +- .../calvin/sequencer/state_machine/apply.rs | 33 +- .../src/calvin/sequencer/validator.rs | 6 +- nodedb-cluster/src/calvin/types/sequencer.rs | 10 +- .../src/calvin/types/transaction.rs | 109 ++++-- nodedb-cluster/src/diag/context.rs | 35 ++ nodedb-cluster/src/diag/inert.rs | 3 + nodedb-cluster/src/diag/mod.rs | 8 +- nodedb-cluster/src/diag/recording.rs | 20 + nodedb-cluster/src/error.rs | 5 + nodedb-cluster/src/routing.rs | 33 +- nodedb-cluster/src/rpc_codec/surrogate.rs | 3 +- nodedb-cluster/src/shard_split.rs | 20 +- nodedb-physical/src/convert_context.rs | 7 +- nodedb-physical/src/surrogate.rs | 15 +- .../src/cluster_harness/node/inspect/crdt.rs | 6 +- .../cluster_harness/node/inspect/snapshot.rs | 8 +- .../cluster_harness/node/inspect/topology.rs | 5 +- nodedb-types/src/id/collection_key.rs | 209 +++++++++++ nodedb-types/src/id/mod.rs | 2 + nodedb-types/src/id/vshard.rs | 39 +- nodedb-types/src/lib.rs | 4 +- nodedb/src/bootstrap/constraint_reconcile.rs | 4 +- nodedb/src/control/array_catalog/persist.rs | 9 +- nodedb/src/control/backup/orchestrator.rs | 11 +- .../control/backup/restore/crdt_reissue.rs | 5 +- nodedb/src/control/backup/restore/durable.rs | 8 +- .../src/control/backup/restore/kv_reissue.rs | 10 +- .../backup/restore/orchestrate/rebind.rs | 3 +- .../backup/restore/redo_reissue/commit.rs | 5 +- .../backup/restore/redo_reissue/documents.rs | 3 +- .../backup/restore/redo_reissue/edges.rs | 9 +- nodedb/src/control/backup/snapshot_keys.rs | 2 +- .../control/catalog_entry/apply/collection.rs | 3 +- .../post_apply/async_dispatch/vector.rs | 4 +- nodedb/src/control/clone/copyup.rs | 21 +- nodedb/src/control/clone/resolver/resolve.rs | 10 +- nodedb/src/control/clone/resolver/rewrite.rs | 6 +- .../calvin/scheduler/driver/core/catch_up.rs | 12 +- .../calvin/scheduler/driver/core/routing.rs | 76 ++-- .../scheduler/driver/core/test_support.rs | 4 +- .../src/control/cluster/snapshot_applier.rs | 13 +- .../src/control/cluster/snapshot_builder.rs | 13 +- nodedb/src/control/crdt_admission.rs | 18 +- nodedb/src/control/exec_receiver/executor.rs | 25 +- .../src/control/exec_receiver/plan_decode.rs | 9 +- .../src/control/gateway/colocation_guard.rs | 13 +- .../error_map/system_dispatch_refusal.rs | 3 +- nodedb/src/control/gateway/router.rs | 43 ++- nodedb/src/control/insert_select/copy_rows.rs | 3 +- .../control/insert_select/expand_staged.rs | 9 +- .../clone_materializer/columnar.rs | 3 +- .../clone_materializer/dispatch.rs | 4 +- .../clone_materializer/document.rs | 12 +- .../maintenance/clone_materializer/kv.rs | 9 +- .../merge_orchestrator/expand_staged_merge.rs | 10 +- .../merge_orchestrator/orchestrator.rs | 6 +- nodedb/src/control/orchestrated_write.rs | 6 +- nodedb/src/control/planner/auto_tier.rs | 4 +- nodedb/src/control/planner/calvin/dispatch.rs | 38 +- .../control/planner/calvin/dispatch_multi.rs | 10 +- nodedb/src/control/planner/calvin/preexec.rs | 4 +- .../control/planner/calvin/submit/local.rs | 3 + .../calvin/tx_class/dependent_builder.rs | 5 +- .../control/planner/calvin/tx_class/shared.rs | 10 +- .../planner/calvin/tx_class/static_builder.rs | 23 +- .../control/planner/implicit_edges/insert.rs | 28 +- .../control/planner/implicit_edges/routed.rs | 52 +-- .../planner/materialized_sum/cross_shard.rs | 3 +- .../control/planner/materialized_sum/recon.rs | 5 +- .../planner/materialized_sum/resolve.rs | 35 +- .../planner/materialized_sum/settle.rs | 22 +- .../src/control/planner/period_lock/lookup.rs | 3 +- .../procedural/executor/core/dispatch.rs | 13 +- .../sql_plan_convert/aggregate/plan.rs | 8 +- .../sql_plan_convert/aggregate/spec.rs | 4 +- .../sql_plan_convert/array_alter_convert.rs | 4 +- .../sql_plan_convert/array_convert/ddl.rs | 6 +- .../sql_plan_convert/array_convert/dml.rs | 10 +- .../array_fn_convert/aggregate.rs | 4 +- .../array_fn_convert/elementwise.rs | 4 +- .../array_fn_convert/maint.rs | 6 +- .../array_fn_convert/project.rs | 4 +- .../array_fn_convert/slice.rs | 4 +- .../planner/sql_plan_convert/convert.rs | 53 ++- .../sql_plan_convert/dml/insert/convert.rs | 9 +- .../sql_plan_convert/dml/insert/identity.rs | 24 +- .../planner/sql_plan_convert/dml/kv_insert.rs | 7 +- .../planner/sql_plan_convert/dml/merge.rs | 5 +- .../dml/update_delete/delete.rs | 7 +- .../dml/update_delete/update.rs | 11 +- .../dml/update_delete/update_from.rs | 4 +- .../planner/sql_plan_convert/dml/upsert.rs | 9 +- .../sql_plan_convert/dml/vector_primary.rs | 23 +- .../planner/sql_plan_convert/scan/core.rs | 21 +- .../planner/sql_plan_convert/scan/join.rs | 6 +- .../sql_plan_convert/scan/recursive.rs | 5 +- .../planner/sql_plan_convert/scan/search.rs | 27 +- .../planner/sql_plan_convert/scan/spatial.rs | 6 +- .../sql_plan_convert/scan/timeseries.rs | 22 +- .../planner/sql_plan_convert/set_ops.rs | 12 +- .../src/control/security/auth_fence/view.rs | 5 +- .../control/security/catalog/surrogate_pk.rs | 231 +++++++----- .../security/permission_tree/reload.rs | 4 +- .../security/permission_tree/sources.rs | 11 +- .../src/control/server/calvin_submit/hook.rs | 10 +- .../server/calvin_submit/inbox_hook.rs | 13 +- .../control/server/dispatch_utils/dispatch.rs | 2 +- .../server/exchange/all_cores/dispatch.rs | 2 +- nodedb/src/control/server/exchange/gather.rs | 4 +- .../control/server/exchange/owning_core.rs | 9 +- .../resolve/exchange/post_process_arm.rs | 3 +- .../control/server/exchange/resolve/peers.rs | 9 +- nodedb/src/control/server/http/routes/crdt.rs | 8 +- .../server/http/routes/promql/remote.rs | 5 +- .../src/control/server/ilp_batch/dispatch.rs | 9 +- .../src/control/server/native/dispatch/ctx.rs | 47 ++- .../server/native/dispatch/direct_ops.rs | 13 +- .../server/native/dispatch/graph_match.rs | 3 +- .../native/dispatch/plan_builder/columnar.rs | 6 +- .../native/dispatch/plan_builder/crdt.rs | 12 +- .../native/dispatch/plan_builder/document.rs | 33 +- .../native/dispatch/plan_builder/graph.rs | 12 +- .../server/native/dispatch/plan_builder/kv.rs | 8 +- .../native/dispatch/plan_builder/vector.rs | 11 +- .../server/native/dispatch/raw_dispatch.rs | 3 +- .../src/control/server/native/dispatch/sql.rs | 10 +- .../control/server/pgwire/handler/facet.rs | 6 +- .../server/pgwire/handler/routing/execute.rs | 13 +- .../control/server/resp/gateway_dispatch.rs | 7 +- .../src/control/server/resp/handler_hash.rs | 6 +- .../server/resp/handler_kv/surrogate.rs | 6 +- .../src/control/server/resp/handler_sorted.rs | 3 +- .../server/response_translate/vector.rs | 7 +- .../server/shared/clone_write/document.rs | 3 +- .../control/server/shared/clone_write/kv.rs | 5 +- .../server/shared/clone_write/probes.rs | 14 +- .../control/server/shared/ddl/engine_apply.rs | 6 +- .../ddl/neutral/collection/dml/insert.rs | 5 +- .../neutral/collection/dml/parse/dispatch.rs | 7 +- .../ddl/neutral/collection/index/build.rs | 2 +- .../ddl/neutral/collection/index/kv_index.rs | 4 +- .../ddl/neutral/collection/index/teardown.rs | 5 +- .../ddl/neutral/collection/purge/dispatch.rs | 2 +- .../ddl/neutral/continuous_agg/create.rs | 3 +- .../shared/ddl/neutral/continuous_agg/drop.rs | 3 +- .../ddl/neutral/continuous_agg/register.rs | 6 +- .../shared/ddl/neutral/continuous_agg/show.rs | 3 +- .../shared/ddl/neutral/convert/driver.rs | 3 +- .../server/shared/ddl/neutral/crdt_ops.rs | 8 +- .../shared/ddl/neutral/dsl/crdt_merge.rs | 8 +- .../shared/ddl/neutral/estimate_count.rs | 2 +- .../shared/ddl/neutral/graph_ops/edge.rs | 62 +--- .../shared/ddl/neutral/kv_atomic/handlers.rs | 22 +- .../ddl/neutral/kv_sorted_index/dispatch.rs | 2 +- .../ddl/neutral/maintenance/vector_index.rs | 6 +- .../shared/ddl/neutral/permission_tree.rs | 3 +- .../neutral/query_functions/balance_as_of.rs | 12 +- .../convert_currency_lookup.rs | 4 +- .../query_functions/temporal_lookup.rs | 4 +- .../neutral/query_functions/verify_balance.rs | 6 +- .../query_functions/verify_hash_chain.rs | 4 +- .../server/shared/ddl/neutral/rate_gate.rs | 12 +- .../ddl/neutral/tenant/move_tenant/cutover.rs | 3 +- .../neutral/tenant/move_tenant/snapshot.rs | 3 +- .../server/shared/ddl/neutral/tenant/purge.rs | 3 +- .../server/shared/ddl/neutral/transfer.rs | 19 +- .../ddl/neutral/tree_ops/create_index.rs | 12 +- .../server/shared/ddl/neutral/tree_ops/sum.rs | 11 +- .../ddl/neutral/version_history/checkpoint.rs | 3 +- .../ddl/neutral/version_history/dispatch.rs | 4 +- .../ddl/neutral/version_history/restore.rs | 6 +- .../shared/ddl/neutral/weighted_pick.rs | 13 +- .../shared/ddl/sync_dispatch/dispatch.rs | 10 +- .../shared/ddl/sync_dispatch/system_task.rs | 12 +- .../server/shared/ddl/user_dispatch.rs | 4 +- .../server/shared/session/commit/metering.rs | 5 +- .../server/shared/session/commit/run.rs | 10 +- .../control/server/shared/session/read_set.rs | 9 +- .../control/server/surrogate_exchange/hook.rs | 15 +- .../server/surrogate_exchange/resolve.rs | 45 +-- .../server/sync/async_dispatch/delta/apply.rs | 5 +- .../control/server/sync/columnar_handler.rs | 6 +- nodedb/src/control/server/sync/fts_handler.rs | 8 +- nodedb/src/control/server/sync/fts_session.rs | 8 +- nodedb/src/control/server/sync/kv_handler.rs | 9 +- .../raft_dispatch/durability_test_support.rs | 2 +- .../server/sync/raft_dispatch/write.rs | 6 +- .../control/server/sync/spatial_handler.rs | 8 +- .../control/server/sync/spatial_session.rs | 8 +- .../control/server/sync/timeseries_handler.rs | 6 +- .../src/control/server/sync/vector_handler.rs | 8 +- .../src/control/server/sync/vector_session.rs | 8 +- .../server/wal_dispatch/write_set_redo.rs | 7 +- .../surrogate/assign/bind_plan/array.rs | 3 +- .../surrogate/assign/bind_plan/binder.rs | 66 ++-- .../surrogate/assign/bind_plan/carried.rs | 5 +- .../surrogate/assign/bind_plan/crdt.rs | 12 +- .../surrogate/assign/bind_plan/document.rs | 21 +- .../surrogate/assign/bind_plan/graph.rs | 18 +- .../control/surrogate/assign/bind_plan/kv.rs | 28 +- .../surrogate/assign/bind_plan/vector.rs | 36 +- .../surrogate/assign/core/assign_ops.rs | 131 +++---- nodedb/src/control/surrogate/physical_impl.rs | 10 +- nodedb/src/control/surrogate/wal_appender.rs | 20 +- .../src/control/target_identity/surrogate.rs | 29 +- nodedb/src/control/trigger/dml_hook.rs | 38 +- .../expand_staged_update_from_join.rs | 8 +- .../control/wal_replication/decode/crdt.rs | 9 +- nodedb/src/control/write_resolve/graph.rs | 6 +- nodedb/src/control/write_resolve/resolver.rs | 13 +- nodedb/src/control/write_resolve/run.rs | 2 +- .../columnar_checkpoint/geometry_restore.rs | 17 +- .../data/executor/core_loop/write_index.rs | 24 +- .../enforcement/materialized_sum/apply.rs | 43 ++- .../materialized_sum/divergence.rs | 15 +- .../data/executor/handlers/point/delete.rs | 5 +- .../data/executor/handlers/point/insert.rs | 5 +- .../src/data/executor/handlers/point/put.rs | 5 +- .../executor/handlers/point/update/exec.rs | 5 +- .../snapshot/restore/tenant_snapshot.rs | 6 +- .../handlers/timeseries_wal_payload.rs | 10 +- .../executor/handlers/upsert/exec/dispatch.rs | 5 +- nodedb/src/data/executor/replay_task.rs | 28 ++ .../data/executor/wal_replay_columnar_dml.rs | 18 +- .../executor/wal_replay_columnar_image.rs | 8 +- .../executor/wal_replay_columnar_truncate.rs | 16 +- nodedb/src/data/executor/wal_replay_fts.rs | 20 +- .../src/data/executor/wal_replay_spatial.rs | 20 +- nodedb/src/data/executor/wal_replay_vector.rs | 9 +- .../data/executor/wal_replay_vector_delete.rs | 8 +- .../data/executor/wal_replay_vector_direct.rs | 24 +- .../executor/wal_replay_vector_extended.rs | 24 +- .../executor/wal_replay_vector_resolved.rs | 8 +- .../data/executor/wal_replay_vector_sparse.rs | 16 +- nodedb/src/engine/bitemporal/enforcement.rs | 3 +- .../timeseries/retention_policy/autowire.rs | 6 +- .../retention_policy/enforcement.rs | 12 +- nodedb/src/error/conversions.rs | 14 +- nodedb/src/event/alert/executor.rs | 6 +- nodedb/src/event/scheduler/executor.rs | 13 +- nodedb/src/event/topic/publish.rs | 4 +- nodedb/src/query/materialized_sum_homing.rs | 59 ++- nodedb/src/query/mod.rs | 2 +- nodedb/src/wal/replay/surrogate/bind.rs | 22 +- nodedb/src/wal/replay/surrogate/dispatch.rs | 32 +- nodedb/tests/crash_harness/vshards.rs | 11 +- .../inproc/cases/calvin_executor_caps.rs | 5 +- .../cases/calvin_executor_dependent_read.rs | 4 +- .../cases/calvin_executor_ollp_property.rs | 5 +- .../inproc/cases/calvin_two_phase_apply.rs | 30 +- .../cases/columnar_natural_pk_identity.rs | 5 +- nodedb/tests/inproc/cases/dml_resolve_pass.rs | 6 +- .../enforcement_materialized_sum_point.rs | 6 +- .../inproc/cases/graph_cross_core_bfs.rs | 4 +- .../inproc/cases/http_query_authorization.rs | 3 +- .../cases/kv_atomic_surrogate_identity.rs | 5 +- .../kv_field_transfer_surrogate_identity.rs | 10 +- .../tests/inproc/cases/snapshot_round_trip.rs | 32 +- .../inproc/cases/snapshot_round_trip_crdt.rs | 5 +- .../cases/snapshot_round_trip_stale_state.rs | 5 +- .../tests/inproc/cases/sorted_index_rows.rs | 7 +- nodedb/tests/inproc/cases/surrogate_pk.rs | 64 ++-- .../inproc/cases/surrogate_wal_recovery.rs | 15 +- nodedb/tests/inproc/cases/wal_catchup.rs | 5 +- .../inproc/cases/write_admission_fence.rs | 6 +- nodedb/tests/native/cases/mod.rs | 1 + ...ative_database_collection_key_placement.rs | 345 ++++++++++++++++++ .../cases/native_gateway_txn_overlay.rs | 6 +- .../cases/native_txn_commit_visibility.rs | 4 +- nodedb/tests/wire/cases/backup_support.rs | 10 +- nodedb/tests/wire/cases/calvin_sql_routing.rs | 4 +- .../wire/cases/graph_timeseries_rls_probe.rs | 4 +- .../graph_vector_write_row_level_security.rs | 4 +- ...ql_transactions_commit_point_visibility.rs | 6 +- ...ql_transactions_cross_shard_read_reject.rs | 4 +- .../cases/sql_transactions_graph_overlay.rs | 2 +- 310 files changed, 2908 insertions(+), 1584 deletions(-) create mode 100644 nodedb-types/src/id/collection_key.rs create mode 100644 nodedb/tests/native/cases/native_database_collection_key_placement.rs diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs index 8fbba3273..4db960948 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_normal.rs @@ -21,10 +21,7 @@ use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, -}; +use nodedb_types::{TenantId, id::DatabaseId}; use tokio::sync::mpsc; use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; @@ -34,7 +31,9 @@ fn two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -48,8 +47,12 @@ fn two_distinct_collections() -> (String, String) { fn make_multishard_txclass() -> (TxClass, u32, u32) { let (col_a, col_b) = two_distinct_collections(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let va = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let vb = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); let write_set = ReadWriteSet::new(vec![ EngineKeySet::Document { collection: col_a, @@ -92,8 +95,12 @@ async fn sequencer_normal_path_commit_on_all_replicas() { // Wire per-vshard receivers on every node. let (tx_a, col_b_name) = two_distinct_collections(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &tx_a).as_u32(); - let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b_name).as_u32(); + let va = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &tx_a) + .vshard() + .as_u32(); + let vb = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b_name) + .vshard() + .as_u32(); let mut vshard_rxs_a: Vec> = Vec::new(); let mut vshard_rxs_b: Vec> = Vec::new(); diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs index 9238b8237..ca445d371 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_3node_shard_failover.rs @@ -45,10 +45,7 @@ use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, -}; +use nodedb_types::{TenantId, id::DatabaseId}; use tokio::sync::mpsc; use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; @@ -59,7 +56,9 @@ fn two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -130,8 +129,12 @@ async fn scheduler_catchup_via_raft_log_replay() { // Wire per-vshard receivers on every node so we can verify fan-out. let (col_a, col_b) = two_distinct_collections(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let va = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let vb = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); let mut vshard_rxs_a: Vec> = Vec::new(); let mut vshard_rxs_b: Vec> = Vec::new(); diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs index 844158677..f71e163d5 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_ollp.rs @@ -39,10 +39,7 @@ use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, -}; +use nodedb_types::{TenantId, id::DatabaseId}; use tokio::sync::mpsc; use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; @@ -54,7 +51,9 @@ fn two_distinct_vshard_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("ollp_col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -137,9 +136,12 @@ async fn ollp_bulk_update_txclass_admitted_and_fanned_out() { // Find two collections that hash to distinct vshards. let (col_static, col_ollp) = two_distinct_vshard_collections(); - let vs_static = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_static).as_u32(); - let vs_ollp = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_ollp).as_u32(); + let vs_static = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_static) + .vshard() + .as_u32(); + let vs_ollp = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_ollp) + .vshard() + .as_u32(); // Wire per-vshard fan-out receivers on every replica. let mut rxs_static: Vec> = Vec::new(); diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs index e74290fff..cd338f3f6 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_e2e_pgwire.rs @@ -22,10 +22,7 @@ use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, types::{EngineKeySet, ReadWriteSet, SchedulerInput, SortedVec, TxClass, VersionedReadSet}, }; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, -}; +use nodedb_types::{TenantId, id::DatabaseId}; use tokio::sync::mpsc; use super::cluster_common::{spawn_with_sequencer, try_recv_txn, wait_for_sequencer_leader}; @@ -38,7 +35,9 @@ fn two_distinct_vshard_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("orders_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -96,8 +95,12 @@ async fn multi_vshard_insert_via_sequencer_admitted_and_replicated() { // Wire per-vshard fan-out receivers on every node. let (col_a, col_b) = two_distinct_vshard_collections(); - let va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let va = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let vb = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); let mut fan_out_rxs_a: Vec> = Vec::new(); let mut fan_out_rxs_b: Vec> = Vec::new(); diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_sequencer_failover.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_sequencer_failover.rs index bcae77704..39362247e 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_sequencer_failover.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/calvin_sequencer_failover.rs @@ -15,10 +15,7 @@ use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass, VersionedReadSet}, }; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, -}; +use nodedb_types::{TenantId, id::DatabaseId}; use super::cluster_common::{spawn_with_sequencer, wait_for_sequencer_leader}; @@ -26,7 +23,9 @@ fn two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs index a631106fb..bff04e519 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs @@ -101,9 +101,8 @@ async fn assign_remote_surrogate_is_authoritative_and_idempotent() { let s1 = assign_surrogate_routed( &coordinator.shared, vshard, - DB, + nodedb_types::CollectionKey::from_bare(DB, &collection), TENANT, - &collection, pk.as_bytes(), TraceId([0u8; 16]), ) @@ -121,9 +120,8 @@ async fn assign_remote_surrogate_is_authoritative_and_idempotent() { let s2 = assign_surrogate_routed( &coordinator.shared, vshard, - DB, + nodedb_types::CollectionKey::from_bare(DB, &collection), TENANT, - &collection, pk.as_bytes(), TraceId([0u8; 16]), ) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs index 78b3a1491..134c037da 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs @@ -71,7 +71,11 @@ fn array_surrogate( shared .credentials .catalog() - .get_surrogate_for_pk(DatabaseId::DEFAULT, tenant, ARRAY, coord_bytes) + .get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, ARRAY), + tenant, + coord_bytes, + ) .ok() .flatten() .map(|s| s.as_u32()) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_crdt_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_crdt_replication.rs index 88898b5b4..35df8bb60 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_crdt_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_crdt_replication.rs @@ -133,7 +133,9 @@ async fn crdt_apply_replicates_and_survives_leader_loss() { // Resolve the collection's data group and its leader from node 0's shared // routing view (same idiom as multi_replica_data_groups). - let vshard = nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + ); let (group_id, group_leader) = { let routing = cluster.nodes[0] .shared diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_surrogate_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_surrogate_replication.rs index 15caa51fd..70e4529d5 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_surrogate_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_surrogate_replication.rs @@ -43,9 +43,8 @@ fn surrogate_for_pk( let catalog = shared.credentials.catalog(); catalog .get_surrogate_for_pk( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), TenantId::new(1), - collection, pk.as_bytes(), ) .ok() diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cross_node_read_occ_abort.rs b/nodedb-cluster-tests/tests/common_suite/cases/cross_node_read_occ_abort.rs index a473802e8..6e53fffa2 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cross_node_read_occ_abort.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cross_node_read_occ_abort.rs @@ -84,14 +84,16 @@ fn admitted_total(node: &TestClusterNode) -> u64 { /// Three `document_schemaless` collection names whose vShard ids are pairwise /// distinct, so a transaction that writes two of them and reads the third is -/// genuinely multi-vShard. Deterministic: `VShardId::from_collection_in_database` +/// genuinely multi-vShard. Deterministic: `VShardId::from_collection` /// is a pure function of the database id + collection-name bytes, so the same /// scan picks the same names every run. fn distinct_vshard_triple() -> (String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..1024 { let name = format!("occ_shard_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 3 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/gather_join_in_txn_occ.rs b/nodedb-cluster-tests/tests/common_suite/cases/gather_join_in_txn_occ.rs index 470293ae1..c4e8ea8b1 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/gather_join_in_txn_occ.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/gather_join_in_txn_occ.rs @@ -76,13 +76,15 @@ use common::occ_shuffle::{ /// Four collection names whose vShard ids are pairwise distinct, so a transaction /// that reads two of them (the join sides) and writes the other two is genuinely /// multi-vShard on both its read set and its write set. Deterministic: -/// `VShardId::from_collection_in_database` is a pure function of the database id + +/// `VShardId::from_collection` is a pure function of the database id + /// collection-name bytes. fn distinct_vshard_quad() -> (String, String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..2048 { let name = format!("gather_join_occ_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 4 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs b/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs index 4dfc93625..c72cca998 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs @@ -63,7 +63,10 @@ fn read_catalog_surrogates( ) -> Vec<(String, u32)> { let catalog = shared.credentials.catalog(); catalog - .scan_surrogates_for_collection(DatabaseId::DEFAULT, TenantId::new(1), collection) + .scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), + TenantId::new(1), + ) .unwrap_or_default() .into_iter() .map(|(pk_bytes, surrogate)| { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs b/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs index dfac58743..881c55de1 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/kv_atomic_autocommit_replicates.rs @@ -18,7 +18,7 @@ use common::cluster_harness::TestCluster; use std::time::{Duration, Instant}; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; const COUNTERS: &str = "repl_kv_ctr"; const BOARD: &str = "repl_kv_board"; @@ -47,7 +47,7 @@ fn counts_three(read: &Result, String>) -> bool { /// The leader node id of the data group that owns `collection`. fn group_leader(cluster: &TestCluster, collection: &str) -> u64 { - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(); let routing = cluster.nodes[0] .shared .cluster_routing diff --git a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_core.rs b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_core.rs index e443d69c8..41663c576 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_core.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_core.rs @@ -27,7 +27,7 @@ use common::cluster_harness::{TestClusterNode, wait_for}; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; const SOURCE: &str = "xc_entries"; @@ -50,8 +50,8 @@ fn pg_detail(e: &tokio_postgres::Error) -> String { #[test] fn source_and_target_home_to_different_vshards() { assert_ne!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, SOURCE), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, TARGET), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, TARGET).vshard(), "this file tests the CROSS-SHARD path; '{SOURCE}' and '{TARGET}' must not be co-resident" ); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_shard.rs b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_shard.rs index 0eed4c862..d860d3442 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_shard.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_cross_shard.rs @@ -19,7 +19,7 @@ use common::cluster_harness::TestCluster; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; /// Source and target, chosen for readability rather than for their hashes — the /// homing assertion below is what makes the choice meaningful. @@ -84,8 +84,8 @@ async fn declare_binding(cluster: &TestCluster) { /// vShard, so every balance below travels on its own task. #[test] fn source_and_target_home_to_different_vshards() { - let source = VShardId::from_collection_in_database(DatabaseId::DEFAULT, SOURCE); - let target = VShardId::from_collection_in_database(DatabaseId::DEFAULT, TARGET); + let source = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE).vshard(); + let target = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, TARGET).vshard(); assert_ne!( source, target, "this file tests the CROSS-SHARD path; '{SOURCE}' and '{TARGET}' must not be co-resident" diff --git a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_replication.rs index 599afe0e8..9a5864bcc 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/materialized_sum_replication.rs @@ -26,7 +26,7 @@ use common::cluster_harness::TestCluster; use std::time::Duration; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; /// Cross-shard fixture: source and target hash to different vShards. const XS_SOURCE: &str = "rep_entries"; @@ -204,8 +204,8 @@ async fn assert_every_replica_agrees( #[test] fn coresident_fixture_shares_one_vshard() { assert_eq!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, CO_SOURCE), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, CO_TARGET), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, CO_SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, CO_TARGET).vshard(), "this fixture must exercise the fold that runs inside the source write's own \ transaction" ); @@ -215,8 +215,8 @@ fn coresident_fixture_shares_one_vshard() { #[test] fn replication_fixture_is_cross_shard() { assert_ne!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, XS_SOURCE), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, XS_TARGET), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, XS_SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, XS_TARGET).vshard(), "this fixture must exercise the replicated cross-shard balance write" ); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs b/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs index 6dbc6fa57..77e6464dd 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/multi_replica_data_groups.rs @@ -121,7 +121,9 @@ async fn data_group_is_multi_replica_and_survives_leader_loss() { .await; // Resolve the collection's data group. - let vshard = nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + ); let group_id = { let routing = cluster.nodes[0] .shared diff --git a/nodedb-cluster-tests/tests/common_suite/cases/native_gather_join_in_txn_occ.rs b/nodedb-cluster-tests/tests/common_suite/cases/native_gather_join_in_txn_occ.rs index 8fd90192d..b69ab7681 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/native_gather_join_in_txn_occ.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/native_gather_join_in_txn_occ.rs @@ -48,13 +48,15 @@ const SERIALIZATION_ABORT: &str = "could not serialize access due to concurrent /// Four collection names whose vShard ids are pairwise distinct, so a transaction /// that reads two of them (the join sides) and writes the other two is genuinely /// multi-vShard on both its read set and its write set. Deterministic: -/// `VShardId::from_collection_in_database` is a pure function of the database id + +/// `VShardId::from_collection` is a pure function of the database id + /// collection-name bytes. fn distinct_vshard_quad() -> (String, String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..2048 { let name = format!("native_gather_join_occ_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 4 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs index bd42b9609..cd1a628f3 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs @@ -17,7 +17,7 @@ use common::cluster_harness::TestCluster; use std::time::{Duration, Instant}; use nodedb::control::wal_replication::{ReplicableWrite, to_replicated_entry}; -use nodedb::types::{DatabaseId, TenantId, VShardId}; +use nodedb::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; const COLL: &str = "dup_proposal_ctr"; @@ -62,7 +62,7 @@ async fn a_proposal_committed_twice_moves_the_counter_once() { .wait_for_full_apply_convergence(Duration::from_secs(15)) .await; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, COLL); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL).vshard(); let deadline = Instant::now() + Duration::from_secs(20); let mut committed = 0; let mut entry_bytes: Option> = None; @@ -99,7 +99,11 @@ async fn a_proposal_committed_twice_moves_the_counter_once() { let surrogate = leader .shared .surrogate_assigner - .assign(DatabaseId::DEFAULT, TenantId::new(TENANT), COLL, b"ctr") + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + TenantId::new(TENANT), + b"ctr", + ) .expect("the seeded key has a surrogate"); let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLL), diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_in_txn_occ.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_in_txn_occ.rs index 0623c8712..45e795052 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_in_txn_occ.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_in_txn_occ.rs @@ -66,13 +66,15 @@ use common::occ_shuffle::{ /// Three `metrics`/`w1`/`w2` collection names whose vShard ids are pairwise /// distinct, so a transaction that writes two of them and reads the third is -/// genuinely multi-vShard. Deterministic: `VShardId::from_collection_in_database` +/// genuinely multi-vShard. Deterministic: `VShardId::from_collection` /// is a pure function of the database id + collection-name bytes. fn distinct_vshard_triple() -> (String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..1024 { let name = format!("shuffle_occ_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 3 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_cost_model.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_cost_model.rs index 8d3055ec3..3c24f1c9d 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_cost_model.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_cost_model.rs @@ -67,8 +67,14 @@ async fn cost_model_auto_selects_shuffle_from_analyze_stats() { const LEFT: &str = "orders"; const RIGHT: &str = "customers"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, LEFT), - vshard_for_collection(DatabaseId::DEFAULT, RIGHT), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + LEFT + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + RIGHT + )), "test collections must hash to different vShards to exercise cross-node shuffle" ); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_end_to_end.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_end_to_end.rs index 9939abf38..e2e29d62e 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_end_to_end.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_end_to_end.rs @@ -66,8 +66,14 @@ async fn distributed_shuffle_join_matches_inner_join() { const LEFT: &str = "orders"; const RIGHT: &str = "customers"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, LEFT), - vshard_for_collection(DatabaseId::DEFAULT, RIGHT), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + LEFT + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + RIGHT + )), "test collections must hash to different vShards to exercise cross-node shuffle" ); @@ -230,8 +236,14 @@ async fn a_shuffle_join_renders_a_time_key_as_the_stored_instant() { const LEFT: &str = "ts_shuffle_events"; const RIGHT: &str = "ts_shuffle_hosts"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, LEFT), - vshard_for_collection(DatabaseId::DEFAULT, RIGHT), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + LEFT + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + RIGHT + )), "test collections must hash to different vShards to exercise cross-node shuffle" ); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_in_txn_occ.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_in_txn_occ.rs index 86ae24058..d2f2d0bfc 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_in_txn_occ.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_join_in_txn_occ.rs @@ -67,13 +67,15 @@ use common::occ_shuffle::{ /// Four collection names whose vShard ids are pairwise distinct, so a transaction /// that reads two of them (the join sides) and writes the other two is genuinely /// multi-vShard on both its read set and its write set. Deterministic: -/// `VShardId::from_collection_in_database` is a pure function of the database id + +/// `VShardId::from_collection` is a pure function of the database id + /// collection-name bytes. fn distinct_vshard_quad() -> (String, String, String, String) { let mut chosen: Vec<(String, u32)> = Vec::new(); for i in 0u32..2048 { let name = format!("shuffle_join_occ_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if chosen.iter().all(|(_, cv)| *cv != v) { chosen.push((name, v)); if chosen.len() == 4 { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs index 05fb8e06c..babc516ea 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs @@ -220,8 +220,10 @@ async fn calvin_commit_publishes_control_changes_at_participant_lsns() { let second = (0..4096) .map(|i| format!("sncgtx_cdc_b_{i}")) .find(|candidate| { - VShardId::from_collection_in_database(nodedb_types::DatabaseId::DEFAULT, candidate) - != VShardId::from_collection_in_database(nodedb_types::DatabaseId::DEFAULT, first) + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, candidate) + .vshard() + != nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, first) + .vshard() }) .expect("collection on a distinct vShard"); for collection in [first, second.as_str()] { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs index 38b7ba642..3c0c447d5 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_hot_key_reservation.rs @@ -19,7 +19,7 @@ use std::time::{Duration, Instant}; use nodedb::control::cluster::calvin::scheduler::lock::LockKey; use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::DatabaseId; use common::cluster_harness::{TestClusterNode, wait_for}; @@ -41,7 +41,9 @@ fn sequencer_leader(node: &TestClusterNode) -> u64 { fn other_vshard_collection(exclude_vshard: u32) -> String { for i in 0u32..4096 { let name = format!("hkr_other_{i}"); - if VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32() + if nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32() != exclude_vshard { return name; @@ -81,7 +83,9 @@ async fn hot_key_read_reservation_installs_self_upgrades_and_releases() { .await; let hot_coll = "hkr_hot_kv"; - let hot_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, hot_coll).as_u32(); + let hot_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, hot_coll) + .vshard() + .as_u32(); let other_coll = other_vshard_collection(hot_vshard); node.client diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs index 606a4ed84..30c8d8043 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs @@ -245,7 +245,7 @@ async fn cluster_sync_columnar_dedup_survives_failover() { const COLL: &str = "csync_failover"; const PRODUCER: u64 = 7777; let tenant = TenantId::new(0); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, COLL); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL).vshard(); cluster .exec_ddl_on_any_leader(&format!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs b/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs index 155e91073..201508c74 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/vshard_names.rs @@ -2,7 +2,7 @@ //! Deterministic name picking for cross-vShard tests. //! -//! vShards are per collection (`VShardId::from_collection_in_database`) and +//! vShards are per collection (`VShardId::from_collection`) and //! per graph endpoint key (`VShardId::from_key`). Both are pure functions of //! their input bytes, so a test can pick names that land on distinct vShards //! without probing the cluster. @@ -16,10 +16,12 @@ const MAX_TRIES: u32 = 512; /// used verbatim; `second` is `{second_prefix}_{i}` for the lowest `i` that /// hashes away from `first`. pub fn distinct_vshard_collections(first: &str, second_prefix: &str) -> (String, String) { - let first_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, first); + let first_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, first).vshard(); for i in 0..MAX_TRIES { let second = format!("{second_prefix}_{i}"); - if VShardId::from_collection_in_database(DatabaseId::DEFAULT, &second) != first_vshard { + if nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &second).vshard() + != first_vshard + { return (first.to_owned(), second); } } @@ -33,7 +35,8 @@ pub fn distinct_vshard_collections(first: &str, second_prefix: &str) -> (String, /// `collection`'s own vShard, so an implicit edge task homed on the key is /// dispatched to a different vShard than the document write. pub fn key_on_other_vshard(collection: &str, prefix: &str) -> String { - let coll_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection); + let coll_vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(); for i in 0..MAX_TRIES { let key = format!("{prefix}_{i}"); if VShardId::from_key(key.as_bytes()) != coll_vshard { diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/calvin_sequencer_starvation.rs b/nodedb-cluster-tests/tests/misc_suite/cases/calvin_sequencer_starvation.rs index e541310bb..716ced9b6 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/calvin_sequencer_starvation.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/calvin_sequencer_starvation.rs @@ -26,16 +26,15 @@ use nodedb_cluster::calvin::sequencer::validator::validate_batch; use nodedb_cluster::calvin::types::{ EngineKeySet, ReadWriteSet, SortedVec, TxClass, VersionedReadSet, }; -use nodedb_types::{ - TenantId, - id::{DatabaseId, VShardId}, -}; +use nodedb_types::{TenantId, id::DatabaseId}; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/join_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/join_cross_node.rs index 452708b45..94a8a8388 100644 --- a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/join_cross_node.rs +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/join_cross_node.rs @@ -53,8 +53,14 @@ async fn cross_node_join_returns_all_matches() { const FACT: &str = "fact"; const DIM: &str = "dim"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, FACT), - vshard_for_collection(DatabaseId::DEFAULT, DIM), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + FACT + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + DIM + )), "test collections must hash to different vShards to exercise cross-node join" ); @@ -177,8 +183,14 @@ async fn cross_node_join_compares_time_keys_in_one_unit() { const EVENTS: &str = "ts_events"; const FEATURES: &str = "ts_features"; assert_ne!( - vshard_for_collection(DatabaseId::DEFAULT, EVENTS), - vshard_for_collection(DatabaseId::DEFAULT, FEATURES), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + EVENTS + )), + vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + FEATURES + )), "test collections must hash to different vShards to exercise cross-node join" ); diff --git a/nodedb-cluster/src/array_routing.rs b/nodedb-cluster/src/array_routing.rs index 1cd89d73a..0085835e2 100644 --- a/nodedb-cluster/src/array_routing.rs +++ b/nodedb-cluster/src/array_routing.rs @@ -36,7 +36,7 @@ pub const VSHARD_COUNT: u32 = 1024; /// /// Array-specific name-only fallback used by `array_sync` paths that route /// before a coordinate or tile extent is known. Uses the same DJB -/// multiply-31 hash as `VShardId::from_collection_in_database`, but without +/// multiply-31 hash as `VShardId::from_collection`, but without /// the database scope — array-sync messages carry their own scoping in the /// op-log header and route per-array by name. pub fn array_vshard_for_name(array_name: &str) -> u32 { diff --git a/nodedb-cluster/src/calvin/sequencer/entry.rs b/nodedb-cluster/src/calvin/sequencer/entry.rs index c0b3656b3..66b232ff6 100644 --- a/nodedb-cluster/src/calvin/sequencer/entry.rs +++ b/nodedb-cluster/src/calvin/sequencer/entry.rs @@ -147,14 +147,16 @@ mod tests { use crate::calvin::types::{EngineKeySet, ReadWriteSet, SequencedTxn, SortedVec, TxClass}; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster/src/calvin/sequencer/inbox.rs b/nodedb-cluster/src/calvin/sequencer/inbox.rs index 9b2d0a7db..630518de9 100644 --- a/nodedb-cluster/src/calvin/sequencer/inbox.rs +++ b/nodedb-cluster/src/calvin/sequencer/inbox.rs @@ -355,7 +355,7 @@ pub fn new_inbox(capacity: usize, config: &SequencerConfig) -> (Inbox, InboxRece mod tests { use super::*; use crate::calvin::types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass}; - use nodedb_types::id::{DatabaseId, VShardId}; + use nodedb_types::id::{CollectionKey, DatabaseId}; fn default_config() -> SequencerConfig { SequencerConfig::default() @@ -365,7 +365,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster/src/calvin/sequencer/replay.rs b/nodedb-cluster/src/calvin/sequencer/replay.rs index 7a30d65cb..79d40796e 100644 --- a/nodedb-cluster/src/calvin/sequencer/replay.rs +++ b/nodedb-cluster/src/calvin/sequencer/replay.rs @@ -104,9 +104,25 @@ impl SequencerStateMachine { continue; } // Re-derive participating_vshards (skipped during serialization) - // exactly as the live apply path does before fan-out. + // exactly as the live apply path does before fan-out. The live + // path skips an entry whose participants cannot be derived and + // files the report, so replay skips it too. + let mut underivable = None; for txn in &mut batch.txns { - txn.tx_class.restore_derived(); + if let Err(err) = txn.tx_class.restore_derived() { + underivable = Some(err); + break; + } + } + if let Some(err) = underivable { + tracing::warn!( + raft_index = entry.index, + epoch = batch.epoch, + error = %err, + "calvin replay: epoch batch carries a transaction with \ + underivable participants; skipping" + ); + continue; } // Shared with the live `apply` EpochBatch arm via @@ -181,7 +197,7 @@ mod tests { }; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; use std::collections::HashMap; use tokio::sync::mpsc; @@ -190,7 +206,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -204,8 +222,12 @@ mod tests { fn make_batch_with_two_vshards() -> (EpochBatch, u32, u32) { let (col_a, col_b) = find_two_distinct_collections(); - let real_va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let real_vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let real_va = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let real_vb = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); let write_set = ReadWriteSet::new(vec![ EngineKeySet::Document { collection: col_a, diff --git a/nodedb-cluster/src/calvin/sequencer/service/core.rs b/nodedb-cluster/src/calvin/sequencer/service/core.rs index 601a6d0c6..3bd1bc201 100644 --- a/nodedb-cluster/src/calvin/sequencer/service/core.rs +++ b/nodedb-cluster/src/calvin/sequencer/service/core.rs @@ -532,14 +532,16 @@ mod tests { use crate::routing::RoutingTable; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs index 4b507b81b..4ff93a135 100644 --- a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs @@ -88,8 +88,25 @@ impl SequencerStateMachine { SequencerEntry::EpochBatch { mut batch } => { // Re-derive the participating_vshards field which is skipped // during serialization (it is computed from write_set collection names). + // A class whose participants cannot be derived makes the entry + // as unusable as one that fails to decode, so it is skipped the + // same way. for txn in &mut batch.txns { - txn.tx_class.restore_derived(); + if let Err(err) = txn.tx_class.restore_derived() { + error!( + epoch = batch.epoch, + raft_index = index, + error = %err, + "sequencer state machine: epoch batch carries a transaction \ + with underivable participants; skipping entry" + ); + crate::diag::sequencer_participants_underivable( + batch.epoch, + index, + &err.to_string(), + ); + return; + } } // A halted state machine has already diverged from the log; @@ -475,14 +492,16 @@ mod tests { }; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -501,8 +520,12 @@ mod tests { // We'll use find_two_distinct_collections and use whatever vshards they hash to. let (col_a, col_b) = find_two_distinct_collections(); let _ = (vshard_a, vshard_b); // actual vshard ids come from the collection hash - let real_va = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let real_vb = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let real_va = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let real_vb = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); let write_set = ReadWriteSet::new(vec![ EngineKeySet::Document { collection: col_a, diff --git a/nodedb-cluster/src/calvin/sequencer/validator.rs b/nodedb-cluster/src/calvin/sequencer/validator.rs index 1c5691acb..188fe6fc2 100644 --- a/nodedb-cluster/src/calvin/sequencer/validator.rs +++ b/nodedb-cluster/src/calvin/sequencer/validator.rs @@ -388,14 +388,16 @@ mod tests { use crate::calvin::types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass}; use nodedb_types::{ TenantId, - id::{DatabaseId, VShardId}, + id::{CollectionKey, DatabaseId}, }; fn find_two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); diff --git a/nodedb-cluster/src/calvin/types/sequencer.rs b/nodedb-cluster/src/calvin/types/sequencer.rs index 65d8a8607..238a3106a 100644 --- a/nodedb-cluster/src/calvin/types/sequencer.rs +++ b/nodedb-cluster/src/calvin/types/sequencer.rs @@ -107,7 +107,7 @@ pub struct EpochBatch { #[cfg(test)] mod tests { use nodedb_types::TenantId; - use nodedb_types::id::{DatabaseId, VShardId}; + use nodedb_types::id::{CollectionKey, DatabaseId}; use super::super::primitives::{EngineKeySet, SortedVec, VersionedReadSet}; use super::super::transaction::ReadWriteSet; @@ -133,7 +133,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -170,7 +172,7 @@ mod tests { }; let bytes = zerompk::to_msgpack_vec(&st).unwrap(); let mut decoded: SequencedTxn = zerompk::from_msgpack(&bytes).unwrap(); - decoded.tx_class.restore_derived(); + decoded.tx_class.restore_derived().expect("restore derived"); assert_eq!(st.epoch, decoded.epoch); assert_eq!(st.position, decoded.position); assert_eq!(st.epoch_system_ms, decoded.epoch_system_ms); @@ -205,7 +207,7 @@ mod tests { let bytes = zerompk::to_msgpack_vec(&batch).unwrap(); let mut decoded: EpochBatch = zerompk::from_msgpack(&bytes).unwrap(); for txn in &mut decoded.txns { - txn.tx_class.restore_derived(); + txn.tx_class.restore_derived().expect("restore derived"); } assert_eq!(batch.epoch, decoded.epoch); assert_eq!(batch.epoch_system_ms, decoded.epoch_system_ms); diff --git a/nodedb-cluster/src/calvin/types/transaction.rs b/nodedb-cluster/src/calvin/types/transaction.rs index e3f0d9770..bd681251e 100644 --- a/nodedb-cluster/src/calvin/types/transaction.rs +++ b/nodedb-cluster/src/calvin/types/transaction.rs @@ -6,7 +6,7 @@ //! representation submitted to the sequencer. use nodedb_types::TenantId; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::{CollectionKey, DatabaseId, VShardId}; use serde::{Deserialize, Serialize}; use crate::error::CalvinError; @@ -58,12 +58,20 @@ impl ReadWriteSet { /// This derivation is re-run on decode rather than serialized, so the /// serialized bytes remain deterministic regardless of how `VShardId` /// is computed. - pub fn participating_vshards(&self) -> Vec { + pub fn participating_vshards(&self) -> Result, CalvinError> { self.participating_vshards_in_database(DatabaseId::DEFAULT) } /// Derive participants using database-scoped collection homes. - pub fn participating_vshards_in_database(&self, database_id: DatabaseId) -> Vec { + /// + /// Key-set collection names are database-qualified, because the Data + /// Plane reads storage by them. Each one is de-qualified into a + /// [`CollectionKey`] before hashing, so the participant set matches the + /// vShard every other path homes the collection to. + pub fn participating_vshards_in_database( + &self, + database_id: DatabaseId, + ) -> Result, CalvinError> { let mut seen = std::collections::HashSet::new(); let mut result = Vec::new(); for engine_set in &self.0 { @@ -80,7 +88,8 @@ impl ReadWriteSet { | EngineKeySet::Vector { .. } | EngineKeySet::Kv { .. } => { let vshard = - VShardId::from_collection_in_database(database_id, engine_set.collection()); + CollectionKey::from_qualified_str(database_id, engine_set.collection())? + .vshard(); if seen.insert(vshard.as_u32()) { result.push(vshard); } @@ -88,7 +97,7 @@ impl ReadWriteSet { } } result.sort_by_key(|v| v.as_u32()); - result + Ok(result) } } @@ -304,7 +313,7 @@ impl TxClass { if write_set.is_empty() { return Err(CalvinError::EmptyWriteSet); } - let mut participating_vshards = write_set.participating_vshards_in_database(database_id); + let mut participating_vshards = write_set.participating_vshards_in_database(database_id)?; let min_participants = if allow_single_vshard { 1 } else { 2 }; // The participant FLOOR is computed from the WRITE set ONLY, and BEFORE // the read-set union below: a txn that writes a single shard but reads N @@ -323,7 +332,7 @@ impl TxClass { // `new_checked` and `restore_derived` — `participating_vshards` is // `#[serde(skip)]` and re-derived on decode, so an encoded and a decoded // `TxClass` would disagree on their participant set if the two diverged. - for v in read_set.participating_vshards_in_database(database_id) { + for v in read_set.participating_vshards_in_database(database_id)? { if !participating_vshards .iter() .any(|e| e.as_u32() == v.as_u32()) @@ -393,17 +402,20 @@ impl TxClass { /// Re-derive fields skipped during serialization. /// /// Call this immediately after deserializing a `TxClass` that came off - /// the wire or out of the Raft log. - pub fn restore_derived(&mut self) { + /// the wire or out of the Raft log. Fails when a key-set collection name + /// lacks the qualifier of `database_id`. Construction rejects such a + /// class, so a failure here means the decoded bytes are not a class any + /// constructor built. + pub fn restore_derived(&mut self) -> Result<(), CalvinError> { let mut vshards = self .write_set - .participating_vshards_in_database(self.database_id); + .participating_vshards_in_database(self.database_id)?; // Union the read set's participating vShards — MUST match `new_checked`'s // union exactly so a decoded `TxClass` derives the identical participant // set the encoder computed (participants are not serialized). for v in self .read_set - .participating_vshards_in_database(self.database_id) + .participating_vshards_in_database(self.database_id)? { if !vshards.iter().any(|e| e.as_u32() == v.as_u32()) { vshards.push(v); @@ -419,6 +431,7 @@ impl TxClass { // Final stable sort after all unions — lockstep with `new_checked`. vshards.sort_by_key(|v| v.as_u32()); self.participating_vshards = vshards; + Ok(()) } /// Set the lock-table owner id propagated to `SequencedTxn.lock_owner`. @@ -462,7 +475,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { return (fname.clone(), name); @@ -491,14 +506,14 @@ mod tests { #[test] fn read_write_set_participating_vshards_distinct() { let ws = multi_vshard_write_set(); - let vshards = ws.participating_vshards(); + let vshards = ws.participating_vshards().expect("participants"); assert!(vshards.len() >= 2, "expected at least 2 distinct vShards"); } #[test] fn read_write_set_participating_vshards_sorted() { let ws = multi_vshard_write_set(); - let vshards = ws.participating_vshards(); + let vshards = ws.participating_vshards().expect("participants"); let ids: Vec = vshards.iter().map(|v| v.as_u32()).collect(); let mut sorted = ids.clone(); sorted.sort(); @@ -509,7 +524,7 @@ mod tests { fn read_write_set_same_collection_counted_once() { // Two EngineKeySets for the same collection: still one vshard. let ws = ReadWriteSet::new(vec![doc_set("users", vec![1]), vec_set("users", vec![1])]); - let vshards = ws.participating_vshards(); + let vshards = ws.participating_vshards().expect("participants"); assert_eq!(vshards.len(), 1); } @@ -611,7 +626,7 @@ mod tests { let first = sonic_rs::to_vec(&tc).unwrap(); let mut restored: TxClass = sonic_rs::from_slice(&first).unwrap(); - restored.restore_derived(); + restored.restore_derived().expect("restore derived"); let second = sonic_rs::to_vec(&restored).unwrap(); assert_eq!(first, second); @@ -624,7 +639,7 @@ mod tests { let tc = make_tx_class(multi_vshard_write_set()); let bytes = zerompk::to_msgpack_vec(&tc).unwrap(); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).unwrap(); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); assert_eq!(tc.tenant_id, decoded.tenant_id); assert_eq!(tc.plans, decoded.plans); assert_eq!(tc.write_set, decoded.write_set); @@ -639,13 +654,19 @@ mod tests { // Pick a vshard id that's different from col_a and col_b. let passive_vshard_id = { - let a = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_a).as_u32(); - let b = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &col_b).as_u32(); + let a = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_a) + .vshard() + .as_u32(); + let b = CollectionKey::from_bare(DatabaseId::DEFAULT, &col_b) + .vshard() + .as_u32(); // Find one that differs from both. let mut candidate = 9999u32; for i in 0u32..64 { let name = format!("passive_col_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if v != a && v != b { candidate = v; break; @@ -730,7 +751,7 @@ mod tests { let bytes = zerompk::to_msgpack_vec(&tx).expect("encode TxClass"); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode TxClass"); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); // Every read_lsn and the Point/Predicate distinction survive exactly. assert_eq!(decoded.versioned_reads, reads); @@ -771,7 +792,7 @@ mod tests { .expect("valid TxClass"); let bytes = zerompk::to_msgpack_vec(&tx).expect("encode"); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode"); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); assert_eq!(decoded.database_id, DatabaseId::new(9)); assert_eq!(decoded.participating_vshards(), tx.participating_vshards()); } @@ -796,7 +817,7 @@ mod tests { let bytes = zerompk::to_msgpack_vec(&legacy).expect("encode legacy"); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode legacy as TxClass"); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); assert!(decoded.versioned_reads.is_empty()); assert!(decoded.dependent_reads.is_none()); @@ -832,7 +853,9 @@ mod tests { // Pick a collection name whose collection-homed vShard differs from // both endpoint homes, to prove routing ignores the collection. - let coll_v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, "follows").as_u32(); + let coll_v = CollectionKey::from_bare(DatabaseId::DEFAULT, "follows") + .vshard() + .as_u32(); let ws = ReadWriteSet::new(vec![EngineKeySet::Edge { collection: "follows".to_owned(), @@ -842,6 +865,7 @@ mod tests { let mut got: Vec = ws .participating_vshards() + .expect("participants derive") .iter() .map(|v| v.as_u32()) .collect(); @@ -867,8 +891,9 @@ mod tests { collection: "users".to_owned(), surrogates: SortedVec::new(vec![7u32]), }]); - let want_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "users").as_u32(); + let want_vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, "users") + .vshard() + .as_u32(); // Strict path still rejects. let strict = TxClass::new( @@ -914,7 +939,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..2048 { let name = format!("coll_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); if let Some((ref fname, fv)) = first { if fv != v { return (fname.clone(), name); @@ -933,8 +960,12 @@ mod tests { // write-only floor still passes via the single-vshard opt-in), and the // union is reproduced identically on decode. let (wcoll, rcoll) = two_distinct_vshard_collections(); - let wv = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &wcoll).as_u32(); - let rv = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &rcoll).as_u32(); + let wv = CollectionKey::from_bare(DatabaseId::DEFAULT, &wcoll) + .vshard() + .as_u32(); + let rv = CollectionKey::from_bare(DatabaseId::DEFAULT, &rcoll) + .vshard() + .as_u32(); assert_ne!(wv, rv); let write_set = ReadWriteSet::new(vec![EngineKeySet::Document { @@ -975,7 +1006,7 @@ mod tests { // set (participants are `#[serde(skip)]`, re-derived in lockstep). let bytes = zerompk::to_msgpack_vec(&tx).expect("encode"); let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode"); - decoded.restore_derived(); + decoded.restore_derived().expect("restore derived"); assert_eq!( tx.participating_vshards(), decoded.participating_vshards(), @@ -1020,9 +1051,27 @@ mod tests { }]); let got: Vec = ws .participating_vshards() + .expect("participants derive") .iter() .map(|v| v.as_u32()) .collect(); assert_eq!(got, vec![only]); } + + #[test] + fn participating_vshards_dequalify_named_database_collections() { + let db = DatabaseId::new(1024); + let qualified = nodedb_types::QualifiedCollection::new(db, "users"); + let ws = ReadWriteSet::new(vec![doc_set(qualified.as_str(), vec![1])]); + let vshards = ws + .participating_vshards_in_database(db) + .expect("participants"); + assert_eq!( + vshards, + vec![CollectionKey::from_bare(db, "users").vshard()] + ); + + let unqualified = ReadWriteSet::new(vec![doc_set("users", vec![1])]); + assert!(unqualified.participating_vshards_in_database(db).is_err()); + } } diff --git a/nodedb-cluster/src/diag/context.rs b/nodedb-cluster/src/diag/context.rs index cb1fb4e37..9ea6e171a 100644 --- a/nodedb-cluster/src/diag/context.rs +++ b/nodedb-cluster/src/diag/context.rs @@ -140,6 +140,41 @@ impl DomainContext for SequencerBackpressureDrop<'_> { } } +/// An epoch batch carries a transaction whose participant set cannot be +/// derived: a key-set collection name lacks the qualifier of the +/// transaction's database. No constructor builds such a class. +pub(super) struct SequencerParticipantsUnderivable<'a> { + pub epoch: u64, + pub raft_index: u64, + pub detail: &'a str, +} + +impl DomainContext for SequencerParticipantsUnderivable<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb_cluster.sequencer_participants_underivable" + } + + fn grouping_key(&self) -> String { + // One bug: a class reached the log without passing construction. The + // epoch, index, and offending name are the occurrence. + "underivable".to_owned() + } + + fn to_json(&self) -> Value { + json!({ + "epoch": self.epoch, + "raft_index": self.raft_index, + "detail": self.detail, + "why_fatal": "the batch is skipped like an undecodable entry, so none of its \ + transactions is fanned out and their completion waiters never \ + resolve until their own deadlines elapse", + "operator_action": "find the proposer that built this transaction class without \ + a TxClass constructor, or the corruption that altered its \ + key-set collection names", + }) + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/nodedb-cluster/src/diag/inert.rs b/nodedb-cluster/src/diag/inert.rs index 772595c95..9086ca417 100644 --- a/nodedb-cluster/src/diag/inert.rs +++ b/nodedb-cluster/src/diag/inert.rs @@ -25,3 +25,6 @@ pub fn sequencer_backpressure_drop( _drops: &[(u32, &'static str)], ) { } + +#[inline] +pub fn sequencer_participants_underivable(_epoch: u64, _raft_index: u64, _detail: &str) {} diff --git a/nodedb-cluster/src/diag/mod.rs b/nodedb-cluster/src/diag/mod.rs index 4bd095d86..301dde73b 100644 --- a/nodedb-cluster/src/diag/mod.rs +++ b/nodedb-cluster/src/diag/mod.rs @@ -30,7 +30,11 @@ mod recording; mod inert; #[cfg(all(feature = "diagnostics", not(target_arch = "wasm32")))] -pub use recording::{sequencer_backpressure_drop, sequencer_epoch_gap}; +pub use recording::{ + sequencer_backpressure_drop, sequencer_epoch_gap, sequencer_participants_underivable, +}; #[cfg(not(all(feature = "diagnostics", not(target_arch = "wasm32"))))] -pub use inert::{sequencer_backpressure_drop, sequencer_epoch_gap}; +pub use inert::{ + sequencer_backpressure_drop, sequencer_epoch_gap, sequencer_participants_underivable, +}; diff --git a/nodedb-cluster/src/diag/recording.rs b/nodedb-cluster/src/diag/recording.rs index 48684ee20..22b341cad 100644 --- a/nodedb-cluster/src/diag/recording.rs +++ b/nodedb-cluster/src/diag/recording.rs @@ -75,3 +75,23 @@ pub fn sequencer_backpressure_drop(epoch: u64, dropped_count: u64, drops: &[(u32 .with_backtrace() .emit(); } + +/// Report an epoch batch skipped because one of its transactions has an +/// underivable participant set. +/// +/// Called from the one site that detects it: `apply`'s `restore_derived` +/// pass over a decoded epoch batch. +pub fn sequencer_participants_underivable(epoch: u64, raft_index: u64, detail: &str) { + let ctx = context::SequencerParticipantsUnderivable { + epoch, + raft_index, + detail, + }; + let _ = Capture::new( + EventKind::InvariantViolation, + "Calvin sequencer: epoch batch carries a transaction with underivable participants", + ) + .domain(&ctx) + .with_backtrace() + .emit(); +} diff --git a/nodedb-cluster/src/error.rs b/nodedb-cluster/src/error.rs index 65c9e1f17..275aad697 100644 --- a/nodedb-cluster/src/error.rs +++ b/nodedb-cluster/src/error.rs @@ -16,6 +16,11 @@ pub enum CalvinError { )] SingleVshardTxn { vshard: u32 }, + /// A key-set collection name lacks the qualifier of the transaction's + /// database, so its vShard cannot be derived. + #[error("calvin key set: {0}")] + CollectionKey(#[from] nodedb_types::CollectionKeyError), + /// A sequencer-layer error. See [`crate::calvin::sequencer::error::SequencerError`] /// for the full variant set. #[error("sequencer error: {0}")] diff --git a/nodedb-cluster/src/routing.rs b/nodedb-cluster/src/routing.rs index 4a471f496..78b93e31f 100644 --- a/nodedb-cluster/src/routing.rs +++ b/nodedb-cluster/src/routing.rs @@ -2,7 +2,7 @@ use std::collections::HashMap; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::{CollectionKey, VShardId}; use crate::error::{ClusterError, Result}; @@ -298,17 +298,14 @@ impl RoutingTable { } } -/// Compute the primary vShard for a `(database, collection)` pair. +/// Compute the primary vShard for a collection. /// -/// Delegates to [`VShardId::from_collection_in_database`] so the cluster -/// routing layer and the types-layer hash function cannot drift. The database -/// id is folded into the hash so the same collection name in two different -/// databases routes to independent vShards — required for multi-database -/// isolation. Passing only the collection name (without `db`) would route -/// every database through the same vShard space and silently corrupt -/// cross-database deployments; the parameter is mandatory by design. -pub fn vshard_for_collection(database_id: DatabaseId, collection: &str) -> u32 { - VShardId::from_collection_in_database(database_id, collection).as_u32() +/// Delegates to [`VShardId::from_collection`], so the cluster routing layer +/// and the types-layer hash cannot drift. The [`CollectionKey`] carries the +/// database id and the bare catalog name, so a qualified name can never reach +/// the hash. +pub fn vshard_for_collection(key: CollectionKey<'_>) -> u32 { + VShardId::from_collection(key).as_u32() } /// FNV-1a 64-bit hash for deterministic key partitioning. @@ -467,11 +464,12 @@ mod tests { // gateway while the data plane still keys them by the correct // hash. This test pins the contract. for db_raw in [0u64, 1, 2, 1024, 999_999] { - let db = DatabaseId::new(db_raw); + let db = nodedb_types::id::DatabaseId::new(db_raw); for name in ["users", "orders", "events", "a", "this_is_a_long_name"] { + let key = CollectionKey::from_bare(db, name); assert_eq!( - vshard_for_collection(db, name), - VShardId::from_collection_in_database(db, name).as_u32(), + vshard_for_collection(key), + VShardId::from_collection(key).as_u32(), "drift detected: db={db_raw} collection={name}" ); } @@ -484,8 +482,11 @@ mod tests { // different vShards (probabilistic; "users" is a canonical // example whose hashes are known to differ across DEFAULT and // DatabaseId(1024)). - let v_default = vshard_for_collection(DatabaseId::DEFAULT, "users"); - let v_other = vshard_for_collection(DatabaseId::new(1024), "users"); + use nodedb_types::id::DatabaseId; + let v_default = + vshard_for_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, "users")); + let v_other = + vshard_for_collection(CollectionKey::from_bare(DatabaseId::new(1024), "users")); assert_ne!( v_default, v_other, "same collection name across databases must route independently" diff --git a/nodedb-cluster/src/rpc_codec/surrogate.rs b/nodedb-cluster/src/rpc_codec/surrogate.rs index c8c00607e..9087f7451 100644 --- a/nodedb-cluster/src/rpc_codec/surrogate.rs +++ b/nodedb-cluster/src/rpc_codec/surrogate.rs @@ -41,7 +41,8 @@ pub struct AssignSurrogateRequest { pub vshard_id: u32, pub database_id: u64, pub tenant_id: u64, - /// Collection the surrogate is scoped to. + /// Bare catalog name of the collection the surrogate is scoped to. With + /// `database_id` it forms the canonical collection key. pub collection: String, /// Primary-key bytes of the endpoint whose surrogate is being resolved. pub pk: Vec, diff --git a/nodedb-cluster/src/shard_split.rs b/nodedb-cluster/src/shard_split.rs index 7500f4508..b20c391d1 100644 --- a/nodedb-cluster/src/shard_split.rs +++ b/nodedb-cluster/src/shard_split.rs @@ -213,16 +213,16 @@ pub fn plan_graph_split( pub fn speculative_prefetch_shards( query_vshards: &[u32], _routing: &RoutingTable, - tenant_collections: &[(nodedb_types::id::DatabaseId, u32, String)], + tenant_collections: &[(u32, nodedb_types::id::CollectionKey<'_>)], ) -> Vec { let mut prefetch: HashSet = HashSet::new(); let queried: HashSet = query_vshards.iter().copied().collect(); - // For each (database, tenant, collection), find all vShards that might hold - // data for the same collection (co-located shards). - for (database_id, _tenant_id, collection) in tenant_collections { - // Hash the (database, collection) pair to find its primary vShard. - let primary = crate::routing::vshard_for_collection(*database_id, collection); + // For each (tenant, collection), find all vShards that might hold data + // for the same collection (co-located shards). + for (_tenant_id, key) in tenant_collections { + // The collection key names its primary vShard. + let primary = crate::routing::vshard_for_collection(*key); // Adjacent vShards (±1, ±2) are likely to hold related data // due to hash distribution locality. @@ -294,7 +294,13 @@ mod tests { let prefetch = speculative_prefetch_shards( &[0, 1], &routing, - &[(nodedb_types::id::DatabaseId::DEFAULT, 1, "users".into())], + &[( + 1, + nodedb_types::id::CollectionKey::from_bare( + nodedb_types::id::DatabaseId::DEFAULT, + "users", + ), + )], ); assert!(prefetch.len() <= 8); // Max prefetch limit. } diff --git a/nodedb-physical/src/convert_context.rs b/nodedb-physical/src/convert_context.rs index d017c09dc..8f1afda29 100644 --- a/nodedb-physical/src/convert_context.rs +++ b/nodedb-physical/src/convert_context.rs @@ -21,10 +21,9 @@ use crate::SurrogateAssigner; /// implementations wrap it (Origin adds catalog/WAL handles, Lite passes /// it through unchanged). pub struct SharedConvertContext { - /// Database scope for vShard computation. All - /// `VShardId::from_collection_in_database` calls inside the converter - /// must use this value so collections in different databases route to - /// distinct shards. + /// Database scope for vShard computation. Every `CollectionKey` the + /// converter builds uses this value, so collections in different + /// databases route to distinct shards. pub database_id: DatabaseId, /// Per-tenant maximum vector dimension (0 = unlimited). Checked during diff --git a/nodedb-physical/src/surrogate.rs b/nodedb-physical/src/surrogate.rs index 05bd6c463..f99dbe1cd 100644 --- a/nodedb-physical/src/surrogate.rs +++ b/nodedb-physical/src/surrogate.rs @@ -8,7 +8,7 @@ //! code paths. Origin's async surrogate-fetch work stays internal to its impl //! and is hidden behind this sync facade. -use nodedb_types::{DatabaseId, Surrogate, TenantId}; +use nodedb_types::{CollectionKey, Surrogate, TenantId}; /// Errors a [`SurrogateAssigner`] may return. /// @@ -38,14 +38,14 @@ pub trait SurrogateAssigner: Send + Sync { /// allocator. Used by CLONE DATABASE to capture an AS-OF cutoff. fn current_hwm(&self) -> u32; - /// Resolve `(database_id, tenant_id, collection, pk_bytes)` to a stable - /// surrogate. Allocate on the first call; return the persisted value on - /// every subsequent call (UPSERT preserves the surrogate). + /// Resolve `(key, tenant_id, pk_bytes)` to a stable surrogate. Allocate + /// on the first call; return the persisted value on every subsequent + /// call (UPSERT preserves the surrogate). `key` carries the database and + /// the bare catalog name. fn assign( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], ) -> Result; @@ -62,8 +62,7 @@ pub trait SurrogateAssigner: Send + Sync { /// verbatim and never re-derives it. fn assign_fresh( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, ) -> Result<(Surrogate, String), SurrogateAssignError>; } diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs b/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs index e633b3130..0c2519152 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs @@ -34,8 +34,7 @@ impl TestClusterNode { let request_id = RequestId::new(HARNESS_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); let vshard_id = VShardId::new(nodedb_cluster::routing::vshard_for_collection( - DatabaseId::DEFAULT, - collection, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), )); let request = Request { request_id, @@ -107,8 +106,7 @@ impl TestClusterNode { let request_id = RequestId::new(HARNESS_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); let vshard_id = VShardId::new(nodedb_cluster::routing::vshard_for_collection( - DatabaseId::DEFAULT, - collection, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), )); let request = Request { request_id, diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs b/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs index 680d3500b..412d5d6a7 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs @@ -36,7 +36,9 @@ impl TestClusterNode { /// error or non-Ok response. pub async fn create_tenant_snapshot(&self, tenant: TenantId) -> Vec { let request_id = RequestId::new(SNAPSHOT_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); - let vshard_id = VShardId::new(vshard_for_collection(DatabaseId::DEFAULT, "__system")); + let vshard_id = VShardId::new(vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), + )); let request = Request { request_id, tenant_id: tenant, @@ -98,7 +100,9 @@ impl TestClusterNode { /// `Ok`. pub async fn restore_tenant_snapshot(&self, snapshot_bytes: Vec) -> bool { let request_id = RequestId::new(SNAPSHOT_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); - let vshard_id = VShardId::new(vshard_for_collection(DatabaseId::DEFAULT, "__system")); + let vshard_id = VShardId::new(vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), + )); let request = Request { request_id, tenant_id: TenantId::new(0), diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs b/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs index de1ab5df2..f01c71164 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs @@ -221,8 +221,9 @@ impl TestClusterNode { /// group mapping yet (e.g. before the CREATE COLLECTION DDL has /// propagated). pub fn group_id_for_collection(&self, collection: &str) -> Option { - let vshard = - nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, collection); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), + ); self.shared .cluster_routing .as_ref()? diff --git a/nodedb-types/src/id/collection_key.rs b/nodedb-types/src/id/collection_key.rs new file mode 100644 index 000000000..7dfac715d --- /dev/null +++ b/nodedb-types/src/id/collection_key.rs @@ -0,0 +1,209 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Canonical collection identity for placement and surrogate identity. +//! +//! A collection's vShard and its surrogate bindings are keyed by the pair +//! `(database_id, bare_name)`. The bare name is the name the catalog keys the +//! collection by. The database-qualified form `"{database_id}/{name}"` names +//! the same collection, but it folds the database into the string a second +//! time. Hashing it gives a different vShard than hashing the bare name. +//! +//! [`CollectionKey`] is the only input the vShard hash and the surrogate +//! allocator accept. It has no `From<&str>`. A caller builds it from a bare +//! catalog name with [`CollectionKey::from_bare`], or from a qualified name +//! with [`CollectionKey::from_qualified`], which strips the qualifier. + +use super::{DatabaseId, QualifiedCollection}; + +/// Error returned when a qualified collection name does not carry the +/// qualifier of the database it is resolved in. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum CollectionKeyError { + /// The name lacks the `"{database_id}/"` prefix that + /// [`QualifiedCollection::new`] writes for a non-default database. + #[error( + "collection name '{name}' is not qualified for database {database_id}; \ + expected the prefix '{database_id}/'" + )] + NotQualified { + /// Raw id of the database the name was resolved in. + database_id: u64, + /// The rejected name. + name: String, + }, +} + +/// The canonical `(database_id, bare_name)` identity of a collection. +/// +/// Borrowed: it never allocates. Build it with [`Self::from_bare`] from a +/// catalog name, or with [`Self::from_qualified`] / +/// [`Self::from_qualified_str`] from a database-qualified name. +/// +/// No `From<&str>` impl exists, so a raw string never converts implicitly: +/// +/// ```compile_fail +/// use nodedb_types::CollectionKey; +/// let key: CollectionKey<'_> = "1024/users".into(); +/// ``` +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct CollectionKey<'a> { + database_id: DatabaseId, + name: &'a str, +} + +impl<'a> CollectionKey<'a> { + /// Build a key from the bare name the catalog keys the collection by. + /// + /// `name` must be the catalog name, never a `"{database_id}/{name}"` + /// string. A qualified name goes through [`Self::from_qualified`]. + pub fn from_bare(database_id: DatabaseId, name: &'a str) -> Self { + Self { database_id, name } + } + + /// Build a key from a [`QualifiedCollection`], stripping its qualifier. + pub fn from_qualified( + database_id: DatabaseId, + qualified: &'a QualifiedCollection, + ) -> Result { + Self::from_qualified_str(database_id, qualified.as_str()) + } + + /// Build a key from a database-qualified name carried as a string on a + /// plan, a WAL record, or the wire. + /// + /// The default database stores names unqualified, so the name is taken + /// as-is. The empty name marks a plan with no routing collection and is + /// the same in both forms, so it is also taken as-is. Any other name in + /// any other database requires the exact `"{database_id}/"` prefix + /// [`QualifiedCollection::new`] writes. Exactly one prefix is stripped, + /// so a bare name that itself contains `/` survives intact. + pub fn from_qualified_str( + database_id: DatabaseId, + qualified: &'a str, + ) -> Result { + if database_id == DatabaseId::DEFAULT || qualified.is_empty() { + return Ok(Self::from_bare(database_id, qualified)); + } + let not_qualified = || CollectionKeyError::NotQualified { + database_id: database_id.as_u64(), + name: qualified.to_owned(), + }; + let (head, bare) = qualified.split_once('/').ok_or_else(not_qualified)?; + if head.is_empty() || !head.bytes().all(|b| b.is_ascii_digit()) { + return Err(not_qualified()); + } + match head.parse::() { + Ok(raw) if raw == database_id.as_u64() && !head.starts_with('0') => { + Ok(Self::from_bare(database_id, bare)) + } + _ => Err(not_qualified()), + } + } + + /// The database the collection lives in. + pub fn database_id(&self) -> DatabaseId { + self.database_id + } + + /// The bare catalog name. + pub fn name(&self) -> &'a str { + self.name + } + + /// The database-qualified name storage engines key data by. + pub fn qualified(&self) -> QualifiedCollection { + QualifiedCollection::new(self.database_id, self.name) + } + + /// The vShard this collection homes to. + pub fn vshard(&self) -> super::VShardId { + super::VShardId::from_collection(*self) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const DB: DatabaseId = DatabaseId::new(1024); + + #[test] + fn qualified_input_yields_the_bare_key() { + let qualified = QualifiedCollection::new(DB, "users"); + let from_qualified = CollectionKey::from_qualified(DB, &qualified).expect("qualified"); + assert_eq!(from_qualified.name(), "users"); + assert_eq!(from_qualified, CollectionKey::from_bare(DB, "users")); + assert_eq!( + from_qualified.vshard(), + CollectionKey::from_bare(DB, "users").vshard() + ); + } + + #[test] + fn qualified_string_cannot_become_a_key_without_dequalifying() { + // The only string-accepting constructors are `from_bare`, which the + // caller names explicitly, and the qualified constructors, which strip + // the qualifier. The qualified path therefore always lands on the bare + // name, never on the qualified string. + let qualified = QualifiedCollection::new(DB, "users"); + let key = CollectionKey::from_qualified_str(DB, qualified.as_str()).expect("qualified"); + assert_ne!(key.name(), qualified.as_str()); + assert_eq!(key.name(), "users"); + assert_eq!(key.qualified(), qualified); + } + + #[test] + fn default_database_names_are_already_bare() { + let key = CollectionKey::from_qualified_str(DatabaseId::DEFAULT, "users").expect("default"); + assert_eq!(key, CollectionKey::from_bare(DatabaseId::DEFAULT, "users")); + } + + #[test] + fn empty_name_is_the_no_collection_sentinel_in_every_database() { + let key = CollectionKey::from_qualified_str(DB, "").expect("empty"); + assert_eq!(key, CollectionKey::from_bare(DB, "")); + } + + #[test] + fn exactly_one_qualifier_is_stripped() { + let qualified = QualifiedCollection::new(DB, "1024/nested"); + let key = CollectionKey::from_qualified(DB, &qualified).expect("qualified"); + assert_eq!(key.name(), "1024/nested"); + } + + #[test] + fn unqualified_name_in_a_named_database_is_rejected() { + let err = CollectionKey::from_qualified_str(DB, "users").expect_err("unqualified"); + assert_eq!( + err, + CollectionKeyError::NotQualified { + database_id: 1024, + name: "users".to_owned(), + } + ); + } + + #[test] + fn foreign_database_qualifier_is_rejected() { + assert!(CollectionKey::from_qualified_str(DB, "7/users").is_err()); + assert!(CollectionKey::from_qualified_str(DB, "+1024/users").is_err()); + assert!(CollectionKey::from_qualified_str(DB, "01024/users").is_err()); + assert!(CollectionKey::from_qualified_str(DB, "/users").is_err()); + } + + #[test] + fn bare_and_qualified_hashes_differ_for_a_named_database() { + // The two strings name one collection. Only the key keeps them on one + // vShard, so hashing the qualified string directly is a routing error. + let mut differs = false; + for i in 0..64 { + let name = format!("coll_{i}"); + let qualified = QualifiedCollection::new(DB, &name); + let key = CollectionKey::from_qualified(DB, &qualified).expect("qualified"); + let raw_qualified = CollectionKey::from_bare(DB, qualified.as_str()); + assert_eq!(key.vshard(), CollectionKey::from_bare(DB, &name).vshard()); + differs |= key.vshard() != raw_qualified.vshard(); + } + assert!(differs); + } +} diff --git a/nodedb-types/src/id/mod.rs b/nodedb-types/src/id/mod.rs index 51c792d1b..138b82395 100644 --- a/nodedb-types/src/id/mod.rs +++ b/nodedb-types/src/id/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: Apache-2.0 pub mod collection; +pub mod collection_key; pub mod database; pub mod document; pub mod edge; @@ -15,6 +16,7 @@ pub mod txn; pub mod vshard; pub use collection::CollectionId; +pub use collection_key::{CollectionKey, CollectionKeyError}; pub use database::DatabaseId; pub use document::DocumentId; pub use edge::{EdgeId, EdgeIdParseError}; diff --git a/nodedb-types/src/id/vshard.rs b/nodedb-types/src/id/vshard.rs index 2ac085db8..0230f2a07 100644 --- a/nodedb-types/src/id/vshard.rs +++ b/nodedb-types/src/id/vshard.rs @@ -39,18 +39,19 @@ impl VShardId { self.0 } - /// Compute vShard from a database + collection name pair. + /// Compute the vShard a collection homes to. /// - /// The database identity is mixed into the hash so that the same collection - /// name in two different databases routes to independent vShards. Uses a - /// DJB-like multiply-31 hash, seeded with the database id bytes, followed - /// by a zero separator byte, followed by the collection name bytes. - pub fn from_collection_in_database(db: crate::id::DatabaseId, collection: &str) -> Self { - let db_bytes = db.as_u64().to_le_bytes(); + /// Takes a [`CollectionKey`](crate::id::CollectionKey), so the hashed name + /// is always the bare catalog name. The database identity is mixed into + /// the hash, so the same name in two databases routes to independent + /// vShards. Uses a DJB-like multiply-31 hash, seeded with the database id + /// bytes, then a zero separator byte, then the bare name bytes. + pub fn from_collection(key: crate::id::CollectionKey<'_>) -> Self { + let db_bytes = key.database_id().as_u64().to_le_bytes(); let hash = db_bytes .iter() .chain(std::iter::once(&0u8)) - .chain(collection.as_bytes().iter()) + .chain(key.name().as_bytes().iter()) .fold(0u32, |h, &b| h.wrapping_mul(31).wrapping_add(b as u32)); Self::new(hash % Self::COUNT) } @@ -114,25 +115,25 @@ mod tests { } #[test] - fn from_collection_in_database_deterministic() { - use crate::id::DatabaseId; + fn from_collection_deterministic() { + use crate::id::{CollectionKey, DatabaseId}; let db = DatabaseId::new(1024); - let a = VShardId::from_collection_in_database(db, "users"); - let b = VShardId::from_collection_in_database(db, "users"); + let a = VShardId::from_collection(CollectionKey::from_bare(db, "users")); + let b = VShardId::from_collection(CollectionKey::from_bare(db, "users")); assert_eq!(a, b); assert!(a.as_u32() < VShardId::COUNT); } #[test] - fn from_collection_in_database_different_dbs_differ() { - use crate::id::DatabaseId; + fn from_collection_different_dbs_differ() { + use crate::id::{CollectionKey, DatabaseId}; let db0 = DatabaseId::DEFAULT; let db1 = DatabaseId::new(1024); // Same collection name in different databases should typically route // to different vShards (probabilistic; collection "users" is a // canonical example and the two hashes are known to differ). - let a = VShardId::from_collection_in_database(db0, "users"); - let b = VShardId::from_collection_in_database(db1, "users"); + let a = VShardId::from_collection(CollectionKey::from_bare(db0, "users")); + let b = VShardId::from_collection(CollectionKey::from_bare(db1, "users")); assert_ne!( a, b, "same collection name, different databases should route differently" @@ -140,9 +141,9 @@ mod tests { } #[test] - fn from_collection_in_database_default_in_range() { - use crate::id::DatabaseId; - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, "orders"); + fn from_collection_default_in_range() { + use crate::id::{CollectionKey, DatabaseId}; + let v = VShardId::from_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, "orders")); assert!(v.as_u32() < VShardId::COUNT); } } diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index 0f4756d38..452c538d3 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -100,8 +100,8 @@ pub use graph::{Direction, GraphStats}; pub use hlc::{ClockSkew, Hlc, HlcClock, MAX_CLOCK_SKEW_NS}; pub use hnsw::{HnswCheckpoint, HnswNodeSnapshot, HnswParams}; pub use id::{ - CollectionId, DatabaseId, DocumentId, EdgeId, EdgeIdParseError, IdError, IdType, NodeId, - QualifiedCollection, ShapeId, TenantId, + CollectionId, CollectionKey, CollectionKeyError, DatabaseId, DocumentId, EdgeId, + EdgeIdParseError, IdError, IdType, NodeId, QualifiedCollection, ShapeId, TenantId, }; pub use identity::KeyRepr; pub use json_msgpack::{ diff --git a/nodedb/src/bootstrap/constraint_reconcile.rs b/nodedb/src/bootstrap/constraint_reconcile.rs index d766099a5..5feb92867 100644 --- a/nodedb/src/bootstrap/constraint_reconcile.rs +++ b/nodedb/src/bootstrap/constraint_reconcile.rs @@ -153,7 +153,9 @@ pub async fn reconcile_once( continue; } - let vshard_id = nodedb_cluster::routing::vshard_for_collection(database_id, &stored.name); + let vshard_id = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(database_id, &stored.name), + ); let entry = ReplicatedEntry::new( stored.tenant_id, database_id.as_u64(), diff --git a/nodedb/src/control/array_catalog/persist.rs b/nodedb/src/control/array_catalog/persist.rs index fe04357e0..4c5689505 100644 --- a/nodedb/src/control/array_catalog/persist.rs +++ b/nodedb/src/control/array_catalog/persist.rs @@ -140,9 +140,8 @@ mod tests { let array = entry("cells"); persist(&cat, &array).unwrap(); cat.put_surrogate( - array.array_id.database_id, + nodedb_types::CollectionKey::from_bare(array.array_id.database_id, &array.name), array.array_id.tenant_id, - &array.name, b"coord:1", nodedb_types::Surrogate::new(42), ) @@ -162,9 +161,8 @@ mod tests { ); assert_eq!( cat.get_surrogate_for_pk( - array.array_id.database_id, + nodedb_types::CollectionKey::from_bare(array.array_id.database_id, &array.name), array.array_id.tenant_id, - &array.name, b"coord:1" ) .unwrap(), @@ -183,9 +181,8 @@ mod tests { ); assert!( cat.get_surrogate_for_pk( - array.array_id.database_id, + nodedb_types::CollectionKey::from_bare(array.array_id.database_id, &array.name), array.array_id.tenant_id, - &array.name, b"coord:1" ) .unwrap() diff --git a/nodedb/src/control/backup/orchestrator.rs b/nodedb/src/control/backup/orchestrator.rs index b003d346f..2808d570f 100644 --- a/nodedb/src/control/backup/orchestrator.rs +++ b/nodedb/src/control/backup/orchestrator.rs @@ -184,7 +184,10 @@ fn filter_node_snapshot( tenant_id, source_vshards, |collection| { - nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, collection) + nodedb_cluster::routing::vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + collection, + )) }, ); zerompk::to_msgpack_vec(&snap).map_err(|e| Error::Internal { @@ -233,9 +236,8 @@ fn push_metadata_sections( let mut binds: Vec = Vec::new(); for coll in &collections { let rows = catalog.scan_surrogates_for_collection( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &coll.name), TenantId::new(tenant_id), - &coll.name, )?; for (pk, surrogate) in rows { binds.push(nodedb_types::backup_envelope::SurrogateBindBlob { @@ -313,8 +315,7 @@ async fn snapshot_self( sync_dispatch::SystemReason::BackupRestore, TenantId::new(tenant_id), // TODO(A8-followup): backup/restore not yet multi-database. - DatabaseId::DEFAULT, - "__system", + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), plan.clone(), ), NODE_SNAPSHOT_TIMEOUT, diff --git a/nodedb/src/control/backup/restore/crdt_reissue.rs b/nodedb/src/control/backup/restore/crdt_reissue.rs index 514aad4ef..c3b44c12f 100644 --- a/nodedb/src/control/backup/restore/crdt_reissue.rs +++ b/nodedb/src/control/backup/restore/crdt_reissue.rs @@ -17,7 +17,7 @@ use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::server::dispatch_utils::{AutocommitWrite, dispatch_autocommit_write}; use crate::control::state::SharedState; use crate::event::EventSource; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::CrdtOp; /// Per-import dispatch timeout. Generous: a collection's Loro snapshot may be @@ -37,7 +37,8 @@ async fn reissue_crdt_collection( collection: &str, bytes: Vec, ) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); + // `collection` is the stored, database-qualified name. + let vshard = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let plan = PhysicalPlan::Crdt(CrdtOp::ImportSnapshot { tenant_id: tenant_id.as_u64(), collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), diff --git a/nodedb/src/control/backup/restore/durable.rs b/nodedb/src/control/backup/restore/durable.rs index f4309799f..8bf326861 100644 --- a/nodedb/src/control/backup/restore/durable.rs +++ b/nodedb/src/control/backup/restore/durable.rs @@ -20,7 +20,8 @@ use crate::types::{DatabaseId, TenantId, VShardId}; /// collection's rows may travel in a single write. const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); -/// Write `plan`, restored into `collection`, durably. +/// Write `plan`, restored into `collection`, durably. `collection` is the +/// bare catalog name in `database_id`. /// /// Branches identically to a normal write: /// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. @@ -35,7 +36,7 @@ pub async fn reissue_plan_durably( collection: &str, plan: PhysicalPlan, ) -> crate::Result<()> { - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); if let Some(proposer) = state.async_raft_proposer() { let entry = crate::control::wal_replication::to_replicated_entry( @@ -89,8 +90,7 @@ pub async fn reissue_plan_durably( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::BackupRestore, tenant_id, - database_id, - collection, + nodedb_types::CollectionKey::from_bare(database_id, collection), plan, ) .with_minted(minted), diff --git a/nodedb/src/control/backup/restore/kv_reissue.rs b/nodedb/src/control/backup/restore/kv_reissue.rs index 55f71afb9..383d03078 100644 --- a/nodedb/src/control/backup/restore/kv_reissue.rs +++ b/nodedb/src/control/backup/restore/kv_reissue.rs @@ -62,7 +62,7 @@ pub(in crate::control::backup::restore) async fn reissue_kv_tables( state, "kv", collection, - crate::types::VShardId::from_collection_in_database(database_id, collection), + nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), rows.len(), ); let now_ms = std::time::SystemTime::now() @@ -73,9 +73,11 @@ pub(in crate::control::backup::restore) async fn reissue_kv_tables( let Some(ttl_ms) = remaining_ttl_ms(expire_at_ms, now_ms) else { continue; }; - let surrogate = state - .surrogate_assigner - .assign(database_id, tenant, &stored, &key)?; + let surrogate = state.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant, + &key, + )?; let plan = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::from_stored(stored.clone()), key, diff --git a/nodedb/src/control/backup/restore/orchestrate/rebind.rs b/nodedb/src/control/backup/restore/orchestrate/rebind.rs index 9444749e4..6ef03e87a 100644 --- a/nodedb/src/control/backup/restore/orchestrate/rebind.rs +++ b/nodedb/src/control/backup/restore/orchestrate/rebind.rs @@ -28,9 +28,8 @@ pub(super) fn rebind_surrogates( let database_id = crate::types::DatabaseId::DEFAULT; for e in binds { state.surrogate_assigner.bind( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &e.collection), TenantId::new(e.tenant_id), - &e.collection, &e.pk, Surrogate::new(e.surrogate), )?; diff --git a/nodedb/src/control/backup/restore/redo_reissue/commit.rs b/nodedb/src/control/backup/restore/redo_reissue/commit.rs index dabb04be9..8192b03ec 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/commit.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/commit.rs @@ -21,7 +21,7 @@ use crate::control::wal_replication::transaction_redo::{ RedoTarget, TransactionRedoPayload, apply_transaction_redo, }; use crate::event::EventSource; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use crate::wal::{RedoRecord, RedoSubRecord}; use super::units::{CollectionUnits, RowUnit}; @@ -139,7 +139,7 @@ pub(super) async fn commit_collection( let target = RedoTarget { tenant_id, database_id, - vshard_id: VShardId::from_collection_in_database(database_id, &collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(), }; let mut records = 0usize; for batch in batch_units(units) { @@ -159,6 +159,7 @@ mod tests { use nodedb_types::Surrogate; use super::*; + use crate::types::VShardId; fn unit(ops: usize, payload_len: usize, pk: &str) -> RowUnit { RowUnit { diff --git a/nodedb/src/control/backup/restore/redo_reissue/documents.rs b/nodedb/src/control/backup/restore/redo_reissue/documents.rs index 17e05f7fe..b433dbd9e 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/documents.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/documents.rs @@ -194,9 +194,8 @@ impl RowBuilder<'_> { /// under that surrogate, over the row it names. fn bind(&self, identity: &RowIdentity, key: StorageKey) -> crate::Result { let surrogate = self.state.surrogate_assigner.bind( - self.database_id, + nodedb_types::CollectionKey::from_bare(self.database_id, self.collection), self.tenant, - self.collection, identity.as_str().as_bytes(), key.surrogate(), )?; diff --git a/nodedb/src/control/backup/restore/redo_reissue/edges.rs b/nodedb/src/control/backup/restore/redo_reissue/edges.rs index 14cf1bf27..899c41eab 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/edges.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/edges.rs @@ -39,10 +39,11 @@ fn node_identity( collection: &str, node_id: &str, ) -> crate::Result { - let surrogate = - state - .surrogate_assigner - .assign(database_id, tenant, collection, node_id.as_bytes())?; + let surrogate = state.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant, + node_id.as_bytes(), + )?; Ok(CarriedIdentity { collection: collection.to_string(), pk_bytes: node_id.as_bytes().to_vec(), diff --git a/nodedb/src/control/backup/snapshot_keys.rs b/nodedb/src/control/backup/snapshot_keys.rs index 7930cbca1..a18519678 100644 --- a/nodedb/src/control/backup/snapshot_keys.rs +++ b/nodedb/src/control/backup/snapshot_keys.rs @@ -80,7 +80,7 @@ pub fn extract_db_scoped_collection(key: &str, tenant_id: u64) -> Option<&str> { /// consumes the output exactly as before. /// /// `vshard_of` maps a collection name to its vshard (the caller passes the -/// canonical `vshard_for_collection(DEFAULT, _)`), matching the Raft snapshot +/// canonical `vshard_for_collection` over the default-database key), matching the Raft snapshot /// SEND builder. Every section kind the snapshot carries is classified here so /// adding a section without updating this filter is impossible to miss: /// diff --git a/nodedb/src/control/catalog_entry/apply/collection.rs b/nodedb/src/control/catalog_entry/apply/collection.rs index 5fa9eafd0..eb5631a6b 100644 --- a/nodedb/src/control/catalog_entry/apply/collection.rs +++ b/nodedb/src/control/catalog_entry/apply/collection.rs @@ -178,9 +178,8 @@ pub fn finalize_purge( catalog, )?; catalog.delete_all_surrogates_for_collection( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, name), nodedb_types::TenantId::new(tenant_id), - name, )?; // An index cannot outlive the collection it indexes. Its identity rows, // its ownership rows, and any engine-side build parameters go with the diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs index b0da81747..25bb5fe9c 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs @@ -33,7 +33,7 @@ use tokio::sync::oneshot; use crate::bridge::envelope::PhysicalPlan; use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::types::{DatabaseId, Lsn, TenantId}; use nodedb_physical::physical_plan::VectorOp; use nodedb_types::StoredVectorIndexParams; @@ -343,7 +343,7 @@ fn append_redo( let owner = RecordOwner { tenant_id: TenantId::new(target.tenant_id), database_id, - vshard_id: VShardId::from_collection_in_database(database_id, target.collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, target.collection).vshard(), }; let outcome = minted.append_plan( &shared.wal, diff --git a/nodedb/src/control/clone/copyup.rs b/nodedb/src/control/clone/copyup.rs index 2508c6614..ebcde51b9 100644 --- a/nodedb/src/control/clone/copyup.rs +++ b/nodedb/src/control/clone/copyup.rs @@ -18,7 +18,7 @@ use nodedb_types::{DatabaseId, Surrogate, TenantId}; use crate::bridge::envelope::{Priority, Request, Status}; use crate::control::state::SharedState; -use crate::types::{ReadConsistency, RequestId, TraceId, VShardId}; +use crate::types::{ReadConsistency, RequestId, TraceId}; use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; /// Parameters for a KV copy-up operation. @@ -47,15 +47,12 @@ pub async fn perform_kv_clone_copyup(params: KvCopyUpParams<'_>) -> crate::Resul source_value_bytes, } = params; - let target_coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( - target_db_id, - target_collection, - ); + let target_key = nodedb_types::CollectionKey::from_bare(target_db_id, target_collection); // Allocate a surrogate for the target KV row. let surrogate = state .surrogate_assigner - .assign(target_db_id, tenant_id, &target_coll_qualified, &kv_key) + .assign(target_key, tenant_id, &kv_key) .map_err(|e| crate::Error::Storage { engine: "clone_kv_copyup".into(), detail: format!("surrogate alloc failed: {e}"), @@ -72,7 +69,7 @@ pub async fn perform_kv_clone_copyup(params: KvCopyUpParams<'_>) -> crate::Resul provenance: None, }); - let vshard_id = VShardId::from_collection_in_database(target_db_id, &target_coll_qualified); + let vshard_id = target_key.vshard(); let req_id = RequestId::new( state .request_id_counter @@ -159,18 +156,14 @@ pub async fn perform_clone_copyup(params: CopyUpParams<'_>) -> crate::Result) -> crate::Result) -> crate::Res let system_time = rewrite_system_time(effective_source_ms, *system_time)?; let Some(source_surrogate) = state .surrogate_assigner - .lookup(source_db_id, tenant_id, source_qualified.as_str(), pk_bytes) + .lookup( + nodedb_types::CollectionKey::from_bare(source_db_id, source_coll), + tenant_id, + pk_bytes, + ) .ok() .flatten() else { diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs index 3c50aac69..41bf6fcfd 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs @@ -237,7 +237,7 @@ mod tests { use nodedb_cluster::calvin::types::{EpochBatch, SchedulerInput, SequencedTxn}; use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerEntry}; use nodedb_types::TenantId; - use nodedb_types::id::{DatabaseId, VShardId}; + use nodedb_types::id::DatabaseId; use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ build_test_scheduler, build_test_scheduler_with_data_side, fill_tenant_inflight, @@ -378,8 +378,9 @@ mod tests { async fn drain_replays_dropped_input_into_lock_table_end_to_end() { // Use the vShard that "test_coll" hashes to, so the batch's fan-out targets — // and its replay decodes for — this scheduler's vShard. - let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "test_coll").as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "test_coll") + .vshard() + .as_u32(); let (mut scheduler, _dir) = build_test_scheduler(vshard); ensure_sequencer_leader(&scheduler); @@ -466,8 +467,9 @@ mod tests { /// epoch 1 (delivered live, in-flight) must be skipped. #[tokio::test] async fn drain_skips_in_flight_overlap_no_double_dispatch_end_to_end() { - let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "test_coll").as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "test_coll") + .vshard() + .as_u32(); let (mut scheduler, _dir) = build_test_scheduler(vshard); ensure_sequencer_leader(&scheduler); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs index 0ada4416a..9a7fa84f0 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs @@ -14,8 +14,7 @@ use nodedb_physical::physical_plan::{ }; use crate::types::{DatabaseId, VShardId}; -#[cfg(test)] -use nodedb_types::QualifiedCollection; +use nodedb_types::{CollectionKey, QualifiedCollection}; /// Where a `PhysicalPlan` routes for Calvin cross-shard scheduling purposes. /// @@ -38,23 +37,35 @@ pub(crate) enum PlanRouting { Unroutable(&'static str), } +/// Whether any read in `reads` homes to `vshard_id`. A read entry carries the +/// plan's database-qualified name. An entry that does not de-qualify homes +/// nowhere, so a txn whose only work is such a read fails loudly as one that +/// homes no local work. pub(crate) fn homes_versioned_read( reads: &nodedb_types::calvin::VersionedReadSet, database_id: DatabaseId, vshard_id: u32, ) -> bool { reads.iter().any(|entry| { - VShardId::from_collection_in_database(database_id, &entry.collection).as_u32() == vshard_id + CollectionKey::from_qualified_str(database_id, &entry.collection) + .is_ok_and(|key| key.vshard().as_u32() == vshard_id) }) } -fn collection_vshard_in_database(database_id: DatabaseId, collection: &str) -> VShardId { - VShardId::from_collection_in_database(database_id, collection) +/// Route a collection-homed write to the vShard of its canonical key. The +/// plan carries the database-qualified name, de-qualified here. +fn collection_routing(database_id: DatabaseId, collection: &QualifiedCollection) -> PlanRouting { + match CollectionKey::from_qualified(database_id, collection) { + Ok(key) => PlanRouting::Vshards(vec![key.vshard()]), + Err(_) => PlanRouting::Unroutable( + "collection name lacks the qualifier of the transaction's database", + ), + } } #[cfg(test)] fn collection_vshard(collection: &str) -> VShardId { - collection_vshard_in_database(DatabaseId::DEFAULT, collection) + CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard() } /// Returns the routing decision for `plan`. Exhaustive over every @@ -105,10 +116,7 @@ fn document_routing(op: &DocumentOp, database_id: DatabaseId) -> PlanRouting { // was derived from homes elsewhere, and the pair is dual-homed by the // two tasks' own vshards rather than by one plan claiming both. | DocumentOp::ApplyBalanceDelta { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) + collection_routing(database_id, collection) } // Never scheduled: the write-resolve orchestrator proposes it through // Raft directly, on the vshard of the collection it resolved. @@ -117,10 +125,7 @@ fn document_routing(op: &DocumentOp, database_id: DatabaseId) -> PlanRouting { ), DocumentOp::InsertSelect { target_collection, .. - } => PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - target_collection.as_str(), - )]), + } => collection_routing(database_id, target_collection), // Both join the target with a DIFFERENT source collection; nothing on // the plan enforces the two live on the same vshard. DocumentOp::Merge { .. } | DocumentOp::UpdateFromJoin { .. } => PlanRouting::Unroutable( @@ -165,10 +170,7 @@ fn kv_routing(op: &KvOp, database_id: DatabaseId) -> PlanRouting { // that collection's vshard like every other single-collection write. | KvOp::PredicateUpdate { collection, .. } | KvOp::PredicateDelete { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) + collection_routing(database_id, collection) } // Source and dest are DIFFERENT collections; no co-location guarantee. KvOp::TransferItem { .. } => PlanRouting::Unroutable( @@ -216,12 +218,7 @@ fn vector_routing(op: &VectorOp, database_id: DatabaseId) -> PlanRouting { | VectorOp::DirectInsertIfAbsent { collection, .. } | VectorOp::DirectDelete { collection, .. } | VectorOp::DirectTruncate { collection, .. } - | VectorOp::DirectUpdate { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) - } + | VectorOp::DirectUpdate { collection, .. } => collection_routing(database_id, collection), // Never scheduled: the write-resolve orchestrator proposes it through // Raft directly, on the vshard of the collection it resolved. VectorOp::ResolvedDirectWrite { .. } => PlanRouting::Unroutable( @@ -305,10 +302,7 @@ fn graph_routing(op: &GraphOp) -> PlanRouting { fn timeseries_routing(op: &TimeseriesOp, database_id: DatabaseId) -> PlanRouting { match op { TimeseriesOp::Ingest { collection, .. } | TimeseriesOp::Truncate { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) + collection_routing(database_id, collection) } // Read-only: it reports the lines the wrapped ingest would store and // mutates nothing. @@ -323,12 +317,7 @@ fn columnar_routing(op: &ColumnarOp, database_id: DatabaseId) -> PlanRouting { | ColumnarOp::Delete { collection, .. } | ColumnarOp::ResolvedUpdate { collection, .. } | ColumnarOp::ResolvedDelete { collection, .. } - | ColumnarOp::Truncate { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) - } + | ColumnarOp::Truncate { collection, .. } => collection_routing(database_id, collection), ColumnarOp::Scan { .. } | ColumnarOp::MaterializeScan { .. } | ColumnarOp::ResolveDml { .. } => PlanRouting::NotAWrite, @@ -347,12 +336,7 @@ fn crdt_routing(op: &CrdtOp, database_id: DatabaseId) -> PlanRouting { | CrdtOp::SetConstraints { collection, .. } | CrdtOp::DropConstraints { collection, .. } | CrdtOp::RestoreToVersion { collection, .. } - | CrdtOp::ImportSnapshot { collection, .. } => { - PlanRouting::Vshards(vec![collection_vshard_in_database( - database_id, - collection.as_str(), - )]) - } + | CrdtOp::ImportSnapshot { collection, .. } => collection_routing(database_id, collection), CrdtOp::Read { .. } | CrdtOp::PreviewApply { .. } | CrdtOp::ReadConstraints { .. } @@ -570,21 +554,23 @@ mod tests { #[test] fn collection_routing_preserves_database_scope() { + let db = DatabaseId::new(7); let collection = (0..2048) .map(|i| format!("db_scoped_{i}")) .find(|name| { - collection_vshard_in_database(DatabaseId::DEFAULT, name) - != collection_vshard_in_database(DatabaseId::new(7), name) + CollectionKey::from_bare(DatabaseId::DEFAULT, name).vshard() + != CollectionKey::from_bare(db, name).vshard() }) .expect("collection whose home differs by database"); let plan = PhysicalPlan::Document(DocumentOp::Truncate { - collection: QualifiedCollection::new(DatabaseId::DEFAULT, &collection), + collection: QualifiedCollection::new(db, &collection), restart_identity: false, resolved_sum_targets: Vec::new(), declared_primary_key: None, }); - let expected = collection_vshard_in_database(DatabaseId::new(7), &collection); - match plan_vshard_in_database(&plan, DatabaseId::new(7)) { + // The plan carries the qualified name; the home is the bare key's. + let expected = CollectionKey::from_bare(db, &collection).vshard(); + match plan_vshard_in_database(&plan, db) { PlanRouting::Vshards(actual) => assert_eq!(actual, vec![expected]), PlanRouting::ControlPlaneOnly | PlanRouting::NotAWrite | PlanRouting::Unroutable(_) => { panic!("document truncate must be database-scoped") diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs index 327790df4..0a91ddf3f 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -175,7 +175,9 @@ pub(super) fn make_sequenced_txn(epoch: u64, position: u32) -> SequencedTxn { /// The vShard that `"test_coll"` homes to in the default database. A /// scheduler built on this vShard owns the reads of [`make_validate_only_txn`]. pub(super) fn test_coll_vshard() -> u32 { - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "test_coll").as_u32() + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "test_coll") + .vshard() + .as_u32() } /// Build a static `SequencedTxn` at `(epoch, position)` that reaches the diff --git a/nodedb/src/control/cluster/snapshot_applier.rs b/nodedb/src/control/cluster/snapshot_applier.rs index a4851dc51..d62a62dea 100644 --- a/nodedb/src/control/cluster/snapshot_applier.rs +++ b/nodedb/src/control/cluster/snapshot_applier.rs @@ -29,7 +29,7 @@ use std::time::Duration; use nodedb_cluster::routing::vshard_for_collection; use nodedb_types::Surrogate; -use nodedb_types::id::DatabaseId; +use nodedb_types::id::{CollectionKey, DatabaseId}; use crate::bridge::envelope::PhysicalPlan; use crate::control::state::SharedState; @@ -108,7 +108,10 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { .map_err(|e| Box::new(e) as Box)?; for coll in collections.iter().filter(|c| { c.is_active - && group_vshards.contains(&vshard_for_collection(DatabaseId::DEFAULT, &c.name)) + && group_vshards.contains(&vshard_for_collection(CollectionKey::from_bare( + DatabaseId::DEFAULT, + &c.name, + ))) }) { collections_to_clear.push((coll.tenant_id, coll.name.clone())); } @@ -133,8 +136,7 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::ClusterSnapshot, TenantId::new(0), - DatabaseId::DEFAULT, - "__system", + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), plan, ), SNAPSHOT_APPLY_TIMEOUT, @@ -158,9 +160,8 @@ impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { for e in &snap.surrogate_pk { catalog .put_surrogate( - DatabaseId::DEFAULT, + CollectionKey::from_bare(DatabaseId::DEFAULT, &e.collection), TenantId::new(e.tenant_id), - &e.collection, &e.pk, Surrogate::new(e.surrogate), ) diff --git a/nodedb/src/control/cluster/snapshot_builder.rs b/nodedb/src/control/cluster/snapshot_builder.rs index 34af3000e..9614e60df 100644 --- a/nodedb/src/control/cluster/snapshot_builder.rs +++ b/nodedb/src/control/cluster/snapshot_builder.rs @@ -58,7 +58,10 @@ impl DataPlaneSnapshotBuilder { /// vshard-of-key logic is never duplicated. Matches the canonical routing /// function (`vshard_for_collection`) used by the RESTORE topology splitter. fn vshard_of(collection: &str) -> u32 { - nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, collection) + nodedb_cluster::routing::vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + collection, + )) } /// Capture PK→surrogate bindings for every active collection whose vshard @@ -81,9 +84,8 @@ impl DataPlaneSnapshotBuilder { .filter(|c| group_vshards.contains(&Self::vshard_of(&c.name))) { let bindings = catalog.scan_surrogates_for_collection( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &coll.name), TenantId::new(coll.tenant_id), - &coll.name, )?; for (pk, surrogate) in bindings { merged.surrogate_pk.push(SurrogateBindEntry { @@ -113,8 +115,7 @@ impl DataPlaneSnapshotBuilder { crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::ClusterSnapshot, TenantId::new(tenant_id), - DatabaseId::DEFAULT, - "__system", + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), plan, ), TENANT_SNAPSHOT_TIMEOUT, @@ -199,7 +200,7 @@ impl DataPlaneSnapshotBuilder { // Graph edges: the versioned edge key embeds the collection as its // FIRST `\x00`-delimited component, and edge writes are homed at - // `vshard_for_collection(DEFAULT, collection)` — the SAME routing + // `vshard_for_collection` over the default-database key — the SAME routing // function `Self::vshard_of` uses. So edges route through the identical // vshard filter every other section uses. The restore path parses the // key and rebuilds CSR, so no key transformation is needed here. diff --git a/nodedb/src/control/crdt_admission.rs b/nodedb/src/control/crdt_admission.rs index ce8d0f407..8dc7b0988 100644 --- a/nodedb/src/control/crdt_admission.rs +++ b/nodedb/src/control/crdt_admission.rs @@ -234,7 +234,9 @@ pub(crate) async fn dispatch_crdt_apply_admitted_outcome( }); } }; - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + // `collection` is the plan's database-qualified name. + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let workflow = CrdtAdmissionWorkflow { state, tenant_id, @@ -296,8 +298,7 @@ async fn preview( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::AdmittedContinuation, workflow.tenant_id, - workflow.database_id, - workflow.collection, + nodedb_types::CollectionKey::from_qualified_str(workflow.database_id, workflow.collection)?, PhysicalPlan::Crdt(CrdtOp::PreviewApply { collection: nodedb_types::QualifiedCollection::from_stored( workflow.collection.to_owned(), @@ -425,7 +426,9 @@ pub(crate) async fn dispatch_crdt_restore_admitted( event_source, policy, } = request; - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + // `collection` is the plan's database-qualified name. + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let workflow = CrdtAdmissionWorkflow { state, tenant_id, @@ -491,8 +494,7 @@ async fn generate_restore_delta( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::AdmittedContinuation, workflow.tenant_id, - workflow.database_id, - workflow.collection, + nodedb_types::CollectionKey::from_qualified_str(workflow.database_id, workflow.collection)?, PhysicalPlan::Crdt(CrdtOp::RestoreToVersion { collection: nodedb_types::QualifiedCollection::from_stored( workflow.collection.to_owned(), @@ -1065,7 +1067,7 @@ mod tests { let task = nodedb_physical::physical_task::PhysicalTask { tenant_id, database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, "docs"), + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "docs").vshard(), plan: apply_plan(), post_set_op: nodedb_physical::physical_task::PostSetOp::None, txn_id: None, @@ -1115,7 +1117,7 @@ mod tests { let task = nodedb_physical::physical_task::PhysicalTask { tenant_id, database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, "docs"), + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "docs").vshard(), plan: apply_plan(), post_set_op: nodedb_physical::physical_task::PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/exec_receiver/executor.rs b/nodedb/src/control/exec_receiver/executor.rs index 94d4830e0..e1dfe540a 100644 --- a/nodedb/src/control/exec_receiver/executor.rs +++ b/nodedb/src/control/exec_receiver/executor.rs @@ -279,14 +279,23 @@ impl LocalPlanExecutor { // // The vshard is not carried on the wire; re-derive it as a pure // function of the plan's primary collection, matching the gateway - // router's `CollectionHomed` arm (`vshard_for_collection`). - let vshard_id = crate::types::VShardId::new( - crate::control::gateway::version_set::touched_collections(&plan) - .into_iter() - .next() - .map(|name| nodedb_cluster::routing::vshard_for_collection(database_id, &name)) - .unwrap_or(0), - ); + // router's `CollectionHomed` arm (`vshard_for_collection`). The plan + // carries the database-qualified name, de-qualified into the + // canonical key before hashing. + let vshard_raw = match crate::control::gateway::version_set::touched_collections(&plan) + .into_iter() + .next() + { + Some(name) => match nodedb_types::CollectionKey::from_qualified_str(database_id, &name) + { + Ok(key) => nodedb_cluster::routing::vshard_for_collection(key), + Err(error) => { + return ExecuteResponse::err(execution_error_to_typed(error.into())); + } + }, + None => 0, + }; + let vshard_id = crate::types::VShardId::new(vshard_raw); if let Err(error) = reject_unadmitted_crdt_apply(&plan) { return ExecuteResponse::err(error); } diff --git a/nodedb/src/control/exec_receiver/plan_decode.rs b/nodedb/src/control/exec_receiver/plan_decode.rs index dd764e5f1..83f22d209 100644 --- a/nodedb/src/control/exec_receiver/plan_decode.rs +++ b/nodedb/src/control/exec_receiver/plan_decode.rs @@ -59,12 +59,9 @@ pub(super) fn decode_plan( ) = &mut plan && *surrogate == nodedb_types::Surrogate::ZERO && !pk_bytes.is_empty() - && let Ok(Some(resolved)) = catalog_ref.get_surrogate_for_pk( - database_id, - crate::types::TenantId::new(tenant_id), - collection.as_str(), - pk_bytes, - ) + && let Ok(key) = nodedb_types::CollectionKey::from_qualified(database_id, collection) + && let Ok(Some(resolved)) = + catalog_ref.get_surrogate_for_pk(key, crate::types::TenantId::new(tenant_id), pk_bytes) { *surrogate = resolved; } diff --git a/nodedb/src/control/gateway/colocation_guard.rs b/nodedb/src/control/gateway/colocation_guard.rs index 408865832..7a62c119d 100644 --- a/nodedb/src/control/gateway/colocation_guard.rs +++ b/nodedb/src/control/gateway/colocation_guard.rs @@ -31,20 +31,21 @@ //! and never assume co-residence. use nodedb_physical::physical_plan::PhysicalPlan; -use nodedb_types::DatabaseId; +use nodedb_types::{CollectionKey, DatabaseId}; use crate::control::router::vshard::VShardRouter; use crate::control::state::SharedState; -use crate::types::VShardId; /// Resolve the Data-Plane core that owns `collection`'s vShard on THIS node, /// using the same `VShardRouter` the dispatch path uses so the guard can never /// drift from the real vShard→core mapping. +/// +/// `collection` is the plan's database-qualified name, de-qualified into the +/// canonical key. A name that does not de-qualify resolves to no core, so the +/// guard refuses the write. fn owning_core(router: &VShardRouter, database_id: DatabaseId, collection: &str) -> Option { - router.resolve(VShardId::from_collection_in_database( - database_id, - collection, - )) + let key = CollectionKey::from_qualified_str(database_id, collection).ok()?; + router.resolve(key.vshard()) } /// True when `coll_a` and `coll_b` resolve to DIFFERENT cores on `router`. diff --git a/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs b/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs index 71b4e2ac4..f63dcf8b0 100644 --- a/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs +++ b/nodedb/src/control/gateway/error_map/system_dispatch_refusal.rs @@ -114,8 +114,7 @@ async fn dispatch_system_keeps_the_refusal_code() { SystemTask::new( SystemReason::DdlApply, TenantId::new(1), - DatabaseId::DEFAULT, - COLLECTION, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLLECTION), set_params_plan(), ), Duration::from_secs(5), diff --git a/nodedb/src/control/gateway/router.rs b/nodedb/src/control/gateway/router.rs index e946b6581..d1bf99e84 100644 --- a/nodedb/src/control/gateway/router.rs +++ b/nodedb/src/control/gateway/router.rs @@ -9,7 +9,8 @@ //! //! 1. Consult the `strategy_fn` closure (backed by the catalog) for the plan's //! primary collection to determine its [`PartitionStrategy`]: -//! - `CollectionHomed` → one vShard derived from [`vshard_for_collection`]. +//! - `CollectionHomed` → one vShard derived from [`vshard_for_collection`] +//! over the collection's canonical key. //! - `KeyPartitioned` → one vShard per distinct key via [`VShardId::from_key`] //! (deduplicated; multiple keys mapping to the same vShard share one route). //! 2. Look up the Raft group leader for each vShard in the routing table. @@ -27,7 +28,7 @@ use nodedb_cluster::routing::{RoutingTable, vshard_for_collection}; use nodedb_types::PartitionStrategy; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::{CollectionKey, DatabaseId, VShardId}; use nodedb_physical::physical_plan::PhysicalPlan; @@ -73,7 +74,7 @@ pub fn route_plan( // In single-node mode every plan runs locally. let Some(routing) = routing else { - let vshard_id = primary_vshard(&plan, database_id); + let vshard_id = primary_vshard(&plan, database_id)?; return Ok(vec![TaskRoute { plan, decision: RouteDecision::Local, @@ -156,10 +157,12 @@ fn route_single_collection( match strategy { PartitionStrategy::CollectionHomed => { // Byte-identical to the original primary_vshard / resolve_decision path. - let vshard_id = primary_name - .as_deref() - .map(|name| vshard_for_collection(database_id, name)) - .unwrap_or(0); + let vshard_id = match primary_name.as_deref() { + Some(name) => { + vshard_for_collection(CollectionKey::from_qualified_str(database_id, name)?) + } + None => 0, + }; let decision = resolve_decision(vshard_id, local_node_id, Some(routing), None); Ok(vec![TaskRoute { plan, @@ -295,15 +298,20 @@ pub fn is_task_vshard_scoped(plan: &PhysicalPlan) -> bool { ) } -/// Determine the primary vShard for a plan by hashing the first collection name. +/// Determine the primary vShard for a plan from its first collection. /// -/// Falls back to vShard 0 for plans that have no named collection (Meta ops). -fn primary_vshard(plan: &PhysicalPlan, database_id: DatabaseId) -> u32 { - touched_collections(plan) - .into_iter() - .next() - .map(|name| vshard_for_collection(database_id, &name)) - .unwrap_or(0) +/// The plan carries the database-qualified name. It is de-qualified into the +/// canonical key before hashing, so the route matches the vShard every other +/// path homes the collection to. Falls back to vShard 0 for plans that have +/// no named collection (Meta ops). +fn primary_vshard(plan: &PhysicalPlan, database_id: DatabaseId) -> Result { + match touched_collections(plan).into_iter().next() { + Some(name) => Ok(vshard_for_collection(CollectionKey::from_qualified_str( + database_id, + &name, + )?)), + None => Ok(0), + } } #[cfg(test)] @@ -462,7 +470,7 @@ mod tests { // would (the collection's owner). assert_eq!( routes[0].vshard_id, - vshard_for_collection(DatabaseId::DEFAULT, "events") + vshard_for_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, "events")) ); } @@ -470,7 +478,8 @@ mod tests { fn find_collection_for_vshard(target: u32) -> String { for i in 0u64.. { let name = format!("col_{i}"); - if vshard_for_collection(DatabaseId::DEFAULT, &name) == target { + if vshard_for_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, &name)) == target + { return name; } } diff --git a/nodedb/src/control/insert_select/copy_rows.rs b/nodedb/src/control/insert_select/copy_rows.rs index ddfcf1b71..8f61f34e7 100644 --- a/nodedb/src/control/insert_select/copy_rows.rs +++ b/nodedb/src/control/insert_select/copy_rows.rs @@ -119,9 +119,8 @@ pub(crate) fn assign_page_rows( }; let surrogate = assign_target_surrogate( state, - database_id, + nodedb_types::CollectionKey::from_qualified_str(database_id, target_collection)?, tenant_id, - target_collection, &spec.target_pk, &value, )?; diff --git a/nodedb/src/control/insert_select/expand_staged.rs b/nodedb/src/control/insert_select/expand_staged.rs index 583c42011..ff8ce7398 100644 --- a/nodedb/src/control/insert_select/expand_staged.rs +++ b/nodedb/src/control/insert_select/expand_staged.rs @@ -21,7 +21,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::insert_select::copy_rows::{assign_page_rows, resolve_copy_spec}; use crate::control::maintenance::clone_materializer::scan_source_page; use crate::control::state::SharedState; -use crate::types::{TxnId, VShardId}; +use crate::types::TxnId; use nodedb_physical::physical_plan::DocumentOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -70,8 +70,11 @@ pub(crate) async fn resolve_and_emit_insert_select_ops( // Recompute the target vShard (rather than reusing the staged task's) // to keep dispatch classification honest, as the MERGE expander does. - let vshard_id = - VShardId::from_collection_in_database(task.database_id, target_collection.as_str()); + let vshard_id = nodedb_types::CollectionKey::from_qualified_str( + task.database_id, + target_collection.as_str(), + )? + .vshard(); // Resolve materialized-sum targets: these ops stage directly, bypassing // statement-level resolution, so without this a bound target collection diff --git a/nodedb/src/control/maintenance/clone_materializer/columnar.rs b/nodedb/src/control/maintenance/clone_materializer/columnar.rs index f8d547564..da1303f48 100644 --- a/nodedb/src/control/maintenance/clone_materializer/columnar.rs +++ b/nodedb/src/control/maintenance/clone_materializer/columnar.rs @@ -126,9 +126,8 @@ pub(super) async fn materialize_columnar_collection( let target_surrogate = state .surrogate_assigner .assign( - db_id, + nodedb_types::CollectionKey::from_bare(db_id, &coll.name), tenant_id, - &target_qualified, &source_surrogate_u32.to_be_bytes(), ) .map_err(|e| crate::Error::Storage { diff --git a/nodedb/src/control/maintenance/clone_materializer/dispatch.rs b/nodedb/src/control/maintenance/clone_materializer/dispatch.rs index 85163ae11..1d46d7211 100644 --- a/nodedb/src/control/maintenance/clone_materializer/dispatch.rs +++ b/nodedb/src/control/maintenance/clone_materializer/dispatch.rs @@ -34,7 +34,9 @@ pub(crate) async fn dispatch_local( plan: PhysicalPlan, txn_id: Option, ) -> crate::Result { - let vshard_id = VShardId::from_collection_in_database(database_id, collection_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection_qualified)? + .vshard(); dispatch_local_on_vshard(state, tenant_id, database_id, vshard_id, plan, txn_id).await } diff --git a/nodedb/src/control/maintenance/clone_materializer/document.rs b/nodedb/src/control/maintenance/clone_materializer/document.rs index 18b94db46..5ff243b4c 100644 --- a/nodedb/src/control/maintenance/clone_materializer/document.rs +++ b/nodedb/src/control/maintenance/clone_materializer/document.rs @@ -106,9 +106,11 @@ pub(super) async fn materialize_document_collection( // would allocate for this (collection, pk) pair. let pk_bytes = catalog .get_pk_for_surrogate( - origin.source_database, + nodedb_types::CollectionKey::from_bare( + origin.source_database, + &origin.source_collection, + ), tenant_id, - &origin.source_collection, source_surrogate, ) .map_err(|e| crate::Error::Storage { @@ -130,7 +132,11 @@ pub(super) async fn materialize_document_collection( // the normal INSERT path would use. let target_surrogate = state .surrogate_assigner - .assign(db_id, tenant_id, &target_qualified, &pk_bytes) + .assign( + nodedb_types::CollectionKey::from_bare(db_id, &coll.name), + tenant_id, + &pk_bytes, + ) .map_err(|e| crate::Error::Storage { engine: "clone_materializer".into(), detail: format!( diff --git a/nodedb/src/control/maintenance/clone_materializer/kv.rs b/nodedb/src/control/maintenance/clone_materializer/kv.rs index 63dd27502..7c73cdb18 100644 --- a/nodedb/src/control/maintenance/clone_materializer/kv.rs +++ b/nodedb/src/control/maintenance/clone_materializer/kv.rs @@ -93,10 +93,11 @@ pub(super) async fn materialize_kv_collection( continue; } - let surrogate = - state - .surrogate_assigner - .assign(db_id, tenant_id, &target_qualified, &key)?; + let surrogate = state.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(db_id, &coll.name), + tenant_id, + &key, + )?; let plan = PhysicalPlan::Kv(KvOp::Put { collection: nodedb_types::QualifiedCollection::new(db_id, &coll.name), key: key.clone(), diff --git a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs index bee1e3ab0..9e424123f 100644 --- a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs +++ b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs @@ -91,8 +91,11 @@ pub(crate) async fn resolve_and_emit_merge_ops( })?; let target_pk = resolve_target_pk(&target, "MERGE")?; - let vshard_id = - VShardId::from_collection_in_database(task.database_id, target_collection.as_str()); + let vshard_id = nodedb_types::CollectionKey::from_qualified_str( + task.database_id, + target_collection.as_str(), + )? + .vshard(); let mut out: Vec = Vec::new(); emit_arms( state, @@ -205,9 +208,8 @@ fn emit_arms( for (_join_key, body) in arms.inserts { let surrogate = assign_target_surrogate( state, - task.database_id, + nodedb_types::CollectionKey::from_qualified_str(task.database_id, target_collection)?, task.tenant_id, - target_collection, target_pk, &body, )?; diff --git a/nodedb/src/control/merge_orchestrator/orchestrator.rs b/nodedb/src/control/merge_orchestrator/orchestrator.rs index f8e024c6b..1d8cb442f 100644 --- a/nodedb/src/control/merge_orchestrator/orchestrator.rs +++ b/nodedb/src/control/merge_orchestrator/orchestrator.rs @@ -230,9 +230,11 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate for (join_key, body) in &insert_rows { let surrogate = assign_target_surrogate( state, - args.database_id, + nodedb_types::CollectionKey::from_qualified_str( + args.database_id, + args.target_collection, + )?, args.tenant_id, - args.target_collection, &target_pk, body, )?; diff --git a/nodedb/src/control/orchestrated_write.rs b/nodedb/src/control/orchestrated_write.rs index 30d038ea9..c618c76e7 100644 --- a/nodedb/src/control/orchestrated_write.rs +++ b/nodedb/src/control/orchestrated_write.rs @@ -20,7 +20,7 @@ use crate::control::state::SharedState; use crate::control::wal_replication::{ ReplicableWrite, propose_replicated_entry, to_replicated_entry, }; -use crate::types::{RequestId, VShardId}; +use crate::types::RequestId; /// Apply `plan`, a resolved write on `collection`, and return the Data-Plane /// response the statement renders. @@ -56,7 +56,9 @@ pub(crate) async fn apply_orchestrated_write( return Ok(resp); }; - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + // `collection` is the plan's database-qualified name. + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let replicable = ReplicableWrite::decide_for_replication(&plan)?; let entry = to_replicated_entry(tenant_id, database_id, vshard_id, &replicable)?.ok_or_else(|| { diff --git a/nodedb/src/control/planner/auto_tier.rs b/nodedb/src/control/planner/auto_tier.rs index eb23c9429..76c85bee2 100644 --- a/nodedb/src/control/planner/auto_tier.rs +++ b/nodedb/src/control/planner/auto_tier.rs @@ -14,7 +14,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::engine::timeseries::retention_policy::RetentionPolicyDef; -use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::TimeseriesOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -229,7 +229,7 @@ fn build_scan_task( } = scope; PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), database_id, plan: PhysicalPlan::Timeseries(TimeseriesOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, collection), diff --git a/nodedb/src/control/planner/calvin/dispatch.rs b/nodedb/src/control/planner/calvin/dispatch.rs index e29bc9145..f365ce63e 100644 --- a/nodedb/src/control/planner/calvin/dispatch.rs +++ b/nodedb/src/control/planner/calvin/dispatch.rs @@ -49,6 +49,7 @@ use crate::control::server::shared::session::read_set::ReadSetEntry; use crate::types::VShardId; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; use nodedb_physical::physical_task::PhysicalTask; +use nodedb_types::CollectionKey; pub use crate::control::planner::calvin::predicate::predicate_class; pub use crate::control::planner::calvin::write_class::is_write_plan; @@ -77,14 +78,22 @@ pub fn is_dependent_predicate(plan: &PhysicalPlan) -> bool { /// /// Each [`ReadSetEntry`] homes to its collection's vShard using the SAME /// collection→vShard map `ReadWriteSet::participating_vshards` uses to derive the -/// `TxClass` read_set's participants. Each read retains its session database so -/// classification and the database-scoped transaction class agree. A read with -/// no extractable collection contributes nothing. -pub fn read_vshards_of(reads: &[ReadSetEntry]) -> BTreeSet { +/// `TxClass` read_set's participants. An entry carries the plan's +/// database-qualified name, so it is de-qualified into a [`CollectionKey`] +/// before hashing. Each read retains its session database so classification +/// and the database-scoped transaction class agree. A read with no extractable +/// collection contributes nothing. +pub fn read_vshards_of(reads: &[ReadSetEntry]) -> crate::Result> { reads .iter() .filter(|e| !e.collection.is_empty()) - .map(|e| VShardId::from_collection_in_database(e.database_id, &e.collection).as_u32()) + .map(|e| { + Ok( + CollectionKey::from_qualified_str(e.database_id, &e.collection)? + .vshard() + .as_u32(), + ) + }) .collect() } @@ -160,7 +169,7 @@ pub(crate) async fn dispatch_calvin_or_fast( // Interactive COMMIT threads its session read-set here; autocommit passes an // empty slice. The read vShards widen both the classification (below) and the // TxClass read_set participants (in `build_static_tx_class`) in lockstep. - let read_vshards = read_vshards_of(reads); + let read_vshards = read_vshards_of(reads)?; let class = classify_dispatch(tasks, &read_vshards); match &class { @@ -454,7 +463,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("dispatch_home_{i}"); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); match first { Some((ref fname, fv)) if fv != vshard => return (fname.clone(), name), None => first = Some((name, vshard)), @@ -484,10 +495,12 @@ mod tests { // cross-node transaction. This test guarantees a future refactor of // `read_vshards_of` / `classify_dispatch` cannot reopen that hole. let (write_coll, read_coll) = two_distinct_vshard_collections(); - let write_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &write_coll).as_u32(); - let read_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &read_coll).as_u32(); + let write_vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &write_coll) + .vshard() + .as_u32(); + let read_vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &read_coll) + .vshard() + .as_u32(); let tasks = vec![doc_insert_task(write_vshard)]; @@ -504,7 +517,8 @@ mod tests { // The homing step under test: a foreign-collection read must home to a // vShard distinct from the write's, contributing a new participant. - let read_vshards = read_vshards_of(std::slice::from_ref(&read_entry)); + let read_vshards = + read_vshards_of(std::slice::from_ref(&read_entry)).expect("read vshards"); assert!( read_vshards.contains(&read_vshard) && !read_vshards.contains(&write_vshard), "read entry for `{read_coll}` must home to vShard {read_vshard}, not the write's {write_vshard}" diff --git a/nodedb/src/control/planner/calvin/dispatch_multi.rs b/nodedb/src/control/planner/calvin/dispatch_multi.rs index 624443634..c9ba806a3 100644 --- a/nodedb/src/control/planner/calvin/dispatch_multi.rs +++ b/nodedb/src/control/planner/calvin/dispatch_multi.rs @@ -162,7 +162,7 @@ pub(crate) async fn dispatch_tasks_to_calvin( reads: &[ReadSetEntry], lock_owner: Option, ) -> crate::Result> { - let read_vshards = read_vshards_of(reads); + let read_vshards = read_vshards_of(reads)?; match classify_dispatch(tasks, &read_vshards) { DispatchClass::MultiShard { .. } => { admit_legacy_multi_shard_dispatch(cross_shard_mode, position)?; @@ -186,12 +186,12 @@ mod tests { use nodedb_cluster::calvin::types::TxnIdWire; use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan}; use nodedb_physical::physical_task::PostSetOp; - use nodedb_types::Surrogate; + use nodedb_types::{CollectionKey, Surrogate}; fn task(collection: &str, surrogate: u32) -> PhysicalTask { PhysicalTask { tenant_id: TenantId::new(1), - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection), + vshard_id: CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(), database_id: DatabaseId::DEFAULT, plan: PhysicalPlan::Document(DocumentOp::PointInsert { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, collection), @@ -225,11 +225,11 @@ mod tests { } fn distinct_collection(from: &str) -> String { - let home = VShardId::from_collection_in_database(DatabaseId::DEFAULT, from); + let home = CollectionKey::from_bare(DatabaseId::DEFAULT, from).vshard(); (0..1024) .map(|index| format!("atomic_{index}")) .find(|candidate| { - VShardId::from_collection_in_database(DatabaseId::DEFAULT, candidate) != home + CollectionKey::from_bare(DatabaseId::DEFAULT, candidate).vshard() != home }) .expect("test routing domain must contain more than one vShard") } diff --git a/nodedb/src/control/planner/calvin/preexec.rs b/nodedb/src/control/planner/calvin/preexec.rs index 2922da5ef..ba91a9c26 100644 --- a/nodedb/src/control/planner/calvin/preexec.rs +++ b/nodedb/src/control/planner/calvin/preexec.rs @@ -21,7 +21,7 @@ use nodedb_types::TenantId; use crate::control::server::dispatch_utils::dispatch_to_data_plane; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; /// One implicit graph edge surfaced from the pre-execution reconnaissance scan. @@ -84,7 +84,7 @@ pub async fn run_preexec_scan( collection: &str, filter_bytes: Vec, ) -> crate::Result { - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + let vshard_id = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); let scan_plan = PhysicalPlan::Document(DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, collection), diff --git a/nodedb/src/control/planner/calvin/submit/local.rs b/nodedb/src/control/planner/calvin/submit/local.rs index b209bdc88..9094d87b6 100644 --- a/nodedb/src/control/planner/calvin/submit/local.rs +++ b/nodedb/src/control/planner/calvin/submit/local.rs @@ -82,6 +82,9 @@ pub async fn submit_and_await_calvin_with_timeout( && tx_class .write_set .participating_vshards_in_database(tx_class.database_id) + .map_err(|e| Error::BadRequest { + detail: format!("Calvin transaction write set: {e}"), + })? .iter() .any(|vshard| { state diff --git a/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs b/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs index f906f561b..e001c3225 100644 --- a/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs +++ b/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs @@ -287,8 +287,9 @@ mod tests { // vshard. This is exactly the shape the contended single-shard // predicate-write routing path builds. let tasks = vec![bulk_delete_task("users")]; - let want_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "users").as_u32(); + let want_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users") + .vshard() + .as_u32(); // Strict builder rejects the single-vshard write set. let strict = build_dependent_tx_class(&tasks, TenantId::new(1), "users", &[7, 8], &[]); diff --git a/nodedb/src/control/planner/calvin/tx_class/shared.rs b/nodedb/src/control/planner/calvin/tx_class/shared.rs index 52072e479..fa6373c3f 100644 --- a/nodedb/src/control/planner/calvin/tx_class/shared.rs +++ b/nodedb/src/control/planner/calvin/tx_class/shared.rs @@ -495,7 +495,7 @@ mod routing_agreement_tests { vec![ task( source_write(), - VShardId::from_collection_in_database(DB, SOURCE), + nodedb_types::CollectionKey::from_bare(DB, SOURCE).vshard(), ), task(balance_write(), crate::query::sum_target_vshard(DB, TARGET)), ] @@ -531,8 +531,8 @@ mod routing_agreement_tests { #[test] fn the_fixture_spans_two_vshards() { assert_ne!( - VShardId::from_collection_in_database(DB, SOURCE), - VShardId::from_collection_in_database(DB, TARGET), + nodedb_types::CollectionKey::from_bare(DB, SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DB, TARGET).vshard(), "the balance-pairing case only exists when source and target hash apart" ); } @@ -560,8 +560,8 @@ mod routing_agreement_tests { let tx = build_static_tx_class(&tasks, TENANT, &[]).expect("build the transaction class"); let mut expected = vec![ - VShardId::from_collection_in_database(DB, SOURCE), - VShardId::from_collection_in_database(DB, TARGET), + nodedb_types::CollectionKey::from_bare(DB, SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DB, TARGET).vshard(), ]; expected.sort_by_key(|v| v.as_u32()); assert_eq!( diff --git a/nodedb/src/control/planner/calvin/tx_class/static_builder.rs b/nodedb/src/control/planner/calvin/tx_class/static_builder.rs index dd7ad68b6..bcf7c7ab8 100644 --- a/nodedb/src/control/planner/calvin/tx_class/static_builder.rs +++ b/nodedb/src/control/planner/calvin/tx_class/static_builder.rs @@ -294,7 +294,9 @@ mod tests { let mut first: Option<(String, u32)> = None; for i in 0u32..1024 { let name = format!("coll_{i}"); - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); match &first { Some((fname, fv)) if *fv != v => return (fname.clone(), name), Some(_) => {} @@ -351,7 +353,8 @@ mod tests { #[test] fn single_vshard_builder_preserves_database_scope() { - let mut task = point_insert_task("db_scoped", 1); + // A plan in a non-default database names its collection qualified. + let mut task = point_insert_task("7/db_scoped", 1); task.database_id = DatabaseId::new(7); let tx = build_single_vshard_tx_class(&[task], TenantId::new(1), &[]) .expect("valid single-vshard TxClass"); @@ -444,14 +447,16 @@ mod tests { .map(|v| v.as_u32()) .collect(); for coll in [col_a.as_str(), col_b.as_str(), "read_col", "scan_col"] { - let v = VShardId::from_collection_in_database(DatabaseId::DEFAULT, coll).as_u32(); + let v = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, coll) + .vshard() + .as_u32(); assert!( participants.contains(&v), "participant set must include the vShard of {coll}" ); } // Every write shard is still present (the read union never drops one). - for v in tx.write_set.participating_vshards() { + for v in tx.write_set.participating_vshards().expect("participants") { assert!( participants.contains(&v.as_u32()), "read union must not drop a write shard" @@ -470,7 +475,10 @@ mod tests { // Participants collapse to the write-derived set when there are no reads. assert_eq!( tx.participating_vshards(), - tx.write_set.participating_vshards().as_slice() + tx.write_set + .participating_vshards() + .expect("participants") + .as_slice() ); } @@ -557,8 +565,9 @@ mod tests { // One point-write task → one collection → one vshard. This is exactly the // shape the contended point-write routing path builds. let tasks = vec![point_insert_task("users", 7)]; - let want_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "users").as_u32(); + let want_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users") + .vshard() + .as_u32(); // Strict builder rejects the single-vshard write set. let strict = build_static_tx_class(&tasks, TenantId::new(1), &[]); diff --git a/nodedb/src/control/planner/implicit_edges/insert.rs b/nodedb/src/control/planner/implicit_edges/insert.rs index 55d1c0083..c7d3e614f 100644 --- a/nodedb/src/control/planner/implicit_edges/insert.rs +++ b/nodedb/src/control/planner/implicit_edges/insert.rs @@ -82,26 +82,14 @@ pub async fn append_implicit_edge_tasks( let vsrc = VShardId::from_key(edge.src.as_bytes()); let vdst = VShardId::from_key(edge.dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - &edge.collection, - edge.src.as_bytes(), - trace_id, - ) - .await?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - &edge.collection, - edge.dst.as_bytes(), - trace_id, - ) - .await?; + // `edge.collection` is the plan's database-qualified name. + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, &edge.collection)?; + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, edge.src.as_bytes(), trace_id) + .await?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, edge.dst.as_bytes(), trace_id) + .await?; let properties = match edge.weight { Some(w) => weight_properties(w), diff --git a/nodedb/src/control/planner/implicit_edges/routed.rs b/nodedb/src/control/planner/implicit_edges/routed.rs index 2cf02c6a3..b77fb13e4 100644 --- a/nodedb/src/control/planner/implicit_edges/routed.rs +++ b/nodedb/src/control/planner/implicit_edges/routed.rs @@ -54,26 +54,12 @@ pub(super) async fn push_edge_delete( let vsrc = VShardId::from_key(src.as_bytes()); let vdst = VShardId::from_key(dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - collection, - src.as_bytes(), - trace_id, - ) - .await?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - collection, - dst.as_bytes(), - trace_id, - ) - .await?; + // `collection` is the plan's database-qualified name. + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?; + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), trace_id).await?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), trace_id).await?; out.push(PhysicalTask { tenant_id, @@ -131,26 +117,12 @@ pub(super) async fn push_edge_put( let vsrc = VShardId::from_key(src.as_bytes()); let vdst = VShardId::from_key(dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - collection, - src.as_bytes(), - trace_id, - ) - .await?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - collection, - dst.as_bytes(), - trace_id, - ) - .await?; + // `collection` is the plan's database-qualified name. + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?; + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), trace_id).await?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), trace_id).await?; out.push(PhysicalTask { tenant_id, diff --git a/nodedb/src/control/planner/materialized_sum/cross_shard.rs b/nodedb/src/control/planner/materialized_sum/cross_shard.rs index 7e80fe91e..6b787cd4e 100644 --- a/nodedb/src/control/planner/materialized_sum/cross_shard.rs +++ b/nodedb/src/control/planner/materialized_sum/cross_shard.rs @@ -77,8 +77,9 @@ pub fn append_cross_shard_balance_tasks( }; let mut for_task = Vec::new(); + let source = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?; for binding in bindings.iter() { - if sum_target_is_co_resident(database_id, collection, &binding.target_collection) { + if sum_target_is_co_resident(source, &binding.target_collection) { continue; } for (join_value, delta) in crate::query::binding_insert_deltas(binding, &docs)? { diff --git a/nodedb/src/control/planner/materialized_sum/recon.rs b/nodedb/src/control/planner/materialized_sum/recon.rs index 8ef97adbc..f1d549da9 100644 --- a/nodedb/src/control/planner/materialized_sum/recon.rs +++ b/nodedb/src/control/planner/materialized_sum/recon.rs @@ -31,7 +31,7 @@ use nodedb_types::{Surrogate, TenantId}; use crate::control::server::dispatch_utils::dispatch_to_data_plane; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TraceId, VShardId}; +use crate::types::{DatabaseId, Lsn, TraceId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; /// What a plan-time reconnaissance read observed, and the version it observed @@ -177,7 +177,8 @@ async fn execute_read( }); } - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let response = dispatch_to_data_plane( state, tenant_id, diff --git a/nodedb/src/control/planner/materialized_sum/resolve.rs b/nodedb/src/control/planner/materialized_sum/resolve.rs index c36dc0e61..24019d3bf 100644 --- a/nodedb/src/control/planner/materialized_sum/resolve.rs +++ b/nodedb/src/control/planner/materialized_sum/resolve.rs @@ -353,14 +353,12 @@ pub(super) async fn lookup_join_value( database_id: DatabaseId, trace_id: TraceId, ) -> crate::Result { - let target = db_qualified(database_id, &binding.target_collection); let vshard = VShardId::from_key(join_value.as_bytes()); lookup_surrogate_routed( state, vshard, - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &binding.target_collection), tenant_id, - &target, join_value.as_bytes(), trace_id, ) @@ -408,13 +406,6 @@ async fn resolve_bodies( Ok(resolved) } -/// Qualify a catalog collection name for the plan / surrogate namespace. -fn db_qualified(database_id: DatabaseId, collection: &str) -> String { - nodedb_types::QualifiedCollection::new(database_id, collection) - .as_str() - .to_owned() -} - /// Strip the `"/"` prefix a planned collection name carries, yielding the /// catalog name the binding index is keyed on. fn strip_db_prefix(database_id: DatabaseId, qualified: &str) -> &str { @@ -544,7 +535,11 @@ mod tests { declare_binding(&state); let target_surrogate = state .surrogate_assigner - .assign(DB, TENANT, "accounts", b"acc-1") + .assign( + nodedb_types::CollectionKey::from_bare(DB, "accounts"), + TENANT, + b"acc-1", + ) .expect("bind target row"); let mut tasks = vec![insert_task("entries", body("acc-1"))]; @@ -577,11 +572,19 @@ mod tests { declare_second_binding(&state); let accounts_row = state .surrogate_assigner - .assign(DB, TENANT, "accounts", b"acc-1") + .assign( + nodedb_types::CollectionKey::from_bare(DB, "accounts"), + TENANT, + b"acc-1", + ) .expect("bind accounts row"); let audit_row = state .surrogate_assigner - .assign(DB, TENANT, "audit_totals", b"acc-1") + .assign( + nodedb_types::CollectionKey::from_bare(DB, "audit_totals"), + TENANT, + b"acc-1", + ) .expect("bind audit_totals row"); assert_ne!( accounts_row, audit_row, @@ -680,7 +683,11 @@ mod tests { declare_binding(&state); let target_surrogate = state .surrogate_assigner - .assign(DB, TENANT, "accounts", b"acc-1") + .assign( + nodedb_types::CollectionKey::from_bare(DB, "accounts"), + TENANT, + b"acc-1", + ) .expect("bind target row"); let mut tasks = vec![PhysicalTask { diff --git a/nodedb/src/control/planner/materialized_sum/settle.rs b/nodedb/src/control/planner/materialized_sum/settle.rs index 28d74dcf5..ab09823b0 100644 --- a/nodedb/src/control/planner/materialized_sum/settle.rs +++ b/nodedb/src/control/planner/materialized_sum/settle.rs @@ -65,8 +65,8 @@ use nodedb_physical::physical_plan::{ resolved_sum_surrogate, }; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -use nodedb_types::Surrogate; use nodedb_types::id::TxnId; +use nodedb_types::{CollectionKey, Surrogate}; use crate::control::server::shared::session::read_set::{ EngineTag, ReadKey, ReadOrigin, ReadSetEntry, @@ -143,12 +143,9 @@ pub(super) fn settle_cross_shard_images( database_id: DatabaseId, ) -> crate::Result { let mut settlement = Settlement::empty(); + let source = CollectionKey::from_qualified_str(database_id, input.source_collection)?; for binding in bindings { - if sum_target_is_co_resident( - database_id, - input.source_collection, - &binding.target_collection, - ) { + if sum_target_is_co_resident(source, &binding.target_collection) { // One core owns both rows: the balance rides the source write's own // transaction and is atomic for free. continue; @@ -314,12 +311,9 @@ pub(super) fn co_resident_target_keys( database_id: DatabaseId, ) -> crate::Result> { let mut keep: Vec = Vec::new(); + let source = CollectionKey::from_qualified_str(database_id, input.source_collection)?; for binding in bindings { - if !sum_target_is_co_resident( - database_id, - input.source_collection, - &binding.target_collection, - ) { + if !sum_target_is_co_resident(source, &binding.target_collection) { continue; } for (old, new) in input.images { @@ -338,8 +332,6 @@ pub(super) fn co_resident_target_keys( mod tests { use super::*; - use crate::types::VShardId; - const TENANT: TenantId = TenantId::new(7); const DB: DatabaseId = DatabaseId::DEFAULT; @@ -347,10 +339,10 @@ mod tests { /// below exercise the path this module exists for. fn cross_shard_pair() -> (String, String) { let source = "settle_entries".to_string(); - let home = VShardId::from_collection_in_database(DB, &source); + let home = CollectionKey::from_bare(DB, &source).vshard(); let target = (0..2048) .map(|i| format!("settle_accounts_{i}")) - .find(|candidate| VShardId::from_collection_in_database(DB, candidate) != home) + .find(|candidate| CollectionKey::from_bare(DB, candidate).vshard() != home) .unwrap_or_else(|| panic!("the routing domain must hold more than one vShard")); (source, target) } diff --git a/nodedb/src/control/planner/period_lock/lookup.rs b/nodedb/src/control/planner/period_lock/lookup.rs index 958e716a1..297243e3a 100644 --- a/nodedb/src/control/planner/period_lock/lookup.rs +++ b/nodedb/src/control/planner/period_lock/lookup.rs @@ -35,9 +35,8 @@ pub(super) async fn lookup_period_surrogate( lookup_surrogate_routed( state, vshard, - database_id, + nodedb_types::CollectionKey::from_bare(database_id, ref_table), tenant_id, - ref_table, period_key.as_bytes(), trace_id, ) diff --git a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs index 875a3abc4..9712c88e3 100644 --- a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs +++ b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs @@ -531,7 +531,9 @@ mod cross_shard_origination_tests { fn remote_homed_name(prefix: &str) -> String { for i in 0..4096u32 { let name = format!("{prefix}_{i}"); - let vshard = nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, &name); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name), + ); if vshard % 2 == 1 { return name; } @@ -544,7 +546,9 @@ mod cross_shard_origination_tests { fn local_homed_name(prefix: &str) -> String { for i in 0..4096u32 { let name = format!("{prefix}_{i}"); - let vshard = nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, &name); + let vshard = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name), + ); if vshard.is_multiple_of(2) { return name; } @@ -610,7 +614,10 @@ mod cross_shard_origination_tests { assert_eq!(req.cascade_depth, 0); assert_eq!( req.target_vshard, - nodedb_cluster::routing::vshard_for_collection(DatabaseId::DEFAULT, &tgt) + nodedb_cluster::routing::vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + &tgt, + )) ); } diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs index ae9ba6e9b..8206e806a 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/plan.rs @@ -13,7 +13,7 @@ use nodedb_sql::types::{EngineType, Filter, SortKey, SqlExpr, SqlPlan}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -93,7 +93,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( join_type.as_str().to_string() }; - let vshard = VShardId::from_collection_in_database(ctx.database_id, &left_collection); + let vshard = + nodedb_types::CollectionKey::from_qualified_str(ctx.database_id, &left_collection)? + .vshard(); return Ok(vec![PhysicalTask { tenant_id, @@ -212,7 +214,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_aggregate( let collection = db_qualified(ctx.database_id, &raw_collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, &raw_collection); - let vshard = VShardId::from_collection_in_database(ctx.database_id, &collection); + let vshard = ctx.collection_key(&raw_collection).vshard(); let group_strs = group_by_to_strings(group_by); let agg_specs: Vec = aggregates.iter().map(agg_expr_to_spec).collect(); diff --git a/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs b/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs index 65f4497f4..fe69c2481 100644 --- a/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs +++ b/nodedb/src/control/planner/sql_plan_convert/aggregate/spec.rs @@ -6,7 +6,7 @@ use nodedb_sql::types::{AggregateExpr, SqlExpr, SqlPlan}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -242,7 +242,7 @@ pub(in crate::control::planner::sql_plan_convert) fn build_input_sourced_aggrega tenant_id, // Coordinator-local: empty collection keeps the task on the // coordinator vshard (the child's rows are not per-shard). - vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), + vshard_id: nodedb_types::CollectionKey::from_bare(ctx.database_id, "").vshard(), database_id: ctx.database_id, plan: PhysicalPlan::Query(QueryOp::Aggregate { collection: nodedb_types::QualifiedCollection::from_stored(raw_collection), diff --git a/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs b/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs index bd383f692..c742523c1 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs @@ -8,7 +8,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::array_catalog::ArrayCatalogEntry; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::MetaOp; use nodedb_types::config::retention::BitemporalRetention; @@ -80,7 +80,7 @@ pub(super) fn convert_alter_array( // durably installed by the authorized dispatch boundary. let _updated = updated; - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs b/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs index a57d8075a..b60359d56 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs @@ -19,7 +19,7 @@ use nodedb_sql::types_array::{ use crate::bridge::envelope::PhysicalPlan; use crate::control::array_catalog::ArrayCatalogEntry; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ArrayOp; use super::super::convert::ConvertContext; @@ -131,7 +131,7 @@ pub(in super::super) fn convert_create_array( // 4. Emit OpenArray so the authorized execution boundary can durably // register it immediately before opening the engine side. - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, @@ -177,7 +177,7 @@ pub(in super::super) fn convert_drop_array( detail: format!("DROP ARRAY {name}: not found"), }); }; - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs b/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs index 8730d2316..f29870c52 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs @@ -9,7 +9,7 @@ use nodedb_sql::types_array::{ArrayCoordLiteral, ArrayInsertRow}; use crate::bridge::envelope::PhysicalPlan; use crate::engine::array::wal::{ArrayDeleteCell, ArrayPutCell}; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::{ArrayOp, ClusterArrayOp}; use super::super::convert::ConvertContext; @@ -44,7 +44,7 @@ pub(in super::super) fn convert_insert_array( })?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); let system_now_ms = chrono::Utc::now().timestamp_millis(); if ctx.cluster_enabled { @@ -57,7 +57,7 @@ pub(in super::super) fn convert_insert_array( format: "msgpack".into(), detail: format!("array coord pk encode: {e}"), })?; - let surrogate = ctx.surrogate_for_pk(name, &pk_bytes)?; + let surrogate = ctx.surrogate_for_pk(ctx.collection_key(name), &pk_bytes)?; let hilbert = encode_hilbert_prefix(&schema, &coord).map_err(|e| crate::Error::PlanError { detail: format!("INSERT INTO ARRAY {name}: Hilbert prefix: {e}"), @@ -111,7 +111,7 @@ pub(in super::super) fn convert_insert_array( format: "msgpack".into(), detail: format!("array coord pk encode: {e}"), })?; - let surrogate = ctx.surrogate_for_pk(name, &pk_bytes)?; + let surrogate = ctx.surrogate_for_pk(ctx.collection_key(name), &pk_bytes)?; cells.push(ArrayPutCell { coord, attrs, @@ -175,7 +175,7 @@ pub(in super::super) fn convert_delete_array( })?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); let system_now_ms = chrono::Utc::now().timestamp_millis(); if ctx.cluster_enabled { diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs index d52fc535b..7b6da2551 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs @@ -8,7 +8,7 @@ use nodedb_sql::temporal::TemporalScope; use nodedb_sql::types_array::ArrayReducerAst; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::{ArrayOp, ClusterArrayOp}; use super::super::convert::ConvertContext; @@ -54,7 +54,7 @@ pub(crate) fn convert_agg( super::helpers::resolve_array_temporal(temporal, "ARRAY_AGG")?; let mapped = map_reducer(reducer); let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); let plan = if ctx.cluster_enabled { // Encode the reducer for the wire. The coordinator decodes it diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/elementwise.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/elementwise.rs index 713f5bacc..17fc2fd4f 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/elementwise.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/elementwise.rs @@ -6,7 +6,7 @@ use nodedb_array::types::ArrayId; use nodedb_sql::types_array::ArrayBinaryOpAst; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ArrayOp; use super::super::convert::ConvertContext; @@ -44,7 +44,7 @@ pub(crate) fn convert_elementwise( } let left = ArrayId::in_database(tenant_id, ctx.database_id, left_name); let right = ArrayId::in_database(tenant_id, ctx.database_id, right_name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, left_name); + let vshard = ctx.collection_key(left_name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/maint.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/maint.rs index a17a422fe..c8e1d68f0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/maint.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/maint.rs @@ -5,7 +5,7 @@ use nodedb_array::types::ArrayId; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ArrayOp; use super::super::convert::ConvertContext; @@ -21,7 +21,7 @@ pub(crate) fn convert_flush( detail: "ARRAY_FLUSH: no WAL wired into convert context".into(), })?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); // A frontier read, not an allocation: `Flush` appends no WAL record, and // this LSN becomes the flushed segment's watermark — every cell it contains // was written by a record below the current frontier. The write funnel mints @@ -47,7 +47,7 @@ pub(crate) fn convert_compact( ) -> crate::Result> { let entry = super::helpers::load_entry(name, tenant_id, ctx)?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/project.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/project.rs index 23de4709a..79b2ed2ec 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/project.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/project.rs @@ -5,7 +5,7 @@ use nodedb_array::types::ArrayId; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ArrayOp; use super::super::convert::ConvertContext; @@ -26,7 +26,7 @@ pub(crate) fn convert_project( }); } let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs index ed997b9f4..0f9bd496a 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs @@ -11,7 +11,7 @@ use nodedb_sql::temporal::TemporalScope; use nodedb_sql::types_array::ArraySliceAst; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::{ArrayOp, ClusterArrayOp}; use super::super::convert::ConvertContext; @@ -59,7 +59,7 @@ pub(crate) fn convert_slice( super::helpers::resolve_array_temporal_scope(temporal, "ARRAY_SLICE")?; let attr_indices = resolve_attr_indices(name, attr_projection, &schema)?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); - let vshard = VShardId::from_collection_in_database(ctx.database_id, name); + let vshard = ctx.collection_key(name).vshard(); let plan = if ctx.cluster_enabled { // In cluster mode emit a ClusterArray variant. The routing loop diff --git a/nodedb/src/control/planner/sql_plan_convert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/convert.rs index 03a901dfc..40930d641 100644 --- a/nodedb/src/control/planner/sql_plan_convert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/convert.rs @@ -80,9 +80,9 @@ pub struct ConvertContext { /// Per-tenant maximum vector dimension (0 = unlimited). Checked in /// `VectorPrimaryInsert` conversion before the task is built. pub max_vector_dim: u32, - /// Database scope for vShard computation. All `VShardId::from_collection_in_database` - /// calls must use this value so that collections in different databases are - /// routed to distinct shards and data-plane isolates them correctly. + /// Database scope for vShard computation. Every `CollectionKey` the + /// converter builds uses this value, so collections in different + /// databases route to distinct shards and the Data Plane isolates them. pub database_id: crate::types::DatabaseId, /// Tenant scope for surrogate identity. Threaded into every surrogate /// `assign`/`lookup` so two tenants with the same primary key in a @@ -136,11 +136,17 @@ impl ConvertContext { self.purpose == PlanningPurpose::Metadata } + /// The canonical key of `bare`, a catalog collection name in this + /// context's database. Placement and surrogate identity use this key. + pub fn collection_key<'a>(&self, bare: &'a str) -> nodedb_types::CollectionKey<'a> { + nodedb_types::CollectionKey::from_bare(self.database_id, bare) + } + /// Resolve an existing surrogate without creating a mapping while planning /// metadata. Execute planning retains the allocating assignment behavior. pub fn surrogate_for_pk( &self, - collection: &str, + key: nodedb_types::CollectionKey<'_>, pk_bytes: &[u8], ) -> crate::Result { let Some(assigner) = self.surrogate_assigner.as_ref() else { @@ -148,10 +154,10 @@ impl ConvertContext { }; if self.is_metadata() { return Ok(assigner - .lookup(self.database_id, self.tenant_id, collection, pk_bytes)? + .lookup(key, self.tenant_id, pk_bytes)? .unwrap_or(nodedb_types::Surrogate::ZERO)); } - assigner.assign(self.database_id, self.tenant_id, collection, pk_bytes) + assigner.assign(key, self.tenant_id, pk_bytes) } /// Resolve an EXISTING pk → surrogate binding read-only, yielding @@ -160,14 +166,14 @@ impl ConvertContext { /// a node-local phantom binding for a key no replica agrees on. pub fn surrogate_for_existing_pk( &self, - collection: &str, + key: nodedb_types::CollectionKey<'_>, pk_bytes: &[u8], ) -> crate::Result { let Some(assigner) = self.surrogate_assigner.as_ref() else { return Ok(nodedb_types::Surrogate::ZERO); }; Ok(assigner - .lookup(self.database_id, self.tenant_id, collection, pk_bytes)? + .lookup(key, self.tenant_id, pk_bytes)? .unwrap_or(nodedb_types::Surrogate::ZERO)) } @@ -180,7 +186,7 @@ impl ConvertContext { /// the same type the allocator renders through. pub fn fresh_surrogate( &self, - collection: &str, + key: nodedb_types::CollectionKey<'_>, ) -> crate::Result<(nodedb_types::Surrogate, String)> { let placeholder = || { let zero = nodedb_types::Surrogate::ZERO; @@ -193,7 +199,7 @@ impl ConvertContext { return placeholder(); } match self.surrogate_assigner.as_ref() { - Some(assigner) => assigner.assign_fresh(self.database_id, self.tenant_id, collection), + Some(assigner) => assigner.assign_fresh(key, self.tenant_id), None => placeholder(), } } @@ -329,26 +335,43 @@ mod tests { assert_eq!( metadata - .surrogate_for_pk("users", b"new-user") + .surrogate_for_pk(metadata.collection_key("users"), b"new-user") + .unwrap() + .as_u32(), + 0 + ); + assert_eq!( + metadata + .fresh_surrogate(metadata.collection_key("users")) .unwrap() + .0 .as_u32(), 0 ); - assert_eq!(metadata.fresh_surrogate("users").unwrap().0.as_u32(), 0); assert_eq!( assigner - .lookup(DatabaseId::DEFAULT, TenantId::new(1), "users", b"new-user") + .lookup( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(1), + b"new-user", + ) .unwrap(), None ); assert_eq!(registry.read().expect("registry").current_hwm(), 0); let execute = context(PlanningPurpose::Execute, Arc::clone(&assigner)); - let allocated = execute.surrogate_for_pk("users", b"new-user").unwrap(); + let allocated = execute + .surrogate_for_pk(execute.collection_key("users"), b"new-user") + .unwrap(); assert_ne!(allocated.as_u32(), 0); assert_eq!( assigner - .lookup(DatabaseId::DEFAULT, TenantId::new(1), "users", b"new-user") + .lookup( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(1), + b"new-user", + ) .unwrap(), Some(allocated) ); diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs index 332605e35..d59c80c02 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs @@ -4,7 +4,7 @@ use nodedb_sql::types::{SqlValue, WriteRoute}; use nodedb_types::Surrogate; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::super::convert::ConvertContext; @@ -45,10 +45,11 @@ pub(in super::super::super) fn convert_insert( tenant_id, ctx, } = args; + let key = ctx.collection_key(collection); let coll_qualified = super::super::super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = key.vshard(); let mut tasks = Vec::new(); let mut columnar_rows: Vec<&Vec<(String, SqlValue)>> = Vec::new(); @@ -106,7 +107,7 @@ pub(in super::super::super) fn convert_insert( let value_bytes = row_to_msgpack(row)?; let (doc_id, surrogate) = resolve_doc_identity_with_declared( ctx, - collection, + key, primary_key, declared_pk.as_deref(), row, @@ -190,7 +191,7 @@ pub(in super::super::super) fn convert_insert( }; let surrogates = columnar_row_surrogates( ctx, - collection, + key, &columnar_rows, primary_key, declared_pk.as_deref(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs index a94261e07..9bf70f12c 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs @@ -1,7 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 use nodedb_sql::types::SqlValue; -use nodedb_types::Surrogate; +use nodedb_types::{CollectionKey, Surrogate}; use super::super::super::convert::ConvertContext; use super::super::super::value::sql_value_to_string; @@ -65,7 +65,7 @@ pub(in super::super::super) fn declared_primary_key_name( /// so no caller mints an identity without the NOT NULL check running first. pub(in super::super) fn resolve_doc_identity_with_declared( ctx: &ConvertContext, - collection: &str, + key: CollectionKey<'_>, primary_key: &str, declared: Option<&str>, row: &[(String, SqlValue)], @@ -75,7 +75,7 @@ pub(in super::super) fn resolve_doc_identity_with_declared( DocId::Present(_) => {} DocId::ExplicitNull | DocId::Absent => { return Err(crate::Error::RejectedConstraint { - collection: collection.to_string(), + collection: key.name().to_string(), constraint: "not_null".to_string(), detail: format!("primary key '{declared}' cannot be NULL or omitted"), }); @@ -84,17 +84,17 @@ pub(in super::super) fn resolve_doc_identity_with_declared( } if is_auto_rowid_pk(primary_key) { - let (s, pk) = assign_fresh(ctx, collection)?; + let (s, pk) = assign_fresh(ctx, key)?; return Ok((pk, s)); } let mint_key: &str = declared.unwrap_or(primary_key); match extract_doc_id(row, mint_key) { DocId::Present(id) => { - let s = assign_for_pk(ctx, collection, id.as_bytes())?; + let s = assign_for_pk(ctx, key, id.as_bytes())?; Ok((id, s)) } DocId::ExplicitNull | DocId::Absent => { - let (s, pk) = assign_fresh(ctx, collection)?; + let (s, pk) = assign_fresh(ctx, key)?; Ok((pk, s)) } } @@ -102,10 +102,10 @@ pub(in super::super) fn resolve_doc_identity_with_declared( pub(in super::super) fn assign_for_pk( ctx: &ConvertContext, - collection: &str, + key: CollectionKey<'_>, pk_bytes: &[u8], ) -> crate::Result { - ctx.surrogate_for_pk(collection, pk_bytes) + ctx.surrogate_for_pk(key, pk_bytes) } /// Allocate a fresh, unique surrogate for a row whose primary key is the @@ -118,9 +118,9 @@ pub(in super::super) fn assign_for_pk( /// Returns the bound identity string. The caller uses it verbatim. pub(super) fn assign_fresh( ctx: &ConvertContext, - collection: &str, + key: CollectionKey<'_>, ) -> crate::Result<(Surrogate, String)> { - ctx.fresh_surrogate(collection) + ctx.fresh_surrogate(key) } /// Whether a collection's declared primary key is the auto-generated `_rowid` @@ -141,7 +141,7 @@ pub(in super::super) fn is_auto_rowid_pk(primary_key: &str) -> bool { /// caller for the whole statement — not re-read here per row. pub(in super::super) fn columnar_row_surrogates( ctx: &ConvertContext, - collection: &str, + key: CollectionKey<'_>, columnar_rows: &[&Vec<(String, SqlValue)>], primary_key: &str, declared_pk: Option<&str>, @@ -155,7 +155,7 @@ pub(in super::super) fn columnar_row_surrogates( let mut out = Vec::with_capacity(columnar_rows.len()); for row in columnar_rows { let (_, surrogate) = - resolve_doc_identity_with_declared(ctx, collection, primary_key, declared, row)?; + resolve_doc_identity_with_declared(ctx, key, primary_key, declared, row)?; out.push(surrogate); } Ok(out) diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs index 48c6a126c..e01cce8b6 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/kv_insert.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{KvInsertIntent, SqlExpr, SqlValue}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::convert::ConvertContext; @@ -25,6 +25,7 @@ pub(in super::super) fn convert_kv_insert( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { + let collection_key = ctx.collection_key(collection); let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); @@ -33,7 +34,7 @@ pub(in super::super) fn convert_kv_insert( } else { assignments_to_update_values(on_conflict_updates)? }; - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let ttl_ms = ttl_secs * 1000; let mut tasks = Vec::with_capacity(entries.len()); for (key_val, value_cols) in entries { @@ -59,7 +60,7 @@ pub(in super::super) fn convert_kv_insert( } buf }; - let surrogate = assign_for_pk(ctx, collection, &key)?; + let surrogate = assign_for_pk(ctx, collection_key, &key)?; let op = match intent { KvInsertIntent::Insert => KvOp::Insert { collection: qualified_collection.clone(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs b/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs index 9d67da961..f5d83986c 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/merge.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{MergeClauseKind, MergePlanAction, MergePlanClause, SqlExpr, SqlPlan}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::DocumentOp; use nodedb_physical::physical_plan::UpdateValue; use nodedb_physical::physical_plan::document::merge_types::{ @@ -45,6 +45,7 @@ pub(in super::super) fn convert_merge( tenant_id, ctx, } = args; + let target_key = ctx.collection_key(target); let target_qualified = super::super::convert::db_qualified(ctx.database_id, target); let qualified_target = nodedb_types::QualifiedCollection::new(ctx.database_id, target); let target = target_qualified.as_str(); @@ -68,7 +69,7 @@ pub(in super::super) fn convert_merge( .map(convert_clause) .collect::>>()?; - let vshard = VShardId::from_collection_in_database(ctx.database_id, target); + let vshard = target_key.vshard(); // A declared PRIMARY KEY implies NOT NULL; the Data Plane checks a MATCHED // or NOT-MATCHED-BY-SOURCE UPDATE arm's post-image against this name. let declared_primary_key = super::declared_primary_key_name(ctx, target)?; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs index 2805683d9..f5f073635 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{EngineType, Filter, SqlValue}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use crate::control::planner::sql_plan_convert::convert::ConvertContext; @@ -25,13 +25,14 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { + let collection_key = ctx.collection_key(collection); let coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( ctx.database_id, collection, ); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); if matches!(engine, EngineType::KeyValue) { // A KV collection has no document store; a WHERE with no primary key @@ -168,7 +169,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( // runs, an unbound row_key affects 0 rows, and the clone CoW // resolver intercepts the ZERO sentinel), but a key this statement // never creates must never mint a binding. - let surrogate = ctx.surrogate_for_existing_pk(collection, &pk_bytes)?; + let surrogate = ctx.surrogate_for_existing_pk(collection_key, &pk_bytes)?; let plan = if is_crdt { PhysicalPlan::Crdt(CrdtOp::DocDelete { collection: qualified_collection.clone(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs index 96138c013..807b59fcb 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{EngineType, Filter, SqlExpr, SqlValue}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use crate::control::planner::sql_plan_convert::convert::ConvertContext; @@ -44,13 +44,14 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( tenant_id, ctx, } = params; + let collection_key = ctx.collection_key(collection); let coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( ctx.database_id, collection, ); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let filter_bytes = serialize_filters(filters)?; let updates = assignments_to_update_values(assignments)?; @@ -122,7 +123,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( let key_bytes = sql_value_to_bytes(key)?; // Content-addressed identity: keeps the surrogate the original insert assigned. // `Surrogate::ZERO` only when no assigner is wired (test / embedded-without-catalog). - let surrogate = ctx.surrogate_for_pk(collection, &key_bytes)?; + let surrogate = ctx.surrogate_for_pk(collection_key, &key_bytes)?; tasks.push(PhysicalTask { tenant_id, vshard_id: vshard, @@ -265,7 +266,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( let plan = if let Some(fields_json) = crdt_fields_json.as_ref() { // An upsert CREATES the row when the key is absent, so it owns // a real identity and allocates one. - let surrogate = ctx.surrogate_for_pk(collection, &pk_bytes)?; + let surrogate = ctx.surrogate_for_pk(collection_key, &pk_bytes)?; PhysicalPlan::Crdt(CrdtOp::DocUpsert { collection: qualified_collection.clone(), document_id: pk_string, @@ -281,7 +282,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( // still runs, an unbound row_key affects 0 rows, and the clone // CoW resolver intercepts the ZERO sentinel), but an UPDATE // creates no row, so it must never mint a binding. - let surrogate = ctx.surrogate_for_existing_pk(collection, &pk_bytes)?; + let surrogate = ctx.surrogate_for_existing_pk(collection_key, &pk_bytes)?; PhysicalPlan::Document(DocumentOp::PointUpdate { collection: qualified_collection.clone(), document_id: pk_string, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update_from.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update_from.rs index 0cd5ef6ce..602f6f6d0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update_from.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update_from.rs @@ -15,7 +15,6 @@ use nodedb_physical::physical_plan::*; use crate::control::planner::sql_plan_convert::convert::ConvertContext; use crate::control::planner::sql_plan_convert::filter::serialize_filters; use crate::control::planner::sql_plan_convert::value::assignments_to_update_values_qualified; -use crate::types::VShardId; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; /// Parameters for [`convert_update_from`], bundled to avoid an unwieldy @@ -47,6 +46,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update_from( tenant_id, ctx, } = params; + let collection_key = ctx.collection_key(collection); let coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( ctx.database_id, collection, @@ -78,7 +78,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update_from( let updates = assignments_to_update_values_qualified(assignments)?; let target_filter_bytes = serialize_filters(target_filters)?; - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); // A declared PRIMARY KEY implies NOT NULL; the Data Plane checks the // post-image against this name once the SET expressions are evaluated. let declared_primary_key = super::super::declared_primary_key_name(ctx, collection)?; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs index f37ecf15a..b632d3e76 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/upsert.rs @@ -9,7 +9,7 @@ use nodedb_sql::types::{SqlExpr, SqlValue, WriteRoute}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::ColumnarInsertIntent; use nodedb_physical::physical_plan::*; @@ -49,10 +49,11 @@ pub(in super::super) fn convert_upsert( tenant_id, ctx, } = args; + let key = ctx.collection_key(collection); let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = key.vshard(); let mut tasks = Vec::new(); // Detect CRDT document collections once. An explicit `ON CONFLICT DO UPDATE @@ -91,7 +92,7 @@ pub(in super::super) fn convert_upsert( let value_bytes = row_to_msgpack(row)?; let (doc_id, surrogate) = resolve_doc_identity_with_declared( ctx, - collection, + key, primary_key, declared_pk.as_deref(), row, @@ -143,7 +144,7 @@ pub(in super::super) fn convert_upsert( let payload = rows_to_msgpack_array(&columnar_rows)?; let surrogates = columnar_row_surrogates( ctx, - collection, + key, &columnar_rows, primary_key, declared_pk.as_deref(), diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs b/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs index a79cde8a2..e57e34c91 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs @@ -9,7 +9,7 @@ //! statement never created mints no binding. use nodedb_sql::types::{Filter, SqlExpr, SqlValue, VectorPrimaryInsertIntent, VectorPrimaryRow}; -use nodedb_types::{RlsWriteCheck, Surrogate}; +use nodedb_types::{CollectionKey, RlsWriteCheck, Surrogate}; use crate::bridge::envelope::PhysicalPlan; use crate::types::{TenantId, VShardId}; @@ -47,24 +47,27 @@ pub(in super::super) struct VectorPrimaryInsertArgs<'a> { } /// The routing every vector-primary task shares. -struct Routing { +struct Routing<'a> { + /// Canonical key: placement and surrogate identity. + key: CollectionKey<'a>, qualified: nodedb_types::QualifiedCollection, collection: String, vshard: VShardId, } -fn routing(ctx: &ConvertContext, collection: &str) -> Routing { +fn routing<'a>(ctx: &ConvertContext, collection: &'a str) -> Routing<'a> { + let key = ctx.collection_key(collection); let qualified = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = db_qualified(ctx.database_id, collection); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection.as_str()); Routing { + key, qualified, collection, - vshard, + vshard: key.vshard(), } } -fn task(tenant_id: TenantId, r: &Routing, ctx: &ConvertContext, op: VectorOp) -> PhysicalTask { +fn task(tenant_id: TenantId, r: &Routing<'_>, ctx: &ConvertContext, op: VectorOp) -> PhysicalTask { PhysicalTask { tenant_id, vshard_id: r.vshard, @@ -150,7 +153,7 @@ pub(in super::super) fn convert_vector_primary_insert( .collect(); let (doc_id, surrogate) = resolve_doc_identity_with_declared( ctx, - collection, + r.key, primary_key, declared.as_deref(), &row_fields, @@ -222,7 +225,7 @@ pub(in super::super) fn convert_vector_primary_insert( /// serialize `filters` for the Data Plane to evaluate on the sidecar rows. fn write_targets( ctx: &ConvertContext, - collection: &str, + collection: CollectionKey<'_>, filters: &[Filter], target_keys: &[SqlValue], ) -> crate::Result { @@ -246,7 +249,7 @@ pub(in super::super) fn convert_vector_primary_delete( ctx: &ConvertContext, ) -> crate::Result> { let r = routing(ctx, collection); - let targets = write_targets(ctx, r.collection.as_str(), filters, target_keys)?; + let targets = write_targets(ctx, r.key, filters, target_keys)?; Ok(vec![task( tenant_id, &r, @@ -321,7 +324,7 @@ pub(in super::super) fn convert_vector_primary_update( }); } let r = routing(ctx, collection); - let targets = write_targets(ctx, r.collection.as_str(), filters, target_keys)?; + let targets = write_targets(ctx, r.key, filters, target_keys)?; let payload_patch: Vec<(String, UpdateValue)> = assignments_to_update_values(assignments)?; Ok(vec![task( tenant_id, diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs index e2256a3ad..bd1e39c4e 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs @@ -6,7 +6,7 @@ use nodedb_sql::types::{EngineType, Filter, SqlValue}; use nodedb_types::SystemTimeScope; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::aggregate::{ @@ -54,7 +54,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( let sort = convert_sort_keys(sort_keys); return Ok(vec![PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(database_id, ""), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, "").vshard(), database_id, plan: PhysicalPlan::Query(QueryOp::ProviderScan { provider: Some(collection.to_string()), @@ -73,13 +73,12 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_scan( }]); } - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); let filter_bytes = serialize_filters(filters)?; let proj_names = extract_projection_names(projection, window_functions); let sort = convert_sort_keys(sort_keys); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); let physical = match engine { EngineType::Timeseries => { @@ -217,12 +216,11 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_document_index_look tenant_id, database_id, } = args; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); let filter_bytes = serialize_filters(filters)?; let proj_names = extract_projection_names(projection, &[]); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); let physical = PhysicalPlan::Document(DocumentOp::IndexedFetch { collection: qualified_collection, path: field.into(), @@ -250,10 +248,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_point_get( tenant_id: TenantId, ctx: &super::super::convert::ConvertContext, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let physical = match engine { EngineType::KeyValue => PhysicalPlan::Kv(KvOp::Get { collection: qualified_collection.clone(), @@ -265,7 +262,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_point_get( let pk_string = sql_value_to_string(key_value); let pk_bytes = pk_string.clone().into_bytes(); let surrogate = match ctx.surrogate_assigner.as_ref() { - Some(a) => match a.lookup(ctx.database_id, ctx.tenant_id, collection, &pk_bytes)? { + Some(a) => match a.lookup(collection_key, ctx.tenant_id, &pk_bytes)? { Some(s) => s, None => { // No surrogate bound in the target database yet. diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/join.rs b/nodedb/src/control/planner/sql_plan_convert/scan/join.rs index 2523ad394..8234f110d 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/join.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/join.rs @@ -7,7 +7,7 @@ use nodedb_sql::planner::bitmap_emit::predicate::BitmapHint; use nodedb_sql::types::SqlPlan; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use nodedb_physical::physical_plan::*; use super::super::aggregate::{ @@ -176,7 +176,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_join( let left_bitmap = raw_left_bm.and_then(|h| bitmap_hint_to_plan(&h, db_id)); let right_bitmap = raw_right_bm.and_then(|h| bitmap_hint_to_plan(&h, db_id)); - let vshard = VShardId::from_collection_in_database(p.ctx.database_id, &left_collection); + let vshard = + nodedb_types::CollectionKey::from_qualified_str(p.ctx.database_id, &left_collection)? + .vshard(); // Shuffle eligibility. A whole-join shuffle is only *structurally* valid // when BOTH sides are plain sharded user collections scanned by name — i.e. diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/recursive.rs b/nodedb/src/control/planner/sql_plan_convert/scan/recursive.rs index a49744ebf..dbcd37f11 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/recursive.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/recursive.rs @@ -13,10 +13,9 @@ use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; pub(in crate::control::planner::sql_plan_convert) fn convert_recursive_scan( p: RecursiveScanParams<'_>, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(p.database_id, p.collection); + let collection_key = nodedb_types::CollectionKey::from_bare(p.database_id, p.collection); let qualified_collection = nodedb_types::QualifiedCollection::new(p.database_id, p.collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(p.database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id: p.tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/search.rs b/nodedb/src/control/planner/sql_plan_convert/scan/search.rs index 79aab126b..7df0e480f 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/search.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/search.rs @@ -4,7 +4,7 @@ //! builder shared across them. use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::filter::serialize_filters; @@ -17,11 +17,10 @@ use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; pub(in crate::control::planner::sql_plan_convert) fn convert_vector_search( p: VectorSearchParams<'_>, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(p.ctx.database_id, p.collection); + let collection_key = nodedb_types::CollectionKey::from_bare(p.ctx.database_id, p.collection); let qualified_collection = nodedb_types::QualifiedCollection::new(p.ctx.database_id, p.collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(p.ctx.database_id, collection); + let vshard = collection_key.vshard(); let filter_bytes = serialize_filters(p.filters)?; let inline_prefilter_plan = match p.array_prefilter { Some(pref) => Some(Box::new(build_array_prefilter_plan( @@ -63,10 +62,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_vector_search( pub(in crate::control::planner::sql_plan_convert) fn convert_sparse_search( p: SparseSearchParams<'_>, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(p.database_id, p.collection); + let collection_key = nodedb_types::CollectionKey::from_bare(p.database_id, p.collection); let qualified_collection = nodedb_types::QualifiedCollection::new(p.database_id, p.collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(p.database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id: p.tenant_id, vshard_id: vshard, @@ -184,10 +182,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_text_search( ) -> crate::Result> { use nodedb_sql::fts_types::FtsQuery; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); // Phrase queries emit a dedicated PhraseSearch op rather than going // through the BM25 plain-string path. Score alias is not meaningful @@ -285,10 +282,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_hybrid_search( tenant_id, database_id, } = p; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, @@ -328,10 +324,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_hybrid_search_tripl tenant_id, database_id, } = p; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/spatial.rs b/nodedb/src/control/planner/sql_plan_convert/scan/spatial.rs index fc12aa38f..0c6d2be70 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/spatial.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/spatial.rs @@ -3,7 +3,6 @@ //! Spatial scan converter. use crate::bridge::envelope::PhysicalPlan; -use crate::types::VShardId; use nodedb_physical::physical_plan::*; use super::super::aggregate::extract_projection_names; @@ -26,10 +25,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_spatial_scan( tenant_id, database_id, } = p; - let coll_qualified = super::super::convert::db_qualified(database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(database_id, collection); + let vshard = collection_key.vshard(); let attr_bytes = serialize_filters(attribute_filters)?; let proj_names = extract_projection_names(projection, &[]); let sp = match predicate { diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs index 0d8521248..9a56408cf 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/timeseries.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::SqlValue; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::super::aggregate::{ @@ -37,16 +37,21 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_scan( ctx, temporal, } = p; - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); - let collection = coll_qualified.as_str(); let filter_bytes = serialize_filters(filters)?; let agg_pairs: Vec<(String, String)> = aggregates.iter().map(agg_expr_to_pair).collect(); // AUTO_TIER: split query across retention tiers if enabled. if *tiered && let Some(registry) = &ctx.retention_registry - && let Some(policy) = registry.get(ctx.database_id.as_u64(), tenant_id.as_u64(), collection) + // Policies are keyed by policy name. A collection's policy is found + // by the bare collection name it targets. + && let Some(policy) = registry.get_for_collection( + ctx.database_id.as_u64(), + tenant_id.as_u64(), + collection_key.name(), + ) && policy.auto_tier { return Ok(super::super::super::auto_tier::plan_tiered_scan( @@ -65,7 +70,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_scan( let proj_names = extract_projection_names(projection, &[]); let computed_bytes = extract_computed_columns(projection, &[], false)?; - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); Ok(vec![PhysicalTask { tenant_id, vshard_id: vshard, @@ -97,10 +102,9 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_ingest( tenant_id: TenantId, ctx: &super::super::convert::ConvertContext, ) -> crate::Result> { - let coll_qualified = super::super::convert::db_qualified(ctx.database_id, collection); + let collection_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); - let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let mut payload = Vec::with_capacity(rows.len() * 128); write_msgpack_array_header(&mut payload, rows.len()); let mut surrogates: Vec = Vec::with_capacity(rows.len()); @@ -119,7 +123,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_timeseries_ingest( // PK collapses every row onto `Surrogate::ZERO` and merges distinct // rows. Nothing looks a timeseries row up by this binding, so the // identity string is discarded. - let (s, _) = ctx.fresh_surrogate(collection)?; + let (s, _) = ctx.fresh_surrogate(collection_key)?; surrogates.push(s); } Ok(vec![PhysicalTask { diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index 0141aac85..5e9c892d0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -5,7 +5,7 @@ use nodedb_sql::types::{EngineType, Projection, SortKey, SqlExpr, SqlPlan, SqlValue, WindowSpec}; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; use nodedb_physical::physical_plan::*; use super::body::convert_body_to_single_plan; @@ -55,7 +55,7 @@ pub(super) fn convert_constant_result( })?; Ok(vec![PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), + vshard_id: nodedb_types::CollectionKey::from_bare(ctx.database_id, "").vshard(), database_id: ctx.database_id, plan: PhysicalPlan::Query(QueryOp::ProviderScan { provider: None, @@ -87,10 +87,11 @@ pub(super) fn convert_truncate( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { + let collection_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, collection); let coll_qualified = super::convert::db_qualified(ctx.database_id, collection); let qualified_collection = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); let collection = coll_qualified.as_str(); - let vshard = VShardId::from_collection_in_database(ctx.database_id, collection); + let vshard = collection_key.vshard(); let plan = match engine { EngineType::DocumentSchemaless | EngineType::DocumentStrict => { PhysicalPlan::Document(DocumentOp::Truncate { @@ -209,6 +210,7 @@ pub(super) fn convert_insert_select( tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { + let target_key = nodedb_types::CollectionKey::from_bare(ctx.database_id, target); let target_qualified = super::convert::db_qualified(ctx.database_id, target); let qualified_target = nodedb_types::QualifiedCollection::new(ctx.database_id, target); let target = target_qualified.as_str(); @@ -254,7 +256,7 @@ pub(super) fn convert_insert_select( let filter_bytes = super::filter::serialize_filters(filters)?; let column_map_bytes = super::aggregate::serialize_column_map(column_map)?; - let vshard = VShardId::from_collection_in_database(ctx.database_id, target); + let vshard = target_key.vshard(); let qualified_source = nodedb_types::QualifiedCollection::new(ctx.database_id, collection); Ok(vec![PhysicalTask { @@ -330,7 +332,7 @@ pub(super) fn convert_subquery( tenant_id, // Coordinator-local: resolved to a `ProviderScan` over the gathered // rows (empty collection, like a constant result), dispatched once. - vshard_id: VShardId::from_collection_in_database(ctx.database_id, ""), + vshard_id: nodedb_types::CollectionKey::from_bare(ctx.database_id, "").vshard(), database_id: ctx.database_id, plan: PhysicalPlan::Query(QueryOp::PostProcess { input: Box::new(child), diff --git a/nodedb/src/control/security/auth_fence/view.rs b/nodedb/src/control/security/auth_fence/view.rs index 1c46b8dc9..595db3fdf 100644 --- a/nodedb/src/control/security/auth_fence/view.rs +++ b/nodedb/src/control/security/auth_fence/view.rs @@ -30,7 +30,7 @@ use tokio::sync::RwLockReadGuard; use crate::control::security::auth_lease::lease_status; use crate::control::security::permission_tree::{PermissionCache, reload}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::types::{DatabaseId, TenantId}; use super::cluster::{behind, group_of_vshard, hosts_group}; @@ -50,7 +50,8 @@ pub async fn permission_view( .filter(|source| source.tenant_id == tenant_id.as_u64()) { let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &source.collection); + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &source.collection) + .vshard(); let group_id = group_of_vshard(state, vshard.as_u32())?; if !hosts_group(state, group_id) { return Err(behind(format!( diff --git a/nodedb/src/control/security/catalog/surrogate_pk.rs b/nodedb/src/control/security/catalog/surrogate_pk.rs index 008c2a4db..5c85bc5d6 100644 --- a/nodedb/src/control/security/catalog/surrogate_pk.rs +++ b/nodedb/src/control/security/catalog/surrogate_pk.rs @@ -8,7 +8,8 @@ //! //! The compound key is `(database_id, tenant_id, collection, pk_bytes)` //! (forward) and `(database_id, tenant_id, collection, surrogate)` (reverse), -//! scoping the PK map to its database + tenant boundary. +//! scoping the PK map to its database + tenant boundary. Every entry point +//! takes a [`CollectionKey`], so `collection` is always the bare catalog name. //! //! ## Migration //! @@ -20,7 +21,7 @@ //! second key component. Both are idempotent: each skips if its target is //! already non-empty. -use nodedb_types::{DatabaseId, Surrogate, TenantId}; +use nodedb_types::{CollectionKey, DatabaseId, Surrogate, TenantId}; use redb::{ReadableDatabase, ReadableTable, ReadableTableMetadata}; #[allow(unused_imports)] // SURROGATE_PK_REV_LEGACY is used only in #[cfg(test)] helpers @@ -37,14 +38,14 @@ impl SystemCatalog { /// pk_bytes)` to the same surrogate is a no-op-on-disk overwrite. pub fn put_surrogate( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], surrogate: Surrogate, ) -> crate::Result<()> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_write() @@ -69,13 +70,13 @@ impl SystemCatalog { /// collection, pk_bytes)`. Returns `None` if no binding exists. pub fn get_surrogate_for_pk( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], ) -> crate::Result> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_read() @@ -96,13 +97,13 @@ impl SystemCatalog { /// surrogate)`. Returns `None` if no binding exists. pub fn get_pk_for_surrogate( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, surrogate: Surrogate, ) -> crate::Result>> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_read() @@ -122,13 +123,13 @@ impl SystemCatalog { /// Remove a surrogate ↔ PK binding atomically. Idempotent. pub fn delete_surrogate( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], ) -> crate::Result<()> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_write() @@ -157,12 +158,12 @@ impl SystemCatalog { /// Returns `Vec<(pk_bytes, surrogate)>` in redb's natural key order. pub fn scan_surrogates_for_collection( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, ) -> crate::Result, Surrogate)>> { - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_read() @@ -233,16 +234,16 @@ impl SystemCatalog { /// collection)` triple. Drains both forward and reverse tables. Idempotent. pub fn delete_all_surrogates_for_collection( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, ) -> crate::Result<()> { - let to_remove = self.scan_surrogates_for_collection(database_id, tenant_id, collection)?; + let to_remove = self.scan_surrogates_for_collection(key, tenant_id)?; if to_remove.is_empty() { return Ok(()); } - let db_id = database_id.as_u64(); + let db_id = key.database_id().as_u64(); let tid = tenant_id.as_u64(); + let collection = key.name(); let txn = self .db .begin_write() @@ -444,25 +445,22 @@ mod max_bound_surrogate_tests { fn floor_is_the_global_maximum_across_every_scope() { let (_dir, cat) = open(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "zzz_last_in_key_order"), TenantId::new(1), - "zzz_last_in_key_order", b"a", Surrogate::new(3), ) .unwrap(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "aaa_first_in_key_order"), TenantId::new(1), - "aaa_first_in_key_order", b"b", Surrogate::new(9_000), ) .unwrap(); cat.put_surrogate( - DatabaseId::new(7), + nodedb_types::CollectionKey::from_bare(DatabaseId::new(7), "other_db"), TenantId::new(2), - "other_db", b"c", Surrogate::new(41), ) @@ -478,9 +476,8 @@ mod max_bound_surrogate_tests { fn floor_outranks_a_stale_hwm_singleton() { let (_dir, cat) = open(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), TenantId::new(1), - "users", b"alice", Surrogate::new(500), ) @@ -519,21 +516,28 @@ mod tests { fn put_then_get_roundtrip() { let (_dir, cat) = open_catalog(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(7), ) .unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice" + ) + .unwrap(), Some(Surrogate::new(7)) ); assert_eq!( - cat.get_pk_for_surrogate(DatabaseId::DEFAULT, T0, "users", Surrogate::new(7)) - .unwrap(), + cat.get_pk_for_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + Surrogate::new(7) + ) + .unwrap(), Some(b"alice".to_vec()) ); } @@ -544,29 +548,35 @@ mod tests { let t1 = TenantId::new(1); let t2 = TenantId::new(2); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), t1, - "users", b"alice", Surrogate::new(10), ) .unwrap(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), t2, - "users", b"alice", Surrogate::new(20), ) .unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, t1, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + t1, + b"alice" + ) + .unwrap(), Some(Surrogate::new(10)) ); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, t2, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + t2, + b"alice" + ) + .unwrap(), Some(Surrogate::new(20)) ); } @@ -575,8 +585,12 @@ mod tests { fn missing_returns_none() { let (_dir, cat) = open_catalog(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "users", b"nobody") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"nobody" + ) + .unwrap(), None ); } @@ -585,56 +599,72 @@ mod tests { fn delete_is_idempotent_and_removes_both_directions() { let (_dir, cat) = open_catalog(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(7), ) .unwrap(); - cat.delete_surrogate(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(); + cat.delete_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice", + ) + .unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice" + ) + .unwrap(), None ); - cat.delete_surrogate(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(); + cat.delete_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice", + ) + .unwrap(); } #[test] fn scan_returns_only_named_collection() { let (_dir, cat) = open_catalog(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(1), ) .unwrap(); - cat.put_surrogate(DatabaseId::DEFAULT, T0, "users", b"bob", Surrogate::new(2)) - .unwrap(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"bob", + Surrogate::new(2), + ) + .unwrap(); + cat.put_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"), T0, - "orders", b"alice", Surrogate::new(3), ) .unwrap(); // A different tenant's same-named collection must not leak into the scan. cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), TenantId::new(9), - "users", b"carol", Surrogate::new(4), ) .unwrap(); let mut got = cat - .scan_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") + .scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + ) .unwrap(); got.sort(); assert_eq!( @@ -650,30 +680,47 @@ mod tests { fn delete_all_wipes_collection_and_leaves_others_intact() { let (_dir, cat) = open_catalog(); cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(1), ) .unwrap(); - cat.put_surrogate(DatabaseId::DEFAULT, T0, "orders", b"o1", Surrogate::new(2)) - .unwrap(); - cat.delete_all_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") - .unwrap(); + cat.put_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"), + T0, + b"o1", + Surrogate::new(2), + ) + .unwrap(); + cat.delete_all_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + ) + .unwrap(); assert!( - cat.scan_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") - .unwrap() - .is_empty() + cat.scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0 + ) + .unwrap() + .is_empty() ); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "orders", b"o1") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"), + T0, + b"o1" + ) + .unwrap(), Some(Surrogate::new(2)) ); // double-delete is a no-op - cat.delete_all_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") - .unwrap(); + cat.delete_all_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + ) + .unwrap(); } // ── Migration tests ─────────────────────────────────────────────────── @@ -702,9 +749,12 @@ mod tests { cat.migrate_surrogate_pk().unwrap(); cat.migrate_surrogate_pk_v3().unwrap(); assert!( - cat.scan_surrogates_for_collection(DatabaseId::DEFAULT, T0, "users") - .unwrap() - .is_empty() + cat.scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0 + ) + .unwrap() + .is_empty() ); } @@ -724,9 +774,8 @@ mod tests { let (_dir, cat) = open_catalog(); // v2 row already exists, written under the default tenant in v3 … cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(7), ) @@ -781,15 +830,18 @@ mod tests { // default identity that wrote them pre-upgrade. let default_tenant = TenantId::new(1); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, default_tenant, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + default_tenant, + b"alice" + ) + .unwrap(), Some(Surrogate::new(7)) ); assert_eq!( cat.get_pk_for_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), default_tenant, - "users", Surrogate::new(7) ) .unwrap(), @@ -802,9 +854,8 @@ mod tests { let (_dir, cat) = open_catalog(); // v3 already populated … cat.put_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), T0, - "users", b"alice", Surrogate::new(7), ) @@ -824,8 +875,12 @@ mod tests { } cat.migrate_surrogate_pk_v3().unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, T0, "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice" + ) + .unwrap(), Some(Surrogate::new(7)) ); } diff --git a/nodedb/src/control/security/permission_tree/reload.rs b/nodedb/src/control/security/permission_tree/reload.rs index f552da0df..a669b8650 100644 --- a/nodedb/src/control/security/permission_tree/reload.rs +++ b/nodedb/src/control/security/permission_tree/reload.rs @@ -21,7 +21,7 @@ use crate::control::server::dispatch_utils::{ LocalRead, dispatch_local_read, reject_data_plane_error, }; use crate::control::state::SharedState; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use super::cache::{PermissionCache, TreeSource, TreeSourceKind}; use super::event_handler::{extract_edge, extract_grant}; @@ -127,7 +127,7 @@ async fn scan_source( state, TenantId::new(source.tenant_id), database_id, - VShardId::from_collection_in_database(database_id, &source.collection), + nodedb_types::CollectionKey::from_bare(database_id, &source.collection).vshard(), LocalRead::DocumentScan { collection: QualifiedCollection::new(database_id, &source.collection), }, diff --git a/nodedb/src/control/security/permission_tree/sources.rs b/nodedb/src/control/security/permission_tree/sources.rs index be05d6eba..ef27af7f1 100644 --- a/nodedb/src/control/security/permission_tree/sources.rs +++ b/nodedb/src/control/security/permission_tree/sources.rs @@ -24,7 +24,7 @@ use std::collections::{HashMap, HashSet}; use std::sync::RwLock; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use super::types::PermissionTreeDef; @@ -43,7 +43,9 @@ impl SourceSet { for collection in [governed.as_str(), def.permission_table.as_str()] { set.collections.insert(collection.to_owned()); set.vshards.insert( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection).as_u32(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection) + .vshard() + .as_u32(), ); } } @@ -150,8 +152,9 @@ mod tests { assert!(index.is_source_collection("docs")); assert!(index.is_source_collection("grants")); assert!(!index.is_source_collection("other")); - let grants_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "grants").as_u32(); + let grants_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "grants") + .vshard() + .as_u32(); assert!(index.is_source_vshard(grants_vshard)); defs.clear(); index.rebuild(&defs); diff --git a/nodedb/src/control/server/calvin_submit/hook.rs b/nodedb/src/control/server/calvin_submit/hook.rs index 9e2334189..57a061d0a 100644 --- a/nodedb/src/control/server/calvin_submit/hook.rs +++ b/nodedb/src/control/server/calvin_submit/hook.rs @@ -76,7 +76,15 @@ impl nodedb_cluster::CalvinSubmit for RegistryCalvinSubmit { }; // Re-derive the participating-vshard set skipped during serialization // (the wire bytes carry only the read/write sets). - tx_class.restore_derived(); + if let Err(e) = tx_class.restore_derived() { + return SubmitCalvinTxnResponse { + error: Some(TypedClusterError::Internal { + code: 0, + message: format!("calvin-submit: TxClass participants underivable: {e}"), + }), + payload_bytes: None, + }; + } let timeout = Duration::from_millis(req.deadline_remaining_ms.max(1)); match submit_and_await_calvin_with_timeout(&self.state, tx_class, timeout).await { diff --git a/nodedb/src/control/server/calvin_submit/inbox_hook.rs b/nodedb/src/control/server/calvin_submit/inbox_hook.rs index 161ca1385..b585f12e4 100644 --- a/nodedb/src/control/server/calvin_submit/inbox_hook.rs +++ b/nodedb/src/control/server/calvin_submit/inbox_hook.rs @@ -86,7 +86,18 @@ impl nodedb_cluster::CalvinSubmitInbox for RegistryCalvinSubmitInbox { }; // Re-derive the participating-vshard set skipped during serialization // (the wire bytes carry only the read/write sets). - tx_class.restore_derived(); + if let Err(e) = tx_class.restore_derived() { + return SubmitCalvinInboxResponse { + inbox_seq: 0, + epoch: 0, + position: 0, + participants: 0, + error: Some(TypedClusterError::Internal { + code: 0, + message: format!("calvin-inbox: TxClass participants underivable: {e}"), + }), + }; + } let timeout = Duration::from_millis(req.deadline_remaining_ms.max(1)); match submit_local_assign(&self.state, tx_class, timeout).await { diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index de5d7eba9..7a69a081c 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -461,7 +461,7 @@ mod tests { PhysicalTask { tenant_id, database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, ARRAY), + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, ARRAY).vshard(), plan: crate::bridge::envelope::PhysicalPlan::Array(ArrayOp::Put { array_id: ArrayId::in_database(tenant_id, DatabaseId::DEFAULT, ARRAY), cells_msgpack: zerompk::to_msgpack_vec(&cells).expect("encode cells"), diff --git a/nodedb/src/control/server/exchange/all_cores/dispatch.rs b/nodedb/src/control/server/exchange/all_cores/dispatch.rs index 7d1ea7905..37112a77b 100644 --- a/nodedb/src/control/server/exchange/all_cores/dispatch.rs +++ b/nodedb/src/control/server/exchange/all_cores/dispatch.rs @@ -223,7 +223,7 @@ async fn generic_gather( && let Some(collection) = plan.collection() { let vshard_id = - crate::types::VShardId::from_collection_in_database(database_id, collection); + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); let resp = dispatch_single_owning_core( state, tenant_id, diff --git a/nodedb/src/control/server/exchange/gather.rs b/nodedb/src/control/server/exchange/gather.rs index 6c171a12e..8848261d0 100644 --- a/nodedb/src/control/server/exchange/gather.rs +++ b/nodedb/src/control/server/exchange/gather.rs @@ -340,8 +340,8 @@ pub(crate) fn gather_all_cores_stream( /// timeseries, spatial, vector, text) /// /// Standard collections are *single-vShard-homed*: all rows for a collection -/// live on exactly one vShard determined by `vshard_for_collection(database_id, -/// &name)`. The data-plane scan is **not** vshard-scoped, so broadcasting the +/// live on exactly one vShard determined by `vshard_for_collection` over the +/// collection's canonical key. The data-plane scan is **not** vshard-scoped, so broadcasting the /// plan to every vShard via `Exchange{Gather}` causes the owning node to return /// the full collection once per route that lands on it — 1 024× duplication. /// diff --git a/nodedb/src/control/server/exchange/owning_core.rs b/nodedb/src/control/server/exchange/owning_core.rs index 94fe23547..e277536b9 100644 --- a/nodedb/src/control/server/exchange/owning_core.rs +++ b/nodedb/src/control/server/exchange/owning_core.rs @@ -51,7 +51,8 @@ pub async fn gather_single_node( return gather_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id).await; } if let Some(collection) = plan.collection() { - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); return gather_single_owning_core( state, tenant_id, @@ -69,8 +70,8 @@ pub async fn gather_single_node( /// Dispatch `plan` to the single Data-Plane core that owns `vshard_id` and /// gather the one bounded response into a [`GatherOutcome`]. /// -/// `vshard_id` is the collection's owning vShard -/// (`VShardId::from_collection_in_database(database_id, collection)`); the +/// `vshard_id` is the collection's owning vShard (the vShard of its +/// canonical `CollectionKey`); the /// dispatcher's `VShardRouter` resolves it to the one core holding the /// collection's rows. /// @@ -78,7 +79,7 @@ pub async fn gather_single_node( /// `read_version_lsn` and exactly one `shard_watermarks` entry keyed to the /// collection's vShard — matching the cluster `dispatch_local` path so an /// in-transaction read records the same OCC read-set entry the write-set uses -/// (writes home to the same `from_collection_in_database` vShard). Aggregate +/// (writes home to the same `CollectionKey` vShard). Aggregate /// finalization (`finalize_aggregate`) is a passthrough over the merged array, /// so one complete aggregate row in yields one row out. pub async fn gather_single_owning_core( diff --git a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs index 8230cffec..b71d33e08 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs @@ -21,7 +21,6 @@ use crate::data::executor::response_codec::{ flatten_hybrid_hits_to_relational_rows, flatten_to_relational_rows, flatten_vector_hits_to_relational_rows, }; -use crate::types::VShardId; use super::dispatch::{ResolveCtx, resolve_exchange}; use super::entry::Resolved; @@ -214,7 +213,7 @@ pub(super) async fn materialize_child_rows( tenant_id, database_id, child, - VShardId::from_collection_in_database(database_id, ""), + nodedb_types::CollectionKey::from_bare(database_id, "").vshard(), trace_id, txn_id, ) diff --git a/nodedb/src/control/server/exchange/resolve/peers.rs b/nodedb/src/control/server/exchange/resolve/peers.rs index 8c19e4bca..2bb1c755c 100644 --- a/nodedb/src/control/server/exchange/resolve/peers.rs +++ b/nodedb/src/control/server/exchange/resolve/peers.rs @@ -16,9 +16,10 @@ use nodedb_cluster::{ }; use crate::control::state::SharedState; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; -/// Producer nodes that own `collection`'s data: resolve its vShard → owning +/// Producer nodes that own `collection`'s data. `collection` is the plan's +/// database-qualified name. Resolve its canonical key's vShard → owning /// group → leader. A user collection is single-vShard-homed, so this is one /// node; returned as a deduped sorted vec for generality. pub(super) fn producer_nodes( @@ -26,7 +27,9 @@ pub(super) fn producer_nodes( database_id: DatabaseId, collection: &str, ) -> crate::Result> { - let vshard = VShardId::from_collection_in_database(database_id, collection).as_u32(); + let vshard = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)? + .vshard() + .as_u32(); let group = routing .group_for_vshard(vshard) .map_err(|e| crate::Error::Internal { diff --git a/nodedb/src/control/server/http/routes/crdt.rs b/nodedb/src/control/server/http/routes/crdt.rs index 511aebcc5..730edb7c8 100644 --- a/nodedb/src/control/server/http/routes/crdt.rs +++ b/nodedb/src/control/server/http/routes/crdt.rs @@ -100,9 +100,8 @@ pub async fn crdt_apply( .shared .surrogate_assigner .assign( - crate::types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(crate::types::DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, body.doc_id.as_bytes(), ) .map_err(|e| ApiError::Internal(e.to_string()))?; @@ -125,10 +124,11 @@ pub async fn crdt_apply( let task = PhysicalTask { tenant_id: identity.tenant_id, - vshard_id: crate::types::VShardId::from_collection_in_database( + vshard_id: nodedb_types::CollectionKey::from_bare( crate::types::DatabaseId::DEFAULT, &collection, - ), + ) + .vshard(), database_id: crate::types::DatabaseId::DEFAULT, plan, post_set_op: PostSetOp::None, diff --git a/nodedb/src/control/server/http/routes/promql/remote.rs b/nodedb/src/control/server/http/routes/promql/remote.rs index e0a70a3d8..8f613cda5 100644 --- a/nodedb/src/control/server/http/routes/promql/remote.rs +++ b/nodedb/src/control/server/http/routes/promql/remote.rs @@ -23,7 +23,7 @@ use crate::control::promql::{self, types::DEFAULT_LOOKBACK_MS}; use crate::control::server::http::admission::admit_without_rate_limit; use crate::control::server::http::auth::{AppState, ResolvedIdentity}; use crate::control::server::http::peer::PeerAddr; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -103,7 +103,8 @@ pub async fn remote_write( continue; } - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let plan = PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), payload: ilp_payload.into_bytes(), diff --git a/nodedb/src/control/server/ilp_batch/dispatch.rs b/nodedb/src/control/server/ilp_batch/dispatch.rs index e75f5dca5..6e18c5e6e 100644 --- a/nodedb/src/control/server/ilp_batch/dispatch.rs +++ b/nodedb/src/control/server/ilp_batch/dispatch.rs @@ -19,7 +19,7 @@ use crate::control::server::ilp_auth::AuthenticatedIlpContext; use crate::control::server::shared::authorization::authorize_task_set; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::TimeseriesOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use nodedb_types::Surrogate; @@ -297,7 +297,8 @@ fn build_ilp_calvin_tasks( Ok(PhysicalTask { tenant_id, database_id, - vshard_id: VShardId::from_collection_in_database(database_id, &group.measurement), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, &group.measurement) + .vshard(), plan: PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: nodedb_types::QualifiedCollection::new( database_id, @@ -338,7 +339,7 @@ mod tests { use crate::control::security::identity::{AuthMethod, AuthenticatedIdentity, DatabaseSet}; use crate::control::security::permission::PermissionStore; use crate::control::security::role::RoleStore; - use crate::types::{DatabaseId, TenantId, VShardId}; + use crate::types::{DatabaseId, TenantId}; use crate::wal::WalManager; use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp}; use nodedb_types::Surrogate; @@ -636,7 +637,7 @@ mod tests { assert_eq!(tasks[0].database_id, database_id); assert_eq!( tasks[0].vshard_id, - VShardId::from_collection_in_database(database_id, "cpu") + nodedb_types::CollectionKey::from_bare(database_id, "cpu").vshard() ); } } diff --git a/nodedb/src/control/server/native/dispatch/ctx.rs b/nodedb/src/control/server/native/dispatch/ctx.rs index 69c8de36e..9b6245dd3 100644 --- a/nodedb/src/control/server/native/dispatch/ctx.rs +++ b/nodedb/src/control/server/native/dispatch/ctx.rs @@ -10,6 +10,9 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; use crate::control::server::shared::session::SessionStore; use crate::control::state::SharedState; +use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; +use nodedb_types::CollectionKey; + use crate::types::{TenantId, VShardId}; /// Dispatch context: holds references needed by all handlers. @@ -52,7 +55,47 @@ impl DispatchCtx<'_> { self.scope.auth() } - pub(super) fn vshard_for_key(&self, key: &str) -> VShardId { - VShardId::from_key(key.as_bytes()) + /// The vShard a direct-op task carries. + /// + /// A graph plan is homed by node key: an edge write by its source node, + /// any other graph op by `document_id`, else the collection name. Every + /// other plan is homed by its collection's canonical key, the bare name + /// in this request's database. That is the vShard the planner, the + /// gateway and the staging gate use for the same collection. + pub(super) fn task_vshard( + &self, + plan: &PhysicalPlan, + document_id: Option<&str>, + collection: &str, + ) -> VShardId { + if let PhysicalPlan::Graph(op) = plan { + let node_key = match op { + GraphOp::EdgePut { src_id, .. } | GraphOp::EdgeDelete { src_id, .. } => { + src_id.as_str() + } + GraphOp::EdgePutBatch { .. } + | GraphOp::ResolveEdgeDelete(_) + | GraphOp::EdgeDeleteBatch { .. } + | GraphOp::Hop { .. } + | GraphOp::Neighbors { .. } + | GraphOp::NeighborsMulti { .. } + | GraphOp::Path { .. } + | GraphOp::Subgraph { .. } + | GraphOp::RagFusion { .. } + | GraphOp::Algo { .. } + | GraphOp::Match { .. } + | GraphOp::MatchContinuation { .. } + | GraphOp::MatchVarLenResume { .. } + | GraphOp::BspSuperstep(_) + | GraphOp::WccSuperstep(_) + | GraphOp::SetNodeLabels { .. } + | GraphOp::RemoveNodeLabels { .. } + | GraphOp::TemporalNeighbors { .. } + | GraphOp::TemporalAlgorithm { .. } + | GraphOp::Stats { .. } => document_id.unwrap_or(collection), + }; + return VShardId::from_key(node_key.as_bytes()); + } + CollectionKey::from_bare(self.database_id(), collection).vshard() } } diff --git a/nodedb/src/control/server/native/dispatch/direct_ops.rs b/nodedb/src/control/server/native/dispatch/direct_ops.rs index a8585200f..bff3d3c9e 100644 --- a/nodedb/src/control/server/native/dispatch/direct_ops.rs +++ b/nodedb/src/control/server/native/dispatch/direct_ops.rs @@ -33,8 +33,6 @@ pub(crate) async fn handle_direct_op( .as_deref() .unwrap_or("default") .to_lowercase(); - let vshard_key = fields.document_id.as_deref().unwrap_or(&collection); - let vshard_id = ctx.vshard_for_key(vshard_key); let tenant_id = ctx.tenant_id(); // CRDT Apply allocates a surrogate while planning, and a KV counter plans @@ -71,6 +69,7 @@ pub(crate) async fn handle_direct_op( Ok(p) => p, Err(e) => return error_to_native_with_sqlstate(seq, "42601", &e), }; + let vshard_id = ctx.task_vshard(&plan, fields.document_id.as_deref(), &collection); // Apply RLS before any special Control-Plane orchestration can observe the plan. if let Err(e) = crate::control::planner::rls_injection::inject_rls_for_single_plan( @@ -359,12 +358,14 @@ pub(crate) async fn handle_direct_op( } // Only reads to widen with are those materialized-sum settlement stamped // on the source rows its shipped balances folded from. + let sum_read_vshards = + match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) { + Ok(vshards) => vshards, + Err(error) => return error_to_native(seq, &error), + }; let route_to_calvin = !in_txn_block && matches!( - classify_dispatch( - &tasks, - &crate::control::planner::calvin::read_vshards_of(&sum_target_reads), - ), + classify_dispatch(&tasks, &sum_read_vshards), DispatchClass::MultiShard { .. } ); if route_to_calvin { diff --git a/nodedb/src/control/server/native/dispatch/graph_match.rs b/nodedb/src/control/server/native/dispatch/graph_match.rs index 401dcaddd..f659ee765 100644 --- a/nodedb/src/control/server/native/dispatch/graph_match.rs +++ b/nodedb/src/control/server/native/dispatch/graph_match.rs @@ -31,8 +31,6 @@ pub(crate) async fn handle_graph_match( .as_deref() .unwrap_or("default") .to_lowercase(); - let vshard_key = fields.document_id.as_deref().unwrap_or(&collection); - let vshard_id = ctx.vshard_for_key(vshard_key); let tenant_id = ctx.tenant_id(); if let Err(error) = super::limits::check_op_limits(ctx.state, fields) { @@ -47,6 +45,7 @@ pub(crate) async fn handle_graph_match( Ok(plan) => plan, Err(error) => return error_to_native_with_sqlstate(seq, "42601", &error), }; + let vshard_id = ctx.task_vshard(&plan, fields.document_id.as_deref(), &collection); if let Err(error) = crate::control::planner::rls_injection::inject_rls_for_single_plan( tenant_id.as_u64(), ctx.database_id(), diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs b/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs index 89e0de06e..5f1d19d5b 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs @@ -97,7 +97,11 @@ fn derive_surrogates( if pk.is_empty() { out.push(Surrogate::ZERO); } else { - out.push(assigner.assign(ctx.database_id(), ctx.tenant_id(), collection, &pk)?); + out.push(assigner.assign( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + &pk, + )?); } } Ok(out) diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs b/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs index 9da6c3706..da2c204c9 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs @@ -42,9 +42,8 @@ pub(crate) fn build_apply( }); let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, document_id.as_bytes(), )?; @@ -144,9 +143,8 @@ pub(crate) fn build_list_insert( })?; let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, document_id.as_bytes(), )?; @@ -170,9 +168,8 @@ pub(crate) fn build_list_delete( let index = require_list_index(fields.list_index, "list_index")?; let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, document_id.as_bytes(), )?; @@ -196,9 +193,8 @@ pub(crate) fn build_list_move( let to_index = require_list_index(fields.list_to_index, "list_to_index")?; let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, document_id.as_bytes(), )?; diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index eb664220a..4916459b3 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -44,7 +44,11 @@ pub(crate) fn build_point_get( let surrogate = ctx .state .surrogate_assigner - .lookup(ctx.database_id(), ctx.tenant_id(), collection, &pk_bytes)? + .lookup( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + &pk_bytes, + )? .unwrap_or(nodedb_types::Surrogate::ZERO); Ok(PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -70,9 +74,8 @@ pub(crate) fn build_point_put( Some(CollectionType::KeyValue(_)) => { let key = doc_id.into_bytes(); let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, &key, )?; Ok(PhysicalPlan::Kv(KvOp::Put { @@ -92,9 +95,8 @@ pub(crate) fn build_point_put( // The line's own surrogate keys its staged row, so a read later in // the same transaction observes it. let (surrogate, _identity) = ctx.state.surrogate_assigner.assign_fresh( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, )?; Ok(PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -116,9 +118,8 @@ pub(crate) fn build_point_put( Some(CollectionType::Document(_)) | None => { let pk_bytes = doc_id.as_bytes().to_vec(); let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, &pk_bytes, )?; Ok(PhysicalPlan::Document(DocumentOp::PointPut { @@ -171,7 +172,11 @@ pub(crate) fn build_point_delete( let surrogate = ctx .state .surrogate_assigner - .lookup(ctx.database_id(), ctx.tenant_id(), collection, &pk_bytes)? + .lookup( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + &pk_bytes, + )? .unwrap_or(nodedb_types::Surrogate::ZERO); Ok(PhysicalPlan::Document(DocumentOp::PointDelete { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -234,9 +239,8 @@ pub(crate) fn build_batch_insert( detail: format!("failed to serialize document '{}': {e}", d.id), })?; let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, d.id.as_bytes(), )?; documents.push((d.id.clone(), value_bytes)); @@ -278,7 +282,11 @@ pub(crate) fn build_update( let surrogate = ctx .state .surrogate_assigner - .lookup(ctx.database_id(), ctx.tenant_id(), collection, &pk_bytes)? + .lookup( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + &pk_bytes, + )? .unwrap_or(nodedb_types::Surrogate::ZERO); Ok(PhysicalPlan::Document(DocumentOp::PointUpdate { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -342,9 +350,8 @@ pub(crate) fn build_upsert( let doc_id = require_doc_id(fields)?; let value = fields.data.clone().unwrap_or_default(); let surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, doc_id.as_bytes(), )?; Ok(PhysicalPlan::Document(DocumentOp::Upsert { diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs b/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs index 2408807fa..a69c0c785 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs @@ -193,15 +193,13 @@ pub(crate) fn build_edge_put( None => String::new(), }; let src_surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, src.as_bytes(), )?; let dst_surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, dst.as_bytes(), )?; Ok(PhysicalPlan::Graph(GraphOp::EdgePut { @@ -247,15 +245,13 @@ pub(crate) fn build_edge_delete( // returns the existing node identities) so a cross-shard delete dual-homes // and locks against a concurrent insert of the same edge. let src_surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, src.as_bytes(), )?; let dst_surrogate = ctx.state.surrogate_assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, dst.as_bytes(), )?; Ok(PhysicalPlan::Graph(GraphOp::EdgeDelete { diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index b3e7c33fa..14232fafe 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -225,9 +225,11 @@ pub(super) fn assign_kv_surrogate( collection: &str, key: &[u8], ) -> crate::Result { - ctx.state - .surrogate_assigner - .assign(ctx.database_id(), ctx.tenant_id(), collection, key) + ctx.state.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + key, + ) } pub(crate) fn build_cas( diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs b/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs index dabd2eafa..37a682aac 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs @@ -75,9 +75,8 @@ pub(crate) fn build_batch_insert( let mut surrogates = Vec::with_capacity(vectors.len()); for _ in &vectors { surrogates.push(assigner.assign_anonymous( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, )?); } @@ -109,15 +108,17 @@ pub(crate) fn build_insert( let (surrogate, pk_bytes) = match fields.document_id.as_deref() { Some(pk) if !pk.is_empty() => ( assigner.assign( - ctx.database_id(), + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), ctx.tenant_id(), - collection, pk.as_bytes(), )?, Some(pk.as_bytes().to_vec()), ), _ => ( - assigner.assign_anonymous(ctx.database_id(), ctx.tenant_id(), collection)?, + assigner.assign_anonymous( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + )?, None, ), }; diff --git a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs index fa521357c..665ee98f9 100644 --- a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs @@ -124,7 +124,8 @@ async fn dispatch_external_crdt_apply( .map_err(crate::Error::from)?; let task = nodedb_physical::physical_task::PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(ctx.database_id(), collection.as_str()), + vshard_id: nodedb_types::CollectionKey::from_qualified(ctx.database_id(), &collection)? + .vshard(), database_id: ctx.database_id(), plan, post_set_op: nodedb_physical::physical_task::PostSetOp::None, diff --git a/nodedb/src/control/server/native/dispatch/sql.rs b/nodedb/src/control/server/native/dispatch/sql.rs index fc2ec9bd7..bb3450bdc 100644 --- a/nodedb/src/control/server/native/dispatch/sql.rs +++ b/nodedb/src/control/server/native/dispatch/sql.rs @@ -316,10 +316,12 @@ async fn execute_planned( // sequencer so it commits atomically. Single-shard (and best-effort) keep // the existing per-task gateway/SPSC dispatch loop below unchanged. // Autocommit single-statement dispatch: no session read-set to widen with. - match classify_dispatch( - &tasks, - &crate::control::planner::calvin::read_vshards_of(&sum_target_reads), - ) { + let sum_read_vshards = match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) + { + Ok(vshards) => vshards, + Err(error) => return resp(error_to_native(seq, &error)), + }; + match classify_dispatch(&tasks, &sum_read_vshards) { DispatchClass::SingleShard { .. } => {} DispatchClass::MultiShard { .. } => { // Dispatching to Calvin here applies the statement durably at diff --git a/nodedb/src/control/server/pgwire/handler/facet.rs b/nodedb/src/control/server/pgwire/handler/facet.rs index dbd7b1d4c..b17488135 100644 --- a/nodedb/src/control/server/pgwire/handler/facet.rs +++ b/nodedb/src/control/server/pgwire/handler/facet.rs @@ -15,7 +15,7 @@ use sonic_rs; use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::SessionId; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use nodedb_physical::physical_plan::QueryOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -37,7 +37,7 @@ pub(super) async fn execute_facet_counts_sql( .sessions .get_current_database(session_id) .unwrap_or(DatabaseId::DEFAULT); - let vshard = VShardId::from_collection_in_database(database_id, &parsed.collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &parsed.collection).vshard(); // Convert filter text to ScanFilter predicates. let filter_bytes = if parsed.filter.is_empty() { @@ -101,7 +101,7 @@ pub(super) async fn execute_search_with_facets_sql( .sessions .get_current_database(session_id) .unwrap_or(DatabaseId::DEFAULT); - let vshard = VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let filter_bytes = if filter_text.is_empty() { Vec::new() diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute.rs b/nodedb/src/control/server/pgwire/handler/routing/execute.rs index 6070bce06..0d890968f 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute.rs @@ -203,7 +203,18 @@ impl NodeDbPgHandler { // Autocommit statement routing: the only reads to widen with are the // ones the materialized-sum settlement stamped on the source rows its // shipped balances were folded from. - let sum_read_vshards = crate::control::planner::calvin::read_vshards_of(&sum_target_reads); + let sum_read_vshards = + match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) { + Ok(vshards) => vshards, + Err(error) => { + let (severity, code, message) = error_to_sqlstate(&error); + return Err(PgWireError::UserError(Box::new(ErrorInfo::new( + severity.to_owned(), + code.to_owned(), + message, + )))); + } + }; match classify_dispatch(&tasks, &sum_read_vshards) { DispatchClass::SingleShard { .. } => { // A single-shard dependent-predicate write (e.g. `DELETE ... diff --git a/nodedb/src/control/server/resp/gateway_dispatch.rs b/nodedb/src/control/server/resp/gateway_dispatch.rs index 3f3b50f97..bea9b39b5 100644 --- a/nodedb/src/control/server/resp/gateway_dispatch.rs +++ b/nodedb/src/control/server/resp/gateway_dispatch.rs @@ -44,7 +44,7 @@ pub(super) async fn dispatch_kv( // `RequestAuthScope::builder` so the dispatched task and `$auth.database_id` // resolve from the same value and cannot drift apart. let database_id = DatabaseId::DEFAULT; - let vshard = VShardId::from_collection_in_database(database_id, &session.collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &session.collection).vshard(); // Extracted before `plan` is moved into `authorize_resp_task`, which // consumes it for RLS injection and task construction — metering needs // the collection/engine shape after dispatch succeeds below, and by then @@ -108,7 +108,7 @@ pub(super) async fn dispatch_kv_write( // DatabaseId::DEFAULT is deliberate here, resolved once and threaded // through `authorize_resp_task` via `RequestAuthScope::builder`. let database_id = DatabaseId::DEFAULT; - let vshard = VShardId::from_collection_in_database(database_id, &session.collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &session.collection).vshard(); // See `dispatch_kv` above: extracted before `authorize_resp_task` moves // `plan`, since metering needs the plan shape after dispatch succeeds. let plan_metering_info = state @@ -454,7 +454,8 @@ mod tests { surrogate_ceiling: None, }); let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &session.collection); + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &session.collection) + .vshard(); let result = authorize_resp_task( &state, diff --git a/nodedb/src/control/server/resp/handler_hash.rs b/nodedb/src/control/server/resp/handler_hash.rs index e33d19ea7..0c4dce8ca 100644 --- a/nodedb/src/control/server/resp/handler_hash.rs +++ b/nodedb/src/control/server/resp/handler_hash.rs @@ -130,9 +130,11 @@ pub(super) async fn handle_hset( // Content-addressed cross-engine identity so the merged row keeps the // surrogate its original insert assigned. let surrogate = match state.surrogate_assigner.assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare( + nodedb_types::DatabaseId::DEFAULT, + &session.collection, + ), session.tenant_id, - &session.collection, &key, ) { Ok(s) => s, diff --git a/nodedb/src/control/server/resp/handler_kv/surrogate.rs b/nodedb/src/control/server/resp/handler_kv/surrogate.rs index 34624523c..7fa6544d1 100644 --- a/nodedb/src/control/server/resp/handler_kv/surrogate.rs +++ b/nodedb/src/control/server/resp/handler_kv/surrogate.rs @@ -19,9 +19,11 @@ pub(super) fn resp_kv_surrogate( state .surrogate_assigner .assign( - crate::types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare( + crate::types::DatabaseId::DEFAULT, + &session.collection, + ), session.tenant_id, - &session.collection, key, ) .map_err(|e| RespValue::err(format!("ERR {e}"))) diff --git a/nodedb/src/control/server/resp/handler_sorted.rs b/nodedb/src/control/server/resp/handler_sorted.rs index b82cda2bc..ce2141017 100644 --- a/nodedb/src/control/server/resp/handler_sorted.rs +++ b/nodedb/src/control/server/resp/handler_sorted.rs @@ -65,9 +65,8 @@ pub(super) async fn handle_zadd( .unwrap_or_default(); let surrogate = match state.surrogate_assigner.assign( - crate::types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(crate::types::DatabaseId::DEFAULT, &index_name), session.tenant_id, - &index_name, &member, ) { Ok(s) => s, diff --git a/nodedb/src/control/server/response_translate/vector.rs b/nodedb/src/control/server/response_translate/vector.rs index 10eac7e97..f12731050 100644 --- a/nodedb/src/control/server/response_translate/vector.rs +++ b/nodedb/src/control/server/response_translate/vector.rs @@ -88,7 +88,9 @@ fn apply_rls_filter(hits: &mut Vec, rls_filters: &[u8], top_k: usize) { /// user's PK through this single catalog call. Returns `None` when the /// catalog has no PK mapping for the surrogate (headless row, or a document /// that was never written) — callers must leave the row's identifier -/// untouched in that case rather than fabricate a value. +/// untouched in that case rather than fabricate a value. `collection` is the +/// plan's database-qualified name, de-qualified into the canonical key; a +/// name that does not de-qualify also resolves to `None`. pub(crate) fn resolve_surrogate_pk( state: &SharedState, database_id: DatabaseId, @@ -97,8 +99,9 @@ pub(crate) fn resolve_surrogate_pk( surrogate: Surrogate, ) -> Option { let catalog = state.credentials.catalog(); + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, collection).ok()?; let pk_bytes = catalog - .get_pk_for_surrogate(database_id, tenant_id, collection, surrogate) + .get_pk_for_surrogate(key, tenant_id, surrogate) .ok()??; String::from_utf8(pk_bytes).ok() } diff --git a/nodedb/src/control/server/shared/clone_write/document.rs b/nodedb/src/control/server/shared/clone_write/document.rs index 095cc4b6c..3469da3f1 100644 --- a/nodedb/src/control/server/shared/clone_write/document.rs +++ b/nodedb/src/control/server/shared/clone_write/document.rs @@ -121,9 +121,8 @@ fn source_surrogate( state .surrogate_assigner .lookup( - source_db_id, + nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)?, tenant_id, - source_coll_qualified, document_id.as_bytes(), ) .map_err(|e| write_err(format!("clone write source surrogate lookup: {e}"))) diff --git a/nodedb/src/control/server/shared/clone_write/kv.rs b/nodedb/src/control/server/shared/clone_write/kv.rs index d534e9221..1fefc79da 100644 --- a/nodedb/src/control/server/shared/clone_write/kv.rs +++ b/nodedb/src/control/server/shared/clone_write/kv.rs @@ -13,7 +13,6 @@ use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::server::shared::authorization::authorize_collection; use crate::control::server::shared::sql::staging_predicates::require_affected_count; use crate::control::state::SharedState; -use crate::types::VShardId; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use nodedb_physical::physical_task::PhysicalTask; @@ -173,7 +172,9 @@ pub(super) async fn intercept_kv_clone_write( rls_filters: rls_filters.clone(), provenance: None, }); - let vshard_id = VShardId::from_collection_in_database(db_id, collection_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)? + .vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, delete_plan) .await .map_err(|e| write_err(format!("clone kv delete dispatch: {e}")))?; diff --git a/nodedb/src/control/server/shared/clone_write/probes.rs b/nodedb/src/control/server/shared/clone_write/probes.rs index 6f7d6385b..071feb778 100644 --- a/nodedb/src/control/server/shared/clone_write/probes.rs +++ b/nodedb/src/control/server/shared/clone_write/probes.rs @@ -38,7 +38,8 @@ pub(super) async fn probe_row_in_target( valid_at_ms: None, }); let plan = with_caller_rls(state, identity, tenant_id, db_id, plan)?; - let vshard_id = VShardId::from_collection_in_database(db_id, collection_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)?.vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, plan).await?; Ok(!resp.payload.is_empty() && resp.status == Status::Ok) } @@ -65,7 +66,9 @@ pub(super) async fn fetch_source_row( valid_at_ms: None, }); let plan = with_caller_rls(state, identity, tenant_id, source_db_id, plan)?; - let vshard_id = VShardId::from_collection_in_database(source_db_id, source_coll_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)? + .vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, source_db_id, plan).await?; if resp.payload.is_empty() || resp.status != Status::Ok { return Ok(None); @@ -94,7 +97,8 @@ pub(super) async fn probe_kv_key_in_target( surrogate_ceiling: None, }); let plan = with_caller_rls(state, identity, tenant_id, db_id, plan)?; - let vshard_id = VShardId::from_collection_in_database(db_id, collection_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)?.vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, plan).await?; Ok(!resp.payload.is_empty() && resp.status == Status::Ok) } @@ -120,7 +124,9 @@ pub(super) async fn fetch_kv_source_value( surrogate_ceiling: None, }); let plan = with_caller_rls(state, identity, tenant_id, source_db_id, plan)?; - let vshard_id = VShardId::from_collection_in_database(source_db_id, source_coll_qualified); + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)? + .vshard(); let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, source_db_id, plan).await?; if resp.payload.is_empty() || resp.status != Status::Ok { return Ok(None); diff --git a/nodedb/src/control/server/shared/ddl/engine_apply.rs b/nodedb/src/control/server/shared/ddl/engine_apply.rs index 14e79df1f..baae6f168 100644 --- a/nodedb/src/control/server/shared/ddl/engine_apply.rs +++ b/nodedb/src/control/server/shared/ddl/engine_apply.rs @@ -35,8 +35,7 @@ pub(crate) async fn apply_in_engine( SystemTask::new( SystemReason::DdlApply, tenant_id, - database_id, - collection, + nodedb_types::CollectionKey::from_bare(database_id, collection), plan, ), timeout, @@ -74,8 +73,7 @@ pub(crate) async fn refuse_materialized_vector_index( SystemTask::new( SystemReason::DdlApply, tenant_id, - database_id, - collection, + nodedb_types::CollectionKey::from_bare(database_id, collection), plan, ), timeout, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs index a7a00536d..a51501b33 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs @@ -260,7 +260,7 @@ pub async fn insert_document( Err(e) => return Some(Err(e)), }; let vec_vshard = - crate::types::VShardId::from_collection_in_database(database_id, &parsed.coll_name); + nodedb_types::CollectionKey::from_bare(database_id, &parsed.coll_name).vshard(); for (field_name, vector) in extract_vector_fields(&fields) { if indexed.contains(&field_name) { continue; @@ -292,9 +292,8 @@ pub async fn insert_document( } } let surrogate = match state.surrogate_assigner.assign( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &parsed.coll_name), tenant_id, - &parsed.coll_name, parsed.doc_id.as_bytes(), ) { Ok(s) => s, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index 85d9a919f..eb1f7ef8d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -266,13 +266,12 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a return Err(ddl_err(sqlstate, message)); } + let sum_read_vshards = crate::control::planner::calvin::read_vshards_of(&sum_target_reads) + .map_err(|error| DdlError::from_error(&error))?; if !in_txn_block && state.sequencer_inbox.get().is_some() && matches!( - crate::control::planner::calvin::classify_dispatch( - &tasks, - &crate::control::planner::calvin::read_vshards_of(&sum_target_reads), - ), + crate::control::planner::calvin::classify_dispatch(&tasks, &sum_read_vshards), crate::control::planner::calvin::DispatchClass::MultiShard { .. } ) { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs index 8a6914853..d60408641 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs @@ -71,7 +71,7 @@ pub(crate) async fn build_secondary_index( // The backfill runs on the local Data Plane (single node) or the leader // (cluster), vShard-local per core. - let vshard = crate::types::VShardId::from_collection_in_database(database_id, collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); let backfill_plan = crate::bridge::envelope::PhysicalPlan::Document( nodedb_physical::physical_plan::DocumentOp::BackfillIndex { collection: nodedb_types::QualifiedCollection::new(database_id, collection), diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs index 681c7b36c..4a4086804 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs @@ -16,7 +16,7 @@ use crate::control::security::catalog::StoredCollection; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, TenantId, TraceId}; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use nodedb_types::QualifiedCollection; @@ -122,7 +122,7 @@ async fn dispatch_durable( crate::control::server::dispatch_utils::AutocommitWrite { tenant_id, database_id, - vshard_id: VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), plan, trace_id: TraceId::ZERO, event_source: crate::event::EventSource::User, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs index cde9fccd0..501ab7f24 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs @@ -182,8 +182,7 @@ async fn vector( // // The record's outcome-floor window opens before the append and closes // from the drop's outcome. - let vshard = - crate::types::VShardId::from_collection_in_database(database_id, &record.collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &record.collection).vshard(); let owner = RecordOwner { tenant_id, database_id, @@ -307,7 +306,7 @@ pub(crate) async fn dispatch( plan: crate::bridge::envelope::PhysicalPlan, minted: Option, ) -> Result<(), DdlError> { - let vshard = crate::types::VShardId::from_collection_in_database(database_id, collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); let response = crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( state, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs index b8b550691..898b05a2d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs @@ -52,7 +52,7 @@ pub async fn dispatch_unregister_collection( // cannot resolve it, so the files are never orphaned. let homing_core = dispatcher .router() - .resolve(VShardId::from_collection_in_database(database, name)) + .resolve(nodedb_types::CollectionKey::from_bare(database, name).vshard()) .unwrap_or(0); for core_id in 0..num_cores { let request_id = state.next_request_id(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs index de45cf73c..8f8540f15 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs @@ -249,8 +249,7 @@ pub async fn create_continuous_aggregate( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::CatalogMaintenance, tenant_id, - database_id, - &def.source, + nodedb_types::CollectionKey::from_bare(database_id, &def.source), plan, ), Duration::from_secs(5), diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs index 319a35247..324a2eecc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs @@ -114,8 +114,7 @@ pub async fn drop_continuous_aggregate( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::CatalogMaintenance, tenant_id, - database_id, - &stored.source, + nodedb_types::CollectionKey::from_bare(database_id, &stored.source), plan, ), Duration::from_secs(5), diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs index a87d441ee..c93d283b6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs @@ -50,8 +50,10 @@ pub async fn register_persisted_continuous_aggregates(state: &SharedState) { sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::CatalogMaintenance, tenant_id, - crate::types::DatabaseId::new(def.database_id), - &def.source, + nodedb_types::CollectionKey::from_bare( + crate::types::DatabaseId::new(def.database_id), + &def.source, + ), plan, ), Duration::from_secs(5), diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs index 2a4a098da..e65c7c397 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs @@ -77,8 +77,7 @@ pub async fn show_continuous_aggregates( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::CatalogMaintenance, tenant_id, - database_id, - "__system", + nodedb_types::CollectionKey::from_bare(database_id, "__system"), PhysicalPlan::Meta(MetaOp::ListContinuousAggregates), ), Duration::from_secs(5), diff --git a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs index 2077ef341..649498ac8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs @@ -103,8 +103,7 @@ pub async fn convert_collection( SystemTask::new( SystemReason::DdlApply, tenant_id, - database_id, - &collection, + nodedb_types::CollectionKey::from_bare(database_id, &collection), plan, ), Duration::from_secs(60), diff --git a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs index f733a5a08..6e652b1f0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs @@ -162,7 +162,11 @@ pub async fn crdt_apply( let surrogate = state .surrogate_assigner - .assign(database_id, tenant_id, collection, document_id.as_bytes()) + .assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + document_id.as_bytes(), + ) .map_err(|e| DdlError::new("XX000", e.to_string()))?; let plan = PhysicalPlan::Crdt(CrdtOp::Apply { @@ -179,7 +183,7 @@ pub async fn crdt_apply( }); let task = PhysicalTask { tenant_id, - vshard_id: crate::types::VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), database_id, plan, post_set_op: PostSetOp::None, diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs index 0d6a1b2a1..8c29757c9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs @@ -97,7 +97,11 @@ pub async fn crdt_merge( let target_surrogate = state .surrogate_assigner - .assign(database_id, tenant_id, collection, target_id.as_bytes()) + .assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + target_id.as_bytes(), + ) .map_err(|e| ddl_err("XX000", e.to_string()))?; let apply_plan = PhysicalPlan::Crdt(CrdtOp::Apply { @@ -114,7 +118,7 @@ pub async fn crdt_merge( }); let task = PhysicalTask { tenant_id, - vshard_id: crate::types::VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), database_id, plan: apply_plan, post_set_op: PostSetOp::None, diff --git a/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs b/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs index b1d278557..faf303eb3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs @@ -51,7 +51,7 @@ pub async fn estimate_count( gate.refuse_if_read_policy(&coll, "ESTIMATE_COUNT")?; gate.refuse_if_field_redacted(&coll, &field, "the estimated count")?; - let vshard = crate::types::VShardId::from_collection_in_database(database_id, &coll); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &coll).vshard(); let plan = PhysicalPlan::Document(DocumentOp::EstimateCount { collection: nodedb_types::QualifiedCollection::new(database_id, &coll), field, diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs index 35d568487..b146666c8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs @@ -80,28 +80,15 @@ pub async fn insert_edge( let vsrc = VShardId::from_key(src.as_bytes()); let vdst = VShardId::from_key(dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - &collection, - src.as_bytes(), - TraceId::ZERO, - ) - .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - &collection, - dst.as_bytes(), - TraceId::ZERO, - ) - .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + let key = nodedb_types::CollectionKey::from_bare(database_id, &collection); + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), TraceId::ZERO) + .await + .map_err(|e| ddl_err("XX000", e.to_string()))?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), TraceId::ZERO) + .await + .map_err(|e| ddl_err("XX000", e.to_string()))?; // Write policy decides the `PROPERTIES` image before staging: this handler // dispatches as trusted internal work, so nothing downstream resolves a policy. @@ -256,28 +243,15 @@ pub async fn delete_edge( let vsrc = VShardId::from_key(src.as_bytes()); let vdst = VShardId::from_key(dst.as_bytes()); - let src_surrogate = assign_surrogate_routed( - state, - vsrc, - database_id, - tenant_id, - &collection, - src.as_bytes(), - TraceId::ZERO, - ) - .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; - let dst_surrogate = assign_surrogate_routed( - state, - vdst, - database_id, - tenant_id, - &collection, - dst.as_bytes(), - TraceId::ZERO, - ) - .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + let key = nodedb_types::CollectionKey::from_bare(database_id, &collection); + let src_surrogate = + assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), TraceId::ZERO) + .await + .map_err(|e| ddl_err("XX000", e.to_string()))?; + let dst_surrogate = + assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), TraceId::ZERO) + .await + .map_err(|e| ddl_err("XX000", e.to_string()))?; // A delete carries no image, so the policy compiles into the plan's write-gate // slot and is decided in the Data Plane against the edge's stored properties. diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs index ed7cf5a72..baba35e3a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs @@ -10,7 +10,7 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use nodedb_physical::physical_plan::{KvCounterShape, KvOp, PhysicalPlan}; use nodedb_sql::planner::dml_helpers::KvCounterKind; @@ -52,13 +52,12 @@ pub async fn kv_incr( let ttl_ms = parse_optional_ttl(&args[3..])?; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, key.as_bytes(), ) .map_err(|e| ddl_err("XX000", e.to_string()))?; @@ -120,13 +119,12 @@ pub async fn kv_incr_float( )); } - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, key.as_bytes(), ) .map_err(|e| ddl_err("XX000", e.to_string()))?; @@ -213,13 +211,12 @@ pub async fn kv_cas( let expected = unquote(&args[2]); let new_value = unquote(&args[3]); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, key.as_bytes(), ) .map_err(|e| ddl_err("XX000", e.to_string()))?; @@ -267,13 +264,12 @@ pub async fn kv_getset( let key = unquote(&args[1]); let new_value = unquote(&args[2]); - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, key.as_bytes(), ) .map_err(|e| ddl_err("XX000", e.to_string()))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs index be8cb49dc..ca2b0d8af 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs @@ -45,7 +45,7 @@ pub struct SortedIndexTarget<'a> { impl SortedIndexTarget<'_> { /// The vShard holding both the collection's rows and its index trees. fn vshard(&self) -> VShardId { - VShardId::from_collection_in_database(self.database_id, self.collection) + nodedb_types::CollectionKey::from_bare(self.database_id, self.collection).vshard() } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs index 3c68e9105..3a593f440 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs @@ -43,7 +43,7 @@ pub async fn handle_show_vector_index( // or: SHOW VECTOR INDEX status ON let (collection, field_name) = parse_collection_column(sql, " ON ")?; let tenant_id = identity.tenant_id; - let vshard = crate::types::VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let plan = PhysicalPlan::Vector(VectorOp::QueryStats { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -140,7 +140,7 @@ pub async fn handle_alter_vector_index_seal( ) -> Result, DdlError> { let (collection, field_name) = parse_collection_column(sql, " ON ")?; let tenant_id = identity.tenant_id; - let vshard = crate::types::VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let plan = PhysicalPlan::Vector(VectorOp::Seal { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -173,7 +173,7 @@ pub async fn handle_alter_vector_index_compact( ) -> Result, DdlError> { let (collection, field_name) = parse_collection_column(sql, " ON ")?; let tenant_id = identity.tenant_id; - let vshard = crate::types::VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let plan = PhysicalPlan::Vector(VectorOp::CompactIndex { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), diff --git a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs index c75e7183d..d7a835cf4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs @@ -135,8 +135,7 @@ async fn source_group_barrier(state: &SharedState, sources: &[String]) -> crate: }; let mut targets: Vec = Vec::new(); for source in sources { - let vshard = - crate::types::VShardId::from_collection_in_database(DatabaseId::DEFAULT, source); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, source).vshard(); let group_id = crate::control::security::auth_fence::cluster::group_of_vshard(state, vshard.as_u32())?; if targets.iter().any(|target| target.group_id == group_id) { diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs index ca33bc44f..042c54865 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -51,11 +51,15 @@ pub async fn balance_as_of( gate.refuse_if_field_redacted(&collection, &column, "the as-of balance")?; // Read current balance from the target document. - let vshard = VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let pk_bytes = key.as_bytes().to_vec(); let surrogate = state .surrogate_assigner - .lookup(database_id, tenant_id, &collection, &pk_bytes) + .lookup( + nodedb_types::CollectionKey::from_bare(database_id, &collection), + tenant_id, + &pk_bytes, + ) .map_err(|e| err("XX000", &format!("surrogate lookup failed: {e}")))? .unwrap_or(nodedb_types::Surrogate::ZERO); let mut get_plan = @@ -115,7 +119,7 @@ pub async fn balance_as_of( // Scan the source collection for rows where join_column = key AND created_at > as_of. let source_vshard = - VShardId::from_collection_in_database(database_id, &mat_def.source_collection); + nodedb_types::CollectionKey::from_bare(database_id, &mat_def.source_collection).vshard(); let mut source_scan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new( diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs index 81fc44d0b..62a1cff67 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -72,7 +72,7 @@ pub async fn convert_currency_lookup( let key_value = format!("{from_ccy}/{to_ccy}"); // Scan rate table to find latest rate where key_column == key_value AND time_column <= as_of. - let vshard = VShardId::from_collection_in_database(database_id, &rate_table); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &rate_table).vshard(); let mut scan_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, &rate_table), limit: usize::MAX, diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs index 2093d0195..2660b40c2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs @@ -10,7 +10,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -46,7 +46,7 @@ pub async fn temporal_lookup( gate.require_document_engine(&table, "TEMPORAL_LOOKUP")?; // Scan the table. - let vshard = VShardId::from_collection_in_database(database_id, &table); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &table).vshard(); let mut scan_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, &table), limit: usize::MAX, diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs index b6e7a25fa..70229796d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -70,7 +70,7 @@ pub async fn verify_balance( gate.refuse_if_any_redaction(&mat_def.source_collection, "the balance verification")?; // Scan all target rows. - let target_vshard = VShardId::from_collection_in_database(database_id, &collection); + let target_vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let mut target_scan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), @@ -107,7 +107,7 @@ pub async fn verify_balance( // Scan all source rows. let source_vshard = - VShardId::from_collection_in_database(database_id, &mat_def.source_collection); + nodedb_types::CollectionKey::from_bare(database_id, &mat_def.source_collection).vshard(); let mut source_scan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new( diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs index 4acda9e41..e97086ffe 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -44,7 +44,7 @@ pub async fn verify_hash_chain( gate.refuse_if_any_redaction(&collection, "the hash chain")?; // Scan all documents. - let vshard = VShardId::from_collection_in_database(database_id, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let mut scan_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), limit: usize::MAX, diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index c5ed236da..d8f4a8f0d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -54,7 +54,8 @@ pub async fn rate_check( let rate_key = format!("_rate:{gate_name}:{key}"); let tenant_id = identity.tenant_id; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, RATE_COLLECTION); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION).vshard(); let ttl_ms = window_secs * 1000; // Fixed-window semantics: TTL is set ONLY on the first call (new key). @@ -94,9 +95,8 @@ pub async fn rate_check( let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION), tenant_id, - RATE_COLLECTION, rate_key.as_bytes(), ) .map_err(|e| super::kv_atomic::ddl_err("XX000", e.to_string()))?; @@ -174,7 +174,8 @@ pub async fn rate_remaining( let rate_key = format!("_rate:{gate_name}:{key}"); let tenant_id = identity.tenant_id; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, RATE_COLLECTION); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION).vshard(); // Read current counter value (non-destructive). let plan = PhysicalPlan::Kv(KvOp::Get { @@ -245,7 +246,8 @@ pub async fn rate_reset( let rate_key = format!("_rate:{gate_name}:{key}"); let tenant_id = identity.tenant_id; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, RATE_COLLECTION); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION).vshard(); let plan = PhysicalPlan::Kv(KvOp::Delete { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, RATE_COLLECTION), diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/cutover.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/cutover.rs index 5dd704a9b..ec0efcdea 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/cutover.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/cutover.rs @@ -136,8 +136,7 @@ async fn dispatch_rename_ops( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::TenantLifecycle, tenant_id, - target_db_id, - "__system", + nodedb_types::CollectionKey::from_bare(target_db_id, "__system"), plan, ), RENAME_DISPATCH_TIMEOUT, diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs index 5699bf45d..e1601d23d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/move_tenant/snapshot.rs @@ -45,8 +45,7 @@ pub async fn run( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::TenantLifecycle, tenant_id, - source_db_id, - "__system", + nodedb_types::CollectionKey::from_bare(source_db_id, "__system"), plan, ), timeout, diff --git a/nodedb/src/control/server/shared/ddl/neutral/tenant/purge.rs b/nodedb/src/control/server/shared/ddl/neutral/tenant/purge.rs index 9d1b9de0e..e3e0417ee 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tenant/purge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tenant/purge.rs @@ -79,8 +79,7 @@ pub async fn purge_tenant( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::TenantLifecycle, tenant_id, - database_id, - "__system", + nodedb_types::CollectionKey::from_bare(database_id, "__system"), plan, ), std::time::Duration::from_secs(300), diff --git a/nodedb/src/control/server/shared/ddl/neutral/transfer.rs b/nodedb/src/control/server/shared/ddl/neutral/transfer.rs index 1fa8279a0..20c632b13 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/transfer.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/transfer.rs @@ -19,7 +19,7 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use super::super::result::{DdlError, DdlResult}; @@ -58,7 +58,7 @@ pub async fn transfer( return Err(ddl_err("42601", "TRANSFER: amount must be positive")); } - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); // Dispatch to Data Plane — entire read+validate+write is atomic (single TPC // core). Routed through the protocol-neutral in-transaction staging gate @@ -76,18 +76,16 @@ pub async fn transfer( let debit_surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, &source_bytes, ) .map_err(|e| ddl_err("XX000", e.to_string()))?; let credit_surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), identity.tenant_id, - &collection, &dest_bytes, ) .map_err(|e| ddl_err("XX000", e.to_string()))?; @@ -141,8 +139,10 @@ pub async fn transfer_item( // Cross-collection transfers must be on the same vshard. // Validate this upfront to prevent silent failures. - let vshard_src = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &source_collection); - let vshard_dst = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &dest_collection); + let vshard_src = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &source_collection).vshard(); + let vshard_dst = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &dest_collection).vshard(); if source_collection != dest_collection && vshard_src != vshard_dst { return Err(ddl_err( "0A000", @@ -163,9 +163,8 @@ pub async fn transfer_item( let surrogate = state .surrogate_assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &dest_collection), identity.tenant_id, - &dest_collection, &dest_bytes, ) .map_err(|e| ddl_err("XX000", e.to_string()))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs index c2b426a27..2b4a0973d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/create_index.rs @@ -210,11 +210,19 @@ pub async fn create_graph_index( let shard = VShardId::from_key(parent.as_bytes()); let src_surrogate = state .surrogate_assigner - .assign(database_id, tenant_id, &collection, parent.as_bytes()) + .assign( + nodedb_types::CollectionKey::from_bare(database_id, &collection), + tenant_id, + parent.as_bytes(), + ) .map_err(|e| ddl_err("XX000", e.to_string()))?; let dst_surrogate = state .surrogate_assigner - .assign(database_id, tenant_id, &collection, child.as_bytes()) + .assign( + nodedb_types::CollectionKey::from_bare(database_id, &collection), + tenant_id, + child.as_bytes(), + ) .map_err(|e| ddl_err("XX000", e.to_string()))?; edges_by_shard.entry(shard).or_default().push(BatchEdge { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), diff --git a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/sum.rs b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/sum.rs index ed725078a..642e692d5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/tree_ops/sum.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/tree_ops/sum.rs @@ -15,7 +15,7 @@ use crate::control::server::response_shape::types::ShapedRows; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; use crate::control::state::SharedState; use crate::engine::graph::traversal_options::GraphTraversalOptions; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; @@ -136,11 +136,16 @@ pub async fn tree_sum( // Without it, we fall back to scanning all tenant collections (O(N×C)). for node_id in &all_ids { for coll_name in &collections_to_search { - let coll_vshard = VShardId::from_collection_in_database(database_id, coll_name); + let coll_vshard = + nodedb_types::CollectionKey::from_bare(database_id, coll_name).vshard(); let pk_bytes = node_id.as_bytes().to_vec(); let surrogate = state .surrogate_assigner - .lookup(database_id, tenant_id, coll_name, &pk_bytes) + .lookup( + nodedb_types::CollectionKey::from_bare(database_id, coll_name), + tenant_id, + &pk_bytes, + ) .map_err(|e| ddl_err("XX000", format!("surrogate lookup: {e}")))? .unwrap_or(nodedb_types::Surrogate::ZERO); let mut get_plan = diff --git a/nodedb/src/control/server/shared/ddl/neutral/version_history/checkpoint.rs b/nodedb/src/control/server/shared/ddl/neutral/version_history/checkpoint.rs index 42c7c568e..9734c163c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/version_history/checkpoint.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/version_history/checkpoint.rs @@ -48,8 +48,7 @@ pub async fn create_checkpoint( SystemTask::new( SystemReason::CatalogMaintenance, tenant_id, - database_id, - &collection, + nodedb_types::CollectionKey::from_bare(database_id, &collection), plan, ), timeout, diff --git a/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs index f63eb717b..a900f72c1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs @@ -22,7 +22,7 @@ use crate::control::server::shared::clone_write::{ use crate::control::server::shared::ddl::sync_dispatch::dispatch_authorized; use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use super::super::super::result::DdlError; @@ -45,7 +45,7 @@ pub(super) async fn dispatch_authorized_read( let task = PhysicalTask { tenant_id: identity.tenant_id, database_id, - vshard_id: VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), plan, post_set_op: PostSetOp::None, txn_id: None, diff --git a/nodedb/src/control/server/shared/ddl/neutral/version_history/restore.rs b/nodedb/src/control/server/shared/ddl/neutral/version_history/restore.rs index 941e7f377..6db2fb94a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/version_history/restore.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/version_history/restore.rs @@ -64,7 +64,11 @@ pub async fn restore_version( let surrogate = state .surrogate_assigner - .assign(database_id, tenant_id, &collection, doc_id.as_bytes()) + .assign( + nodedb_types::CollectionKey::from_bare(database_id, &collection), + tenant_id, + doc_id.as_bytes(), + ) .map_err(|e| err("XX000", format!("surrogate assign: {e}")))?; let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); diff --git a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs index 27a10e040..417a3d84f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs @@ -88,7 +88,7 @@ pub async fn weighted_pick( } let tenant_id = identity.tenant_id; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); // `collection` is a caller argument, so the scan it names is authorized and // row-filtered here — a pick is a read of every row in the collection. The @@ -165,9 +165,11 @@ pub async fn weighted_pick( let audit_surrogate = state .surrogate_assigner .assign( - crate::types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare( + crate::types::DatabaseId::DEFAULT, + "_system_random_audit", + ), tenant_id, - "_system_random_audit", &audit_key_bytes, ) .map_err(|e| ddl_err("XX000", format!("WEIGHTED_PICK: audit surrogate bind: {e}")))?; @@ -193,10 +195,11 @@ pub async fn weighted_pick( crate::control::server::dispatch_utils::AutocommitWrite { tenant_id, database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::from_collection_in_database( + vshard_id: nodedb_types::CollectionKey::from_bare( DatabaseId::DEFAULT, "_system_random_audit", - ), + ) + .vshard(), plan: audit_plan, trace_id: TraceId::ZERO, event_source: crate::event::EventSource::User, diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs index 698f9eed9..ccf648cb3 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/dispatch.rs @@ -55,17 +55,17 @@ pub(crate) async fn dispatch_system_response_with_source( timeout: Duration, event_source: crate::event::EventSource, ) -> crate::Result { - let vshard_id = VShardId::from_collection_in_database(task.database_id, task.collection); + let vshard_id = task.collection.vshard(); tracing::trace!( reason = task.reason.label(), - collection = task.collection, + collection = task.collection.name(), "system-initiated data plane dispatch" ); dispatch_plan( state, PlanDispatch { tenant_id: task.tenant_id, - database_id: task.database_id, + database_id: task.collection.database_id(), vshard_id, plan: task.plan, timeout, @@ -92,6 +92,8 @@ pub(crate) async fn dispatch_system_response_with_source( /// mints that type, so an entry point that authorizes without running clone-write /// and clone-read interception fails to compile instead of silently reading a /// `Shadowed` clone target-locally. +/// +/// `collection` is the bare catalog name in the task's database. pub(crate) async fn dispatch_authorized( state: &SharedState, checked: CloneCheckedTask, @@ -99,7 +101,7 @@ pub(crate) async fn dispatch_authorized( timeout: Duration, ) -> crate::Result> { let task = checked.into_authorized().into_physical_task(); - let vshard_id = VShardId::from_collection_in_database(task.database_id, collection); + let vshard_id = nodedb_types::CollectionKey::from_bare(task.database_id, collection).vshard(); let tenant_id = task.tenant_id; let resp = dispatch_plan( state, diff --git a/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs b/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs index 8a0977df9..c70f2f4b5 100644 --- a/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs +++ b/nodedb/src/control/server/shared/ddl/sync_dispatch/system_task.rs @@ -18,7 +18,8 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::server::dispatch_utils::MintedRecords; -use crate::types::{DatabaseId, TenantId}; +use crate::types::TenantId; +use nodedb_types::CollectionKey; /// Why a Data-Plane dispatch carries no user identity. /// @@ -87,8 +88,9 @@ impl SystemReason { pub(crate) struct SystemTask<'a> { pub(super) reason: SystemReason, pub(super) tenant_id: TenantId, - pub(super) database_id: DatabaseId, - pub(super) collection: &'a str, + /// Canonical key of the collection the task homes to. Its database is the + /// task's database. + pub(super) collection: CollectionKey<'a>, pub(super) plan: PhysicalPlan, /// Records the caller appended for this task, under their outcome-floor /// window. `None` when the task appends nothing. @@ -104,14 +106,12 @@ impl<'a> SystemTask<'a> { pub(crate) fn new( reason: SystemReason, tenant_id: TenantId, - database_id: DatabaseId, - collection: &'a str, + collection: CollectionKey<'a>, plan: PhysicalPlan, ) -> Self { Self { reason, tenant_id, - database_id, collection, plan, minted: None, diff --git a/nodedb/src/control/server/shared/ddl/user_dispatch.rs b/nodedb/src/control/server/shared/ddl/user_dispatch.rs index b829bf933..137f4e351 100644 --- a/nodedb/src/control/server/shared/ddl/user_dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/user_dispatch.rs @@ -22,7 +22,7 @@ use crate::control::server::shared::metering::{ }; use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; -use crate::types::{DatabaseId, VShardId}; +use crate::types::DatabaseId; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::sync_dispatch::dispatch_authorized; @@ -210,7 +210,7 @@ async fn authorize_for_identity( let task = PhysicalTask { tenant_id: identity.tenant_id, - vshard_id: VShardId::from_collection_in_database(scope.database_id(), collection), + vshard_id: nodedb_types::CollectionKey::from_bare(scope.database_id(), collection).vshard(), database_id: scope.database_id(), plan, post_set_op: PostSetOp::None, diff --git a/nodedb/src/control/server/shared/session/commit/metering.rs b/nodedb/src/control/server/shared/session/commit/metering.rs index 63827869e..2320f95c5 100644 --- a/nodedb/src/control/server/shared/session/commit/metering.rs +++ b/nodedb/src/control/server/shared/session/commit/metering.rs @@ -56,7 +56,7 @@ mod tests { use crate::control::security::identity::{ AuthMethod, AuthenticatedIdentity, DatabaseSet, Role, }; - use crate::types::{DatabaseId, TenantId, VShardId}; + use crate::types::{DatabaseId, TenantId}; use crate::wal::WalManager; use super::*; @@ -93,7 +93,8 @@ mod tests { fn buffered_task(plan: PhysicalPlan) -> PhysicalTask { PhysicalTask { tenant_id: TenantId::new(1), - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, "widgets"), + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "widgets") + .vshard(), database_id: DatabaseId::DEFAULT, plan, post_set_op: PostSetOp::None, diff --git a/nodedb/src/control/server/shared/session/commit/run.rs b/nodedb/src/control/server/shared/session/commit/run.rs index 34873c1b3..ce36a6e52 100644 --- a/nodedb/src/control/server/shared/session/commit/run.rs +++ b/nodedb/src/control/server/shared/session/commit/run.rs @@ -103,7 +103,15 @@ pub async fn run_commit( // The interactive-COMMIT read-set widens dispatch classification: a txn that // writes shard X but read shard Y participates in {X, Y} and must route // through Calvin with Y as a participant. Autocommit has no session read-set. - let read_vshards = read_vshards_of(&read_set); + let read_vshards = match read_vshards_of(&read_set) { + Ok(vshards) => vshards, + Err(error) => { + reservation_release::release_and_rollback(state, sessions, session_id).await; + return CommitOutcome::Aborted { + reason: AbortReason::Dispatch(error), + }; + } + }; // In-transaction `MERGE`, `UPDATE ... FROM `, and `INSERT ... SELECT` // are resolved + staged into concrete, surrogate-carrying point writes diff --git a/nodedb/src/control/server/shared/session/read_set.rs b/nodedb/src/control/server/shared/session/read_set.rs index b119d5d7d..28bf1faec 100644 --- a/nodedb/src/control/server/shared/session/read_set.rs +++ b/nodedb/src/control/server/shared/session/read_set.rs @@ -267,7 +267,14 @@ pub async fn record_read_set( // Route the shared reservation to the SAME vshard the commit batch will use // for this key (write/commit routing derives the shard identically), so the // self-upgrade at commit finds the shared lock on the right scheduler. - let vshard = VShardId::from_collection_in_database(database_id, &collection).as_u32(); + // `collection` is the plan's database-qualified name. A name that does + // not de-qualify takes no reservation, and the read proceeds under OCC. + let Ok(collection_key) = + nodedb_types::CollectionKey::from_qualified_str(database_id, &collection) + else { + return; + }; + let vshard = collection_key.vshard().as_u32(); // Reuse the transaction's single reservation owner (None on the first hot-key // read; the assignment mints it, and `record_reservation` adopts it). Guard // dropped inside the accessor — nothing is held across the await below. diff --git a/nodedb/src/control/server/surrogate_exchange/hook.rs b/nodedb/src/control/server/surrogate_exchange/hook.rs index 34099fd58..d227768b2 100644 --- a/nodedb/src/control/server/surrogate_exchange/hook.rs +++ b/nodedb/src/control/server/surrogate_exchange/hook.rs @@ -67,14 +67,15 @@ impl nodedb_cluster::AssignRemoteSurrogate for RegistryAssignRemoteSurrogate { async fn on_assign_surrogate(&self, req: AssignSurrogateRequest) -> AssignSurrogateResponse { let database_id = DatabaseId::from(req.database_id); let tenant_id = TenantId::new(req.tenant_id); + // The request carries the bare catalog name. + let key = nodedb_types::CollectionKey::from_bare(database_id, &req.collection); if req.lookup_only == Some(true) { - return match self.state.surrogate_assigner.lookup( - database_id, - tenant_id, - &req.collection, - &req.pk, - ) { + return match self + .state + .surrogate_assigner + .lookup(key, tenant_id, &req.pk) + { Ok(Some(surrogate)) => AssignSurrogateResponse { surrogate: surrogate.as_u32(), error: None, @@ -99,7 +100,7 @@ impl nodedb_cluster::AssignRemoteSurrogate for RegistryAssignRemoteSurrogate { match self .state .surrogate_assigner - .assign(database_id, tenant_id, &req.collection, &req.pk) + .assign(key, tenant_id, &req.pk) { Ok(surrogate) => AssignSurrogateResponse { surrogate: surrogate.as_u32(), diff --git a/nodedb/src/control/server/surrogate_exchange/resolve.rs b/nodedb/src/control/server/surrogate_exchange/resolve.rs index f88b87d4a..982a1ae88 100644 --- a/nodedb/src/control/server/surrogate_exchange/resolve.rs +++ b/nodedb/src/control/server/surrogate_exchange/resolve.rs @@ -36,11 +36,11 @@ use std::sync::Arc; use nodedb_cluster::{ AssignSurrogateRequest, AssignSurrogateResponse, NexarTransport, RaftRpc, RoutingTable, }; -use nodedb_types::Surrogate; +use nodedb_types::{CollectionKey, Surrogate}; use crate::control::server::exchange::resolve::register_peers_from_topology; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use crate::types::{TenantId, TraceId, VShardId}; /// Where a routed surrogate exchange for a given home vShard must run. enum Route<'a> { @@ -123,9 +123,8 @@ fn leader_for( /// its `collection`, or either without the scoping ids, names no row. struct ExchangeKey<'a> { vshard: VShardId, - database_id: DatabaseId, + collection: CollectionKey<'a>, tenant_id: TenantId, - collection: &'a str, pk: &'a [u8], trace_id: TraceId, } @@ -138,9 +137,8 @@ fn build_request( ) -> AssignSurrogateRequest { let ExchangeKey { vshard, - database_id, - tenant_id, collection, + tenant_id, pk, trace_id, } = key; @@ -151,9 +149,11 @@ fn build_request( ); AssignSurrogateRequest { vshard_id: vshard.as_u32(), - database_id: database_id.as_u64(), + database_id: collection.database_id().as_u64(), tenant_id: tenant_id.as_u64(), - collection: collection.to_string(), + // The bare catalog name: the leader rebuilds the canonical key from + // it and `database_id`. + collection: collection.name().to_string(), pk: pk.to_vec(), deadline_remaining_ms, trace_id: trace_id.0, @@ -198,7 +198,7 @@ async fn send_to_leader( /// vShard's leader when this node is not the leader. /// /// `vshard` is the endpoint key's home vShard (the caller resolves it from the -/// key, e.g. via [`VShardId::from_key`]). `database_id` / `tenant_id` scope the +/// key, e.g. via [`VShardId::from_key`]). `collection` and `tenant_id` scope the /// identity; `trace_id` is propagated to the leader-side handler for tracing. /// /// Returns the authoritative `Surrogate` (`Surrogate::ZERO` only in the @@ -207,24 +207,20 @@ async fn send_to_leader( pub async fn assign_surrogate_routed( state: &SharedState, vshard: VShardId, - database_id: DatabaseId, + collection: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk: &[u8], trace_id: TraceId, ) -> crate::Result { - match route_for(state, vshard, collection)? { - Route::Local => state - .surrogate_assigner - .assign(database_id, tenant_id, collection, pk), + match route_for(state, vshard, collection.name())? { + Route::Local => state.surrogate_assigner.assign(collection, tenant_id, pk), Route::Remote { leader, transport } => { let req = build_request( state, ExchangeKey { vshard, - database_id, - tenant_id, collection, + tenant_id, pk, trace_id, }, @@ -252,24 +248,20 @@ pub async fn assign_surrogate_routed( pub async fn lookup_surrogate_routed( state: &SharedState, vshard: VShardId, - database_id: DatabaseId, + collection: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk: &[u8], trace_id: TraceId, ) -> crate::Result> { - match route_for(state, vshard, collection)? { - Route::Local => state - .surrogate_assigner - .lookup(database_id, tenant_id, collection, pk), + match route_for(state, vshard, collection.name())? { + Route::Local => state.surrogate_assigner.lookup(collection, tenant_id, pk), Route::Remote { leader, transport } => { let req = build_request( state, ExchangeKey { vshard, - database_id, - tenant_id, collection, + tenant_id, pk, trace_id, }, @@ -287,7 +279,8 @@ pub async fn lookup_surrogate_routed( None => Err(crate::Error::Internal { detail: format!( "surrogate-exchange: leader node {leader} answered a lookup for \ - '{collection}' without a found flag; cannot tell a hit from a miss" + '{}' without a found flag; cannot tell a hit from a miss", + collection.name() ), }), } diff --git a/nodedb/src/control/server/sync/async_dispatch/delta/apply.rs b/nodedb/src/control/server/sync/async_dispatch/delta/apply.rs index 392e989be..cff9ed63c 100644 --- a/nodedb/src/control/server/sync/async_dispatch/delta/apply.rs +++ b/nodedb/src/control/server/sync/async_dispatch/delta/apply.rs @@ -263,9 +263,8 @@ pub(crate) async fn apply_delta_and_finalize( } let surrogate = match shared.surrogate_assigner.assign( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &delta_msg.collection), tenant_id, - &delta_msg.collection, delta_msg.document_id.as_bytes(), ) { Ok(s) => s, @@ -333,7 +332,7 @@ pub(crate) async fn apply_delta_and_finalize( }); let vshard_id = - crate::types::VShardId::from_collection_in_database(database_id, &delta_msg.collection); + nodedb_types::CollectionKey::from_bare(database_id, &delta_msg.collection).vshard(); let authorized = super::super::super::raft_dispatch::authorize_sync_task( shared, Some(identity), diff --git a/nodedb/src/control/server/sync/columnar_handler.rs b/nodedb/src/control/server/sync/columnar_handler.rs index 57fdc7a19..7562bc592 100644 --- a/nodedb/src/control/server/sync/columnar_handler.rs +++ b/nodedb/src/control/server/sync/columnar_handler.rs @@ -175,9 +175,8 @@ impl<'a> ColumnarDispatcher for SharedStateColumnarDispatcher<'a> { surrogates.push(nodedb_types::Surrogate::ZERO); } else { surrogates.push(self.shared.surrogate_assigner.assign( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &collection), tenant_id, - &collection, &pk, )?); } @@ -341,7 +340,8 @@ impl SyncSession { let decoded = decoded_rows.len() as u64; let tenant_id = self.tenant_id.unwrap_or(TenantId::new(0)); - let vshard = VShardId::from_collection_in_database(self.database_id(), &msg.collection); + let vshard = + nodedb_types::CollectionKey::from_bare(self.database_id(), &msg.collection).vshard(); debug!( session = %self.session_id, diff --git a/nodedb/src/control/server/sync/fts_handler.rs b/nodedb/src/control/server/sync/fts_handler.rs index 587460a9e..b39130131 100644 --- a/nodedb/src/control/server/sync/fts_handler.rs +++ b/nodedb/src/control/server/sync/fts_handler.rs @@ -197,9 +197,11 @@ impl<'a> FtsDispatcher for SharedStateFtsDispatcher<'a> { collection: &str, doc_id: &str, ) -> crate::Result { - self.shared - .surrogate_assigner - .assign(database_id, tenant_id, collection, doc_id.as_bytes()) + self.shared.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + doc_id.as_bytes(), + ) } } diff --git a/nodedb/src/control/server/sync/fts_session.rs b/nodedb/src/control/server/sync/fts_session.rs index f66150f14..a1314c777 100644 --- a/nodedb/src/control/server/sync/fts_session.rs +++ b/nodedb/src/control/server/sync/fts_session.rs @@ -12,7 +12,7 @@ use nodedb_types::sync::wire::AckStatus; use super::fts_handler::FtsDispatcher; use super::session::SyncSession; use super::wire::*; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; impl SyncSession { /// Process a `FtsIndexMsg`: allocate surrogate, WAL-append on CP, dispatch @@ -85,7 +85,8 @@ impl SyncSession { }; let tenant_id = self.tenant_id.unwrap_or(TenantId::new(0)); - let vshard = VShardId::from_collection_in_database(self.database_id(), &msg.collection); + let vshard = + nodedb_types::CollectionKey::from_bare(self.database_id(), &msg.collection).vshard(); debug!( session = %self.session_id, @@ -216,7 +217,8 @@ impl SyncSession { }; let tenant_id = self.tenant_id.unwrap_or(TenantId::new(0)); - let vshard = VShardId::from_collection_in_database(self.database_id(), &msg.collection); + let vshard = + nodedb_types::CollectionKey::from_bare(self.database_id(), &msg.collection).vshard(); debug!( session = %self.session_id, diff --git a/nodedb/src/control/server/sync/kv_handler.rs b/nodedb/src/control/server/sync/kv_handler.rs index accee6289..52ecd987d 100644 --- a/nodedb/src/control/server/sync/kv_handler.rs +++ b/nodedb/src/control/server/sync/kv_handler.rs @@ -25,7 +25,7 @@ use crate::control::server::shared::clone_write::{ CloneCheckedOutcome, InterceptAndAuthorizeParams, intercept_and_authorize, }; use crate::event::EventSource; -use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, TenantId, TraceId}; /// The write a KV push applies. #[derive(Debug, Clone, PartialEq, Eq)] @@ -101,9 +101,8 @@ impl KvPushDispatcher for SharedStateKvDispatcher<'_> { let mut plan = match op { KvPushWriteOp::Put { body, ttl_ms } => { let surrogate = self.shared.surrogate_assigner.assign( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &collection), tenant_id, - &collection, &key, )?; PhysicalPlan::Kv(KvOp::Put { @@ -136,7 +135,7 @@ impl KvPushDispatcher for SharedStateKvDispatcher<'_> { let task = nodedb_physical::physical_task::PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(database_id, &collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(), database_id, plan, post_set_op: nodedb_physical::physical_task::PostSetOp::None, @@ -189,7 +188,7 @@ impl KvPushDispatcher for SharedStateKvDispatcher<'_> { let owner = RecordOwner { tenant_id, database_id, - vshard_id: VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), }; // A delete of no keys applies nothing. Its provenance moves the mark, // and its records make the move durable. diff --git a/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs b/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs index 764e21a02..567c7f780 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/durability_test_support.rs @@ -31,7 +31,7 @@ pub(super) fn tenant() -> TenantId { } pub(super) fn vshard() -> VShardId { - VShardId::from_collection_in_database(DatabaseId::DEFAULT, COLLECTION) + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLLECTION).vshard() } /// A `SharedState` whose bridge's Data-Plane side the test drives by hand. diff --git a/nodedb/src/control/server/sync/raft_dispatch/write.rs b/nodedb/src/control/server/sync/raft_dispatch/write.rs index 258babb9c..d36eb3e0f 100644 --- a/nodedb/src/control/server/sync/raft_dispatch/write.rs +++ b/nodedb/src/control/server/sync/raft_dispatch/write.rs @@ -11,7 +11,6 @@ use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; use crate::control::wal_replication::{ReplicableWrite, to_replicated_entry}; use crate::event::EventSource; -use crate::types::VShardId; use super::admission_guard::reject_unadmitted_crdt_apply; use super::outcome::SyncDispatchOutcome; @@ -88,7 +87,7 @@ pub(crate) async fn dispatch_write_replicated( vshard_id, }; let refused = reject_unadmitted_crdt_apply(&plan).and_then(|()| { - if vshard_id == VShardId::from_collection_in_database(database_id, collection) { + if vshard_id == nodedb_types::CollectionKey::from_bare(database_id, collection).vshard() { Ok(()) } else { Err(crate::Error::Internal { @@ -139,8 +138,7 @@ pub(crate) async fn dispatch_write_replicated( let task = crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::AdmittedContinuation, tenant_id, - database_id, - collection, + nodedb_types::CollectionKey::from_bare(database_id, collection), plan, ); let resp = if local_frontier_mutation { diff --git a/nodedb/src/control/server/sync/spatial_handler.rs b/nodedb/src/control/server/sync/spatial_handler.rs index ece8ca732..4bade8530 100644 --- a/nodedb/src/control/server/sync/spatial_handler.rs +++ b/nodedb/src/control/server/sync/spatial_handler.rs @@ -207,9 +207,11 @@ impl<'a> SpatialDispatcher for SharedStateSpatialDispatcher<'a> { collection: &str, doc_id: &str, ) -> crate::Result { - self.shared - .surrogate_assigner - .assign(database_id, tenant_id, collection, doc_id.as_bytes()) + self.shared.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + doc_id.as_bytes(), + ) } } diff --git a/nodedb/src/control/server/sync/spatial_session.rs b/nodedb/src/control/server/sync/spatial_session.rs index 96886b6f8..500cb36a6 100644 --- a/nodedb/src/control/server/sync/spatial_session.rs +++ b/nodedb/src/control/server/sync/spatial_session.rs @@ -14,7 +14,7 @@ use nodedb_types::sync::wire::AckStatus; use super::session::SyncSession; use super::spatial_handler::{SpatialDispatcher, SpatialInsertTarget}; use super::wire::*; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; impl SyncSession { /// Process a `SpatialInsertMsg`: deserialise geometry, allocate surrogate, @@ -105,7 +105,8 @@ impl SyncSession { }; let tenant_id = self.tenant_id.unwrap_or(TenantId::new(0)); - let vshard = VShardId::from_collection_in_database(self.database_id(), &msg.collection); + let vshard = + nodedb_types::CollectionKey::from_bare(self.database_id(), &msg.collection).vshard(); debug!( session = %self.session_id, @@ -245,7 +246,8 @@ impl SyncSession { }; let tenant_id = self.tenant_id.unwrap_or(TenantId::new(0)); - let vshard = VShardId::from_collection_in_database(self.database_id(), &msg.collection); + let vshard = + nodedb_types::CollectionKey::from_bare(self.database_id(), &msg.collection).vshard(); debug!( session = %self.session_id, diff --git a/nodedb/src/control/server/sync/timeseries_handler.rs b/nodedb/src/control/server/sync/timeseries_handler.rs index eaf5b7dc8..17d3dcbc2 100644 --- a/nodedb/src/control/server/sync/timeseries_handler.rs +++ b/nodedb/src/control/server/sync/timeseries_handler.rs @@ -253,7 +253,7 @@ impl SyncSession { "timeseries push decoded, dispatching to Data Plane" ); - let vshard = VShardId::from_collection_in_database(database_id, &msg.collection); + let vshard = nodedb_types::CollectionKey::from_bare(database_id, &msg.collection).vshard(); match dispatcher .dispatch_ingest( @@ -639,7 +639,7 @@ mod tests { assert_eq!(calls[0].1, database_id); assert_eq!( calls[0].2, - VShardId::from_collection_in_database(database_id, "metrics") + nodedb_types::CollectionKey::from_bare(database_id, "metrics").vshard() ); } @@ -663,7 +663,7 @@ mod tests { assert_eq!(calls[0].1, DatabaseId::DEFAULT); assert_eq!( calls[0].2, - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "metrics") + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "metrics").vshard() ); assert_eq!(calls[0].3, "metrics"); // ILP payload must contain the collection name and lite_id. diff --git a/nodedb/src/control/server/sync/vector_handler.rs b/nodedb/src/control/server/sync/vector_handler.rs index 27a6949ca..627aaf6f6 100644 --- a/nodedb/src/control/server/sync/vector_handler.rs +++ b/nodedb/src/control/server/sync/vector_handler.rs @@ -227,9 +227,11 @@ impl<'a> VectorDispatcher for SharedStateVectorDispatcher<'a> { collection: &str, doc_id: &str, ) -> crate::Result { - self.shared - .surrogate_assigner - .assign(database_id, tenant_id, collection, doc_id.as_bytes()) + self.shared.surrogate_assigner.assign( + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + doc_id.as_bytes(), + ) } } diff --git a/nodedb/src/control/server/sync/vector_session.rs b/nodedb/src/control/server/sync/vector_session.rs index db104acc5..f7b929450 100644 --- a/nodedb/src/control/server/sync/vector_session.rs +++ b/nodedb/src/control/server/sync/vector_session.rs @@ -13,7 +13,7 @@ use nodedb_types::sync::wire::AckStatus; use super::session::SyncSession; use super::vector_handler::{VectorDispatcher, VectorInsertParams}; use super::wire::*; -use crate::types::{TenantId, VShardId}; +use crate::types::TenantId; impl SyncSession { /// Process a `VectorInsertMsg`: allocate surrogate, dispatch to Data Plane, @@ -106,7 +106,8 @@ impl SyncSession { }; let tenant_id = self.tenant_id.unwrap_or(TenantId::new(0)); - let vshard = VShardId::from_collection_in_database(self.database_id(), &msg.collection); + let vshard = + nodedb_types::CollectionKey::from_bare(self.database_id(), &msg.collection).vshard(); debug!( session = %self.session_id, @@ -245,7 +246,8 @@ impl SyncSession { }; let tenant_id = self.tenant_id.unwrap_or(TenantId::new(0)); - let vshard = VShardId::from_collection_in_database(self.database_id(), &msg.collection); + let vshard = + nodedb_types::CollectionKey::from_bare(self.database_id(), &msg.collection).vshard(); debug!( session = %self.session_id, diff --git a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs index f5e33123c..8195bb7c7 100644 --- a/nodedb/src/control/server/wal_dispatch/write_set_redo.rs +++ b/nodedb/src/control/server/wal_dispatch/write_set_redo.rs @@ -76,7 +76,8 @@ pub fn append_write_set_redo( // A cross-collection entry homes to a different vShard, so it's re-derived // per entry rather than reusing the caller-hoisted `vshard_id`. let entry_vshard_id = match &entry.collection { - Some(c) => VShardId::from_collection_in_database(database_id, c), + // A write-set entry names its storage collection, the qualified name. + Some(c) => nodedb_types::CollectionKey::from_qualified_str(database_id, c)?.vshard(), None => vshard_id, }; let lsn = if entry.is_delete { @@ -112,7 +113,9 @@ pub fn mint_dispatch_local_redo( if resp.status != Status::Ok || resp.write_set.is_empty() { return Ok(()); } - let vshard_id = VShardId::from_collection_in_database(database_id, collection); + // `collection` is the plan's database-qualified name. + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); append_write_set_redo( wal, tenant_id, diff --git a/nodedb/src/control/surrogate/assign/bind_plan/array.rs b/nodedb/src/control/surrogate/assign/bind_plan/array.rs index 6bdf135ef..c70ee84d0 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/array.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/array.rs @@ -28,7 +28,8 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut ArrayOp) -> crate::Resu detail: format!("array coord pk encode: {e}"), } })?; - let bound = binder.resolve(&array_id.name, &pk_bytes, cell.surrogate)?; + let bound = + binder.resolve(binder.bare_key(&array_id.name), &pk_bytes, cell.surrogate)?; if bound != cell.surrogate { cell.surrogate = bound; rewritten = true; diff --git a/nodedb/src/control/surrogate/assign/bind_plan/binder.rs b/nodedb/src/control/surrogate/assign/bind_plan/binder.rs index a1a36a40c..b6f9fe622 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/binder.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/binder.rs @@ -5,7 +5,7 @@ use std::cell::RefCell; use nodedb_physical::physical_plan::PhysicalPlan; -use nodedb_types::Surrogate; +use nodedb_types::{CollectionKey, Surrogate}; use super::super::SurrogateAssigner; use super::carried::CarriedIdentity; @@ -52,17 +52,32 @@ impl<'a> IdentityBinder<'a> { self.recorded.map(RefCell::into_inner).unwrap_or_default() } - fn record(&self, collection: &str, pk_bytes: &[u8], surrogate: Surrogate) { + /// The canonical key of a collection named on a plan. Plans carry the + /// database-qualified name, so it is de-qualified here. + pub(super) fn plan_key<'c>(&self, qualified: &'c str) -> crate::Result> { + Ok(CollectionKey::from_qualified_str( + self.database_id, + qualified, + )?) + } + + /// The canonical key of a collection named by its bare catalog name, as + /// an array plan names its array. + pub(super) fn bare_key<'c>(&self, bare: &'c str) -> CollectionKey<'c> { + CollectionKey::from_bare(self.database_id, bare) + } + + fn record(&self, key: CollectionKey<'_>, pk_bytes: &[u8], surrogate: Surrogate) { if let Some(recorded) = &self.recorded { recorded.borrow_mut().push(CarriedIdentity { - collection: collection.to_string(), + collection: key.name().to_string(), pk_bytes: pk_bytes.to_vec(), surrogate, }); } } - /// The authoritative surrogate for `(collection, pk_bytes)`. + /// The authoritative surrogate for `(key, pk_bytes)`. /// /// A non-ZERO `carried` value came from the coordinator that planned the /// write: install it first-wins and return the bound value, which is the @@ -71,24 +86,18 @@ impl<'a> IdentityBinder<'a> { /// only: an existing binding or ZERO. ZERO is never written. pub(super) fn resolve( &self, - collection: &str, + key: CollectionKey<'_>, pk_bytes: &[u8], carried: Surrogate, ) -> crate::Result { if carried != Surrogate::ZERO { - let bound = self.assigner.bind( - self.database_id, - self.tenant_id, - collection, - pk_bytes, - carried, - )?; - self.record(collection, pk_bytes, bound); + let bound = self.assigner.bind(key, self.tenant_id, pk_bytes, carried)?; + self.record(key, pk_bytes, bound); return Ok(bound); } Ok(self .assigner - .lookup(self.database_id, self.tenant_id, collection, pk_bytes)? + .lookup(key, self.tenant_id, pk_bytes)? .unwrap_or(Surrogate::ZERO)) } @@ -96,30 +105,30 @@ impl<'a> IdentityBinder<'a> { /// its own big-endian bytes, the same key `assign_anonymous` binds under. pub(super) fn resolve_self_keyed( &self, - collection: &str, + key: CollectionKey<'_>, carried: Surrogate, ) -> crate::Result { - self.resolve(collection, &carried.as_u32().to_be_bytes(), carried) + self.resolve(key, &carried.as_u32().to_be_bytes(), carried) } /// [`Self::resolve`] writing the authoritative value back into `slot`. pub(super) fn resolve_in_place( &self, - collection: &str, + key: CollectionKey<'_>, pk_bytes: &[u8], slot: &mut Surrogate, ) -> crate::Result<()> { - *slot = self.resolve(collection, pk_bytes, *slot)?; + *slot = self.resolve(key, pk_bytes, *slot)?; Ok(()) } /// [`Self::resolve_self_keyed`] writing the authoritative value back into `slot`. pub(super) fn resolve_self_keyed_in_place( &self, - collection: &str, + key: CollectionKey<'_>, slot: &mut Surrogate, ) -> crate::Result<()> { - *slot = self.resolve_self_keyed(collection, *slot)?; + *slot = self.resolve_self_keyed(key, *slot)?; Ok(()) } @@ -128,27 +137,24 @@ impl<'a> IdentityBinder<'a> { /// this; a live apply always carries a non-ZERO value and binds it. pub(super) fn resolve_or_assign_in_place( &self, - collection: &str, + key: CollectionKey<'_>, document_id: &str, slot: &mut Surrogate, ) -> crate::Result<()> { if *slot != Surrogate::ZERO { - return self.resolve_in_place(collection, document_id.as_bytes(), slot); + return self.resolve_in_place(key, document_id.as_bytes(), slot); } tracing::warn!( database_id = self.database_id.as_u64(), tenant_id = self.tenant_id.as_u64(), - collection, + collection = key.name(), document_id, "CRDT apply carries no surrogate; allocating locally, which can diverge across replicas" ); - *slot = self.assigner.assign( - self.database_id, - self.tenant_id, - collection, - document_id.as_bytes(), - )?; - self.record(collection, document_id.as_bytes(), *slot); + *slot = self + .assigner + .assign(key, self.tenant_id, document_id.as_bytes())?; + self.record(key, document_id.as_bytes(), *slot); Ok(()) } } diff --git a/nodedb/src/control/surrogate/assign/bind_plan/carried.rs b/nodedb/src/control/surrogate/assign/bind_plan/carried.rs index c0289e22c..5f270488c 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/carried.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/carried.rs @@ -19,6 +19,8 @@ use crate::types::{DatabaseId, TenantId}; /// One `(collection, pk_bytes) → surrogate` identity. #[derive(Debug, Clone, PartialEq, Eq)] pub struct CarriedIdentity { + /// The bare catalog name. With the binding database it forms the + /// canonical `CollectionKey`. pub collection: String, pub pk_bytes: Vec, pub surrogate: Surrogate, @@ -55,9 +57,8 @@ pub fn bind_carried_identities( ) -> crate::Result<()> { for identity in identities { let bound = assigner.bind( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &identity.collection), tenant_id, - &identity.collection, &identity.pk_bytes, identity.surrogate, )?; diff --git a/nodedb/src/control/surrogate/assign/bind_plan/crdt.rs b/nodedb/src/control/surrogate/assign/bind_plan/crdt.rs index dabb0c3ab..204f8bd68 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/crdt.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/crdt.rs @@ -21,7 +21,11 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut CrdtOp) -> crate::Resul document_id, surrogate, .. - } => binder.resolve_or_assign_in_place(collection.as_str(), document_id, surrogate), + } => binder.resolve_or_assign_in_place( + binder.plan_key(collection.as_str())?, + document_id, + surrogate, + ), CrdtOp::DocUpsert { collection, document_id, @@ -57,7 +61,11 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut CrdtOp) -> crate::Resul document_id, surrogate, .. - } => binder.resolve_in_place(collection.as_str(), document_id.as_bytes(), surrogate), + } => binder.resolve_in_place( + binder.plan_key(collection.as_str())?, + document_id.as_bytes(), + surrogate, + ), // Reads, snapshots, constraints, policies and previews create no row. CrdtOp::Read { .. } | CrdtOp::ImportSnapshot { .. } diff --git a/nodedb/src/control/surrogate/assign/bind_plan/document.rs b/nodedb/src/control/surrogate/assign/bind_plan/document.rs index 7b6c1d7f4..256d81e08 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/document.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/document.rs @@ -38,7 +38,11 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut DocumentOp) -> crate::R document_id, surrogate, .. - } => binder.resolve_in_place(collection.as_str(), document_id.as_bytes(), surrogate), + } => binder.resolve_in_place( + binder.plan_key(collection.as_str())?, + document_id.as_bytes(), + surrogate, + ), DocumentOp::BatchInsert { collection, documents, @@ -60,7 +64,11 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut DocumentOp) -> crate::R }); } for ((document_id, _value), surrogate) in documents.iter().zip(surrogates.iter_mut()) { - binder.resolve_in_place(collection.as_str(), document_id.as_bytes(), surrogate)?; + binder.resolve_in_place( + binder.plan_key(collection.as_str())?, + document_id.as_bytes(), + surrogate, + )?; } Ok(()) } @@ -78,8 +86,11 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut DocumentOp) -> crate::R resolved_insert_identities.iter_mut().enumerate() { let carried = nodedb_types::Surrogate::new(*surrogate); - let bound = - binder.resolve(target_collection.as_str(), document_id.as_bytes(), carried)?; + let bound = binder.resolve( + binder.plan_key(target_collection.as_str())?, + document_id.as_bytes(), + carried, + )?; if bound == carried { continue; } @@ -110,7 +121,7 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut DocumentOp) -> crate::R surrogate, .. } => binder.resolve_in_place( - collection.as_str(), + binder.plan_key(collection.as_str())?, document_id.as_bytes(), surrogate, )?, diff --git a/nodedb/src/control/surrogate/assign/bind_plan/graph.rs b/nodedb/src/control/surrogate/assign/bind_plan/graph.rs index 8248c3979..0ca0acafb 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/graph.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/graph.rs @@ -24,8 +24,16 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut GraphOp) -> crate::Resu dst_surrogate, .. } => { - binder.resolve_in_place(collection.as_str(), src_id.as_bytes(), src_surrogate)?; - binder.resolve_in_place(collection.as_str(), dst_id.as_bytes(), dst_surrogate) + binder.resolve_in_place( + binder.plan_key(collection.as_str())?, + src_id.as_bytes(), + src_surrogate, + )?; + binder.resolve_in_place( + binder.plan_key(collection.as_str())?, + dst_id.as_bytes(), + dst_surrogate, + ) } GraphOp::EdgePutBatch { edges } | GraphOp::EdgeDeleteBatch { edges } => { for edge in edges.iter_mut() { @@ -57,7 +65,7 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut GraphOp) -> crate::Resu } fn bind_edge(binder: &IdentityBinder<'_>, edge: &mut BatchEdge) -> crate::Result<()> { - let collection = edge.collection.as_str(); - binder.resolve_in_place(collection, edge.src_id.as_bytes(), &mut edge.src_surrogate)?; - binder.resolve_in_place(collection, edge.dst_id.as_bytes(), &mut edge.dst_surrogate) + let key = binder.plan_key(edge.collection.as_str())?; + binder.resolve_in_place(key, edge.src_id.as_bytes(), &mut edge.src_surrogate)?; + binder.resolve_in_place(key, edge.dst_id.as_bytes(), &mut edge.dst_surrogate) } diff --git a/nodedb/src/control/surrogate/assign/bind_plan/kv.rs b/nodedb/src/control/surrogate/assign/bind_plan/kv.rs index 204dd2436..40fb7d31c 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/kv.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/kv.rs @@ -61,7 +61,7 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut KvOp) -> crate::Result< key, surrogate, .. - } => binder.resolve_in_place(collection.as_str(), key, surrogate), + } => binder.resolve_in_place(binder.plan_key(collection.as_str())?, key, surrogate), KvOp::BatchPut { collection, entries, @@ -83,7 +83,7 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut KvOp) -> crate::Result< }); } for ((key, _value), surrogate) in entries.iter().zip(surrogates.iter_mut()) { - binder.resolve_in_place(collection.as_str(), key, surrogate)?; + binder.resolve_in_place(binder.plan_key(collection.as_str())?, key, surrogate)?; } Ok(()) } @@ -95,8 +95,16 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut KvOp) -> crate::Result< credit_surrogate, .. } => { - binder.resolve_in_place(collection.as_str(), source_key, debit_surrogate)?; - binder.resolve_in_place(collection.as_str(), dest_key, credit_surrogate) + binder.resolve_in_place( + binder.plan_key(collection.as_str())?, + source_key, + debit_surrogate, + )?; + binder.resolve_in_place( + binder.plan_key(collection.as_str())?, + dest_key, + credit_surrogate, + ) } // The moved item lands under `dest_key` in the destination collection. KvOp::TransferItem { @@ -104,7 +112,11 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut KvOp) -> crate::Result< dest_key, surrogate, .. - } => binder.resolve_in_place(dest_collection.as_str(), dest_key, surrogate), + } => binder.resolve_in_place( + binder.plan_key(dest_collection.as_str())?, + dest_key, + surrogate, + ), KvOp::ResolveWrite(inner) => bind(binder, inner), KvOp::ResolvedWrite { mutations, .. } => { for mutation in mutations.iter_mut() { @@ -114,7 +126,11 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut KvOp) -> crate::Result< key, surrogate, .. - } => binder.resolve_in_place(collection.as_str(), key, surrogate)?, + } => binder.resolve_in_place( + binder.plan_key(collection.as_str())?, + key, + surrogate, + )?, // Named by key; the row's identity was bound when it was put. KvResolvedMutation::Delete { .. } | KvResolvedMutation::Expire { .. } diff --git a/nodedb/src/control/surrogate/assign/bind_plan/vector.rs b/nodedb/src/control/surrogate/assign/bind_plan/vector.rs index 6f9e68176..c1003d425 100644 --- a/nodedb/src/control/surrogate/assign/bind_plan/vector.rs +++ b/nodedb/src/control/surrogate/assign/bind_plan/vector.rs @@ -15,8 +15,12 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut VectorOp) -> crate::Res pk_bytes, .. } => match pk_bytes { - Some(pk) => binder.resolve_in_place(collection.as_str(), pk, surrogate), - None => binder.resolve_self_keyed_in_place(collection.as_str(), surrogate), + Some(pk) => { + binder.resolve_in_place(binder.plan_key(collection.as_str())?, pk, surrogate) + } + None => { + binder.resolve_self_keyed_in_place(binder.plan_key(collection.as_str())?, surrogate) + } }, VectorOp::BatchInsert { collection, @@ -24,7 +28,10 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut VectorOp) -> crate::Res .. } => { for surrogate in surrogates.iter_mut() { - binder.resolve_self_keyed_in_place(collection.as_str(), surrogate)?; + binder.resolve_self_keyed_in_place( + binder.plan_key(collection.as_str())?, + surrogate, + )?; } Ok(()) } @@ -32,7 +39,8 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut VectorOp) -> crate::Res collection, document_surrogate, .. - } => binder.resolve_self_keyed_in_place(collection.as_str(), document_surrogate), + } => binder + .resolve_self_keyed_in_place(binder.plan_key(collection.as_str())?, document_surrogate), VectorOp::DirectUpsert { collection, surrogate, @@ -50,7 +58,12 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut VectorOp) -> crate::Res surrogate, pk_bytes, .. - } => bind_keyed_or_self(binder, collection.as_str(), pk_bytes, surrogate), + } => bind_keyed_or_self( + binder, + binder.plan_key(collection.as_str())?, + pk_bytes, + surrogate, + ), VectorOp::ResolveDirectWrite(inner) => bind(binder, inner), VectorOp::ResolvedDirectWrite { collection, @@ -63,7 +76,12 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut VectorOp) -> crate::Res surrogate, pk_bytes, .. - } => bind_keyed_or_self(binder, collection.as_str(), pk_bytes, surrogate)?, + } => bind_keyed_or_self( + binder, + binder.plan_key(collection.as_str())?, + pk_bytes, + surrogate, + )?, // Named by a surrogate bound when the row was inserted. VectorResolvedMutation::Delete { .. } | VectorResolvedMutation::Update { .. } => {} @@ -98,12 +116,12 @@ pub(super) fn bind(binder: &IdentityBinder<'_>, op: &mut VectorOp) -> crate::Res /// An empty `pk_bytes` is a headless row: self-key it. fn bind_keyed_or_self( binder: &IdentityBinder<'_>, - collection: &str, + key: nodedb_types::CollectionKey<'_>, pk_bytes: &[u8], surrogate: &mut nodedb_types::Surrogate, ) -> crate::Result<()> { if pk_bytes.is_empty() { - return binder.resolve_self_keyed_in_place(collection, surrogate); + return binder.resolve_self_keyed_in_place(key, surrogate); } - binder.resolve_in_place(collection, pk_bytes, surrogate) + binder.resolve_in_place(key, pk_bytes, surrogate) } diff --git a/nodedb/src/control/surrogate/assign/core/assign_ops.rs b/nodedb/src/control/surrogate/assign/core/assign_ops.rs index 18ebf64ed..ce623ef40 100644 --- a/nodedb/src/control/surrogate/assign/core/assign_ops.rs +++ b/nodedb/src/control/surrogate/assign/core/assign_ops.rs @@ -3,7 +3,7 @@ //! The read/allocate/bind operations on [`super::SurrogateAssigner`]: //! `assign`, `assign_fresh`, `assign_anonymous`, `bind`, and `lookup`. -use nodedb_types::{DatabaseId, TenantId}; +use nodedb_types::{CollectionKey, TenantId}; use nodedb_types::Surrogate; @@ -24,9 +24,8 @@ impl SurrogateAssigner { /// persisted PK row cannot diverge under concurrent assigners. pub fn assign( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], ) -> crate::Result { let catalog = self.credential_store.catalog(); @@ -34,9 +33,7 @@ impl SurrogateAssigner { // Fast-path: existing binding. Done under a read lock — most // production calls land here once the per-collection working // set has been observed. - if let Some(s) = - catalog.get_surrogate_for_pk(database_id, tenant_id, collection, pk_bytes)? - { + if let Some(s) = catalog.get_surrogate_for_pk(key, tenant_id, pk_bytes)? { return Ok(s); } @@ -59,9 +56,7 @@ impl SurrogateAssigner { let registry = self.registry_write()?; // Re-check inside the lock: another assigner may have raced // us between the read above and the lock acquisition. - if let Some(s) = - catalog.get_surrogate_for_pk(database_id, tenant_id, collection, pk_bytes)? - { + if let Some(s) = catalog.get_surrogate_for_pk(key, tenant_id, pk_bytes)? { return Ok(s); } let surrogate = match self.alloc_locked(®istry)? { @@ -84,7 +79,7 @@ impl SurrogateAssigner { continue; } }; - catalog.put_surrogate(database_id, tenant_id, collection, pk_bytes, surrogate)?; + catalog.put_surrogate(key, tenant_id, pk_bytes, surrogate)?; // Emit a durable WAL bind before the lock releases. Order is // load-bearing: a crash between catalog write and bind append // is invisible (the catalog row is already on disk via redb's @@ -92,13 +87,8 @@ impl SurrogateAssigner { // to recover; a crash between bind append and lock release is // recovered by replaying the bind into the catalog (idempotent // via the two-table overwrite). - self.wal_appender.record_bind_to_wal( - database_id, - tenant_id, - surrogate.as_u32(), - collection, - pk_bytes, - )?; + self.wal_appender + .record_bind_to_wal(key, tenant_id, surrogate.as_u32(), pk_bytes)?; self.maybe_flush(®istry, catalog)?; return Ok(surrogate); } @@ -121,9 +111,8 @@ impl SurrogateAssigner { /// which the caller uses verbatim. pub fn assign_fresh( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, ) -> crate::Result<(Surrogate, String)> { let catalog = self.credential_store.catalog(); @@ -146,14 +135,9 @@ impl SurrogateAssigner { let pk = crate::engine::document::store::RowIdentity::for_surrogate(surrogate).into_string(); let pk_bytes = pk.as_bytes(); - catalog.put_surrogate(database_id, tenant_id, collection, pk_bytes, surrogate)?; - self.wal_appender.record_bind_to_wal( - database_id, - tenant_id, - surrogate.as_u32(), - collection, - pk_bytes, - )?; + catalog.put_surrogate(key, tenant_id, pk_bytes, surrogate)?; + self.wal_appender + .record_bind_to_wal(key, tenant_id, surrogate.as_u32(), pk_bytes)?; self.maybe_flush(®istry, catalog)?; return Ok((surrogate, pk)); } @@ -172,13 +156,12 @@ impl SurrogateAssigner { /// rather than synthesized as `Surrogate::ZERO`. pub fn lookup( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], ) -> crate::Result> { let catalog = self.credential_store.catalog(); - catalog.get_surrogate_for_pk(database_id, tenant_id, collection, pk_bytes) + catalog.get_surrogate_for_pk(key, tenant_id, pk_bytes) } /// Bind `(collection, pk_bytes)` to a *carried* surrogate without ever @@ -216,9 +199,8 @@ impl SurrogateAssigner { /// surrogate and diverge from the coordinator. pub fn bind( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, pk_bytes: &[u8], surrogate: Surrogate, ) -> crate::Result { @@ -228,9 +210,7 @@ impl SurrogateAssigner { // installed (replay, retry, or a competing coordinator's carried // value applied first) it is authoritative — return it, never // overwrite, discard the carried value even if it differs. - if let Some(existing) = - catalog.get_surrogate_for_pk(database_id, tenant_id, collection, pk_bytes)? - { + if let Some(existing) = catalog.get_surrogate_for_pk(key, tenant_id, pk_bytes)? { return Ok(existing); } @@ -244,19 +224,12 @@ impl SurrogateAssigner { // Re-check under the lock (TOCTOU): a concurrent `assign`/`bind` on // this node could have written between the pre-check and the lock. // First-wins still applies — return the existing value. - if let Some(existing) = - catalog.get_surrogate_for_pk(database_id, tenant_id, collection, pk_bytes)? - { + if let Some(existing) = catalog.get_surrogate_for_pk(key, tenant_id, pk_bytes)? { return Ok(existing); } - catalog.put_surrogate(database_id, tenant_id, collection, pk_bytes, surrogate)?; - self.wal_appender.record_bind_to_wal( - database_id, - tenant_id, - surrogate.as_u32(), - collection, - pk_bytes, - )?; + catalog.put_surrogate(key, tenant_id, pk_bytes, surrogate)?; + self.wal_appender + .record_bind_to_wal(key, tenant_id, surrogate.as_u32(), pk_bytes)?; // Advance the local watermark past the carried value so a later // LOCAL `assign`/`assign_anonymous` on this node can never re-issue // it. Idempotent and monotonic — never lowers, never advances the @@ -278,9 +251,8 @@ impl SurrogateAssigner { /// catalog single-shaped — no special-case "unbound" rows. pub fn assign_anonymous( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, - collection: &str, ) -> crate::Result { let catalog = self.credential_store.catalog(); @@ -299,12 +271,11 @@ impl SurrogateAssigner { } }; let self_bytes = surrogate.as_u32().to_be_bytes(); - catalog.put_surrogate(database_id, tenant_id, collection, &self_bytes, surrogate)?; + catalog.put_surrogate(key, tenant_id, &self_bytes, surrogate)?; self.wal_appender.record_bind_to_wal( - database_id, + key, tenant_id, surrogate.as_u32(), - collection, &self_bytes, )?; self.maybe_flush(®istry, catalog)?; @@ -338,10 +309,18 @@ mod tests { fn assign_is_idempotent_for_same_pk() { let (_dir, a) = open_test(); let s1 = a - .assign(DatabaseId::DEFAULT, T0, "users", b"alice") + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice", + ) .unwrap(); let s2 = a - .assign(DatabaseId::DEFAULT, T0, "users", b"alice") + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice", + ) .unwrap(); assert_eq!(s1, s2); assert_eq!(s1, Surrogate::new(1)); @@ -353,10 +332,18 @@ mod tests { let t1 = TenantId::new(1); let t2 = TenantId::new(2); let s1 = a - .assign(DatabaseId::DEFAULT, t1, "users", b"alice") + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + t1, + b"alice", + ) .unwrap(); let s2 = a - .assign(DatabaseId::DEFAULT, t2, "users", b"alice") + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + t2, + b"alice", + ) .unwrap(); assert_ne!(s1, s2); } @@ -365,9 +352,19 @@ mod tests { fn assign_distinct_pks_returns_distinct_surrogates() { let (_dir, a) = open_test(); let s1 = a - .assign(DatabaseId::DEFAULT, T0, "users", b"alice") + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice", + ) + .unwrap(); + let s2 = a + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"bob", + ) .unwrap(); - let s2 = a.assign(DatabaseId::DEFAULT, T0, "users", b"bob").unwrap(); assert_ne!(s1, s2); } @@ -375,12 +372,20 @@ mod tests { fn assign_writes_reverse_binding() { let (_dir, a) = open_test(); let s = a - .assign(DatabaseId::DEFAULT, T0, "users", b"alice") + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + b"alice", + ) .unwrap(); let cat = a.credential_store.catalog(); assert_eq!( - cat.get_pk_for_surrogate(DatabaseId::DEFAULT, T0, "users", s) - .unwrap(), + cat.get_pk_for_surrogate( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + s + ) + .unwrap(), Some(b"alice".to_vec()) ); } @@ -393,7 +398,11 @@ mod tests { for i in 0..n { let pk = format!("u{i}"); let _ = a - .assign(DatabaseId::DEFAULT, T0, "users", pk.as_bytes()) + .assign( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + T0, + pk.as_bytes(), + ) .unwrap(); } // Either threshold (1024 ops or 200 ms elapsed) may fire diff --git a/nodedb/src/control/surrogate/physical_impl.rs b/nodedb/src/control/surrogate/physical_impl.rs index 067c03d63..9d858a249 100644 --- a/nodedb/src/control/surrogate/physical_impl.rs +++ b/nodedb/src/control/surrogate/physical_impl.rs @@ -15,22 +15,20 @@ impl PhysicalSurrogateAssigner for SurrogateAssigner { fn assign( &self, - database_id: nodedb_types::DatabaseId, + key: nodedb_types::CollectionKey<'_>, tenant_id: nodedb_types::TenantId, - collection: &str, pk_bytes: &[u8], ) -> Result { - Self::assign(self, database_id, tenant_id, collection, pk_bytes) + Self::assign(self, key, tenant_id, pk_bytes) .map_err(|e| SurrogateAssignError::Backend(e.to_string())) } fn assign_fresh( &self, - database_id: nodedb_types::DatabaseId, + key: nodedb_types::CollectionKey<'_>, tenant_id: nodedb_types::TenantId, - collection: &str, ) -> Result<(nodedb_types::Surrogate, String), SurrogateAssignError> { - Self::assign_fresh(self, database_id, tenant_id, collection) + Self::assign_fresh(self, key, tenant_id) .map_err(|e| SurrogateAssignError::Backend(e.to_string())) } } diff --git a/nodedb/src/control/surrogate/wal_appender.rs b/nodedb/src/control/surrogate/wal_appender.rs index b4920836d..c4f713d53 100644 --- a/nodedb/src/control/surrogate/wal_appender.rs +++ b/nodedb/src/control/surrogate/wal_appender.rs @@ -15,7 +15,7 @@ use std::sync::Arc; -use nodedb_types::{DatabaseId, TenantId}; +use nodedb_types::{CollectionKey, TenantId}; use crate::wal::WalManager; use crate::wal::manager::NO_APPLY_KEY; @@ -33,14 +33,14 @@ pub trait SurrogateWalAppender: Send + Sync { /// `(surrogate, collection, pk_bytes)` triple. Called by /// `SurrogateAssigner::assign` after the catalog two-table txn so /// the binding is durable before the write-lock is released. The - /// `(database_id, tenant_id)` scope is stamped into the record header - /// so replay re-keys the binding under the same scope. + /// record carries the key's bare name. The `(database_id, tenant_id)` + /// scope is stamped into the record header so replay re-keys the + /// binding under the same canonical key. fn record_bind_to_wal( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, surrogate: u32, - collection: &str, pk_bytes: &[u8], ) -> crate::Result<()>; } @@ -67,17 +67,16 @@ impl SurrogateWalAppender for WalSurrogateAppender { fn record_bind_to_wal( &self, - database_id: DatabaseId, + key: CollectionKey<'_>, tenant_id: TenantId, surrogate: u32, - collection: &str, pk_bytes: &[u8], ) -> crate::Result<()> { self.wal.appender(NO_APPLY_KEY).append_surrogate_bind( - database_id, + key.database_id(), tenant_id, surrogate, - collection, + key.name(), pk_bytes, )?; // Force the record to disk before the assigner releases its @@ -99,10 +98,9 @@ impl SurrogateWalAppender for NoopWalAppender { fn record_bind_to_wal( &self, - _database_id: DatabaseId, + _key: CollectionKey<'_>, _tenant_id: TenantId, _surrogate: u32, - _collection: &str, _pk_bytes: &[u8], ) -> crate::Result<()> { Ok(()) diff --git a/nodedb/src/control/target_identity/surrogate.rs b/nodedb/src/control/target_identity/surrogate.rs index f77b9894e..9ae2ae2f9 100644 --- a/nodedb/src/control/target_identity/surrogate.rs +++ b/nodedb/src/control/target_identity/surrogate.rs @@ -3,43 +3,36 @@ //! Assign a fresh, catalog-registered surrogate for a row written into a //! target collection on behalf of another operation. -use nodedb_types::{DatabaseId, Surrogate, TenantId, extract_pk_value}; +use nodedb_types::{CollectionKey, Surrogate, TenantId, extract_pk_value}; use super::pk::TargetPk; use crate::control::state::SharedState; /// Assign a fresh, registered surrogate for one written row on the TARGET's -/// primary key. +/// primary key. `target` is the target collection's canonical key. pub(crate) fn assign_target_surrogate( state: &SharedState, - database_id: DatabaseId, + target: CollectionKey<'_>, tenant_id: TenantId, - target_collection: &str, target_pk: &TargetPk, body: &[u8], ) -> crate::Result { match target_pk { TargetPk::AutoRowId => { - let (surrogate, _) = - state - .surrogate_assigner - .assign_fresh(database_id, tenant_id, target_collection)?; + let (surrogate, _) = state.surrogate_assigner.assign_fresh(target, tenant_id)?; Ok(surrogate) } TargetPk::Field { name, declared } => match extract_pk_value(body, name) { // The empty string is a key like any other. Minting a fresh // surrogate for it would let two rows share it. - Some(pk) => state.surrogate_assigner.assign( - database_id, - tenant_id, - target_collection, - pk.as_bytes(), - ), + Some(pk) => state + .surrogate_assigner + .assign(target, tenant_id, pk.as_bytes()), // No usable key value on a DDL-declared PRIMARY KEY: NOT NULL is // implied, so refuse rather than mint a surrogate for a row that // plain INSERT would already reject. None if *declared => Err(crate::Error::RejectedConstraint { - collection: target_collection.to_string(), + collection: target.name().to_string(), constraint: "not_null".to_string(), detail: format!("primary key '{name}' cannot be NULL or omitted"), }), @@ -48,11 +41,7 @@ pub(crate) fn assign_target_surrogate( // binding. The row's identity is its document storage key, so // the allocator binds the hex form. _ => { - let (surrogate, _) = state.surrogate_assigner.assign_fresh( - database_id, - tenant_id, - target_collection, - )?; + let (surrogate, _) = state.surrogate_assigner.assign_fresh(target, tenant_id)?; Ok(surrogate) } }, diff --git a/nodedb/src/control/trigger/dml_hook.rs b/nodedb/src/control/trigger/dml_hook.rs index 19217f497..25dc59d07 100644 --- a/nodedb/src/control/trigger/dml_hook.rs +++ b/nodedb/src/control/trigger/dml_hook.rs @@ -15,7 +15,7 @@ use crate::control::security::auth_context::AuthContext; use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::server::shared::authorization::authorize_collection; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, TenantId, TraceId}; use nodedb_physical::physical_plan::DocumentOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -271,22 +271,16 @@ pub async fn fetch_old_row( }); } - // Catalog/permission/surrogate lookups key on the bare name plus a - // separate `database_id`, never the qualified string — recover it by - // stripping the same prefix `collection` was qualified with. - let bare_collection = if database_id == DatabaseId::DEFAULT { - collection.as_str() - } else { - collection - .as_str() - .strip_prefix(&format!("{}/", database_id.as_u64())) - .ok_or_else(|| crate::Error::RejectedAuthz { + // Catalog, permission, surrogate and vShard lookups use the canonical + // key: the bare name plus a separate `database_id`. + let key = + nodedb_types::CollectionKey::from_qualified(database_id, collection).map_err(|error| { + crate::Error::RejectedAuthz { tenant_id, - resource: format!( - "OLD-row fetch: '{collection}' is not qualified for database {database_id}" - ), - })? - }; + resource: format!("OLD-row fetch: {error}"), + } + })?; + let bare_collection = key.name(); let audit = ArcAuditEmitter(std::sync::Arc::clone(&state.audit)); authorize_collection( @@ -300,11 +294,7 @@ pub async fn fetch_old_row( )?; let pk_bytes = document_id.as_bytes().to_vec(); - let Some(surrogate) = - state - .surrogate_assigner - .lookup(database_id, tenant_id, bare_collection, &pk_bytes)? - else { + let Some(surrogate) = state.surrogate_assigner.lookup(key, tenant_id, &pk_bytes)? else { return Ok(HashMap::new()); }; let mut plan = crate::bridge::envelope::PhysicalPlan::Document(DocumentOp::PointGet { @@ -334,7 +324,8 @@ pub async fn fetch_old_row( let task = PhysicalTask { tenant_id, database_id, - vshard_id: VShardId::from_key(document_id.as_bytes()), + // A document is homed by its collection, never by its own key. + vshard_id: key.vshard(), plan, post_set_op: PostSetOp::None, txn_id: None, @@ -479,9 +470,8 @@ mod tests { state .surrogate_assigner .lookup( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, collection), identity.tenant_id, - collection, document_id.as_bytes() ) .expect("inspect surrogate binding after denial"), diff --git a/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs b/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs index 28c82eccb..4f24ed34a 100644 --- a/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs +++ b/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs @@ -14,7 +14,6 @@ use crate::control::target_identity::{ bare_collection_name, derive_document_id, resolve_target_pk, }; use crate::query::ResolvedUpdateRowWire; -use crate::types::VShardId; use nodedb_physical::physical_plan::DocumentOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -67,8 +66,11 @@ pub(crate) async fn resolve_and_emit_update_from_join_ops( // Recomputed rather than reusing the staged task's vShard, keeping dispatch // classification honest, like the MERGE / INSERT SELECT expanders. - let vshard_id = - VShardId::from_collection_in_database(task.database_id, target_collection.as_str()); + let vshard_id = nodedb_types::CollectionKey::from_qualified_str( + task.database_id, + target_collection.as_str(), + )? + .vshard(); // A join-column rewrite debits the target left and credits the one joined — // resolving post-images alone would leave the abandoned target overstated. diff --git a/nodedb/src/control/wal_replication/decode/crdt.rs b/nodedb/src/control/wal_replication/decode/crdt.rs index 15d4ab2a9..4100eac56 100644 --- a/nodedb/src/control/wal_replication/decode/crdt.rs +++ b/nodedb/src/control/wal_replication/decode/crdt.rs @@ -658,9 +658,8 @@ mod tests { for i in 0..5 { assigner .assign( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "docs"), tenant, - "docs", format!("burn-{i}").as_bytes(), ) .expect("burn allocation"); @@ -713,7 +712,11 @@ mod tests { assert_eq!( assigner - .lookup(DatabaseId::DEFAULT, tenant, "docs", b"doc-1") + .lookup( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "docs"), + tenant, + b"doc-1" + ) .expect("catalog lookup"), Some(leader_surrogate), "the carried surrogate must be installed in the local catalog" diff --git a/nodedb/src/control/write_resolve/graph.rs b/nodedb/src/control/write_resolve/graph.rs index a2febb9e9..4d9fa15ba 100644 --- a/nodedb/src/control/write_resolve/graph.rs +++ b/nodedb/src/control/write_resolve/graph.rs @@ -82,8 +82,8 @@ impl EngineWriteResolver for GraphWriteResolver { } /// Key-homed on the source endpoint, where the forward edge row lives. - fn vshard(&self, _database_id: DatabaseId) -> VShardId { - VShardId::from_key(self.src_id.as_bytes()) + fn vshard(&self, _database_id: DatabaseId) -> crate::Result { + Ok(VShardId::from_key(self.src_id.as_bytes())) } fn build_resolve_op(&self) -> PhysicalPlan { @@ -103,7 +103,7 @@ impl EngineWriteResolver for GraphWriteResolver { state, ctx.tenant_id, ctx.database_id, - self.vshard(ctx.database_id), + self.vshard(ctx.database_id)?, op, None, ) diff --git a/nodedb/src/control/write_resolve/resolver.rs b/nodedb/src/control/write_resolve/resolver.rs index bc42e33b4..b749f9b39 100644 --- a/nodedb/src/control/write_resolve/resolver.rs +++ b/nodedb/src/control/write_resolve/resolver.rs @@ -33,10 +33,15 @@ pub trait EngineWriteResolver: Send + Sync { fn collection(&self) -> &str; /// Control Plane. Home vShard the resolved write is proposed to. - /// Collection-homed by default; the graph resolver overrides this since - /// an edge is key-homed on its source endpoint instead. - fn vshard(&self, database_id: DatabaseId) -> VShardId { - VShardId::from_collection_in_database(database_id, self.collection()) + /// Collection-homed by default: `collection()` is the plan's + /// database-qualified name, de-qualified into the canonical key. The + /// graph resolver overrides this since an edge is key-homed on its source + /// endpoint instead. + fn vshard(&self, database_id: DatabaseId) -> crate::Result { + Ok( + nodedb_types::CollectionKey::from_qualified_str(database_id, self.collection())? + .vshard(), + ) } /// Control Plane. Pure, no I/O: the read-only op that resolves this write. diff --git a/nodedb/src/control/write_resolve/run.rs b/nodedb/src/control/write_resolve/run.rs index 60e0d922a..f4ad3201d 100644 --- a/nodedb/src/control/write_resolve/run.rs +++ b/nodedb/src/control/write_resolve/run.rs @@ -46,7 +46,7 @@ pub async fn run_write_resolve( let resolved = resolver.resolve(state, ctx, resolve_op).await?; let resolved_plan = resolver.apply(resolved)?; - let vshard_id = resolver.vshard(ctx.database_id); + let vshard_id = resolver.vshard(ctx.database_id)?; match propose_resolved(state, ctx, resolver.collection(), vshard_id, resolved_plan).await? { ProposeOutcome::Applied(response) => return Ok(response), ProposeOutcome::RetryRequired => { diff --git a/nodedb/src/data/executor/columnar_checkpoint/geometry_restore.rs b/nodedb/src/data/executor/columnar_checkpoint/geometry_restore.rs index fe7d0680f..1568b629f 100644 --- a/nodedb/src/data/executor/columnar_checkpoint/geometry_restore.rs +++ b/nodedb/src/data/executor/columnar_checkpoint/geometry_restore.rs @@ -51,7 +51,7 @@ use tracing::warn; use super::super::core_loop::CoreLoop; use super::super::scan_normalize::decoded_col_to_value; use crate::bridge::envelope::PhysicalPlan; -use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp}; use nodedb_types::RlsWriteCheck; use nodedb_types::columnar::ColumnType; @@ -115,10 +115,23 @@ impl CoreLoop { // rows that are already durable, and is not itself a write. Noting an // LSN here would raise the core watermark during boot from a path that // applied no record. + // The checkpoint key carries the stored, database-qualified name. + let vshard = match nodedb_types::CollectionKey::from_qualified_str(*db_id, collection) { + Ok(key) => key.vshard(), + Err(e) => { + warn!( + %collection, + error = %e, + "columnar checkpoint restore: collection name does not de-qualify; its \ + geometry rows are absent from the rebuilt R-tree" + ); + return 0; + } + }; let task = Self::replay_task( *tenant_id, *db_id, - VShardId::from_collection_in_database(*db_id, collection), + vshard, PhysicalPlan::Columnar(ColumnarOp::Insert { collection: nodedb_types::QualifiedCollection::from_stored(collection.clone()), payload: Vec::new(), diff --git a/nodedb/src/data/executor/core_loop/write_index.rs b/nodedb/src/data/executor/core_loop/write_index.rs index 3c8bfe54f..ec7694cdb 100644 --- a/nodedb/src/data/executor/core_loop/write_index.rs +++ b/nodedb/src/data/executor/core_loop/write_index.rs @@ -24,7 +24,7 @@ use std::collections::HashMap; use nodedb_types::calvin::{ReadKeyIdent, VersionedReadEntry}; use nodedb_types::{DatabaseId, TenantId}; -use crate::types::{Lsn, VShardId}; +use crate::types::Lsn; use super::CoreLoop; @@ -333,21 +333,21 @@ impl CoreLoop { let db = task.request.database_id; let tenant = TenantId::new(tid); let local_vshard = task.request.vshard_id.as_u32(); - versioned_reads - .iter() - .filter(|entry| { - VShardId::from_collection_in_database(db, &entry.collection).as_u32() - == local_vshard - }) - .all(|entry| { - self.write_index.read_is_valid( + // An entry carries the plan's database-qualified name. One that does not + // de-qualify cannot be homed or validated, so the read set fails closed. + versioned_reads.iter().all(|entry| { + match nodedb_types::CollectionKey::from_qualified_str(db, &entry.collection) { + Err(_) => false, + Ok(key) if key.vshard().as_u32() != local_vshard => true, + Ok(_) => self.write_index.read_is_valid( db, tenant, &entry.collection, &entry.key, entry.read_lsn, - ) - }) + ), + } + }) } /// Record a committed document/vector write's version, keyed by the @@ -1141,7 +1141,7 @@ pub(crate) mod tests { /// The vShard `collection` homes to in the default database — mirrors the /// homing `read_set_still_current` filters entries by. fn local_vshard(collection: &str) -> VShardId { - VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection) + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard() } /// Some vShard other than `than`, for exercising the cross-shard filter. diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs index f065404ee..e4e259456 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/apply.rs @@ -128,17 +128,19 @@ impl CoreLoop { images: &RowImages<'_>, ) -> crate::Result> { let mut writes: Vec = Vec::new(); + // The source's canonical key, de-qualified from the plan's name once. + let source = nodedb_types::CollectionKey::from_qualified_str( + DatabaseId::new(ctx.database_id), + ctx.collection, + )?; for binding in bindings { // Whether one core owns both rows. The Control Plane asked the SAME // question, from the same plane-neutral function, when it decided // whether the balance could ride this transaction — so the two // cannot disagree about which bindings this core is responsible // for. - let co_resident = crate::query::sum_target_is_co_resident( - DatabaseId::new(ctx.database_id), - ctx.collection, - &binding.target_collection, - ); + let co_resident = + crate::query::sum_target_is_co_resident(source, &binding.target_collection); // A binding the plan DEFERRED is applied by its own // `ApplyBalanceDelta` task on the target's core. Applying it here as // well would double-count it — and this transaction belongs to the @@ -384,11 +386,17 @@ mod tests { #[test] fn the_local_fixture_is_co_resident() { assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, TARGET), + crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + TARGET, + ), "'{SOURCE}' and '{TARGET}' must share a vShard for the inline fold to be observable" ); assert!( - !crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, REMOTE_TARGET), + !crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + REMOTE_TARGET, + ), "'{REMOTE_TARGET}' must NOT share '{SOURCE}'s vShard; it pins the deferred path" ); } @@ -777,13 +785,19 @@ mod tests { #[test] fn the_local_fixture_is_co_resident_for_path_level_tests() { assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, TARGET), + crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + TARGET, + ), "'{SOURCE}' and '{TARGET}' must share a vShard: the inline fold writes the target \ inside the source's transaction, on the source's core, and each core opens its own \ document store" ); assert!( - !crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, REMOTE_TARGET), + !crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + REMOTE_TARGET, + ), "'{REMOTE_TARGET}' must NOT share '{SOURCE}'s vShard; it pins the deferred rule" ); } @@ -1375,8 +1389,10 @@ mod tests { if [SOURCE, TARGET, JOIN_SOURCE].contains(&candidate.as_str()) { continue; } - if crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, &candidate) - { + if crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + &candidate, + ) { return candidate; } } @@ -1458,7 +1474,10 @@ mod tests { // path instead and leave the collision this test exists for uncovered. for target in [TARGET, second_target.as_str()] { assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, target), + crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + target, + ), "'{target}' must share '{SOURCE}'s vShard for its inline fold to be observable" ); } diff --git a/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs b/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs index 9115eb9dd..f19fca434 100644 --- a/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs +++ b/nodedb/src/data/executor/enforcement/materialized_sum/divergence.rs @@ -89,6 +89,15 @@ impl CoreLoop { let Some(config) = self.doc_configs.get(&key) else { return false; }; + // A name that will not de-qualify fails the statement in the write + // path with the same typed error. That is not a prediction drift, and + // reporting it as one would retry the same failing statement forever. + let Ok(source) = nodedb_types::CollectionKey::from_qualified_str( + DatabaseId::new(check.database_id), + check.collection, + ) else { + return false; + }; for binding in &config.enforcement.materialized_sum_sources { // A CROSS-SHARD binding's join values are deliberately ABSENT from // the resolution: the Control Plane settled their deltas at plan @@ -101,11 +110,7 @@ impl CoreLoop { // instead: the images the shipped deltas were folded from are // stamped as a read, and this core votes ABORT before any mutation // if they have moved. - if !crate::query::sum_target_is_co_resident( - DatabaseId::new(check.database_id), - check.collection, - &binding.target_collection, - ) { + if !crate::query::sum_target_is_co_resident(source, &binding.target_collection) { continue; } // An assignment that will not evaluate fails the statement in the diff --git a/nodedb/src/data/executor/handlers/point/delete.rs b/nodedb/src/data/executor/handlers/point/delete.rs index e24d63d59..2d42fcd1a 100644 --- a/nodedb/src/data/executor/handlers/point/delete.rs +++ b/nodedb/src/data/executor/handlers/point/delete.rs @@ -330,7 +330,10 @@ mod tests { #[test] fn the_fixture_is_co_resident() { assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, TARGET), + crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + TARGET, + ), "'{SOURCE}' and '{TARGET}' must share a vShard: a cross-shard binding's balance \ travels on its own task and is never folded into the source write's transaction" ); diff --git a/nodedb/src/data/executor/handlers/point/insert.rs b/nodedb/src/data/executor/handlers/point/insert.rs index 8ef89caf4..52fc57f09 100644 --- a/nodedb/src/data/executor/handlers/point/insert.rs +++ b/nodedb/src/data/executor/handlers/point/insert.rs @@ -358,7 +358,10 @@ mod tests { #[test] fn the_fixture_is_co_resident() { assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, TARGET), + crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + TARGET, + ), "'{SOURCE}' and '{TARGET}' must share a vShard: a cross-shard binding's balance \ travels on its own task and is never folded into the source write's transaction" ); diff --git a/nodedb/src/data/executor/handlers/point/put.rs b/nodedb/src/data/executor/handlers/point/put.rs index caa666913..de575f4d4 100644 --- a/nodedb/src/data/executor/handlers/point/put.rs +++ b/nodedb/src/data/executor/handlers/point/put.rs @@ -267,7 +267,10 @@ mod tests { #[test] fn the_fixture_is_co_resident() { assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, TARGET), + crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + TARGET, + ), "'{SOURCE}' and '{TARGET}' must share a vShard: a cross-shard binding's balance \ travels on its own task and is never folded into the source write's transaction" ); diff --git a/nodedb/src/data/executor/handlers/point/update/exec.rs b/nodedb/src/data/executor/handlers/point/update/exec.rs index e061079ab..ed7354874 100644 --- a/nodedb/src/data/executor/handlers/point/update/exec.rs +++ b/nodedb/src/data/executor/handlers/point/update/exec.rs @@ -363,7 +363,10 @@ mod tests { #[test] fn the_fixture_is_co_resident() { assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, TARGET), + crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + TARGET, + ), "'{SOURCE}' and '{TARGET}' must share a vShard: a cross-shard binding's balance \ travels on its own task and is never folded into the source write's transaction" ); diff --git a/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs b/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs index f27ad9c6d..3cba8e583 100644 --- a/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs +++ b/nodedb/src/data/executor/handlers/snapshot/restore/tenant_snapshot.rs @@ -492,10 +492,8 @@ mod tests { let task = CoreLoop::replay_vector_task( crate::types::TenantId::new(0), nodedb_types::DatabaseId::DEFAULT, - crate::types::VShardId::from_collection_in_database( - nodedb_types::DatabaseId::DEFAULT, - "emb", - ), + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "emb") + .vshard(), PhysicalPlan::Meta(MetaOp::WalAppend { payload: Vec::new(), }), diff --git a/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs b/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs index 645e9f0d9..72652c8a7 100644 --- a/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs +++ b/nodedb/src/data/executor/handlers/timeseries_wal_payload.rs @@ -92,10 +92,13 @@ impl CoreLoop { "msgpack" } }); + let Some(vshard) = self.replay_vshard("timeseries", record_lsn, db_id, collection) else { + return 0; + }; let mut task = Self::replay_task( tid, db_id, - crate::types::VShardId::from_collection_in_database(db_id, collection), + vshard, PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), payload: payload.to_vec(), @@ -197,10 +200,13 @@ impl CoreLoop { // tenant_id, request_id}` — it never inspects the embedded plan. // Embed empty vecs for the plan-level surrogates/provenance to avoid // cloning the owned values we need to pass as explicit args below. + let Some(vshard) = self.replay_vshard("columnar", record_lsn, db_id, collection) else { + return 0; + }; let task = Self::replay_task( tid, db_id, - crate::types::VShardId::from_collection_in_database(db_id, collection), + vshard, PhysicalPlan::Columnar(ColumnarOp::Insert { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), payload: payload.to_vec(), diff --git a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs index 651f32639..4924d202b 100644 --- a/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs +++ b/nodedb/src/data/executor/handlers/upsert/exec/dispatch.rs @@ -185,7 +185,10 @@ mod tests { #[test] fn the_fixture_is_co_resident() { assert!( - crate::query::sum_target_is_co_resident(DatabaseId::DEFAULT, SOURCE, TARGET), + crate::query::sum_target_is_co_resident( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE), + TARGET, + ), "'{SOURCE}' and '{TARGET}' must share a vShard: a cross-shard binding's balance \ travels on its own task and is never folded into the source write's transaction" ); diff --git a/nodedb/src/data/executor/replay_task.rs b/nodedb/src/data/executor/replay_task.rs index d223b5604..3abef70fc 100644 --- a/nodedb/src/data/executor/replay_task.rs +++ b/nodedb/src/data/executor/replay_task.rs @@ -48,4 +48,32 @@ impl CoreLoop { resolved_now_ms: None, } } + + /// The vShard a replayed record's collection homes to. + /// + /// A replay record carries the stored, database-qualified collection + /// name. It is de-qualified into the canonical `CollectionKey`, so the + /// replay task carries the same vShard the live write did. A name that + /// does not de-qualify is reported as a record that cannot be applied, + /// and the caller skips the record. + pub(in crate::data::executor) fn replay_vshard( + &mut self, + engine: &str, + record_lsn: u64, + database_id: DatabaseId, + stored_collection: &str, + ) -> Option { + match nodedb_types::CollectionKey::from_qualified_str(database_id, stored_collection) { + Ok(key) => Some(key.vshard()), + Err(error) => { + self.replay_record_unapplied( + engine, + "collection key", + record_lsn, + &error.to_string(), + ); + None + } + } + } } diff --git a/nodedb/src/data/executor/wal_replay_columnar_dml.rs b/nodedb/src/data/executor/wal_replay_columnar_dml.rs index 25b0e0632..1267c8000 100644 --- a/nodedb/src/data/executor/wal_replay_columnar_dml.rs +++ b/nodedb/src/data/executor/wal_replay_columnar_dml.rs @@ -45,7 +45,7 @@ use super::core_loop::CoreLoop; use crate::bridge::envelope::{PhysicalPlan, Status}; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::types::{DatabaseId, Lsn, TenantId}; use nodedb_physical::physical_plan::ColumnarOp; use nodedb_types::RlsWriteCheck; use nodedb_types::Value; @@ -102,7 +102,13 @@ impl CoreLoop { } let tid = TenantId::new(tenant_id); - let vshard_id = VShardId::from_collection_in_database(database_id, &record.collection); + // The record decoded as this shape, so a skip is `Some(0)`, never a + // fall-through to the row-payload decoders. + let Some(vshard_id) = + self.replay_vshard("columnar", record_lsn, database_id, &record.collection) + else { + return Some(0); + }; // The task carries the real predicate even though today's handlers read // only `task.request.{database_id, tenant_id}`. A placeholder plan would @@ -260,7 +266,13 @@ impl CoreLoop { } let tid = TenantId::new(tenant_id); - let vshard_id = VShardId::from_collection_in_database(database_id, &record.collection); + // The record decoded as this shape, so a skip is `Some(0)`, never a + // fall-through to the row-payload decoders. + let Some(vshard_id) = + self.replay_vshard("columnar", record_lsn, database_id, &record.collection) + else { + return Some(0); + }; let replay_check = RlsWriteCheck::already_decided_elsewhere(); let response = if record.is_update { diff --git a/nodedb/src/data/executor/wal_replay_columnar_image.rs b/nodedb/src/data/executor/wal_replay_columnar_image.rs index 54a059441..8e48f478f 100644 --- a/nodedb/src/data/executor/wal_replay_columnar_image.rs +++ b/nodedb/src/data/executor/wal_replay_columnar_image.rs @@ -25,7 +25,7 @@ use crate::bridge::envelope::{ErrorCode, PhysicalPlan}; use crate::data::executor::handlers::columnar_write::ndb_field_to_value; use crate::data::executor::handlers::transaction::undo::UndoEntry; use crate::data::executor::task::ExecutionTask; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::types::{DatabaseId, Lsn, TenantId}; /// One decoded row of a `columnar_image` record. struct ImageRow { @@ -146,10 +146,14 @@ impl CoreLoop { return Some(0); } }; + let Some(vshard) = self.replay_vshard("columnar", record_lsn, database_id, collection) + else { + return Some(0); + }; let task = Self::replay_task( tid, database_id, - VShardId::from_collection_in_database(database_id, collection), + vshard, PhysicalPlan::Columnar(ColumnarOp::Insert { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), payload: plan_payload, diff --git a/nodedb/src/data/executor/wal_replay_columnar_truncate.rs b/nodedb/src/data/executor/wal_replay_columnar_truncate.rs index 8ec0a0fe7..685320780 100644 --- a/nodedb/src/data/executor/wal_replay_columnar_truncate.rs +++ b/nodedb/src/data/executor/wal_replay_columnar_truncate.rs @@ -26,7 +26,7 @@ use nodedb_wal::record::RecordType; use super::core_loop::CoreLoop; use super::timeseries_checkpoint::stamp::TsReplayStamp; use crate::bridge::envelope::{PhysicalPlan, Status}; -use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::types::{DatabaseId, Lsn, TenantId}; use nodedb_physical::physical_plan::{ColumnarOp, TimeseriesOp}; /// The key of one columnar-family collection on a core. @@ -213,10 +213,15 @@ impl CoreLoop { if self.claim_for_validation() { return false; } + let Some(vshard) = + self.replay_vshard("columnar", record_lsn, database_id, &record.collection) + else { + return false; + }; let task = Self::replay_task( TenantId::new(tenant_id), database_id, - VShardId::from_collection_in_database(database_id, &record.collection), + vshard, PhysicalPlan::Columnar(ColumnarOp::Truncate { collection: nodedb_types::QualifiedCollection::from_stored( record.collection.clone(), @@ -287,10 +292,15 @@ impl CoreLoop { if self.claim_for_validation() { return false; } + let Some(vshard) = + self.replay_vshard("timeseries", record_lsn, database_id, &record.collection) + else { + return false; + }; let task = Self::replay_task( tid, database_id, - VShardId::from_collection_in_database(database_id, &record.collection), + vshard, PhysicalPlan::Timeseries(TimeseriesOp::Truncate { collection: nodedb_types::QualifiedCollection::from_stored( record.collection.clone(), diff --git a/nodedb/src/data/executor/wal_replay_fts.rs b/nodedb/src/data/executor/wal_replay_fts.rs index 9f22e556a..d815552a5 100644 --- a/nodedb/src/data/executor/wal_replay_fts.rs +++ b/nodedb/src/data/executor/wal_replay_fts.rs @@ -193,10 +193,12 @@ impl CoreLoop { } } - let vshard = crate::types::VShardId::from_collection_in_database( - database_id, - &payload.collection, - ); + let Some(vshard) = + self.replay_vshard("fts", record_lsn, database_id, &payload.collection) + else { + skipped += 1; + continue; + }; let task = Self::replay_fts_task( nodedb_types::TenantId::new(tenant_id), database_id, @@ -295,10 +297,12 @@ impl CoreLoop { } } - let vshard = crate::types::VShardId::from_collection_in_database( - database_id, - &payload.collection, - ); + let Some(vshard) = + self.replay_vshard("fts", record_lsn, database_id, &payload.collection) + else { + skipped += 1; + continue; + }; let task = Self::replay_fts_task( nodedb_types::TenantId::new(tenant_id), database_id, diff --git a/nodedb/src/data/executor/wal_replay_spatial.rs b/nodedb/src/data/executor/wal_replay_spatial.rs index 6681c5c8e..d08f6bdb7 100644 --- a/nodedb/src/data/executor/wal_replay_spatial.rs +++ b/nodedb/src/data/executor/wal_replay_spatial.rs @@ -226,10 +226,12 @@ impl CoreLoop { continue; } - let vshard = crate::types::VShardId::from_collection_in_database( - database_id, - &payload.collection, - ); + let Some(vshard) = + self.replay_vshard("spatial", record_lsn, database_id, &payload.collection) + else { + skipped += 1; + continue; + }; let task = Self::replay_spatial_task( nodedb_types::TenantId::new(tenant_id), database_id, @@ -327,10 +329,12 @@ impl CoreLoop { continue; } - let vshard = crate::types::VShardId::from_collection_in_database( - database_id, - &payload.collection, - ); + let Some(vshard) = + self.replay_vshard("spatial", record_lsn, database_id, &payload.collection) + else { + skipped += 1; + continue; + }; let task = Self::replay_spatial_task( nodedb_types::TenantId::new(tenant_id), database_id, diff --git a/nodedb/src/data/executor/wal_replay_vector.rs b/nodedb/src/data/executor/wal_replay_vector.rs index 4079e75c0..95f52e410 100644 --- a/nodedb/src/data/executor/wal_replay_vector.rs +++ b/nodedb/src/data/executor/wal_replay_vector.rs @@ -172,10 +172,15 @@ impl CoreLoop { // compat doc-id slot (always `None` on this write path) // maps straight through to `pk_bytes` for fidelity. let pk_bytes = doc_id.as_ref().map(|d| d.as_bytes().to_vec()); - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + skipped += 1; + continue; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), diff --git a/nodedb/src/data/executor/wal_replay_vector_delete.rs b/nodedb/src/data/executor/wal_replay_vector_delete.rs index 87af61862..2678a0b24 100644 --- a/nodedb/src/data/executor/wal_replay_vector_delete.rs +++ b/nodedb/src/data/executor/wal_replay_vector_delete.rs @@ -64,10 +64,14 @@ impl CoreLoop { ) { return false; } - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), diff --git a/nodedb/src/data/executor/wal_replay_vector_direct.rs b/nodedb/src/data/executor/wal_replay_vector_direct.rs index 571348ee7..86cfd7b54 100644 --- a/nodedb/src/data/executor/wal_replay_vector_direct.rs +++ b/nodedb/src/data/executor/wal_replay_vector_direct.rs @@ -61,10 +61,14 @@ impl CoreLoop { ) { return false; } - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), @@ -141,10 +145,14 @@ impl CoreLoop { return false; } } - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), @@ -217,10 +225,14 @@ impl CoreLoop { ) { return false; } - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), diff --git a/nodedb/src/data/executor/wal_replay_vector_extended.rs b/nodedb/src/data/executor/wal_replay_vector_extended.rs index 14ad54613..8830aa5e6 100644 --- a/nodedb/src/data/executor/wal_replay_vector_extended.rs +++ b/nodedb/src/data/executor/wal_replay_vector_extended.rs @@ -220,10 +220,14 @@ impl CoreLoop { ) { return false; } - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), @@ -323,10 +327,14 @@ impl CoreLoop { ) { return false; } - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), @@ -411,10 +419,14 @@ impl CoreLoop { ) { return false; } - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), diff --git a/nodedb/src/data/executor/wal_replay_vector_resolved.rs b/nodedb/src/data/executor/wal_replay_vector_resolved.rs index d67b30766..c396686d0 100644 --- a/nodedb/src/data/executor/wal_replay_vector_resolved.rs +++ b/nodedb/src/data/executor/wal_replay_vector_resolved.rs @@ -71,10 +71,14 @@ impl CoreLoop { ) { return false; } - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let rls_write_check = nodedb_types::RlsWriteCheck::already_decided_elsewhere(); let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), diff --git a/nodedb/src/data/executor/wal_replay_vector_sparse.rs b/nodedb/src/data/executor/wal_replay_vector_sparse.rs index 9700d8b43..a09953701 100644 --- a/nodedb/src/data/executor/wal_replay_vector_sparse.rs +++ b/nodedb/src/data/executor/wal_replay_vector_sparse.rs @@ -82,10 +82,14 @@ impl CoreLoop { return false; } self.record_sparse_doc_undo(database_id, tenant_id, (&collection, &field_name), &doc_id); - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "sparse_vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), @@ -149,10 +153,14 @@ impl CoreLoop { return false; } self.record_sparse_doc_undo(database_id, tenant_id, (&collection, &field_name), &doc_id); - let vshard = crate::types::VShardId::from_collection_in_database( + let Some(vshard) = self.replay_vshard( + "sparse_vector", + record_lsn, DatabaseId::new(database_id), &collection, - ); + ) else { + return false; + }; let task = Self::replay_vector_task( nodedb_types::TenantId::new(tenant_id), DatabaseId::new(database_id), diff --git a/nodedb/src/engine/bitemporal/enforcement.rs b/nodedb/src/engine/bitemporal/enforcement.rs index 67aff6bce..74e99de1b 100644 --- a/nodedb/src/engine/bitemporal/enforcement.rs +++ b/nodedb/src/engine/bitemporal/enforcement.rs @@ -153,8 +153,7 @@ async fn run_one(state: &Arc, entry: &Entry) { crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::RetentionEnforcement, tenant_id, - entry.database_id, - &entry.collection, + nodedb_types::CollectionKey::from_bare(entry.database_id, &entry.collection), plan, ), Duration::from_secs(DISPATCH_DEADLINE_SECS), diff --git a/nodedb/src/engine/timeseries/retention_policy/autowire.rs b/nodedb/src/engine/timeseries/retention_policy/autowire.rs index 77b426e97..820ceb63b 100644 --- a/nodedb/src/engine/timeseries/retention_policy/autowire.rs +++ b/nodedb/src/engine/timeseries/retention_policy/autowire.rs @@ -72,8 +72,7 @@ pub async fn register_tiers( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::RetentionEnforcement, tenant_id, - DatabaseId::new(def.database_id), - &source, + nodedb_types::CollectionKey::from_bare(DatabaseId::new(def.database_id), &source), plan, ), Duration::from_secs(5), @@ -118,8 +117,7 @@ pub async fn unregister_tiers( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::RetentionEnforcement, tenant_id, - DatabaseId::new(def.database_id), - route_collection, + nodedb_types::CollectionKey::from_bare(DatabaseId::new(def.database_id), route_collection), plan, ), Duration::from_secs(5), diff --git a/nodedb/src/engine/timeseries/retention_policy/enforcement.rs b/nodedb/src/engine/timeseries/retention_policy/enforcement.rs index ff79dbf7b..6e51082e4 100644 --- a/nodedb/src/engine/timeseries/retention_policy/enforcement.rs +++ b/nodedb/src/engine/timeseries/retention_policy/enforcement.rs @@ -102,8 +102,7 @@ async fn enforcement_loop( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::RetentionEnforcement, tenant_id, - DatabaseId::new(policy.database_id), - &policy.collection, + nodedb_types::CollectionKey::from_bare(DatabaseId::new(policy.database_id), &policy.collection), plan, ), Duration::from_secs(30), @@ -133,8 +132,7 @@ async fn enforcement_loop( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::RetentionEnforcement, tenant_id, - DatabaseId::new(policy.database_id), - &policy.collection, + nodedb_types::CollectionKey::from_bare(DatabaseId::new(policy.database_id), &policy.collection), plan, ), Duration::from_secs(30), @@ -193,8 +191,10 @@ async fn check_watermark_coverage( crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( crate::control::server::shared::ddl::sync_dispatch::SystemReason::RetentionEnforcement, tenant_id, - DatabaseId::new(policy.database_id), - &policy.collection, + nodedb_types::CollectionKey::from_bare( + DatabaseId::new(policy.database_id), + &policy.collection, + ), plan, ), Duration::from_secs(10), diff --git a/nodedb/src/error/conversions.rs b/nodedb/src/error/conversions.rs index 3899a4769..a05f6aa0a 100644 --- a/nodedb/src/error/conversions.rs +++ b/nodedb/src/error/conversions.rs @@ -1,7 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 //! `From` impls that build a [`super::Error`] from `nodedb-physical` error -//! types (wire decoding, physical-plan conversion). Kept apart from the enum +//! types (wire decoding, physical-plan conversion) and from the +//! `nodedb-types` collection-key error. Kept apart from the enum //! definition in `types.rs` so a new physical-layer error source has one //! obvious home instead of growing the enum file further. @@ -41,3 +42,14 @@ impl From for Error { } } } + +/// A qualified collection name that does not carry its database's qualifier +/// reached a placement or surrogate path. Every qualified name is built by +/// `QualifiedCollection::new`, so this is an internal invariant break. +impl From for Error { + fn from(e: nodedb_types::CollectionKeyError) -> Self { + Error::Internal { + detail: e.to_string(), + } + } +} diff --git a/nodedb/src/event/alert/executor.rs b/nodedb/src/event/alert/executor.rs index 1df1d23f7..aa0747fd8 100644 --- a/nodedb/src/event/alert/executor.rs +++ b/nodedb/src/event/alert/executor.rs @@ -192,8 +192,10 @@ async fn execute_aggregate_scan( sync_dispatch::SystemTask::new( sync_dispatch::SystemReason::EventPlane, tenant_id, - crate::types::DatabaseId::new(alert.database_id), - &alert.collection, + nodedb_types::CollectionKey::from_bare( + crate::types::DatabaseId::new(alert.database_id), + &alert.collection, + ), plan, ), Duration::from_secs(30), diff --git a/nodedb/src/event/scheduler/executor.rs b/nodedb/src/event/scheduler/executor.rs index 711002184..16d79ad62 100644 --- a/nodedb/src/event/scheduler/executor.rs +++ b/nodedb/src/event/scheduler/executor.rs @@ -316,10 +316,11 @@ fn should_fire_on_this_node(sched: &ScheduleDef, state: &SharedState) -> bool { if let Some(ref collection) = sched.target_collection { // Collection-targeted schedule: fire only on the shard leader in // the database that owns the schedule definition. - let vshard_id = nodedb_cluster::routing::vshard_for_collection( - nodedb_types::id::DatabaseId::new(sched.database_id), - collection, - ); + let vshard_id = + nodedb_cluster::routing::vshard_for_collection(nodedb_types::CollectionKey::from_bare( + nodedb_types::id::DatabaseId::new(sched.database_id), + collection, + )); let routing = routing_lock.read().unwrap_or_else(|p| p.into_inner()); match routing.leader_for_vshard(vshard_id) { Ok(leader) => leader == node_id, @@ -365,10 +366,10 @@ fn is_raft_group_healthy(sched: &ScheduleDef, state: &SharedState) -> bool { .target_collection .as_ref() .map(|c| { - nodedb_cluster::routing::vshard_for_collection( + nodedb_cluster::routing::vshard_for_collection(nodedb_types::CollectionKey::from_bare( nodedb_types::id::DatabaseId::new(sched.database_id), c, - ) + )) }) .unwrap_or(0); // Cross-collection → coordinator vShard 0. diff --git a/nodedb/src/event/topic/publish.rs b/nodedb/src/event/topic/publish.rs index 9fc924706..7827d92b9 100644 --- a/nodedb/src/event/topic/publish.rs +++ b/nodedb/src/event/topic/publish.rs @@ -142,7 +142,9 @@ fn get_or_create_topic_buffer( /// Determine the home node for a topic. fn topic_home_node(state: &SharedState, database_id: DatabaseId, topic_name: &str) -> Option { let routing_lock = state.cluster_routing.as_ref()?; - let vshard_id = nodedb_cluster::routing::vshard_for_collection(database_id, topic_name); + let vshard_id = nodedb_cluster::routing::vshard_for_collection( + nodedb_types::CollectionKey::from_bare(database_id, topic_name), + ); let routing = routing_lock.read().unwrap_or_else(|p| p.into_inner()); routing.leader_for_vshard(vshard_id).ok() } diff --git a/nodedb/src/query/materialized_sum_homing.rs b/nodedb/src/query/materialized_sum_homing.rs index f2db1805e..f73f6c7f5 100644 --- a/nodedb/src/query/materialized_sum_homing.rs +++ b/nodedb/src/query/materialized_sum_homing.rs @@ -17,43 +17,27 @@ //! that both planes call. It reads no catalog and touches no storage, which is //! also what lets the Data Plane use it without holding Control-Plane state. -use crate::types::{DatabaseId, VShardId}; +use nodedb_types::CollectionKey; -/// Qualify a catalog collection name into the db-scoped name every plan carries. -/// -/// Homing hashes the name AS IT APPEARS ON THE PLAN, so a target named only by -/// the catalog has to be qualified the same way before it can be compared with a -/// source that already is. -pub fn db_qualified(database_id: DatabaseId, collection: &str) -> String { - nodedb_types::QualifiedCollection::new(database_id, collection) - .as_str() - .to_owned() -} +use crate::types::{DatabaseId, VShardId}; /// The vShard a materialized-sum target collection homes to. /// -/// `target_collection` is the CATALOG name carried on the binding; it is -/// qualified here so the result matches how the target's own writes route. +/// `target_collection` is the bare CATALOG name carried on the binding. It +/// forms the same [`CollectionKey`] the target's own writes route by. pub fn sum_target_vshard(database_id: DatabaseId, target_collection: &str) -> VShardId { - VShardId::from_collection_in_database( - database_id, - &db_qualified(database_id, target_collection), - ) + CollectionKey::from_bare(database_id, target_collection).vshard() } /// Whether the balance write may ride the source write's transaction. /// -/// `source_collection` is the source's name as it appears on the plan — the same -/// string its own task is homed on. `true` means one core owns both rows and the -/// derived write is atomic for free; `false` means the balance needs its own -/// task on the target's vShard, dual-homed with the source through Calvin. -pub fn sum_target_is_co_resident( - database_id: DatabaseId, - source_collection: &str, - target_collection: &str, -) -> bool { - VShardId::from_collection_in_database(database_id, source_collection) - == sum_target_vshard(database_id, target_collection) +/// `source` is the source collection's key, de-qualified from the name on the +/// plan by the caller. The target shares the source's database. `true` means +/// one core owns both rows and the derived write is atomic for free. `false` +/// means the balance needs its own task on the target's vShard, dual-homed +/// with the source through Calvin. +pub fn sum_target_is_co_resident(source: CollectionKey<'_>, target_collection: &str) -> bool { + source.vshard() == sum_target_vshard(source.database_id(), target_collection) } #[cfg(test)] @@ -68,9 +52,9 @@ mod tests { for i in 0..256 { let source = format!("src_{i}"); let target = format!("dst_{i}"); - let co_resident = sum_target_is_co_resident(db, &source, &target); - let same_home = VShardId::from_collection_in_database(db, &source) - == sum_target_vshard(db, &target); + let source_key = CollectionKey::from_bare(db, &source); + let co_resident = sum_target_is_co_resident(source_key, &target); + let same_home = source_key.vshard() == sum_target_vshard(db, &target); assert_eq!(co_resident, same_home); } } @@ -80,8 +64,9 @@ mod tests { #[test] fn a_collection_is_co_resident_with_itself() { for db in [DatabaseId::DEFAULT, DatabaseId::new(7)] { - let qualified = db_qualified(db, "ledger"); - assert!(sum_target_is_co_resident(db, &qualified, "ledger")); + let qualified = nodedb_types::QualifiedCollection::new(db, "ledger"); + let source = CollectionKey::from_qualified(db, &qualified).expect("qualified"); + assert!(sum_target_is_co_resident(source, "ledger")); } } @@ -92,7 +77,13 @@ mod tests { fn distinct_collections_are_usually_cross_shard() { let db = DatabaseId::DEFAULT; let cross = (0..512) - .filter(|i| !sum_target_is_co_resident(db, &format!("src_{i}"), &format!("dst_{i}"))) + .filter(|i| { + let source = format!("src_{i}"); + !sum_target_is_co_resident( + CollectionKey::from_bare(db, &source), + &format!("dst_{i}"), + ) + }) .count(); assert!( cross > 256, diff --git a/nodedb/src/query/mod.rs b/nodedb/src/query/mod.rs index 0030fc169..d80469664 100644 --- a/nodedb/src/query/mod.rs +++ b/nodedb/src/query/mod.rs @@ -18,7 +18,7 @@ pub use fusion::{ pub use materialized_sum_delta::{ binding_amount, binding_insert_deltas, binding_join_value, json_to_decimal, }; -pub use materialized_sum_homing::{db_qualified, sum_target_is_co_resident, sum_target_vshard}; +pub use materialized_sum_homing::{sum_target_is_co_resident, sum_target_vshard}; pub use materialized_sum_images::{ BindingDelta, apply_conflict_assignments, apply_update_assignments, binding_image_deltas, coalesce_binding_deltas, diff --git a/nodedb/src/wal/replay/surrogate/bind.rs b/nodedb/src/wal/replay/surrogate/bind.rs index 98cd000e6..15922e584 100644 --- a/nodedb/src/wal/replay/surrogate/bind.rs +++ b/nodedb/src/wal/replay/surrogate/bind.rs @@ -22,9 +22,8 @@ pub fn apply_surrogate_bind( let parsed = SurrogateBindPayload::from_bytes(payload).map_err(crate::Error::Wal)?; let surrogate = Surrogate::new(parsed.surrogate); catalog.put_surrogate( - database_id, + nodedb_types::CollectionKey::from_bare(database_id, &parsed.collection), tenant_id, - &parsed.collection, &parsed.pk_bytes, surrogate, )?; @@ -64,15 +63,18 @@ mod tests { .unwrap(); apply_surrogate_bind(&payload, DatabaseId::DEFAULT, TenantId::new(0), &cat, ®).unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, TenantId::new(0), "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(0), + b"alice" + ) + .unwrap(), Some(Surrogate::new(7)) ); assert_eq!( cat.get_pk_for_surrogate( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), TenantId::new(0), - "users", Surrogate::new(7) ) .unwrap(), @@ -90,8 +92,12 @@ mod tests { apply_surrogate_bind(&payload, DatabaseId::DEFAULT, TenantId::new(0), &cat, ®).unwrap(); apply_surrogate_bind(&payload, DatabaseId::DEFAULT, TenantId::new(0), &cat, ®).unwrap(); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, TenantId::new(0), "users", b"bob") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(0), + b"bob" + ) + .unwrap(), Some(Surrogate::new(3)) ); assert_eq!(reg.read().unwrap().current_hwm(), 3); diff --git a/nodedb/src/wal/replay/surrogate/dispatch.rs b/nodedb/src/wal/replay/surrogate/dispatch.rs index 76c6896c8..a982f41fa 100644 --- a/nodedb/src/wal/replay/surrogate/dispatch.rs +++ b/nodedb/src/wal/replay/surrogate/dispatch.rs @@ -176,8 +176,12 @@ mod tests { assert_eq!(stats.binds, 1); assert_eq!(stats.binds_skipped, 0); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, TenantId::new(0), "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(0), + b"alice" + ) + .unwrap(), Some(nodedb_types::Surrogate::new(7)) ); } @@ -201,8 +205,12 @@ mod tests { assert_eq!(stats.binds, 0); assert_eq!(stats.binds_skipped, 1); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, TenantId::new(0), "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(0), + b"alice" + ) + .unwrap(), None, "a pre-drop binding must stay deleted" ); @@ -225,8 +233,12 @@ mod tests { .expect("replay"); assert_eq!(stats.binds, 1); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, TenantId::new(0), "users", b"bob") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(0), + b"bob" + ) + .unwrap(), Some(nodedb_types::Surrogate::new(11)) ); } @@ -245,8 +257,12 @@ mod tests { hwm_after_first ); assert_eq!( - cat.get_surrogate_for_pk(DatabaseId::DEFAULT, TenantId::new(0), "users", b"alice") - .unwrap(), + cat.get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), + TenantId::new(0), + b"alice" + ) + .unwrap(), Some(nodedb_types::Surrogate::new(7)) ); } diff --git a/nodedb/tests/crash_harness/vshards.rs b/nodedb/tests/crash_harness/vshards.rs index eb411d462..43575f574 100644 --- a/nodedb/tests/crash_harness/vshards.rs +++ b/nodedb/tests/crash_harness/vshards.rs @@ -2,7 +2,7 @@ //! Collection names that home on distinct vShards. -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::DatabaseId; /// One collection name per prefix, `_`, each on a vShard no other /// returned name homes on. A transaction that writes two of them spans two @@ -13,8 +13,9 @@ pub fn names_on_distinct_vshards(prefixes: [&str; N]) -> [String let (name, vshard) = (0..512u32) .map(|i| { let name = format!("{prefix}_{i}"); - let vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, &name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); (name, vshard) }) .find(|(_, vshard)| !taken.contains(vshard)) @@ -30,7 +31,9 @@ pub const SINGLE_NODE_DATA_GROUPS: u32 = 4; /// The data Raft group a single-node server applies writes to `name` in. pub fn single_node_data_group(name: &str) -> u32 { - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, name) + .vshard() + .as_u32(); 1 + vshard % SINGLE_NODE_DATA_GROUPS } diff --git a/nodedb/tests/inproc/cases/calvin_executor_caps.rs b/nodedb/tests/inproc/cases/calvin_executor_caps.rs index 3e73c8a7e..f4546864a 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_caps.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_caps.rs @@ -15,7 +15,7 @@ use nodedb_cluster::calvin::types::{ DependentReadSpec, EngineKeySet, PassiveReadKey, ReadWriteSet, SortedVec, TxClass, VersionedReadSet, }; -use nodedb_types::{TenantId, id::VShardId}; +use nodedb_types::TenantId; /// Find two collection names whose vShards differ. fn two_distinct_collections() -> (String, String) { @@ -23,7 +23,8 @@ fn two_distinct_collections() -> (String, String) { for i in 0u32..512 { let name = format!("col_{i}"); let vshard = - VShardId::from_collection_in_database(nodedb::types::DatabaseId::DEFAULT, &name) + nodedb_types::CollectionKey::from_bare(nodedb::types::DatabaseId::DEFAULT, &name) + .vshard() .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { diff --git a/nodedb/tests/inproc/cases/calvin_executor_dependent_read.rs b/nodedb/tests/inproc/cases/calvin_executor_dependent_read.rs index 5fd3a4aa0..0d510760f 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_dependent_read.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_dependent_read.rs @@ -22,7 +22,6 @@ use nodedb_cluster::calvin::types::{ DependentReadSpec, EngineKeySet, PassiveReadKey, ReadWriteSet, SequencedTxn, SortedVec, TxClass, VersionedReadSet, }; -use nodedb_types::id::VShardId; use nodedb_types::{DatabaseId, QualifiedCollection}; fn two_distinct_collections() -> (String, String) { @@ -30,7 +29,8 @@ fn two_distinct_collections() -> (String, String) { for i in 0u32..512 { let name = format!("col_{i}"); let vshard = - VShardId::from_collection_in_database(nodedb::types::DatabaseId::DEFAULT, &name) + nodedb_types::CollectionKey::from_bare(nodedb::types::DatabaseId::DEFAULT, &name) + .vshard() .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { diff --git a/nodedb/tests/inproc/cases/calvin_executor_ollp_property.rs b/nodedb/tests/inproc/cases/calvin_executor_ollp_property.rs index 973b46a50..1a9852afa 100644 --- a/nodedb/tests/inproc/cases/calvin_executor_ollp_property.rs +++ b/nodedb/tests/inproc/cases/calvin_executor_ollp_property.rs @@ -21,14 +21,15 @@ use nodedb_cluster::calvin::{ sequencer::{SequencerConfig, new_inbox}, types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass, VersionedReadSet}, }; -use nodedb_types::{TenantId, id::VShardId}; +use nodedb_types::TenantId; fn two_distinct_collections() -> (String, String) { let mut first: Option<(String, u32)> = None; for i in 0u32..512 { let name = format!("col_{i}"); let vshard = - VShardId::from_collection_in_database(nodedb::types::DatabaseId::DEFAULT, &name) + nodedb_types::CollectionKey::from_bare(nodedb::types::DatabaseId::DEFAULT, &name) + .vshard() .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { diff --git a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs index 3f82e35bb..5d85bd0b8 100644 --- a/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs +++ b/nodedb/tests/inproc/cases/calvin_two_phase_apply.rs @@ -337,8 +337,9 @@ fn drop_discards_invalid_staged_calvin_write() { // The read entry's collection must home to the staged request's vShard for // the read-set check to consider it. - let read_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "dropcoll").as_u32(); + let read_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "dropcoll") + .vshard() + .as_u32(); // A read of `dropcoll` observed at LSN 50 — stale against the seed's write // at LSN 100 → the read-set is no longer current → abort vote. @@ -439,8 +440,9 @@ fn point_read_at_write_lsn_commits_and_flush_applies() { ); assert_eq!(seed.status, Status::Ok, "seed write must commit: {seed:?}"); - let point_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "pointcoll").as_u32(); + let point_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "pointcoll") + .vshard() + .as_u32(); // A Point read of the exact same key observed at LSN 10 (== the write) is // still current: no write happened AFTER the read. @@ -521,8 +523,9 @@ fn stale_point_read_of_kv_key_aborts_stage_and_drop_discards() { ); assert_eq!(seed.status, Status::Ok, "seed write must commit: {seed:?}"); - let stale_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "stalecoll").as_u32(); + let stale_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "stalecoll") + .vshard() + .as_u32(); // A Point read of the exact same key observed at LSN 5 — stale against the // write at LSN 10 → the read-set is no longer current → abort vote. @@ -607,8 +610,9 @@ fn stale_point_read_of_kv_key_aborts_stage_and_drop_discards() { fn absent_kv_key_phantom_insert_causes_abort() { let (mut core, mut tx, mut rx, _dir) = make_core(); - let phantom_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "phantomkv").as_u32(); + let phantom_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "phantomkv") + .vshard() + .as_u32(); // The key was absent when read at (the then-current watermark) LSN 5 — no // write is seeded yet. @@ -683,8 +687,9 @@ fn absent_kv_key_phantom_insert_causes_abort() { fn absent_document_phantom_insert_is_caught() { let (mut core, mut tx, mut rx, _dir) = make_core(); - let doc_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "phantomdocs").as_u32(); + let doc_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "phantomdocs") + .vshard() + .as_u32(); // The document was absent when read: capture degraded the miss to a // collection-scoped predicate on "phantomdocs" at read_lsn 5. @@ -757,8 +762,9 @@ fn absent_document_phantom_insert_is_caught() { fn absent_document_read_without_matching_insert_still_commits() { let (mut core, mut tx, mut rx, _dir) = make_core(); - let doc_vshard = - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "phantomdocs").as_u32(); + let doc_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "phantomdocs") + .vshard() + .as_u32(); let absent_doc_read = VersionedReadEntry { engine: EngineTag::Document, diff --git a/nodedb/tests/inproc/cases/columnar_natural_pk_identity.rs b/nodedb/tests/inproc/cases/columnar_natural_pk_identity.rs index 309480ce0..aea537927 100644 --- a/nodedb/tests/inproc/cases/columnar_natural_pk_identity.rs +++ b/nodedb/tests/inproc/cases/columnar_natural_pk_identity.rs @@ -68,7 +68,10 @@ async fn columnar_natural_pk_rows_get_distinct_surrogates() { // the time the INSERT returns. let catalog = server.shared.credentials.catalog(); let bindings = catalog - .scan_surrogates_for_collection(DatabaseId::DEFAULT, TenantId::new(1), "parts") + .scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "parts"), + TenantId::new(1), + ) .expect("scan persisted surrogate bindings for parts"); assert_eq!( diff --git a/nodedb/tests/inproc/cases/dml_resolve_pass.rs b/nodedb/tests/inproc/cases/dml_resolve_pass.rs index 55255a66b..bedc0f0fc 100644 --- a/nodedb/tests/inproc/cases/dml_resolve_pass.rs +++ b/nodedb/tests/inproc/cases/dml_resolve_pass.rs @@ -9,7 +9,7 @@ use nodedb_test_support::pgwire_harness::TestServer; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; /// Binding target (holds the balance) and binding source (drives it) for /// [`autocommit_update_from_join_on_a_sum_source_folds_the_balance`]. The names @@ -26,8 +26,8 @@ const JOIN_SOURCE: &str = "atm_adjust"; #[test] fn materialized_sum_source_and_target_are_co_resident() { assert_eq!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, SUM_SOURCE), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, SUM_TARGET), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SUM_SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SUM_TARGET).vshard(), "the autocommit sum test must exercise the CO-RESIDENT path; \ rename the collections until the two hashes agree again" ); diff --git a/nodedb/tests/inproc/cases/enforcement_materialized_sum_point.rs b/nodedb/tests/inproc/cases/enforcement_materialized_sum_point.rs index dd945fc0a..77afa0c62 100644 --- a/nodedb/tests/inproc/cases/enforcement_materialized_sum_point.rs +++ b/nodedb/tests/inproc/cases/enforcement_materialized_sum_point.rs @@ -21,7 +21,7 @@ use nodedb_test_support::pgwire_harness::TestServer; -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; /// Target and source. The names are chosen for their HASHES: they collide on /// one vShard, which is what makes every test below the co-resident path. See @@ -35,8 +35,8 @@ const SOURCE: &str = "uo_entries"; #[test] fn materialized_sum_source_and_target_are_co_resident() { assert_eq!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, SOURCE), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, TARGET), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, SOURCE).vshard(), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, TARGET).vshard(), "the point-shape tests in this file must exercise the CO-RESIDENT path; \ rename the collections until the two hashes agree again" ); diff --git a/nodedb/tests/inproc/cases/graph_cross_core_bfs.rs b/nodedb/tests/inproc/cases/graph_cross_core_bfs.rs index e1482c04c..dec23770d 100644 --- a/nodedb/tests/inproc/cases/graph_cross_core_bfs.rs +++ b/nodedb/tests/inproc/cases/graph_cross_core_bfs.rs @@ -20,7 +20,7 @@ use nodedb::control::server::broadcast::broadcast_call_count; use nodedb::control::server::shared::clone_write::{ CloneCheckedOutcome, InterceptAndAuthorizeParams, intercept_and_authorize, }; -use nodedb::types::{DatabaseId, TenantId, TraceId, VShardId}; +use nodedb::types::{DatabaseId, TenantId, TraceId}; use nodedb_physical::physical_plan::{BatchEdge, GraphOp, PhysicalPlan}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use nodedb_test_support::pgwire_harness::TestServer; @@ -31,7 +31,7 @@ async fn seed_star(server: &TestServer, collection: &str, leaf_prefix: &str, cou let database_id = DatabaseId::DEFAULT; let task = PhysicalTask { tenant_id, - vshard_id: VShardId::from_collection_in_database(database_id, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(), database_id, plan: PhysicalPlan::Graph(GraphOp::EdgePutBatch { edges: (0..count) diff --git a/nodedb/tests/inproc/cases/http_query_authorization.rs b/nodedb/tests/inproc/cases/http_query_authorization.rs index de86d2a7d..13e1b45dc 100644 --- a/nodedb/tests/inproc/cases/http_query_authorization.rs +++ b/nodedb/tests/inproc/cases/http_query_authorization.rs @@ -279,9 +279,8 @@ async fn crdt_apply_rejects_ungranted_custom_role_before_surrogate_assignment() srv.shared .surrogate_assigner .lookup( - DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection), TenantId::new(1), - collection, doc_id.as_bytes(), ) .expect("look up CRDT document surrogate"), diff --git a/nodedb/tests/inproc/cases/kv_atomic_surrogate_identity.rs b/nodedb/tests/inproc/cases/kv_atomic_surrogate_identity.rs index 43ea77f0e..bc9321ddb 100644 --- a/nodedb/tests/inproc/cases/kv_atomic_surrogate_identity.rs +++ b/nodedb/tests/inproc/cases/kv_atomic_surrogate_identity.rs @@ -52,7 +52,10 @@ async fn kv_incr_on_fresh_key_persists_a_real_surrogate() { // binding synchronously during plan conversion. let catalog = server.shared.credentials.catalog(); let bindings = catalog - .scan_surrogates_for_collection(DatabaseId::DEFAULT, TenantId::new(1), "c") + .scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "c"), + TenantId::new(1), + ) .expect("scan persisted surrogate bindings for c"); assert_eq!( diff --git a/nodedb/tests/inproc/cases/kv_field_transfer_surrogate_identity.rs b/nodedb/tests/inproc/cases/kv_field_transfer_surrogate_identity.rs index b95614ded..de499259a 100644 --- a/nodedb/tests/inproc/cases/kv_field_transfer_surrogate_identity.rs +++ b/nodedb/tests/inproc/cases/kv_field_transfer_surrogate_identity.rs @@ -52,7 +52,10 @@ async fn kv_field_set_on_existing_key_persists_a_real_surrogate() { let catalog = server.shared.credentials.catalog(); let bindings = catalog - .scan_surrogates_for_collection(DatabaseId::DEFAULT, TenantId::new(1), "cf") + .scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "cf"), + TenantId::new(1), + ) .expect("scan persisted surrogate bindings for cf"); assert_eq!( @@ -94,7 +97,10 @@ async fn kv_transfer_persists_two_distinct_surrogates() { let catalog = server.shared.credentials.catalog(); let bindings = catalog - .scan_surrogates_for_collection(DatabaseId::DEFAULT, TenantId::new(1), "ct") + .scan_surrogates_for_collection( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "ct"), + TenantId::new(1), + ) .expect("scan persisted surrogate bindings for ct"); // Fix: two bindings (debit from the insert, credit from the transfer), diff --git a/nodedb/tests/inproc/cases/snapshot_round_trip.rs b/nodedb/tests/inproc/cases/snapshot_round_trip.rs index 26f0847c8..3a0e6825b 100644 --- a/nodedb/tests/inproc/cases/snapshot_round_trip.rs +++ b/nodedb/tests/inproc/cases/snapshot_round_trip.rs @@ -42,7 +42,10 @@ async fn snapshot_round_trip_builder_to_applier() { let pks = ["pk0", "pk1", "pk2", "pk3", "pk4"]; // ── Sanity: the collection's vShard belongs to the data group we build. ─── - let vshard = vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + COLL, + )); let routing = single_node_routing(); assert!( routing.vshards_for_group(DATA_GROUP_ID).contains(&vshard), @@ -87,7 +90,11 @@ async fn snapshot_round_trip_builder_to_applier() { // The source must have a surrogate binding for pk0 (proves inserts allocated // identities the snapshot will carry). let source_surrogate = source_catalog - .get_surrogate_for_pk(DatabaseId::DEFAULT, tid, COLL, pks[0].as_bytes()) + .get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + tid, + pks[0].as_bytes(), + ) .expect("source get_surrogate_for_pk") .expect("source must have a surrogate for pk0"); @@ -154,7 +161,11 @@ async fn snapshot_round_trip_builder_to_applier() { // rebound on apply. let target_catalog = target.shared.credentials.catalog().clone(); let target_surrogate = target_catalog - .get_surrogate_for_pk(DatabaseId::DEFAULT, tid, COLL, pks[0].as_bytes()) + .get_surrogate_for_pk( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + tid, + pks[0].as_bytes(), + ) .expect("target get_surrogate_for_pk") .expect("target must have a rebound surrogate for pk0"); assert_eq!( @@ -186,7 +197,10 @@ async fn snapshot_round_trip_timeseries() { ]; // ── Sanity: the collection's vShard belongs to the data group we build. ─── - let vshard = vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + COLL, + )); assert!( single_node_routing() .vshards_for_group(DATA_GROUP_ID) @@ -278,7 +292,10 @@ async fn snapshot_round_trip_vector() { ]; // ── Sanity: the collection's vShard belongs to the data group we build. ─── - let vshard = vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + COLL, + )); assert!( single_node_routing() .vshards_for_group(DATA_GROUP_ID) @@ -370,7 +387,10 @@ async fn snapshot_round_trip_edges() { const FANOUT: usize = 8; // ── Sanity: the collection's vShard belongs to the data group we build. ─── - let vshard = vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + COLL, + )); assert!( single_node_routing() .vshards_for_group(DATA_GROUP_ID) diff --git a/nodedb/tests/inproc/cases/snapshot_round_trip_crdt.rs b/nodedb/tests/inproc/cases/snapshot_round_trip_crdt.rs index 88ca3764e..83321e8a1 100644 --- a/nodedb/tests/inproc/cases/snapshot_round_trip_crdt.rs +++ b/nodedb/tests/inproc/cases/snapshot_round_trip_crdt.rs @@ -36,7 +36,10 @@ async fn snapshot_round_trip_crdt() { const DOC: &str = "doc1"; // ── Sanity: the collection's vShard belongs to the data group we build. ─── - let vshard = vshard_for_collection(DatabaseId::DEFAULT, COLL); + let vshard = vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + COLL, + )); assert!( single_node_routing() .vshards_for_group(DATA_GROUP_ID) diff --git a/nodedb/tests/inproc/cases/snapshot_round_trip_stale_state.rs b/nodedb/tests/inproc/cases/snapshot_round_trip_stale_state.rs index cc1b2c660..b5c70af6c 100644 --- a/nodedb/tests/inproc/cases/snapshot_round_trip_stale_state.rs +++ b/nodedb/tests/inproc/cases/snapshot_round_trip_stale_state.rs @@ -40,7 +40,10 @@ async fn snapshot_round_trip_stale_state() { // ── Sanity: both collections' vShards belong to the data group we build. ── let routing = single_node_routing(); for coll in [COLL_A, COLL_B] { - let vshard = vshard_for_collection(DatabaseId::DEFAULT, coll); + let vshard = vshard_for_collection(nodedb_types::CollectionKey::from_bare( + DatabaseId::DEFAULT, + coll, + )); assert!( routing.vshards_for_group(DATA_GROUP_ID).contains(&vshard), "collection {coll} vShard {vshard} must belong to data group {DATA_GROUP_ID}" diff --git a/nodedb/tests/inproc/cases/sorted_index_rows.rs b/nodedb/tests/inproc/cases/sorted_index_rows.rs index 0ce43a85d..a59613489 100644 --- a/nodedb/tests/inproc/cases/sorted_index_rows.rs +++ b/nodedb/tests/inproc/cases/sorted_index_rows.rs @@ -21,7 +21,7 @@ //! Authorization for these same functions is covered by //! `sorted_index_authorization.rs`. -use nodedb::types::{DatabaseId, VShardId}; +use nodedb::types::DatabaseId; use nodedb_test_support::pgwire_harness::TestServer; /// Data-Plane cores the test server runs. More than one is the point: it is @@ -34,7 +34,10 @@ const INDEX: &str = "sidx_rows_lb"; /// The core a name is routed to, mirroring `VShardRouter::round_robin`. fn owning_core(name: &str) -> usize { - VShardId::from_collection_in_database(DatabaseId::DEFAULT, name).as_u32() as usize % CORES + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, name) + .vshard() + .as_u32() as usize + % CORES } /// A leaderboard with `rows` already stored, then indexed. Returns the server. diff --git a/nodedb/tests/inproc/cases/surrogate_pk.rs b/nodedb/tests/inproc/cases/surrogate_pk.rs index 0404e7899..483fa2289 100644 --- a/nodedb/tests/inproc/cases/surrogate_pk.rs +++ b/nodedb/tests/inproc/cases/surrogate_pk.rs @@ -37,25 +37,22 @@ fn assign_is_idempotent_for_same_pk() { let a = make_assigner(creds.clone(), wal); let s1 = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"alice", ) .unwrap(); let s2 = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"alice", ) .unwrap(); let s3 = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"alice", ) .unwrap(); @@ -64,9 +61,8 @@ fn assign_is_idempotent_for_same_pk() { let cat = creds.catalog(); assert_eq!( cat.get_surrogate_for_pk( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"alice" ) .unwrap(), @@ -74,9 +70,8 @@ fn assign_is_idempotent_for_same_pk() { ); assert_eq!( cat.get_pk_for_surrogate( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", s1 ) .unwrap(), @@ -91,25 +86,22 @@ fn assign_distinct_pks_returns_distinct_surrogates() { let a = make_assigner(creds, wal); let s1 = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"alice", ) .unwrap(); let s2 = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"bob", ) .unwrap(); let s3 = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"carol", ) .unwrap(); @@ -127,34 +119,30 @@ fn drop_collection_wipes_surrogate_map() { let a = make_assigner(creds.clone(), wal); let _ = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"alice", ) .unwrap(); let _ = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"bob", ) .unwrap(); let s_other = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "orders"), nodedb_types::TenantId::new(0), - "orders", b"o1", ) .unwrap(); let cat = creds.catalog(); assert_eq!( cat.scan_surrogates_for_collection( - nodedb_types::DatabaseId::DEFAULT, - nodedb_types::TenantId::new(0), - "users" + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), + nodedb_types::TenantId::new(0) ) .unwrap() .len(), @@ -162,25 +150,22 @@ fn drop_collection_wipes_surrogate_map() { ); cat.delete_all_surrogates_for_collection( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", ) .unwrap(); assert!( cat.scan_surrogates_for_collection( - nodedb_types::DatabaseId::DEFAULT, - nodedb_types::TenantId::new(0), - "users" + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), + nodedb_types::TenantId::new(0) ) .unwrap() .is_empty() ); assert_eq!( cat.get_surrogate_for_pk( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "orders"), nodedb_types::TenantId::new(0), - "orders", b"o1" ) .unwrap(), @@ -214,10 +199,9 @@ impl SurrogateWalAppender for CountingAppender { fn record_bind_to_wal( &self, - _database_id: nodedb_types::DatabaseId, + _key: nodedb_types::CollectionKey<'_>, _tenant_id: nodedb_types::TenantId, _surrogate: u32, - _collection: &str, _pk_bytes: &[u8], ) -> nodedb::Result<()> { self.binds.fetch_add(1, std::sync::atomic::Ordering::AcqRel); @@ -237,9 +221,8 @@ fn flush_emits_wal_record_at_threshold() { let pk = format!("u{i:08}"); let _ = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", pk.as_bytes(), ) .unwrap(); @@ -278,9 +261,8 @@ fn assigns_persist_across_reopen() { let a = make_assigner(creds, wal); s_persisted = a .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"alice", ) .unwrap(); @@ -290,9 +272,8 @@ fn assigns_persist_across_reopen() { let cat = creds.catalog(); assert_eq!( cat.get_surrogate_for_pk( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", b"alice" ) .unwrap(), @@ -300,9 +281,8 @@ fn assigns_persist_across_reopen() { ); assert_eq!( cat.get_pk_for_surrogate( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, "users"), nodedb_types::TenantId::new(0), - "users", s_persisted ) .unwrap(), diff --git a/nodedb/tests/inproc/cases/surrogate_wal_recovery.rs b/nodedb/tests/inproc/cases/surrogate_wal_recovery.rs index e10336420..326c55298 100644 --- a/nodedb/tests/inproc/cases/surrogate_wal_recovery.rs +++ b/nodedb/tests/inproc/cases/surrogate_wal_recovery.rs @@ -73,9 +73,8 @@ fn kill_restart_recovers_all_bindings_and_hwm() { for (coll, pk) in &inserts { let s = assigner .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, coll), nodedb_types::TenantId::new(0), - coll, pk, ) .unwrap(); @@ -120,9 +119,8 @@ fn kill_restart_recovers_all_bindings_and_hwm() { assert_eq!( catalog .get_surrogate_for_pk( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, coll), nodedb_types::TenantId::new(0), - coll, pk ) .unwrap(), @@ -132,9 +130,8 @@ fn kill_restart_recovers_all_bindings_and_hwm() { assert_eq!( catalog .get_pk_for_surrogate( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, coll), nodedb_types::TenantId::new(0), - coll, *surrogate ) .unwrap(), @@ -189,9 +186,11 @@ fn kill_restart_after_hwm_flush_threshold_recovers_via_alloc_record() { last_surrogate = Some( assigner .assign( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare( + nodedb_types::DatabaseId::DEFAULT, + "users", + ), nodedb_types::TenantId::new(0), - "users", pk.as_bytes(), ) .unwrap(), diff --git a/nodedb/tests/inproc/cases/wal_catchup.rs b/nodedb/tests/inproc/cases/wal_catchup.rs index 067f5282e..1f0b8af69 100644 --- a/nodedb/tests/inproc/cases/wal_catchup.rs +++ b/nodedb/tests/inproc/cases/wal_catchup.rs @@ -91,7 +91,8 @@ impl TestStack { let task = PhysicalTask { tenant_id, database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection), + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection) + .vshard(), plan, post_set_op: PostSetOp::None, txn_id: None, @@ -165,7 +166,7 @@ impl TestStack { .appender(NO_APPLY_KEY) .append_timeseries_batch( TenantId::new(1), - VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection), + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(), DatabaseId::DEFAULT, &wal_payload, ) diff --git a/nodedb/tests/inproc/cases/write_admission_fence.rs b/nodedb/tests/inproc/cases/write_admission_fence.rs index ba2627cd3..834801233 100644 --- a/nodedb/tests/inproc/cases/write_admission_fence.rs +++ b/nodedb/tests/inproc/cases/write_admission_fence.rs @@ -51,7 +51,7 @@ fn register_lock_manager( shared: &SharedState, collection: &str, ) -> (Arc>, VShardId) { - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(); let lm = Arc::new(Mutex::new(LockManager::new())); shared .calvin @@ -192,7 +192,7 @@ async fn single_node_point_write_uses_global_keyed_order_lock() { let (shared, _dir) = build_shared(); let coll = "single_node_coll"; // Deliberately DO NOT register a lock manager — the single-node path. - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, coll); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, coll).vshard(); let plan = kv_put(coll, b"K"); let lock = match admit(&shared, &target(vshard, &plan)) { @@ -219,7 +219,7 @@ async fn single_node_point_write_uses_global_keyed_order_lock() { async fn single_node_same_key_serializes_fifo() { let (shared, _dir) = build_shared(); let coll = "single_node_fifo"; - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, coll); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, coll).vshard(); let plan = kv_put(coll, b"K"); let (key, lock) = match admit(&shared, &target(vshard, &plan)) { diff --git a/nodedb/tests/native/cases/mod.rs b/nodedb/tests/native/cases/mod.rs index d494b9a13..484c07822 100644 --- a/nodedb/tests/native/cases/mod.rs +++ b/nodedb/tests/native/cases/mod.rs @@ -3,6 +3,7 @@ mod native_clone_read_intercept; mod native_clone_write_intercept; mod native_create_then_dml; +mod native_database_collection_key_placement; mod native_direct_op_txn_overlay; mod native_dml_affected_counts; mod native_dml_outcome_conformance; diff --git a/nodedb/tests/native/cases/native_database_collection_key_placement.rs b/nodedb/tests/native/cases/native_database_collection_key_placement.rs new file mode 100644 index 000000000..755e7d513 --- /dev/null +++ b/nodedb/tests/native/cases/native_database_collection_key_placement.rs @@ -0,0 +1,345 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A collection in a non-default database homes to one vShard, and each row +//! keeps one surrogate, on every write and read path. +//! +//! The vShard hash and the surrogate allocator take the canonical key +//! `(database_id, bare_name)`. The SQL planner carries the qualified name +//! `"{database_id}/{name}"` and de-qualifies it. The native direct ops carry +//! the bare name. Each collection here is picked so the two strings hash to +//! different Data Plane cores. A path that hashed the qualified string would +//! place its rows on a core the other path never reads. It would also bind +//! their surrogates under a key the other path never looks up. +//! +//! Rows are written through SQL (pgwire) and through native `PointPut`. Each +//! row is then read and overwritten through the other path. + +use std::collections::BTreeMap; + +use nodedb_test_support::native_harness::{do_handshake, send_request}; +use nodedb_test_support::pgwire_harness::TestServer; + +use nodedb_types::protocol::opcodes::ResponseStatus; +use nodedb_types::protocol::text_fields::TextFields; +use nodedb_types::protocol::{AuthMethod, HelloFrame, NativeResponse, OpCode}; +use nodedb_types::{CollectionKey, DatabaseId, QualifiedCollection, Surrogate, TenantId}; +use tokio::net::TcpStream; + +const NUM_CORES: usize = 4; + +/// The tenant the harness trust superuser belongs to. +const TENANT: TenantId = TenantId::new(1); + +/// Rows the SQL path writes, with their first values. +const SQL_ROWS: [(&str, i64); 2] = [("s1", 7101), ("s2", 7102)]; + +/// Rows the native path writes, with their first values. +const NATIVE_ROWS: [(&str, i64); 2] = [("n1", 7201), ("n2", 7202)]; + +/// The engine under test and the name of its primary-key column. +#[derive(Clone, Copy)] +enum Engine { + Document, + Kv, +} + +impl Engine { + fn pk_column(self) -> &'static str { + match self { + Engine::Document => "id", + Engine::Kv => "key", + } + } + + fn create_sql(self, name: &str) -> String { + match self { + Engine::Document => format!( + "CREATE COLLECTION {name} (id TEXT PRIMARY KEY, v INT) \ + WITH (engine='document_schemaless')" + ), + Engine::Kv => { + format!("CREATE COLLECTION {name} (key TEXT PRIMARY KEY, v INT) WITH (engine='kv')") + } + } + } + + /// The value bytes a native `PointPut` carries for row `pk`. + fn native_value(self, pk: &str, v: i64) -> Vec { + let value = match self { + Engine::Document => serde_json::json!({ "id": pk, "v": v }), + Engine::Kv => serde_json::json!({ "v": v }), + }; + nodedb_types::json_to_msgpack(&value).expect("encode native value") + } +} + +/// The Data Plane core a vShard runs on. +fn core_of(vshard: nodedb_types::id::VShardId) -> usize { + vshard.as_u32() as usize % NUM_CORES +} + +/// A collection name whose canonical key and raw qualified string hash to +/// different cores in `database_id`. Without the canonical key, the SQL path +/// and the native path would place its rows on different cores. +fn split_placement_name(database_id: DatabaseId, prefix: &str) -> String { + for i in 0..256u32 { + let name = format!("{prefix}_{i}"); + let qualified = QualifiedCollection::new(database_id, &name); + let canonical = CollectionKey::from_bare(database_id, &name).vshard(); + let raw_qualified = CollectionKey::from_bare(database_id, qualified.as_str()).vshard(); + if core_of(canonical) != core_of(raw_qualified) { + return name; + } + } + panic!("no collection name under '{prefix}' splits bare and qualified hashes across cores"); +} + +/// Open a native session bound to `database`. +async fn native_session(srv: &TestServer, database: &str) -> TcpStream { + let username = srv + .shared + .credentials + .configured_trust_superuser() + .expect("read configured trust superuser") + .expect("harness runs in trust mode"); + let addr = format!("127.0.0.1:{}", srv.native_port) + .parse() + .expect("native addr"); + let (mut stream, _ack) = do_handshake(addr, &HelloFrame::current()) + .await + .expect("native handshake"); + let auth = send_request( + &mut stream, + 1, + OpCode::Auth, + TextFields { + auth: Some(AuthMethod::Trust { username }), + database: Some(database.to_owned()), + ..Default::default() + }, + ) + .await; + assert_eq!(auth.status, ResponseStatus::Ok, "native auth: {auth:?}"); + stream +} + +async fn native_put( + stream: &mut TcpStream, + seq: u64, + engine: Engine, + collection: &str, + pk: &str, + v: i64, +) { + let put = send_request( + stream, + seq, + OpCode::PointPut, + TextFields { + collection: Some(collection.to_owned()), + document_id: Some(pk.to_owned()), + data: Some(engine.native_value(pk, v)), + ..Default::default() + }, + ) + .await; + assert_eq!( + put.status, + ResponseStatus::Ok, + "native PointPut {pk}: {put:?}" + ); +} + +async fn native_get( + stream: &mut TcpStream, + seq: u64, + collection: &str, + pk: &str, +) -> NativeResponse { + let get = send_request( + stream, + seq, + OpCode::PointGet, + TextFields { + collection: Some(collection.to_owned()), + document_id: Some(pk.to_owned()), + ..Default::default() + }, + ) + .await; + assert_eq!( + get.status, + ResponseStatus::Ok, + "native PointGet {pk}: {get:?}" + ); + get +} + +/// Assert a native `PointGet` of `pk` finds the row and carries `v`. +async fn assert_native_reads(stream: &mut TcpStream, seq: u64, collection: &str, pk: &str, v: i64) { + let get = native_get(stream, seq, collection, pk).await; + let rows = get.rows.clone().unwrap_or_default(); + assert!( + !rows.is_empty(), + "native PointGet must find row '{pk}' in '{collection}': {get:?}" + ); + assert!( + format!("{rows:?}").contains(&v.to_string()), + "native PointGet of '{pk}' must carry v = {v}: {get:?}" + ); +} + +/// Assert a SQL point read of `pk` finds exactly one row with value `v`. +async fn assert_sql_reads(srv: &TestServer, engine: Engine, collection: &str, pk: &str, v: i64) { + let pk_column = engine.pk_column(); + let rows = srv + .query_rows(&format!( + "SELECT {pk_column}, v FROM {collection} WHERE {pk_column} = '{pk}'" + )) + .await + .unwrap_or_else(|e| panic!("SQL point read of '{pk}': {e}")); + assert_eq!( + rows, + vec![vec![pk.to_owned(), v.to_string()]], + "SQL point read must find row '{pk}' in '{collection}' with v = {v}" + ); +} + +/// The surrogate bound to every row, keyed by primary key. Each binding +/// lives under the canonical key, never under the raw qualified string. +fn surrogates( + srv: &TestServer, + database_id: DatabaseId, + collection: &str, +) -> BTreeMap { + let canonical = CollectionKey::from_bare(database_id, collection); + let qualified = QualifiedCollection::new(database_id, collection); + let raw_qualified = CollectionKey::from_bare(database_id, qualified.as_str()); + let mut out = BTreeMap::new(); + for (pk, _) in SQL_ROWS.iter().chain(NATIVE_ROWS.iter()) { + let surrogate = srv + .shared + .surrogate_assigner + .lookup(canonical, TENANT, pk.as_bytes()) + .expect("surrogate lookup") + .unwrap_or_else(|| panic!("row '{pk}' has no surrogate under the canonical key")); + let stray = srv + .shared + .surrogate_assigner + .lookup(raw_qualified, TENANT, pk.as_bytes()) + .expect("stray surrogate lookup"); + assert_eq!( + stray, None, + "row '{pk}' must not bind a surrogate under the qualified string" + ); + out.insert((*pk).to_owned(), surrogate); + } + out +} + +/// Run the cross-path scenario for one engine. +async fn cross_path_rows_share_placement_and_identity(engine: Engine, prefix: &str) { + let srv = TestServer::start_multicores(NUM_CORES).await; + let database = format!("{prefix}_db"); + srv.exec(&format!("CREATE DATABASE {database}")) + .await + .expect("create database"); + srv.exec(&format!("USE DATABASE {database}")) + .await + .expect("use database"); + let database_id = srv + .shared + .credentials + .catalog() + .get_database_id_by_name(&database) + .expect("read database id") + .expect("database exists"); + assert_ne!(database_id, DatabaseId::DEFAULT); + + let collection = split_placement_name(database_id, prefix); + srv.exec(&engine.create_sql(&collection)) + .await + .expect("create collection"); + + let pk_column = engine.pk_column(); + for (pk, v) in SQL_ROWS { + srv.exec(&format!( + "INSERT INTO {collection} ({pk_column}, v) VALUES ('{pk}', {v})" + )) + .await + .unwrap_or_else(|e| panic!("SQL INSERT of '{pk}': {e}")); + } + + let mut stream = native_session(&srv, &database).await; + let mut seq = 2; + for (pk, v) in NATIVE_ROWS { + native_put(&mut stream, seq, engine, &collection, pk, v).await; + seq += 1; + } + + // Each path reads the rows the other path wrote. + for (pk, v) in SQL_ROWS { + assert_native_reads(&mut stream, seq, &collection, pk, v).await; + seq += 1; + } + for (pk, v) in NATIVE_ROWS { + assert_sql_reads(&srv, engine, &collection, pk, v).await; + } + + let before = surrogates(&srv, database_id, &collection); + + // Each path overwrites a row the other path wrote. The overwrite must + // land on the stored row, never beside it. + let (sql_pk, _) = SQL_ROWS[0]; + let native_over_sql = 7301; + native_put( + &mut stream, + seq, + engine, + &collection, + sql_pk, + native_over_sql, + ) + .await; + seq += 1; + let (native_pk, _) = NATIVE_ROWS[0]; + let sql_over_native = 7302; + srv.exec(&format!( + "UPSERT INTO {collection} ({pk_column}, v) VALUES ('{native_pk}', {sql_over_native})" + )) + .await + .unwrap_or_else(|e| panic!("SQL UPSERT of '{native_pk}': {e}")); + + assert_sql_reads(&srv, engine, &collection, sql_pk, native_over_sql).await; + assert_native_reads(&mut stream, seq, &collection, native_pk, sql_over_native).await; + + let after = surrogates(&srv, database_id, &collection); + assert_eq!( + before, after, + "every row must keep its surrogate when the other path overwrites it" + ); + + let mut keys: Vec = srv + .query_rows(&format!("SELECT {pk_column} FROM {collection}")) + .await + .expect("full scan") + .into_iter() + .filter_map(|row| row.into_iter().next()) + .collect(); + keys.sort(); + assert_eq!( + keys, + vec!["n1", "n2", "s1", "s2"], + "each row must be stored once, on one core" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn document_rows_share_placement_and_surrogate_across_sql_and_native() { + cross_path_rows_share_placement_and_identity(Engine::Document, "ckp_doc").await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn kv_rows_share_placement_and_surrogate_across_sql_and_native() { + cross_path_rows_share_placement_and_identity(Engine::Kv, "ckp_kv").await; +} diff --git a/nodedb/tests/native/cases/native_gateway_txn_overlay.rs b/nodedb/tests/native/cases/native_gateway_txn_overlay.rs index 7b7654bfc..4208deba1 100644 --- a/nodedb/tests/native/cases/native_gateway_txn_overlay.rs +++ b/nodedb/tests/native/cases/native_gateway_txn_overlay.rs @@ -242,9 +242,11 @@ async fn native_commit_makes_strict_row_visible_to_every_read_path() { .shared .surrogate_assigner .lookup( - nodedb_types::DatabaseId::DEFAULT, + nodedb_types::CollectionKey::from_bare( + nodedb_types::DatabaseId::DEFAULT, + "native_committed_visibility" + ), nodedb_types::TenantId::new(1), - "native_committed_visibility", b"a1", ) .expect("lookup committed PK binding") diff --git a/nodedb/tests/native/cases/native_txn_commit_visibility.rs b/nodedb/tests/native/cases/native_txn_commit_visibility.rs index 298cd90bf..a5490c3a8 100644 --- a/nodedb/tests/native/cases/native_txn_commit_visibility.rs +++ b/nodedb/tests/native/cases/native_txn_commit_visibility.rs @@ -17,7 +17,6 @@ use std::time::Duration; use nodedb_test_support::native_harness::{do_handshake, read_frame, write_frame}; use nodedb_test_support::pgwire_harness::TestServer; -use nodedb_types::id::VShardId; use nodedb_types::protocol::opcodes::ResponseStatus; use nodedb_types::protocol::text_fields::TextFields; use nodedb_types::protocol::{HelloFrame, NativeRequest, NativeResponse, OpCode, RequestFields}; @@ -82,7 +81,8 @@ fn collection_on_nonzero_core() -> String { for i in 0..64u32 { let name = format!("native_txn_vis_{i}"); let vshard = - VShardId::from_collection_in_database(nodedb::types::DatabaseId::DEFAULT, &name) + nodedb_types::CollectionKey::from_bare(nodedb::types::DatabaseId::DEFAULT, &name) + .vshard() .as_u32(); if !(vshard as usize).is_multiple_of(NUM_CORES) { return name; diff --git a/nodedb/tests/wire/cases/backup_support.rs b/nodedb/tests/wire/cases/backup_support.rs index 5db8db111..fc5e274ad 100644 --- a/nodedb/tests/wire/cases/backup_support.rs +++ b/nodedb/tests/wire/cases/backup_support.rs @@ -4,7 +4,7 @@ use bytes::Bytes; use futures::{SinkExt, StreamExt}; -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::DatabaseId; /// Take a backup of `tenant` over `client` and return the envelope bytes. pub async fn drain_backup(client: &tokio_postgres::Client, tenant: u64) -> Result, String> { @@ -48,11 +48,15 @@ pub async fn push_restore( /// Calvin scheduler. pub fn names_on_two_vshards(prefix: &str) -> (String, String) { let first = format!("{prefix}_a"); - let first_vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, &first).as_u32(); + let first_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &first) + .vshard() + .as_u32(); let second = (0..512u32) .map(|i| format!("{prefix}_b_{i}")) .find(|name| { - VShardId::from_collection_in_database(DatabaseId::DEFAULT, name).as_u32() + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, name) + .vshard() + .as_u32() != first_vshard }) .unwrap_or_else(|| panic!("no {prefix}_b name on another vShard in 512 tries")); diff --git a/nodedb/tests/wire/cases/calvin_sql_routing.rs b/nodedb/tests/wire/cases/calvin_sql_routing.rs index c3baed9f5..47405df8b 100644 --- a/nodedb/tests/wire/cases/calvin_sql_routing.rs +++ b/nodedb/tests/wire/cases/calvin_sql_routing.rs @@ -6,7 +6,6 @@ //! a real or mocked sequencer with full Raft. use crate::harness::TestServer; -use nodedb_types::id::VShardId; /// Find two collection names whose vShards differ. Deterministic within a process. fn find_two_distinct_collections() -> (String, String) { @@ -14,7 +13,8 @@ fn find_two_distinct_collections() -> (String, String) { for i in 0u32..512 { let name = format!("col_{i}"); let vshard = - VShardId::from_collection_in_database(nodedb_types::id::DatabaseId::DEFAULT, &name) + nodedb_types::CollectionKey::from_bare(nodedb_types::id::DatabaseId::DEFAULT, &name) + .vshard() .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { diff --git a/nodedb/tests/wire/cases/graph_timeseries_rls_probe.rs b/nodedb/tests/wire/cases/graph_timeseries_rls_probe.rs index a2f289253..8145c99aa 100644 --- a/nodedb/tests/wire/cases/graph_timeseries_rls_probe.rs +++ b/nodedb/tests/wire/cases/graph_timeseries_rls_probe.rs @@ -36,7 +36,9 @@ fn edge_endpoints_are_co_resident() { #[test] fn the_timeseries_collection_homes_on_one_vshard() { assert!( - VShardId::from_collection_in_database(DatabaseId::DEFAULT, "g_ts_probe_metrics").as_u32() + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "g_ts_probe_metrics") + .vshard() + .as_u32() < VShardId::COUNT ); } diff --git a/nodedb/tests/wire/cases/graph_vector_write_row_level_security.rs b/nodedb/tests/wire/cases/graph_vector_write_row_level_security.rs index 445b0e091..bfa4c572e 100644 --- a/nodedb/tests/wire/cases/graph_vector_write_row_level_security.rs +++ b/nodedb/tests/wire/cases/graph_vector_write_row_level_security.rs @@ -322,7 +322,9 @@ async fn a_vector_write_omitting_the_governed_column_is_denied() { /// keeps the statement single-shard, so what decides it is the write gate, /// which is the thing under test. fn node_in_collection_shard(collection: &str, prefix: &str) -> String { - let home = VShardId::from_collection_in_database(DatabaseId::DEFAULT, collection).as_u32(); + let home = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection) + .vshard() + .as_u32(); (0..100_000u32) .map(|n| format!("{prefix}{n}")) .find(|node| VShardId::from_key(node.as_bytes()).as_u32() == home) diff --git a/nodedb/tests/wire/cases/sql_transactions_commit_point_visibility.rs b/nodedb/tests/wire/cases/sql_transactions_commit_point_visibility.rs index aa006f9d6..0e9541590 100644 --- a/nodedb/tests/wire/cases/sql_transactions_commit_point_visibility.rs +++ b/nodedb/tests/wire/cases/sql_transactions_commit_point_visibility.rs @@ -7,7 +7,7 @@ //! A write staged on the wrong core leaves the owning core's overlay empty, //! so COMMIT would resolve and install nothing there. -use nodedb_types::id::{DatabaseId, VShardId}; +use nodedb_types::id::DatabaseId; use crate::harness::TestServer; @@ -17,7 +17,9 @@ fn collection_on_nonzero_core(prefix: &str) -> String { (0..64u32) .map(|i| format!("{prefix}_{i}")) .find(|name| { - let vshard = VShardId::from_collection_in_database(DatabaseId::DEFAULT, name).as_u32(); + let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, name) + .vshard() + .as_u32(); !(vshard as usize).is_multiple_of(NUM_CORES) }) .expect("a candidate collection hashes off core 0") diff --git a/nodedb/tests/wire/cases/sql_transactions_cross_shard_read_reject.rs b/nodedb/tests/wire/cases/sql_transactions_cross_shard_read_reject.rs index 6c020c6a0..f8b558af7 100644 --- a/nodedb/tests/wire/cases/sql_transactions_cross_shard_read_reject.rs +++ b/nodedb/tests/wire/cases/sql_transactions_cross_shard_read_reject.rs @@ -13,7 +13,6 @@ //! `sequencer_inbox` is unset, which does not happen on this harness. use crate::harness::TestServer; -use nodedb_types::id::VShardId; /// Find two collection names whose vShards differ. Deterministic within a /// process. @@ -22,7 +21,8 @@ fn find_two_distinct_collections() -> (String, String) { for i in 0u32..512 { let name = format!("xrd_col_{i}"); let vshard = - VShardId::from_collection_in_database(nodedb_types::id::DatabaseId::DEFAULT, &name) + nodedb_types::CollectionKey::from_bare(nodedb_types::id::DatabaseId::DEFAULT, &name) + .vshard() .as_u32(); if let Some((ref fname, fv)) = first { if fv != vshard { diff --git a/nodedb/tests/wire/cases/sql_transactions_graph_overlay.rs b/nodedb/tests/wire/cases/sql_transactions_graph_overlay.rs index c0b513643..2ca07e146 100644 --- a/nodedb/tests/wire/cases/sql_transactions_graph_overlay.rs +++ b/nodedb/tests/wire/cases/sql_transactions_graph_overlay.rs @@ -8,7 +8,7 @@ //! _type }` is mirrored by the Control Plane as an implicit `GraphOp::EdgePut` //! task appended to the SAME task list as the document write //! (`append_implicit_edge_tasks`). The two tasks do NOT share a `VShardId`: -//! the document write homes on `VShardId::from_collection_in_database`, the +//! the document write homes on `VShardId::from_collection`, the //! edge on `VShardId::from_key(src)` -- for `g_tx` / `staged_rollback` that is //! 348 vs 797. A self-loop does not change that, and cannot: the two homing //! functions take different inputs. So the statement classifies as From 3c94bb4c87ad9c1e28283b9a7e7f6f9bb47a23d4 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 16:54:00 +0800 Subject: [PATCH 51/64] feat(diagnostics): capture a descriptor lease the renewal loop drops Escalate the renewal loop's re-acquire and release failures from a warn log to error plus a faultbox Capture, grouped by step and error class so a retry every tick files one report instead of storming. An unrenewed lease expires under a holder that still plans against it; an unreleased one blocks every DDL drain on its descriptor. --- nodedb/src/control/lease/renewal.rs | 32 ++++++++++--- nodedb/src/diag/context/lease.rs | 73 +++++++++++++++++++++++++++++ nodedb/src/diag/context/mod.rs | 2 + nodedb/src/diag/mod.rs | 21 +++++---- nodedb/src/diag/recording/lease.rs | 37 +++++++++++++++ nodedb/src/diag/recording/mod.rs | 2 + 6 files changed, 150 insertions(+), 17 deletions(-) create mode 100644 nodedb/src/diag/context/lease.rs create mode 100644 nodedb/src/diag/recording/lease.rs diff --git a/nodedb/src/control/lease/renewal.rs b/nodedb/src/control/lease/renewal.rs index 83e6ea12b..7274b6872 100644 --- a/nodedb/src/control/lease/renewal.rs +++ b/nodedb/src/control/lease/renewal.rs @@ -38,7 +38,7 @@ use nodedb_cluster::LoopMetrics; use nodedb_types::config::tuning::ClusterTransportTuning; use tokio::sync::watch; use tokio::task::JoinHandle; -use tracing::{debug, info, warn}; +use tracing::{debug, error, info}; use crate::control::state::SharedState; @@ -156,8 +156,9 @@ impl LeaseRenewalLoop { } /// One iteration: snapshot the near-expiry lease set under a - /// short read lock, then re-acquire each one. Errors are - /// logged at warn — the next tick retries automatically. + /// short read lock, then re-acquire each one. A failed step is + /// logged at error and recorded as a diagnostic capture. The + /// next tick retries it. /// /// **Why we use wall-clock nanoseconds, not `hlc_clock.peek()`**: /// `peek` returns the last HLC the clock observed, which may @@ -193,21 +194,38 @@ impl LeaseRenewalLoop { version, self.config.full_duration, ) { - warn!( + error!( descriptor = ?id, version, error = %e, - "descriptor lease renewal: re-acquire failed" + "descriptor lease renewal: re-acquire failed; the lease \ + expires unless a later tick refreshes it" + ); + crate::diag::descriptor_lease_not_renewed( + &e, + "renew", + &id, + version, + shared.node_id, ); self.loop_metrics.record_error("renew"); } } None => { if let Err(e) = super::release::release_leases(&shared, vec![id.clone()]) { - warn!( + error!( descriptor = ?id, + held_version, error = %e, - "descriptor lease renewal: release after drop failed" + "descriptor lease renewal: release after drop failed; the \ + lease blocks DDL drains on its descriptor until it expires" + ); + crate::diag::descriptor_lease_not_renewed( + &e, + "release", + &id, + held_version, + shared.node_id, ); self.loop_metrics.record_error("release"); } diff --git a/nodedb/src/diag/context/lease.rs b/nodedb/src/diag/context/lease.rs new file mode 100644 index 000000000..47f5d36ed --- /dev/null +++ b/nodedb/src/diag/context/lease.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Forensic payloads for descriptor lease capture sites. +//! +//! A lease the renewal loop cannot refresh or release expires under a +//! holder that still plans against it, or stays in the lease map and holds +//! every DDL drain on its descriptor until expiry. + +use faultbox::DomainContext; +use faultbox::serde_json::{Value, json}; + +/// A descriptor lease the renewal loop could not refresh or release. +pub(in crate::diag) struct DescriptorLeaseNotRenewed<'a> { + /// `"renew"` or `"release"`: the step that failed. + pub step: &'static str, + /// Debug form of the descriptor id. + pub descriptor: &'a str, + /// Version the step proposed. + pub version: u64, + pub node_id: u64, + /// What failed, without the per-occurrence detail. + pub error_class: &'a str, +} + +impl DomainContext for DescriptorLeaseNotRenewed<'_> { + fn domain_kind(&self) -> &'static str { + "nodedb.descriptor_lease_not_renewed" + } + + fn grouping_key(&self) -> String { + // The step and error class name the bug. The descriptor, version + // and node are the occurrence, so a retry every tick files one report. + format!("step={} cause={}", self.step, self.error_class) + } + + fn to_json(&self) -> Value { + json!({ + "step": self.step, + "descriptor": self.descriptor, + "version": self.version, + "node_id": self.node_id, + "error_class": self.error_class, + "why_fatal": "a lease that is not refreshed expires while this node still \ + plans against its descriptor, and a lease that is not released \ + blocks every DDL drain on the descriptor until it expires", + "operator_action": "check metadata-group leadership and the applied index on \ + this node, then clear the error the step names", + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn grouping_ignores_the_descriptor_identity() { + let first = DescriptorLeaseNotRenewed { + step: "renew", + descriptor: "a", + version: 1, + node_id: 1, + error_class: "descriptor lease grant did not apply within 5s", + }; + let second = DescriptorLeaseNotRenewed { + descriptor: "b", + version: 9, + node_id: 3, + ..first + }; + assert_eq!(first.grouping_key(), second.grouping_key()); + } +} diff --git a/nodedb/src/diag/context/mod.rs b/nodedb/src/diag/context/mod.rs index a33246f6e..198b4e1f1 100644 --- a/nodedb/src/diag/context/mod.rs +++ b/nodedb/src/diag/context/mod.rs @@ -11,6 +11,7 @@ mod crdt; mod data_plane; mod index_rebuild; mod ingest; +mod lease; mod outcome_floor; mod quota; mod raft_apply; @@ -32,6 +33,7 @@ pub(in crate::diag) use data_plane::{ pub(in crate::diag) use index_rebuild::IndexRebuildNotInstalled; pub(in crate::diag) use ingest::IlpAcceptedLinesDropped; pub use ingest::IlpFlushOutcome; +pub(in crate::diag) use lease::DescriptorLeaseNotRenewed; pub(in crate::diag) use outcome_floor::{WriteWindowHeld, WriteWindowLeaked}; pub use quota::{DATABASE_SCOPE, TENANT_SCOPE}; pub(in crate::diag) use quota::{ diff --git a/nodedb/src/diag/mod.rs b/nodedb/src/diag/mod.rs index 918afd60d..22112dd73 100644 --- a/nodedb/src/diag/mod.rs +++ b/nodedb/src/diag/mod.rs @@ -13,14 +13,15 @@ pub use recording::{ IndexRebuildTarget, VectorBuildTarget, batch_insert_without_surrogates, calvin_apply_halted, calvin_completion_timeout, catalog_apply_orphan_row, collection_purge_row_missing, consumer_group_offsets_retained, data_plane_core_fail_stopped, data_plane_response_lost, - data_plane_responses_lost, entry_kind, fts_index_update_failed, history_compaction_not_applied, - ilp_invalid_utf8_drop, ilp_line_read_drop, index_rebuild_not_installed, metadata_apply_wedged, - orphaned_index_entry_after_delete, quota_row_invalid, quota_row_undecodable, - quota_row_write_failed, quota_scope_purge_incomplete, quota_scope_replay_aborted, - raft_entries_reapplied, raft_entry_reapplied, replay_record_unapplied, replicated_write_parked, - replicated_writes_parked, retention_autowire_orphaned, scope_quota_not_installed, - strict_row_undecodable, synonym_group_not_applied, vector_build_failed, - vector_builder_disconnected, vector_builder_spawn_failed, vector_index_not_applied, - vector_rebuild_unreadable, wal_archival_failed_truncation_held, write_acked_without_durability, - write_window_held, write_window_leaked, + data_plane_responses_lost, descriptor_lease_not_renewed, entry_kind, fts_index_update_failed, + history_compaction_not_applied, ilp_invalid_utf8_drop, ilp_line_read_drop, + index_rebuild_not_installed, metadata_apply_wedged, orphaned_index_entry_after_delete, + quota_row_invalid, quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, + quota_scope_replay_aborted, raft_entries_reapplied, raft_entry_reapplied, + replay_record_unapplied, replicated_write_parked, replicated_writes_parked, + retention_autowire_orphaned, scope_quota_not_installed, strict_row_undecodable, + synonym_group_not_applied, vector_build_failed, vector_builder_disconnected, + vector_builder_spawn_failed, vector_index_not_applied, vector_rebuild_unreadable, + wal_archival_failed_truncation_held, write_acked_without_durability, write_window_held, + write_window_leaked, }; diff --git a/nodedb/src/diag/recording/lease.rs b/nodedb/src/diag/recording/lease.rs new file mode 100644 index 000000000..867de1d9e --- /dev/null +++ b/nodedb/src/diag/recording/lease.rs @@ -0,0 +1,37 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capture sites for descriptor leases the renewal loop could not maintain. + +use faultbox::{Capture, EventKind, error_chain_of}; + +use super::shared::error_class; +use crate::diag::context; + +/// Report a descriptor lease the renewal loop could not refresh or release. +/// Called only from the renewal loop's error arms. `step` is `"renew"` or +/// `"release"`. +pub fn descriptor_lease_not_renewed( + err: &crate::Error, + step: &'static str, + descriptor: &nodedb_cluster::DescriptorId, + version: u64, + node_id: u64, +) { + let class = error_class(err); + let descriptor = format!("{descriptor:?}"); + let ctx = context::DescriptorLeaseNotRenewed { + step, + descriptor: &descriptor, + version, + node_id, + error_class: &class, + }; + let _ = Capture::new( + EventKind::Error, + "descriptor lease renewal: the lease was neither refreshed nor released", + ) + .error_chain(error_chain_of(err)) + .domain(&ctx) + .with_backtrace() + .emit(); +} diff --git a/nodedb/src/diag/recording/mod.rs b/nodedb/src/diag/recording/mod.rs index 8a14ff74f..d5a8992ff 100644 --- a/nodedb/src/diag/recording/mod.rs +++ b/nodedb/src/diag/recording/mod.rs @@ -13,6 +13,7 @@ mod crdt; mod data_plane; mod index_rebuild; mod ingest; +mod lease; mod outcome_floor; mod quota; mod raft_apply; @@ -33,6 +34,7 @@ pub use data_plane::{ }; pub use index_rebuild::{IndexRebuildTarget, index_rebuild_not_installed}; pub use ingest::{ilp_invalid_utf8_drop, ilp_line_read_drop}; +pub use lease::descriptor_lease_not_renewed; pub use outcome_floor::{write_window_held, write_window_leaked}; pub use quota::{ quota_row_invalid, quota_row_undecodable, quota_row_write_failed, quota_scope_purge_incomplete, From a457bd2589d0e5b2dcc15fc1c7604d1181f9b738 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 20:09:08 +0800 Subject: [PATCH 52/64] feat(errors): keep a Data-Plane refusal's typed code across the cluster and DDL paths Array cluster writes and reads, backup restore, merge/update-from-join resolve passes, calvin pre-execution scans, materialized-sum reconnaissance reads, and several DDL neutral functions (rate gate, weighted pick, KV index, collection purge) now propagate a refused Data-Plane response as its own typed ErrorCode instead of folding it into a generic Internal or Storage error. A shard's Data-Plane verdict crosses the cluster wire as a new VShardRefusal RPC frame carrying the coded ErrorCode verbatim, so the coordinator rebuilds ClusterError::DataPlane and renders the same SQLSTATE a single-node execution would. Such a refusal never counts against the array shard's circuit breaker, since it comes from a shard that answered. HTTP DDL errors now derive their status from the SQLSTATE through a shared sqlstate-to-HTTP status table instead of a two-way if/else, and carry their typed cause in the response body via a new HttpError.cause field, matching native and pgwire. RetryableSchemaChanged now classifies as a write conflict and renders 40001 (serialization failure, retried by drivers) instead of an internal 503/XX000. SessionTokenExpired now classifies as auth_expired and renders 401/28000 (invalid authorization) instead of a bad request, on native, pgwire, and HTTP alike. --- .../src/distributed_array/scatter.rs | 9 +- nodedb-cluster/src/error.rs | 10 ++ .../src/raft_loop/handle_rpc/plan_dispatch.rs | 13 +- nodedb-cluster/src/rpc_codec/discriminants.rs | 4 + nodedb-cluster/src/rpc_codec/mod.rs | 1 + nodedb-cluster/src/rpc_codec/raft_rpc.rs | 6 + nodedb-cluster/src/rpc_codec/vshard.rs | 54 +++++- .../control/array_sync/raft_apply/common.rs | 102 ++++++++--- .../backup/restore/orchestrate/restore.rs | 14 +- .../backup/restore/redo_reissue/commit.rs | 8 +- .../cluster/array_cluster_exec/dispatch.rs | 5 + .../control/cluster/array_cluster_helpers.rs | 25 +++ .../src/control/cluster/array_executor/mod.rs | 2 + .../control/cluster/array_executor/read.rs | 28 +-- .../control/cluster/array_executor/refusal.rs | 95 ++++++++++ .../control/cluster/array_executor/write.rs | 19 +- .../control/gateway/error_map/class_parity.rs | 90 +++++++++- nodedb/src/control/gateway/error_map/http.rs | 14 +- nodedb/src/control/gateway/error_map/mod.rs | 1 + .../src/control/gateway/error_map/pgwire.rs | 6 +- .../control/gateway/error_map/remote_code.rs | 1 + .../gateway/error_map/sqlstate_status.rs | 114 ++++++++++++ .../merge_orchestrator/expand_staged_merge.rs | 15 +- nodedb/src/control/planner/calvin/preexec.rs | 68 ++++++-- .../control/planner/materialized_sum/recon.rs | 73 ++++++-- nodedb/src/control/server/http/auth.rs | 8 + .../http/routes/query/materialized/encode.rs | 96 +++++++++- .../server/http/routes/result_shape.rs | 1 + nodedb/src/control/server/http/types.rs | 15 ++ .../server/native/dispatch/conversion.rs | 9 +- .../control/server/pgwire/types/error_map.rs | 15 +- .../ddl/neutral/collection/index/kv_index.rs | 22 +-- .../ddl/neutral/collection/purge/dispatch.rs | 17 +- .../shared/ddl/neutral/kv_atomic/mod.rs | 2 +- .../server/shared/ddl/neutral/rate_gate.rs | 165 ++++++++++++------ .../shared/ddl/neutral/weighted_pick.rs | 45 ++--- .../control/server/shared/response_payload.rs | 41 +++++ .../expand_staged_update_from_join.rs | 15 +- nodedb/src/error_classify.rs | 8 +- nodedb/tests/crash_harness/pgwire.rs | 8 +- 40 files changed, 988 insertions(+), 256 deletions(-) create mode 100644 nodedb/src/control/cluster/array_executor/refusal.rs create mode 100644 nodedb/src/control/gateway/error_map/sqlstate_status.rs diff --git a/nodedb-cluster/src/distributed_array/scatter.rs b/nodedb-cluster/src/distributed_array/scatter.rs index 547b479f0..fef1f4453 100644 --- a/nodedb-cluster/src/distributed_array/scatter.rs +++ b/nodedb-cluster/src/distributed_array/scatter.rs @@ -81,12 +81,15 @@ use super::rpc::ShardRpcDispatch; /// answers `WrongOwner` until the coordinator's routing table catches up (the /// single retry in `call_with_wrong_owner_retry` already re-reads the live /// table). Counting it as a liveness failure would open the shared breaker and -/// then fast-fail healthy shards' slice/put/agg/delete with `CircuitOpen`. Only -/// `WrongOwner` is excluded — every genuine transport/timeout/unreachable error -/// still counts, mirroring `RetryPolicy::is_retryable`'s conservative policy. +/// then fast-fail healthy shards' slice/put/agg/delete with `CircuitOpen`. +/// `WrongOwner` and a typed Data-Plane verdict are excluded. Every genuine +/// transport/timeout/unreachable error still counts, mirroring +/// `RetryPolicy::is_retryable`'s conservative policy. fn counts_against_breaker(err: &ClusterError) -> bool { match err { ClusterError::WrongOwner { .. } => false, + // A typed Data-Plane verdict comes from a healthy shard that answered. + ClusterError::DataPlane { .. } => false, // An unresponsive peer is exactly what the breaker exists to shed // load from, so a shard timeout counts like any other liveness failure. ClusterError::ShardTimeout { .. } => true, diff --git a/nodedb-cluster/src/error.rs b/nodedb-cluster/src/error.rs index 275aad697..7d490cd46 100644 --- a/nodedb-cluster/src/error.rs +++ b/nodedb-cluster/src/error.rs @@ -118,6 +118,16 @@ pub enum ClusterError { #[error("storage error: {detail}")] Storage { detail: String }, + /// A shard's Data Plane refused the request with a typed verdict. + /// + /// The code crosses the node hop verbatim as `RaftRpc::VShardRefusal`, so + /// the coordinator renders the SQLSTATE a single-node execution renders. + /// The message uses the code's `Debug` form for logs only. + #[error("data plane refused the request: {code:?}")] + DataPlane { + code: crate::rpc_codec::DataPlaneErrorCode, + }, + #[error("codec error: {detail}")] Codec { detail: String }, diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs index a9d825a23..5571f7535 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs @@ -9,7 +9,7 @@ use crate::forward::{ChunkSink, PlanExecutor}; use crate::multi_raft::MultiRaft; use crate::rpc_codec::{ DataProposeRequest, DataProposeResponse, ExecuteRequest, MetadataProposeRequest, ProposeTarget, - RaftRpc, TypedClusterError, + RaftRpc, TypedClusterError, VShardRefusal, }; use super::super::loop_core::{CommitApplier, RaftLoop}; @@ -52,10 +52,17 @@ impl RaftLoop { } // VShardEnvelope — dispatch to registered handler (Event Plane, etc.). + // A typed Data-Plane verdict answers as a `VShardRefusal` frame, so the + // caller rebuilds the same code. Any other handler error closes the stream. pub(super) async fn handle_vshard_envelope_rpc(&self, bytes: Vec) -> Result { if let Some(ref handler) = self.vshard_handler { - let response_bytes = handler(bytes).await?; - Ok(RaftRpc::VShardEnvelope(response_bytes)) + match handler(bytes).await { + Ok(response_bytes) => Ok(RaftRpc::VShardEnvelope(response_bytes)), + Err(ClusterError::DataPlane { code }) => { + Ok(RaftRpc::VShardRefusal(VShardRefusal { code })) + } + Err(other) => Err(other), + } } else { Err(ClusterError::Transport { detail: "VShardEnvelope handler not configured".into(), diff --git a/nodedb-cluster/src/rpc_codec/discriminants.rs b/nodedb-cluster/src/rpc_codec/discriminants.rs index 94db52e5b..446a55818 100644 --- a/nodedb-cluster/src/rpc_codec/discriminants.rs +++ b/nodedb-cluster/src/rpc_codec/discriminants.rs @@ -151,6 +151,10 @@ pub const RPC_AUTH_LEASE_RENEW_RESP: u8 = 50; /// `RPC_AUTH_BARRIER_RESP`. pub const RPC_AUTH_BARRIER_REQ: u8 = 51; pub const RPC_AUTH_BARRIER_RESP: u8 = 52; +/// Answer to an `RPC_VSHARD_ENVELOPE` request whose handler returned a typed +/// Data-Plane verdict. It carries the verdict code in place of a response +/// envelope. +pub const RPC_VSHARD_REFUSAL: u8 = 53; // VShardMessageType discriminants for distributed array ops (u16, range 80-89). // These mirror `crate::wire::VShardMessageType` repr values and are declared diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index da881d23d..0c0a88c7f 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -66,3 +66,4 @@ pub use shuffle::{ ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, SortKey, }; pub use surrogate::{AssignSurrogateRequest, AssignSurrogateResponse}; +pub use vshard::VShardRefusal; diff --git a/nodedb-cluster/src/rpc_codec/raft_rpc.rs b/nodedb-cluster/src/rpc_codec/raft_rpc.rs index 40d993067..7550d845a 100644 --- a/nodedb-cluster/src/rpc_codec/raft_rpc.rs +++ b/nodedb-cluster/src/rpc_codec/raft_rpc.rs @@ -32,6 +32,7 @@ use super::shuffle::{ ShufflePushEnd, ShufflePushRequest, }; use super::surrogate::{AssignSurrogateRequest, AssignSurrogateResponse}; +use super::vshard::VShardRefusal; use super::{ auth_lease, calvin_submit, cluster_mgmt, data_propose, execute, metadata, raft_msgs, read_index, reservation, shuffle, surrogate, vshard, @@ -157,6 +158,9 @@ pub enum RaftRpc { AuthLeaseRenewResponse(AuthLeaseRenewResponse), AuthBarrierRequest(AuthBarrierRequest), AuthBarrierResponse(AuthBarrierResponse), + // Answer to a `VShardEnvelope` request whose handler refused with a + // typed Data-Plane verdict. + VShardRefusal(VShardRefusal), } /// Encode a [`RaftRpc`] into a framed binary message stamped with `epoch`. @@ -232,6 +236,7 @@ pub fn encode(rpc: &RaftRpc, epoch: &crate::cluster_epoch::ClusterEpochState) -> RaftRpc::AuthLeaseRenewResponse(m) => auth_lease::encode_renew_resp(m, &mut out), RaftRpc::AuthBarrierRequest(m) => auth_lease::encode_barrier_req(m, &mut out), RaftRpc::AuthBarrierResponse(m) => auth_lease::encode_barrier_resp(m, &mut out), + RaftRpc::VShardRefusal(m) => vshard::encode_vshard_refusal(m, &mut out), }?; super::header::stamp_epoch(&mut out, epoch)?; Ok(out) @@ -333,6 +338,7 @@ pub fn decode(data: &[u8], epoch: &crate::cluster_epoch::ClusterEpochState) -> R RPC_AUTH_LEASE_RENEW_RESP => auth_lease::decode_renew_resp(payload), RPC_AUTH_BARRIER_REQ => auth_lease::decode_barrier_req(payload), RPC_AUTH_BARRIER_RESP => auth_lease::decode_barrier_resp(payload), + RPC_VSHARD_REFUSAL => vshard::decode_vshard_refusal(payload), _ => Err(ClusterError::Codec { detail: format!("unknown rpc_type: {rpc_type}"), }), diff --git a/nodedb-cluster/src/rpc_codec/vshard.rs b/nodedb-cluster/src/rpc_codec/vshard.rs index 1c441b622..ba249c5f2 100644 --- a/nodedb-cluster/src/rpc_codec/vshard.rs +++ b/nodedb-cluster/src/rpc_codec/vshard.rs @@ -6,11 +6,21 @@ //! retention, and archival messages. The inner VShardMessageType determines //! the handler. The envelope bytes are passed through raw (already serialized //! in their own binary format). +//! +//! A handler that refuses with a typed Data-Plane verdict answers with a +//! [`VShardRefusal`] frame instead of a response envelope. -use super::discriminants::RPC_VSHARD_ENVELOPE; +use super::data_plane_error::DataPlaneErrorCode; +use super::discriminants::{RPC_VSHARD_ENVELOPE, RPC_VSHARD_REFUSAL}; use super::header::write_frame; use super::raft_rpc::RaftRpc; -use crate::error::Result; +use crate::error::{ClusterError, Result}; + +/// A typed Data-Plane verdict that answers a VShardEnvelope request. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct VShardRefusal { + pub code: DataPlaneErrorCode, +} pub(super) fn encode_vshard_envelope(bytes: &[u8], out: &mut Vec) -> Result<()> { write_frame(RPC_VSHARD_ENVELOPE, bytes, out) @@ -20,3 +30,43 @@ pub(super) fn decode_vshard_envelope(payload: &[u8]) -> Result { // VShardEnvelope is already in its own binary format — pass through raw. Ok(RaftRpc::VShardEnvelope(payload.to_vec())) } + +pub(super) fn encode_vshard_refusal(msg: &VShardRefusal, out: &mut Vec) -> Result<()> { + let bytes = rkyv::to_bytes::(msg).map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize VShardRefusal: {e}"), + })?; + write_frame(RPC_VSHARD_REFUSAL, &bytes, out) +} + +pub(super) fn decode_vshard_refusal(payload: &[u8]) -> Result { + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity(payload.len()); + aligned.extend_from_slice(payload); + let refusal = + rkyv::from_bytes::(&aligned).map_err(|e| { + ClusterError::Codec { + detail: format!("rkyv deserialize VShardRefusal: {e}"), + } + })?; + Ok(RaftRpc::VShardRefusal(refusal)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + #[test] + fn a_refusal_survives_the_wire() { + let code = DataPlaneErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + let epoch = ClusterEpochState::default(); + let rpc = RaftRpc::VShardRefusal(VShardRefusal { code: code.clone() }); + let encoded = encode(&rpc, &epoch).expect("encode"); + match decode(&encoded, &epoch).expect("decode") { + RaftRpc::VShardRefusal(refusal) => assert_eq!(refusal.code, code), + other => panic!("decoded the wrong variant: {other:?}"), + } + } +} diff --git a/nodedb/src/control/array_sync/raft_apply/common.rs b/nodedb/src/control/array_sync/raft_apply/common.rs index 7ef31acf0..e309c1e25 100644 --- a/nodedb/src/control/array_sync/raft_apply/common.rs +++ b/nodedb/src/control/array_sync/raft_apply/common.rs @@ -74,7 +74,9 @@ pub(super) struct ArrayWriteSubmit { /// /// An error-status response is surfaced as a typed error: a committed entry that /// failed to apply must reach the propose waiter as a failure, not an empty -/// success, and must NOT advance the floor. +/// success, and must NOT advance the floor. A coded refusal is +/// `crate::Error::DataPlane`, so the waiter classifies it and a final refusal +/// records its marker. The funnel's own errors keep their variant. pub(super) async fn submit_array_write( state: &Arc, params: ArrayWriteSubmit, @@ -125,19 +127,11 @@ pub(super) async fn submit_array_write( change_feed: ChangeFeedOwner::Unowned, }, ) - .await - .map_err(|e| crate::Error::Internal { - detail: format!("{op_label}: {e}"), - })?; + .await?; let response = outcome.response; if response.status != Status::Ok { - let detail = response - .error_code - .as_ref() - .map(|c| format!("{op_label} error: {c:?}")) - .unwrap_or_else(|| format!("{op_label} returned error status")); - return Err(crate::Error::Internal { detail }); + return Err(apply_refusal(op_label, &response)); } // The response carries the write-version this replica stamped alongside the // payload; an array plan names no user collection, so it is `Lsn::ZERO` here @@ -288,22 +282,16 @@ pub(super) fn build_array_request( } } -/// Await a Data Plane response, mapping timeout / channel-closed / error-status -/// into `crate::Error::Internal` with a contextual `op_label`. +/// Await a Data Plane response. An error status becomes [`apply_refusal`]. +/// A timeout or a closed channel becomes `crate::Error::Internal` with a +/// contextual `op_label`. pub(super) async fn await_data_plane( rx: impl std::future::Future>, op_label: &str, ) -> ProposeResult { match tokio::time::timeout(Duration::from_secs(30), rx).await { Ok(Ok(resp)) if resp.status == Status::Ok => Ok(AppliedWrite::from_response(&resp)), - Ok(Ok(resp)) => { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{op_label} error: {c:?}")) - .unwrap_or_else(|| format!("{op_label} returned error status")); - Err(crate::Error::Internal { detail }) - } + Ok(Ok(resp)) => Err(apply_refusal(op_label, &resp)), Ok(Err(_)) => Err(crate::Error::Internal { detail: format!("{op_label}: response channel closed"), }), @@ -312,3 +300,75 @@ pub(super) async fn await_data_plane( }), } } + +/// The typed error for a Data-Plane response with a non-`Ok` status. +/// +/// A coded refusal is `crate::Error::DataPlane` with its own code. Only a +/// refusal with no code is `crate::Error::Internal`. +pub(super) fn apply_refusal(op_label: &str, response: &Response) -> crate::Error { + match response.error_code.as_deref() { + Some(code) => crate::Error::DataPlane(code.clone()), + None => crate::Error::Internal { + detail: format!("{op_label} returned error status"), + }, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::types::{Lsn, RequestId}; + + fn refusal(code: Option) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A coded refusal of a committed array write keeps its code, so the + /// final-refusal check sees it. + #[test] + fn a_coded_refusal_keeps_its_code() { + let code = ErrorCode::RejectedPrevalidation { + reason: "cell out of bounds".into(), + }; + let error = apply_refusal("array cell write", &refusal(Some(code.clone()))); + match &error { + crate::Error::DataPlane(kept) => assert_eq!(kept, &code), + other => panic!("expected the typed refusal, got {other:?}"), + } + assert!(crate::control::server::dispatch_utils::error_is_final_refusal(&error)); + } + + #[test] + fn a_refusal_with_no_code_is_internal() { + match apply_refusal("OpenArray", &refusal(None)) { + crate::Error::Internal { detail } => assert!(detail.starts_with("OpenArray")), + other => panic!("expected an internal error, got {other:?}"), + } + } + + /// The Data-Plane response await keeps the code as well. + #[tokio::test] + async fn awaiting_a_coded_refusal_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + let response = refusal(Some(code.clone())); + let result = await_data_plane(async move { Ok::<_, ()>(response) }, "OpenArray").await; + match result { + Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), + other => panic!("expected the typed refusal, got {other:?}"), + } + } +} diff --git a/nodedb/src/control/backup/restore/orchestrate/restore.rs b/nodedb/src/control/backup/restore/orchestrate/restore.rs index cb10686e7..18d076ed5 100644 --- a/nodedb/src/control/backup/restore/orchestrate/restore.rs +++ b/nodedb/src/control/backup/restore/orchestrate/restore.rs @@ -109,13 +109,17 @@ pub async fn restore_tenant( // applier's own register hook and a later boot seed are both // idempotent with it. A registration failure fails the restore. for coll in &restored_collections { + // A Data-Plane verdict keeps its code. dispatch_register_from_stored(state, coll) .await - .map_err(|e| Error::Internal { - detail: format!( - "restore: Data Plane registration of collection '{}' failed: {e}", - coll.name - ), + .map_err(|e| match e { + Error::DataPlane(_) => e, + other => Error::Internal { + detail: format!( + "restore: Data Plane registration of collection '{}' failed: {other}", + coll.name + ), + }, })?; } } diff --git a/nodedb/src/control/backup/restore/redo_reissue/commit.rs b/nodedb/src/control/backup/restore/redo_reissue/commit.rs index 8192b03ec..3021a300d 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/commit.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/commit.rs @@ -144,10 +144,14 @@ pub(super) async fn commit_collection( let mut records = 0usize; for batch in batch_units(units) { let payload = batch_payload(&collection, batch); + // A Data-Plane verdict keeps its code. commit_record(state, target, &payload) .await - .map_err(|e| crate::Error::Internal { - detail: format!("restore: re-issuing rows of '{collection}' failed: {e}"), + .map_err(|e| match e { + crate::Error::DataPlane(_) => e, + other => crate::Error::Internal { + detail: format!("restore: re-issuing rows of '{collection}' failed: {other}"), + }, })?; records += 1; } diff --git a/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs b/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs index 93a921ea6..ba84f2ffe 100644 --- a/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs +++ b/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs @@ -145,6 +145,11 @@ impl NexarArrayDispatch { detail: "array shard response: failed to decode VShardEnvelope".into(), } }), + // The shard's Data Plane refused with a typed verdict. It keeps + // its code, the same error the local short-circuit returns. + RaftRpc::VShardRefusal(refusal) => { + Err(nodedb_cluster::error::ClusterError::DataPlane { code: refusal.code }) + } other => Err(nodedb_cluster::error::ClusterError::Transport { detail: format!( "array shard RPC: unexpected response type {:?}", diff --git a/nodedb/src/control/cluster/array_cluster_helpers.rs b/nodedb/src/control/cluster/array_cluster_helpers.rs index 16aa80c1b..80c83f233 100644 --- a/nodedb/src/control/cluster/array_cluster_helpers.rs +++ b/nodedb/src/control/cluster/array_cluster_helpers.rs @@ -102,6 +102,9 @@ pub(super) fn cluster_err(e: nodedb_cluster::error::ClusterError) -> Error { nodedb_cluster::error::ClusterError::ShardTimeout { .. } => Error::DeadlineExceeded { request_id: crate::types::RequestId::new(0), }, + // A shard's Data-Plane verdict keeps its code, so the statement + // renders the SQLSTATE a single-node execution renders. + nodedb_cluster::error::ClusterError::DataPlane { code } => Error::DataPlane(code.into()), other => Error::Internal { detail: format!("array cluster: {other}"), }, @@ -129,3 +132,25 @@ pub(super) fn array_resp_msg_type(opcode: u32) -> Option { _ => None, } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::ErrorCode; + + /// A shard verdict that crossed the cluster keeps its code at the + /// coordinator, never `Internal`. + #[test] + fn a_shard_verdict_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + let wire = nodedb_cluster::error::ClusterError::DataPlane { + code: code.clone().into(), + }; + match cluster_err(wire) { + Error::DataPlane(rebuilt) => assert_eq!(rebuilt, code), + other => panic!("expected the typed verdict, got {other:?}"), + } + } +} diff --git a/nodedb/src/control/cluster/array_executor/mod.rs b/nodedb/src/control/cluster/array_executor/mod.rs index 5c89db0b1..e65d997cc 100644 --- a/nodedb/src/control/cluster/array_executor/mod.rs +++ b/nodedb/src/control/cluster/array_executor/mod.rs @@ -28,10 +28,12 @@ //! - [`read`]: read/scan handlers (slice, aggregate, surrogate-bitmap scan) plus //! the response-row parsers. //! - [`write`]: write handlers (put, delete). +//! - [`refusal`]: the typed cluster error a Data-Plane refusal answers with. pub mod cells; mod executor; mod read; +mod refusal; mod trait_impl; mod write; diff --git a/nodedb/src/control/cluster/array_executor/read.rs b/nodedb/src/control/cluster/array_executor/read.rs index 3eb9c6472..4ffdf3107 100644 --- a/nodedb/src/control/cluster/array_executor/read.rs +++ b/nodedb/src/control/cluster/array_executor/read.rs @@ -15,6 +15,7 @@ use crate::types::{TxnId, VShardId}; use nodedb_types::SurrogateBitmap; use super::executor::DataPlaneArrayExecutor; +use super::refusal::refusal_error; use crate::data::executor::response_codec::ArraySliceResponse; use nodedb_physical::physical_plan::{ArrayOp, ArrayReducer, PhysicalPlan}; @@ -62,14 +63,7 @@ impl DataPlaneArrayExecutor { .await?; if resp.status == crate::bridge::envelope::Status::Error { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| "unknown Data Plane error".into()); - return Err(ClusterError::Storage { - detail: format!("array slice Data Plane error: {detail}"), - }); + return Err(refusal_error("array slice", &resp)); } // Decode the structured `ArraySliceResponse` envelope, then split the @@ -137,14 +131,7 @@ impl DataPlaneArrayExecutor { .await?; if resp.status == crate::bridge::envelope::Status::Error { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| "unknown Data Plane error".into()); - return Err(ClusterError::Storage { - detail: format!("array agg Data Plane error: {detail}"), - }); + return Err(refusal_error("array agg", &resp)); } if resp.payload.is_empty() { @@ -189,14 +176,7 @@ impl DataPlaneArrayExecutor { .await?; if resp.status == crate::bridge::envelope::Status::Error { - let detail = resp - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| "unknown Data Plane error".into()); - return Err(ClusterError::Storage { - detail: format!("surrogate bitmap scan Data Plane error: {detail}"), - }); + return Err(refusal_error("surrogate bitmap scan", &resp)); } collect_surrogate_bitmap(&resp.payload) diff --git a/nodedb/src/control/cluster/array_executor/refusal.rs b/nodedb/src/control/cluster/array_executor/refusal.rs new file mode 100644 index 000000000..1844772c5 --- /dev/null +++ b/nodedb/src/control/cluster/array_executor/refusal.rs @@ -0,0 +1,95 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The cluster error an array shard answers with when its Data Plane refuses. +//! +//! A coded refusal crosses as `ClusterError::DataPlane`, so the coordinator +//! rebuilds `crate::Error::DataPlane(code)` and renders the SQLSTATE a +//! single-node execution renders. Only a refusal with no code is a storage +//! error. + +use nodedb_cluster::error::ClusterError; + +use crate::bridge::envelope::Response; + +/// The cluster error for a Data-Plane response with an error status. +pub(super) fn refusal_error(context: &str, response: &Response) -> ClusterError { + match response.error_code.as_deref() { + Some(code) => ClusterError::DataPlane { + code: code.clone().into(), + }, + None => ClusterError::Storage { + detail: format!("{context}: data plane returned an error status with no error code"), + }, + } +} + +/// The cluster error for a local-execution error. A Data-Plane verdict keeps +/// its code. Every other error is a storage error with `context` before its +/// message. +pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterError { + match error { + crate::Error::DataPlane(code) => ClusterError::DataPlane { code: code.into() }, + other => ClusterError::Storage { + detail: format!("{context}: {other}"), + }, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload, Status}; + use crate::types::{Lsn, RequestId}; + + fn refusal(code: Option) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + fn unsupported() -> ErrorCode { + ErrorCode::Unsupported { + detail: "not on this engine".into(), + } + } + + /// A coded refusal keeps its code through the cluster error and back to + /// the coordinator's typed error. + #[test] + fn a_coded_refusal_keeps_its_code() { + match refusal_error("array slice", &refusal(Some(unsupported()))) { + ClusterError::DataPlane { code } => { + assert_eq!(ErrorCode::from(code), unsupported()); + } + other => panic!("expected the typed refusal, got {other:?}"), + } + } + + #[test] + fn a_refusal_with_no_code_is_a_storage_error() { + match refusal_error("array slice", &refusal(None)) { + ClusterError::Storage { detail } => assert!(detail.starts_with("array slice: ")), + other => panic!("expected a storage error, got {other:?}"), + } + } + + #[test] + fn an_execution_verdict_keeps_its_code() { + let error = crate::Error::DataPlane(unsupported()); + match execution_error("array put", error) { + ClusterError::DataPlane { code } => { + assert_eq!(ErrorCode::from(code), unsupported()); + } + other => panic!("expected the typed refusal, got {other:?}"), + } + } +} diff --git a/nodedb/src/control/cluster/array_executor/write.rs b/nodedb/src/control/cluster/array_executor/write.rs index 8292ac612..6d2982a89 100644 --- a/nodedb/src/control/cluster/array_executor/write.rs +++ b/nodedb/src/control/cluster/array_executor/write.rs @@ -14,6 +14,7 @@ use nodedb_cluster::error::{ClusterError, Result}; use super::cells::flatten_blob_vec; use super::executor::DataPlaneArrayExecutor; +use super::refusal::{execution_error, refusal_error}; use crate::control::server::dispatch_utils::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; @@ -130,9 +131,7 @@ impl DataPlaneArrayExecutor { entry, ) .await - .map_err(|e| ClusterError::Storage { - detail: format!("{op_label} raft propose: {e}"), - })?; + .map_err(|e| execution_error(&format!("{op_label} raft propose"), e))?; let affected = require_affected_count(&apply_payload).map_err(|e| ClusterError::Storage { detail: format!("{op_label}: {e}"), @@ -154,20 +153,10 @@ impl DataPlaneArrayExecutor { single_node_submit(array_id, VShardId::new(local_vshard_id), plan), ) .await - .map_err(|e| ClusterError::Storage { - detail: format!("{op_label}: {e}"), - })?; + .map_err(|e| execution_error(op_label, e))?; if outcome.response.status == crate::bridge::envelope::Status::Error { - let detail = outcome - .response - .error_code - .as_ref() - .map(|c| format!("{c:?}")) - .unwrap_or_else(|| "unknown Data Plane error".into()); - return Err(ClusterError::Storage { - detail: format!("{op_label} Data Plane error: {detail}"), - }); + return Err(refusal_error(op_label, &outcome.response)); } // Ack with the LSN the funnel actually minted. `None` would mean the diff --git a/nodedb/src/control/gateway/error_map/class_parity.rs b/nodedb/src/control/gateway/error_map/class_parity.rs index 8adc049cb..d11d0ff20 100644 --- a/nodedb/src/control/gateway/error_map/class_parity.rs +++ b/nodedb/src/control/gateway/error_map/class_parity.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Every Data-Plane `ErrorCode` answers one class on native, pgwire and HTTP. +//! Every Data-Plane `ErrorCode`, and each Control-Plane error a client acts +//! on, answers one class on native, pgwire and HTTP. //! //! pgwire renders a Data-Plane verdict as its SQLSTATE. A native client reads //! the numeric `nodedb_types` code on the frame. The two agree when the @@ -275,6 +276,23 @@ fn classified_data_plane_codes_are_not_http_500() { } } +/// The SQLSTATE status table agrees with the gateway status table for every +/// Data-Plane code. A DDL error and a query error of one class answer one +/// HTTP status. +#[test] +fn sqlstate_status_agrees_with_the_gateway_status() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(pg_state), + status, + "{code:?} is {pg_state} on pgwire" + ); + } +} + /// `Unsupported` is feature-not-supported on every surface. #[test] fn unsupported_is_feature_not_supported_everywhere() { @@ -287,3 +305,73 @@ fn unsupported_is_feature_not_supported_everywhere() { assert_eq!(native.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); assert_eq!(GatewayErrorMap::to_http(&err).0, 501); } + +/// Control-Plane errors a client acts on. Each has a class of its own. +fn control_plane_samples() -> Vec { + vec![ + crate::Error::RetryableSchemaChanged { + descriptor: "orders".into(), + }, + crate::Error::SessionTokenExpired, + ] +} + +/// A Control-Plane error has one class on native and pgwire, and its native +/// numeric code renders in that class. +#[test] +fn control_plane_errors_have_one_class_on_native_and_pgwire() { + for err in control_plane_samples() { + let (_, pg_state, _) = error_to_sqlstate(&err); + assert_ne!(pg_state, sqlstate::INTERNAL_ERROR, "{err:?} has no class"); + + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, pg_state, "native SQLSTATE for {err:?}"); + let native_state = numeric_code_to_sqlstate(native.code); + assert_eq!( + class(native_state), + class(pg_state), + "{err:?}: pgwire sends {pg_state}, native code {} renders {native_state}", + native.code + ); + } +} + +/// A schema change the server could not absorb is the retryable +/// serialization class on every surface. +#[test] +fn schema_change_is_a_retryable_serialization_failure() { + let err = crate::Error::RetryableSchemaChanged { + descriptor: "orders".into(), + }; + assert_eq!(error_to_sqlstate(&err).1, sqlstate::SERIALIZATION_FAILURE); + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, sqlstate::SERIALIZATION_FAILURE); + assert_eq!(native.code, nodedb_types::error::ErrorCode::WRITE_CONFLICT); + assert!(crate::error_classify::classify(&err).is_retriable()); + let status = GatewayErrorMap::to_http(&err).0; + assert_eq!(status, 409); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(sqlstate::SERIALIZATION_FAILURE), + status + ); +} + +/// An expired session token is invalid authorization on every surface. +#[test] +fn expired_session_token_is_invalid_authorization_everywhere() { + let err = crate::Error::SessionTokenExpired; + assert_eq!(error_to_sqlstate(&err).1, sqlstate::INVALID_AUTHORIZATION); + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, sqlstate::INVALID_AUTHORIZATION); + assert_eq!(native.code, nodedb_types::error::ErrorCode::AUTH_EXPIRED); + assert_eq!( + numeric_code_to_sqlstate(native.code), + sqlstate::INVALID_AUTHORIZATION + ); + let status = GatewayErrorMap::to_http(&err).0; + assert_eq!(status, 401); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(sqlstate::INVALID_AUTHORIZATION), + status + ); +} diff --git a/nodedb/src/control/gateway/error_map/http.rs b/nodedb/src/control/gateway/error_map/http.rs index b194b72c2..a1ad0c5e8 100644 --- a/nodedb/src/control/gateway/error_map/http.rs +++ b/nodedb/src/control/gateway/error_map/http.rs @@ -4,13 +4,22 @@ use super::gateway_map::GatewayErrorMap; use super::remote_code::remote_code_to_http_status; +use super::sqlstate_status::sqlstate_to_http_status; use crate::Error; impl GatewayErrorMap { + /// Map a SQLSTATE into an HTTP status, for an error that reaches HTTP as + /// a SQLSTATE, such as a DDL error. Each class takes the status + /// [`Self::to_http`] gives the gateway errors of that class. + pub fn sqlstate_to_http(sqlstate: &str) -> u16 { + sqlstate_to_http_status(sqlstate) + } + /// Map a gateway error into `(http_status_code, message)` for HTTP. /// /// Uses standard HTTP status semantics: /// - 400 Bad Request for client-side errors (bad SQL, not found) + /// - 401 Unauthorized for an expired session token /// - 403 Forbidden for authz errors /// - 409 Conflict for write-conflict / constraint violations /// - 429 Too Many Requests for a rate-gate refusal @@ -25,11 +34,14 @@ impl GatewayErrorMap { format!("cluster in leader election; leader hint: {leader_addr}"), ), Error::DeadlineExceeded { .. } => (504, err.to_string()), - Error::RetryableSchemaChanged { .. } => (503, err.to_string()), + // The retryable serialization class, the status a write conflict + // takes on every path. + Error::RetryableSchemaChanged { .. } => (409, err.to_string()), Error::CollectionNotFound { collection, .. } => { (404, format!("collection \"{collection}\" does not exist")) } Error::RejectedAuthz { .. } => (403, err.to_string()), + Error::SessionTokenExpired => (401, err.to_string()), Error::BadRequest { detail } => (400, detail.clone()), Error::PlanError { detail } => (400, detail.clone()), Error::RejectedConstraint { detail, .. } => (409, detail.clone()), diff --git a/nodedb/src/control/gateway/error_map/mod.rs b/nodedb/src/control/gateway/error_map/mod.rs index 095a7de80..640544fe6 100644 --- a/nodedb/src/control/gateway/error_map/mod.rs +++ b/nodedb/src/control/gateway/error_map/mod.rs @@ -14,6 +14,7 @@ mod native; mod pgwire; mod remote_code; mod resp; +mod sqlstate_status; #[cfg(test)] mod system_dispatch_refusal; #[cfg(test)] diff --git a/nodedb/src/control/gateway/error_map/pgwire.rs b/nodedb/src/control/gateway/error_map/pgwire.rs index f643056b7..7b7e80567 100644 --- a/nodedb/src/control/gateway/error_map/pgwire.rs +++ b/nodedb/src/control/gateway/error_map/pgwire.rs @@ -37,12 +37,12 @@ mod tests { assert_eq!(code, sqlstate::QUERY_CANCELED); } - /// Both paths answer `INTERNAL_ERROR` from the shared table, with the - /// variant's own message naming the descriptor. + /// Both paths answer `SERIALIZATION_FAILURE`, the SQLSTATE a client + /// retries on, with the variant's own message naming the descriptor. #[test] fn pgwire_schema_changed() { let (code, msg) = GatewayErrorMap::to_pgwire(&schema_changed()); - assert_eq!(code, sqlstate::INTERNAL_ERROR); + assert_eq!(code, sqlstate::SERIALIZATION_FAILURE); assert!(msg.contains("users")); } diff --git a/nodedb/src/control/gateway/error_map/remote_code.rs b/nodedb/src/control/gateway/error_map/remote_code.rs index 6dc2f4d64..8f9d94cb2 100644 --- a/nodedb/src/control/gateway/error_map/remote_code.rs +++ b/nodedb/src/control/gateway/error_map/remote_code.rs @@ -23,6 +23,7 @@ pub(super) fn remote_code_to_http_status(code: nodedb_types::error::ErrorCode) - Ec::DEADLINE_EXCEEDED => 504, Ec::COLLECTION_NOT_FOUND | Ec::DOCUMENT_NOT_FOUND => 404, Ec::AUTHORIZATION_DENIED => 403, + Ec::AUTH_EXPIRED => 401, Ec::BAD_REQUEST | Ec::PLAN_ERROR | Ec::TYPE_MISMATCH diff --git a/nodedb/src/control/gateway/error_map/sqlstate_status.rs b/nodedb/src/control/gateway/error_map/sqlstate_status.rs new file mode 100644 index 000000000..5a181f25e --- /dev/null +++ b/nodedb/src/control/gateway/error_map/sqlstate_status.rs @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! SQLSTATE to HTTP status, for an error that reaches HTTP as a SQLSTATE. +//! +//! A DDL error carries a SQLSTATE and a numeric code. Several DDL SQLSTATEs +//! share one code, so the status follows the SQLSTATE. Each class takes the +//! status `to_http` gives the gateway errors of that class. + +use nodedb_types::error::sqlstate; + +/// Map a SQLSTATE to an HTTP status. +/// +/// - 5xx for an internal or system error, and for an unavailable server +/// - 4xx per class for a request the client must change +/// - 409 for a conflict or a constraint violation +/// - 429 for a rate limit +/// - 501 for an unsupported feature +pub(super) fn sqlstate_to_http_status(state: &str) -> u16 { + match state { + sqlstate::INSUFFICIENT_PRIVILEGE => 403, + sqlstate::UNDEFINED_TABLE => 404, + // The object already exists: the request conflicts with the catalog. + "42710" | "42P07" | "42723" => 409, + sqlstate::TOO_MANY_CONNECTIONS => 429, + // A configured quota the request exceeds. A retry fails the same way. + sqlstate::CONFIGURATION_LIMIT_EXCEEDED => 400, + // No leader or no quorum answered. A retry succeeds later. + sqlstate::LOCK_NOT_AVAILABLE => 503, + sqlstate::QUERY_CANCELED => 504, + _ => class_status(state), + } +} + +/// The status of a SQLSTATE class. +fn class_status(state: &str) -> u16 { + match state.get(..2).unwrap_or(state) { + // No data, or an unknown database. + "02" | "3D" => 404, + "0A" => 501, + // Connection failure, insufficient resources, operator intervention. + "08" | "53" | "57" => 503, + // Data exception, invalid transaction state, syntax or access rule, + // program limit. + "22" | "25" | "42" | "54" => 400, + // Constraint violation, dependent objects, transaction rollback, + // object not in prerequisite state. + "23" | "2B" | "40" | "55" => 409, + "28" => 401, + // `XX` internal error, `58` system error, and any other class. + _ => 500, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn internal_errors_are_server_faults() { + assert_eq!(sqlstate_to_http_status(sqlstate::INTERNAL_ERROR), 500); + assert_eq!(sqlstate_to_http_status(sqlstate::IO_ERROR), 500); + } + + #[test] + fn feature_not_supported_is_not_implemented() { + assert_eq!( + sqlstate_to_http_status(sqlstate::FEATURE_NOT_SUPPORTED), + 501 + ); + } + + #[test] + fn conflicts_and_constraints_are_409() { + for state in [ + sqlstate::UNIQUE_VIOLATION, + sqlstate::NOT_NULL_VIOLATION, + sqlstate::CHECK_VIOLATION, + sqlstate::SERIALIZATION_FAILURE, + sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, + "42P07", + "2BP01", + ] { + assert_eq!(sqlstate_to_http_status(state), 409, "{state}"); + } + } + + #[test] + fn rate_limits_are_429() { + assert_eq!(sqlstate_to_http_status(sqlstate::TOO_MANY_CONNECTIONS), 429); + } + + #[test] + fn client_errors_take_their_class_status() { + assert_eq!(sqlstate_to_http_status(sqlstate::SYNTAX_ERROR), 400); + assert_eq!(sqlstate_to_http_status(sqlstate::DATA_EXCEPTION), 400); + assert_eq!( + sqlstate_to_http_status(sqlstate::INSUFFICIENT_PRIVILEGE), + 403 + ); + assert_eq!( + sqlstate_to_http_status(sqlstate::INVALID_AUTHORIZATION), + 401 + ); + assert_eq!(sqlstate_to_http_status(sqlstate::UNDEFINED_TABLE), 404); + assert_eq!(sqlstate_to_http_status(sqlstate::INVALID_CATALOG_NAME), 404); + } + + #[test] + fn unavailability_is_503_and_a_deadline_is_504() { + assert_eq!(sqlstate_to_http_status(sqlstate::SERVER_OVERLOAD), 503); + assert_eq!(sqlstate_to_http_status(sqlstate::LOCK_NOT_AVAILABLE), 503); + assert_eq!(sqlstate_to_http_status(sqlstate::QUERY_CANCELED), 504); + } +} diff --git a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs index 9e424123f..c83443852 100644 --- a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs +++ b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs @@ -18,7 +18,7 @@ use nodedb_types::TenantId; -use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::bridge::envelope::PhysicalPlan; use crate::control::maintenance::clone_materializer::{dispatch_local, read_all_source_rows}; use crate::control::state::SharedState; use crate::types::VShardId; @@ -181,15 +181,10 @@ async fn resolve_merge_arms( task.txn_id, ) .await?; - if resolve_resp.status != Status::Ok { - return Err(crate::Error::Dispatch { - detail: format!( - "in-transaction MERGE resolve failed: {:?}", - resolve_resp.error_code - ), - }); - } - decode_resolve(&resolve_resp.payload) + // A refused resolve keeps its Data-Plane code. + let payload = + crate::control::server::shared::response_payload::payload_or_typed_error(resolve_resp)?; + decode_resolve(&payload) } /// Rewrite the three resolved arms into concrete point-write tasks appended diff --git a/nodedb/src/control/planner/calvin/preexec.rs b/nodedb/src/control/planner/calvin/preexec.rs index ba91a9c26..41584da62 100644 --- a/nodedb/src/control/planner/calvin/preexec.rs +++ b/nodedb/src/control/planner/calvin/preexec.rs @@ -128,13 +128,8 @@ pub async fn run_preexec_scan( database_id, txn_id: None, }; - let payloads = gateway - .execute_internal(&gw_ctx, scan_plan) - .await - .map_err(|e| crate::Error::Storage { - engine: "preexec-scan".into(), - detail: format!("pre-execution scan failed: {e}"), - })?; + // A shard verdict keeps its own typed error. + let payloads = gateway.execute_internal(&gw_ctx, scan_plan).await?; // A single-collection scan routes to one vshard → one payload. An // absent payload means zero matching rows. let payload = payloads.into_iter().next().unwrap_or_default(); @@ -151,16 +146,17 @@ pub async fn run_preexec_scan( ) .await?; - // A shard verdict keeps its own typed error, so a scan the statement's - // deadline cut short reports the deadline rather than a storage fault. - crate::control::server::dispatch_utils::reject_data_plane_error(&response)?; - if response.status != crate::bridge::envelope::Status::Ok { - return Err(crate::Error::Storage { - engine: "preexec-scan".into(), - detail: format!("pre-execution scan failed: {:?}", response.error_code), - }); - } + scan_from_response(&response) +} +/// Decode a local scan response. +/// +/// A shard verdict keeps its own typed error, so a scan the statement's +/// deadline cut short reports the deadline rather than a storage fault. +/// `reject_data_plane_error` passes only a `NotFound` refusal. Its payload is +/// empty, so it decodes as no matches, the answer the gateway path gives. +fn scan_from_response(response: &crate::bridge::envelope::Response) -> crate::Result { + crate::control::server::dispatch_utils::reject_data_plane_error(response)?; Ok(decode_scan(&response.payload)) } @@ -276,6 +272,46 @@ fn decode_scan_json(json_str: &str) -> PreexecScan { #[cfg(test)] mod tests { use super::*; + use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; + use crate::types::{Lsn, RequestId}; + + fn refusal(code: ErrorCode) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: Some(Box::new(code)), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A refused scan keeps its code, never a storage error. + #[test] + fn a_refused_scan_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + match scan_from_response(&refusal(code.clone())) { + Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), + Err(other) => panic!("expected the typed refusal, got {other:?}"), + Ok(_) => panic!("a refused scan must fail"), + } + } + + /// A `NotFound` refusal means the shard holds no slice of the collection, + /// so the scan matched nothing. + #[test] + fn a_not_found_scan_matches_nothing() { + let scan = scan_from_response(&refusal(ErrorCode::NotFound)) + .expect("a NotFound refusal reads as an empty scan"); + assert!(scan.surrogates.is_empty()); + assert!(scan.edges.is_empty()); + } #[test] fn decode_empty_payload_returns_empty() { diff --git a/nodedb/src/control/planner/materialized_sum/recon.rs b/nodedb/src/control/planner/materialized_sum/recon.rs index f1d549da9..16ff2cc82 100644 --- a/nodedb/src/control/planner/materialized_sum/recon.rs +++ b/nodedb/src/control/planner/materialized_sum/recon.rs @@ -164,13 +164,10 @@ async fn execute_read( database_id, txn_id: None, }; + // A shard verdict keeps its own typed error. let (payloads, _watermarks, read_version_lsn) = gateway .execute_internal_with_watermarks(&gw_ctx, plan) - .await - .map_err(|e| crate::Error::Storage { - engine: "materialized-sum-recon".into(), - detail: format!("reconnaissance read failed: {e}"), - })?; + .await?; return Ok(ReconRead { rows: payloads, read_version_lsn, @@ -188,15 +185,19 @@ async fn execute_read( TraceId::ZERO, ) .await?; - // A shard verdict keeps its own typed error, so a read the statement's - // deadline cut short reports the deadline rather than a storage fault. - crate::control::server::dispatch_utils::reject_data_plane_error(&response)?; - if response.status != crate::bridge::envelope::Status::Ok { - return Err(crate::Error::Storage { - engine: "materialized-sum-recon".into(), - detail: format!("reconnaissance read failed: {:?}", response.error_code), - }); - } + read_from_response(&response) +} + +/// The rows of a local read response. +/// +/// A shard verdict keeps its own typed error, so a read the statement's +/// deadline cut short reports the deadline rather than a storage fault. +/// `reject_data_plane_error` passes only a `NotFound` refusal. Its payload is +/// empty, so it reads as no rows, the answer the gateway path gives. +fn read_from_response( + response: &crate::bridge::envelope::Response, +) -> crate::Result>>> { + crate::control::server::dispatch_utils::reject_data_plane_error(response)?; Ok(ReconRead { read_version_lsn: response.read_version_lsn, rows: vec![response.payload.to_vec()], @@ -217,3 +218,47 @@ fn decode_rows(payload: &[u8]) -> Vec { .filter_map(|(_, body)| nodedb_types::json_from_msgpack(&body).ok()) .collect() } + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; + use crate::types::RequestId; + + fn refusal(code: ErrorCode) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: Some(Box::new(code)), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A refused read keeps its code, never a storage error. + #[test] + fn a_refused_read_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + match read_from_response(&refusal(code.clone())) { + Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), + Err(other) => panic!("expected the typed refusal, got {other:?}"), + Ok(_) => panic!("a refused read must fail"), + } + } + + /// A `NotFound` refusal reads as one empty payload, which decodes to no + /// rows. + #[test] + fn a_not_found_read_has_no_rows() { + let read = read_from_response(&refusal(ErrorCode::NotFound)) + .expect("a NotFound refusal reads as no rows"); + assert!(read.rows.iter().all(|payload| payload.is_empty())); + } +} diff --git a/nodedb/src/control/server/http/auth.rs b/nodedb/src/control/server/http/auth.rs index a40e3f2ef..dc2aa268c 100644 --- a/nodedb/src/control/server/http/auth.rs +++ b/nodedb/src/control/server/http/auth.rs @@ -320,6 +320,9 @@ pub enum ApiError { status: StatusCode, message: String, code: nodedb_types::error::ErrorCode, + /// The typed error that caused this one, such as the Data-Plane + /// refusal behind a DDL phase failure. `None` when there is none. + cause: Option>, }, } @@ -343,8 +346,13 @@ impl IntoResponse for ApiError { status, message, code, + cause, } => { let body = HttpError::with_code(message, code.to_string()); + let body = match cause { + Some(cause) => body.caused_by(&cause), + None => body, + }; (status, axum::Json(body)).into_response() } other => { diff --git a/nodedb/src/control/server/http/routes/query/materialized/encode.rs b/nodedb/src/control/server/http/routes/query/materialized/encode.rs index e8effeb3d..138bfcfba 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/encode.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/encode.rs @@ -4,16 +4,17 @@ use super::super::super::super::auth::ApiError; +/// Map a DDL error to the HTTP error the client reads. The status follows the +/// SQLSTATE through the gateway status table. The code and the typed cause +/// travel in the body, as they do on native and pgwire. pub(super) fn ddl_error_to_api(error: crate::control::server::shared::ddl::DdlError) -> ApiError { - let status = if error.sqlstate == "42501" { - axum::http::StatusCode::FORBIDDEN - } else { - axum::http::StatusCode::BAD_REQUEST - }; + let status = crate::control::gateway::GatewayErrorMap::sqlstate_to_http(&error.sqlstate); ApiError::Coded { - status, + status: axum::http::StatusCode::from_u16(status) + .unwrap_or(axum::http::StatusCode::INTERNAL_SERVER_ERROR), message: error.message, code: error.code, + cause: error.cause, } } @@ -45,7 +46,7 @@ mod tests { assert!(matches!( ddl_error_to_api(error), - ApiError::Coded { status, message, code } + ApiError::Coded { status, message, code, .. } if status == axum::http::StatusCode::FORBIDDEN && message == "write permission denied" && code == nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED @@ -72,5 +73,86 @@ mod tests { nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED.to_string() ); assert_eq!(json["error"], "write permission denied"); + assert!(json.get("cause").is_none(), "no cause, no cause field"); + } + + async fn response_json(error: ApiError) -> (axum::http::StatusCode, serde_json::Value) { + let response = error.into_response(); + let status = response.status(); + let body = axum::body::to_bytes(response.into_body(), usize::MAX) + .await + .expect("read response body"); + let json = serde_json::from_slice(&body).expect("valid JSON body"); + (status, json) + } + + /// An internal DDL error is a server fault, and its typed cause reaches + /// the client with its own message and code. + #[tokio::test] + async fn internal_ddl_error_is_500_with_its_cause() { + let cause = nodedb_types::NodeDbError::from(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::Unsupported { + detail: "not on this engine".into(), + }, + )); + let mut error = crate::control::server::shared::ddl::DdlError::move_tenant_cutover_failed( + "MOVE TENANT cutover failed", + ); + error.cause = Some(Box::new(cause)); + + let (status, json) = response_json(ddl_error_to_api(error)).await; + assert_eq!(status, axum::http::StatusCode::INTERNAL_SERVER_ERROR); + assert_eq!(json["error"], "MOVE TENANT cutover failed"); + assert_eq!( + json["code"], + nodedb_types::error::ErrorCode::MOVE_TENANT_CUTOVER_FAILED.to_string() + ); + assert_eq!( + json["cause"]["code"], + nodedb_types::error::ErrorCode::SQL_NOT_ENABLED.to_string() + ); + assert_eq!(json["cause"]["error"], "not on this engine"); + } + + /// A plain `XX000` DDL error is a server fault, never a client error. + #[tokio::test] + async fn xx000_ddl_error_is_500() { + let error = + crate::control::server::shared::ddl::DdlError::new("XX000", "catalog write failed"); + let (status, _) = response_json(ddl_error_to_api(error)).await; + assert_eq!(status, axum::http::StatusCode::INTERNAL_SERVER_ERROR); + } + + /// A feature-not-supported DDL error is 501 Not Implemented. + #[tokio::test] + async fn feature_not_supported_ddl_error_is_501() { + let error = crate::control::server::shared::ddl::DdlError::new( + "0A000", + "changing vector index params is not supported", + ); + let (status, json) = response_json(ddl_error_to_api(error)).await; + assert_eq!(status, axum::http::StatusCode::NOT_IMPLEMENTED); + assert_eq!( + json["code"], + nodedb_types::error::ErrorCode::SQL_NOT_ENABLED.to_string() + ); + } + + /// A conflict, a constraint violation and a rate limit take their own + /// status, never a blanket 400. + #[test] + fn ddl_conflicts_and_rate_limits_take_their_class_status() { + for (state, expected) in [ + ("42P07", axum::http::StatusCode::CONFLICT), + ("23505", axum::http::StatusCode::CONFLICT), + ("53300", axum::http::StatusCode::TOO_MANY_REQUESTS), + ("42P01", axum::http::StatusCode::NOT_FOUND), + ] { + let error = crate::control::server::shared::ddl::DdlError::new(state, "refused"); + match ddl_error_to_api(error) { + ApiError::Coded { status, .. } => assert_eq!(status, expected, "{state}"), + other => panic!("expected a coded error for {state}, got {other:?}"), + } + } } } diff --git a/nodedb/src/control/server/http/routes/result_shape.rs b/nodedb/src/control/server/http/routes/result_shape.rs index 50fcbf793..28ee3e84f 100644 --- a/nodedb/src/control/server/http/routes/result_shape.rs +++ b/nodedb/src/control/server/http/routes/result_shape.rs @@ -44,6 +44,7 @@ pub(super) fn shape_error_to_api(e: NodeDbError) -> ApiError { status, message: e.message().to_string(), code, + cause: e.cause().map(|cause| Box::new(cause.clone())), } } diff --git a/nodedb/src/control/server/http/types.rs b/nodedb/src/control/server/http/types.rs index c692aadbe..df8abe506 100644 --- a/nodedb/src/control/server/http/types.rs +++ b/nodedb/src/control/server/http/types.rs @@ -126,6 +126,10 @@ pub struct HttpError { pub error: String, #[serde(skip_serializing_if = "Option::is_none")] pub code: Option, + /// The typed error that caused this one, as its message and code. Absent + /// when there is none. + #[serde(skip_serializing_if = "Option::is_none")] + pub cause: Option>, } impl HttpError { @@ -133,6 +137,7 @@ impl HttpError { Self { error: error.into(), code: None, + cause: None, } } @@ -140,8 +145,18 @@ impl HttpError { Self { error: error.into(), code: Some(code.into()), + cause: None, } } + + /// Carry `cause` as this error's cause, with its own message and code. + pub fn caused_by(mut self, cause: &nodedb_types::NodeDbError) -> Self { + self.cause = Some(Box::new(Self::with_code( + cause.message(), + cause.code().to_string(), + ))); + self + } } #[cfg(test)] diff --git a/nodedb/src/control/server/native/dispatch/conversion.rs b/nodedb/src/control/server/native/dispatch/conversion.rs index 5c2abe257..a3592317b 100644 --- a/nodedb/src/control/server/native/dispatch/conversion.rs +++ b/nodedb/src/control/server/native/dispatch/conversion.rs @@ -51,11 +51,12 @@ pub(crate) fn native_error_fields(e: &crate::Error) -> NativeErrorFields { crate::Error::CollectionNotFound { collection, .. } => { ("42P01", format!("collection '{collection}' not found")) } - // Same SQLSTATE as the "not authenticated" responses in - // `session::request`: the client's stored bearer token expired - // mid-connection and it must re-authenticate with a fresh Auth frame. + // The SQLSTATE pgwire renders, and the one the "not authenticated" + // responses in `session::request` carry. The client's stored bearer + // token expired mid-connection and it must re-authenticate with a + // fresh Auth frame. The message names the native Auth frame. crate::Error::SessionTokenExpired => ( - "28000", + nodedb_types::error::sqlstate::INVALID_AUTHORIZATION, "OIDC bearer token expired; re-authenticate with a fresh Auth request".into(), ), // A cross-shard Calvin OCC abort is a serialization failure (40001) — diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 875092e82..9e9798904 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -204,6 +204,16 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::SourceFrozen { .. } => { ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) } + // A descriptor changed under the statement and the server's own + // retries ran out. The client retries the statement, so it takes + // SERIALIZATION_FAILURE (40001), the SQLSTATE drivers retry on. + crate::Error::RetryableSchemaChanged { .. } => { + ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) + } + // The session's bearer token expired. The client re-authenticates. + crate::Error::SessionTokenExpired => { + ("ERROR", sqlstate::INVALID_AUTHORIZATION, err.to_string()) + } crate::Error::CloneWriteRequiresMaterialize { .. } => ( "ERROR", sqlstate::CLONE_WRITE_REQUIRES_MATERIALIZE.0, @@ -298,7 +308,8 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> // Mirrors the `RejectedConstraint` arm. Ec::CONSTRAINT_VIOLATION => sqlstate::UNIQUE_VIOLATION, // Mirrors the `ConflictRetry` / `CalvinSerializationConflict` / - // `SourceFrozen` arms, and `OllpExhausted` when it exhausted on drift. + // `SourceFrozen` / `RetryableSchemaChanged` arms, and `OllpExhausted` + // when it exhausted on drift. Ec::WRITE_CONFLICT => sqlstate::SERIALIZATION_FAILURE, // Mirrors the `DeadlineExceeded` arm. Ec::DEADLINE_EXCEEDED => sqlstate::QUERY_CANCELED, @@ -328,6 +339,8 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> Ec::FAN_OUT_EXCEEDED => sqlstate::STATEMENT_TOO_COMPLEX, // Mirrors the `RejectedAuthz` arm. Ec::AUTHORIZATION_DENIED => sqlstate::INSUFFICIENT_PRIVILEGE, + // Mirrors the `SessionTokenExpired` arm. + Ec::AUTH_EXPIRED => sqlstate::INVALID_AUTHORIZATION, // Mirrors the `RateExceeded` arm. Ec::RATE_EXCEEDED => sqlstate::TOO_MANY_CONNECTIONS, // Mirrors the `MemoryExhausted` / `Backpressure` arms. diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs index 4a4086804..5222ebec0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/kv_index.rs @@ -108,7 +108,8 @@ pub(crate) async fn drop_kv_index( } /// Dispatch a KV index plan through the autocommit write funnel, which -/// appends its WAL record, and fail on a refused reply. +/// appends its WAL record, and fail on a refused reply. A refusal keeps its +/// SQLSTATE and code. async fn dispatch_durable( state: &SharedState, tenant_id: TenantId, @@ -117,7 +118,7 @@ async fn dispatch_durable( plan: PhysicalPlan, step: &str, ) -> Result<(), DdlError> { - let response = crate::control::server::dispatch_utils::dispatch_autocommit_write( + crate::control::server::dispatch_utils::dispatch_autocommit_write( state, crate::control::server::dispatch_utils::AutocommitWrite { tenant_id, @@ -130,23 +131,10 @@ async fn dispatch_durable( }, ) .await + .and_then(crate::control::server::shared::response_payload::payload_or_typed_error) .map_err(|e| { - err( - "XX000", - format!("key-value index {step} on '{collection}': {e}"), - ) + DdlError::from_error_in_context(&format!("key-value index {step} on '{collection}'"), &e) })?; - - if response.status == crate::bridge::envelope::Status::Error { - let detail = match response.error_code.as_deref() { - Some(code) => format!("{code:?}"), - None => String::from_utf8_lossy(&response.payload).into_owned(), - }; - return Err(err( - "XX000", - format!("key-value index {step} on '{collection}' was refused: {detail}"), - )); - } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs index 898b05a2d..f92339b5c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs @@ -105,14 +105,17 @@ pub async fn dispatch_unregister_collection( .ok_or_else(|| crate::Error::Dispatch { detail: format!("collection reclaim channel closed on core {core_id}"), })?; + // A coded refusal keeps its Data-Plane code. if response.status != Status::Ok { - return Err(crate::Error::Storage { - engine: "collection-purge".into(), - detail: format!( - "UnregisterCollection for tenant {tenant_id} collection '{name}' \ - failed on core {core_id}: {:?}", - response.error_code - ), + return Err(match response.error_code { + Some(code) => crate::Error::DataPlane(*code), + None => crate::Error::Storage { + engine: "collection-purge".into(), + detail: format!( + "UnregisterCollection for tenant {tenant_id} collection '{name}' \ + failed on core {core_id} with no error code" + ), + }, }); } Ok(()) diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/mod.rs index 43ce99325..2bb312865 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/mod.rs @@ -19,6 +19,6 @@ pub mod dispatch; pub mod handlers; pub(crate) use dispatch::{ - ddl_err, dispatch_and_respond, parse_function_args, single_text_col, split_args, unquote, + dispatch_and_respond, parse_function_args, single_text_col, split_args, unquote, }; pub use handlers::{kv_cas, kv_getset, kv_incr, kv_incr_float}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index d8f4a8f0d..6b79f5a8e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -16,6 +16,7 @@ use crate::bridge::envelope::{PhysicalPlan, Status}; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; use crate::types::{DatabaseId, TraceId, VShardId}; use nodedb_physical::physical_plan::KvOp; @@ -99,7 +100,7 @@ pub async fn rate_check( tenant_id, rate_key.as_bytes(), ) - .map_err(|e| super::kv_atomic::ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error_in_context("RATE_CHECK", &e))?; let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, RATE_COLLECTION), key: rate_key.as_bytes().to_vec(), @@ -116,40 +117,33 @@ pub async fn rate_check( shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); - match dispatch_counter_write(state, tenant_id, vshard, plan).await { - Ok(resp) if resp.status == Status::Ok => { - let payload_text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - let current: i64 = sonic_rs::from_str::(&payload_text) - .ok() - .and_then(|v| v.get("value")?.as_i64()) - .unwrap_or(1); - - if current > max_count { - // Read TTL to compute retry_after_ms. - let ttl_remaining = read_ttl_ms(state, tenant_id, vshard, &rate_key).await; - Err(ddl_err( - "53300", - format!( - "rate limit exceeded for {gate_name}:{key}, retry after {ttl_remaining}ms (current={current}, max={max_count})" - ), - )) - } else { - let result = serde_json::json!({ - "allowed": true, - "current": current, - "max_count": max_count, - "remaining": max_count - current, - }); - Ok(vec![single_text_col("rate_check", result.to_string())]) - } - } - Ok(resp) => { - let payload_text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - Err(ddl_err("XX000", payload_text)) - } - Err(e) => Err(ddl_err("XX000", e.to_string())), + let payload = counter_write_payload( + "RATE_CHECK", + dispatch_counter_write(state, tenant_id, vshard, plan).await, + )?; + let payload_text = crate::data::executor::response_codec::decode_payload_to_json(&payload); + let current: i64 = sonic_rs::from_str::(&payload_text) + .ok() + .and_then(|v| v.get("value")?.as_i64()) + .unwrap_or(1); + + if current > max_count { + // Read TTL to compute retry_after_ms. + let ttl_remaining = read_ttl_ms(state, tenant_id, vshard, &rate_key).await; + Err(ddl_err( + "53300", + format!( + "rate limit exceeded for {gate_name}:{key}, retry after {ttl_remaining}ms (current={current}, max={max_count})" + ), + )) + } else { + let result = serde_json::json!({ + "allowed": true, + "current": current, + "max_count": max_count, + "remaining": max_count - current, + }); + Ok(vec![single_text_col("rate_check", result.to_string())]) } } @@ -259,26 +253,17 @@ pub async fn rate_reset( provenance: None, }); - match dispatch_counter_write(state, tenant_id, vshard, plan).await { - Ok(resp) if resp.status == Status::Ok => { - let result = serde_json::json!({ - "gate": gate_name, - "key": key, - "reset": true, - }); - Ok(vec![single_text_col("rate_reset", result.to_string())]) - } - // A refusal arrives as an error status inside an `Ok` response. The - // counter is still there, so the reset did not happen. - Ok(resp) => Err(ddl_err( - "XX000", - format!( - "RATE_RESET: the counter delete was refused: {:?}", - resp.error_code - ), - )), - Err(e) => Err(ddl_err("XX000", e.to_string())), - } + // A refusal means the counter is still there, so the reset did not happen. + counter_write_payload( + "RATE_RESET", + dispatch_counter_write(state, tenant_id, vshard, plan).await, + )?; + let result = serde_json::json!({ + "gate": gate_name, + "key": key, + "reset": true, + }); + Ok(vec![single_text_col("rate_reset", result.to_string())]) } // ── Helpers ──────────────────────────────────────────────────────────── @@ -308,6 +293,19 @@ async fn dispatch_counter_write( .await } +/// The payload of a counter write, or its error with `context` before the +/// message. A refusal arrives as an error status inside an `Ok` response. A +/// coded refusal keeps its SQLSTATE and code. Only a refusal with no code is +/// `XX000`. +fn counter_write_payload( + context: &str, + result: crate::Result, +) -> Result, DdlError> { + result + .and_then(payload_or_typed_error) + .map_err(|e| DdlError::from_error_in_context(context, &e)) +} + /// Read TTL remaining for a KV key (in milliseconds). async fn read_ttl_ms( state: &SharedState, @@ -376,3 +374,60 @@ fn parse_u64(s: &str, func: &str, param: &str) -> Result { fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } + +#[cfg(test)] +mod tests { + use nodedb_types::error::sqlstate; + + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload, Response}; + use crate::types::{Lsn, RequestId}; + + fn refusal(code: Option) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + /// A refused counter write keeps its SQLSTATE and code, with the function + /// name before the message. + #[test] + fn a_coded_refusal_keeps_its_sqlstate() { + let refused = refusal(Some(ErrorCode::Unsupported { + detail: "not on this engine".into(), + })); + let err = counter_write_payload("RATE_CHECK", Ok(refused)) + .expect_err("a refused counter write fails the call"); + assert_eq!(err.sqlstate, sqlstate::FEATURE_NOT_SUPPORTED, "{err:?}"); + assert_eq!(err.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); + assert!(err.message.starts_with("RATE_CHECK: "), "{}", err.message); + } + + /// A dispatch error keeps its own class too. + #[test] + fn a_dispatch_error_keeps_its_sqlstate() { + let deadline = crate::Error::DeadlineExceeded { + request_id: RequestId::new(1), + }; + let err = counter_write_payload("RATE_RESET", Err(deadline)) + .expect_err("a failed dispatch fails the call"); + assert_eq!(err.sqlstate, sqlstate::QUERY_CANCELED, "{err:?}"); + } + + /// A refusal with no code has no class of its own. + #[test] + fn a_refusal_with_no_code_is_internal() { + let err = counter_write_payload("RATE_RESET", Ok(refusal(None))) + .expect_err("a refused counter write fails the call"); + assert_eq!(err.sqlstate, sqlstate::INTERNAL_ERROR, "{err:?}"); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs index 417a3d84f..403adf5c1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/weighted_pick.rs @@ -16,9 +16,10 @@ use serde_json::{Map, Value as JsonValue}; -use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; use crate::engine::random::alias::AliasTable; use crate::engine::random::csprng::SeedableRng; @@ -172,7 +173,9 @@ pub async fn weighted_pick( tenant_id, &audit_key_bytes, ) - .map_err(|e| ddl_err("XX000", format!("WEIGHTED_PICK: audit surrogate bind: {e}")))?; + .map_err(|e| { + DdlError::from_error_in_context("WEIGHTED_PICK: audit surrogate bind", &e) + })?; let audit_plan = PhysicalPlan::Kv(KvOp::Put { collection: nodedb_types::QualifiedCollection::new( DatabaseId::DEFAULT, @@ -190,7 +193,8 @@ pub async fn weighted_pick( // once its audit record is durable: Raft in cluster mode, else the // write funnel's `AppendHere`. A pick returned without its record is // an unaudited pick reported as an audited one. - let resp = crate::control::server::dispatch_utils::dispatch_durable_autocommit_write( + // A refused audit write keeps its SQLSTATE and code. + crate::control::server::dispatch_utils::dispatch_durable_autocommit_write( state, crate::control::server::dispatch_utils::AutocommitWrite { tenant_id, @@ -207,16 +211,8 @@ pub async fn weighted_pick( }, ) .await - .map_err(|e| ddl_err("XX000", format!("WEIGHTED_PICK: audit write: {e}")))?; - if resp.status != crate::bridge::envelope::Status::Ok { - return Err(ddl_err( - "XX000", - format!( - "WEIGHTED_PICK: the audit write was refused: {:?}", - resp.error_code - ), - )); - } + .and_then(payload_or_typed_error) + .map_err(|e| DdlError::from_error_in_context("WEIGHTED_PICK: audit write", &e))?; } // Step 5: Build response rows. @@ -265,7 +261,9 @@ async fn scan_all_entries( }); gate.inject_rls(&mut plan)?; - let resp = crate::control::server::dispatch_utils::dispatch_to_data_plane( + // A refused scan is not an empty collection: a pick from it draws from + // rows the scan never returned. It keeps its SQLSTATE and code. + let payload = crate::control::server::dispatch_utils::dispatch_to_data_plane( state, tenant_id, crate::types::DatabaseId::DEFAULT, @@ -274,22 +272,13 @@ async fn scan_all_entries( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; - - // A refused scan is not an empty collection: a pick from it draws from - // rows the scan never returned. - if resp.status != Status::Ok { - return Err(ddl_err( - "XX000", - format!( - "WEIGHTED_PICK: the scan of '{collection}' was refused: {:?}", - resp.error_code - ), - )); - } + .and_then(payload_or_typed_error) + .map_err(|e| { + DdlError::from_error_in_context(&format!("WEIGHTED_PICK: the scan of '{collection}'"), &e) + })?; // KV scan returns a flat msgpack array of entry maps. - let payload_text = crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); + let payload_text = crate::data::executor::response_codec::decode_payload_to_json(&payload); let json: serde_json::Value = sonic_rs::from_str(&payload_text).map_err(|e| { ddl_err( "XX000", diff --git a/nodedb/src/control/server/shared/response_payload.rs b/nodedb/src/control/server/shared/response_payload.rs index 710681367..1c1f04a4b 100644 --- a/nodedb/src/control/server/shared/response_payload.rs +++ b/nodedb/src/control/server/shared/response_payload.rs @@ -28,3 +28,44 @@ pub(crate) fn payload_or_typed_error(response: Response) -> crate::Result, payload: &[u8]) -> Response { + Response { + request_id: RequestId::new(1), + status: Status::Error, + attempt: 1, + partial: false, + payload: Payload::from_vec(payload.to_vec()), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + #[test] + fn a_coded_refusal_keeps_its_code() { + let code = ErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + match payload_or_typed_error(refusal(Some(code.clone()), b"")) { + Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), + other => panic!("expected the typed refusal, got {other:?}"), + } + } + + #[test] + fn a_refusal_with_no_code_reports_its_payload() { + match payload_or_typed_error(refusal(None, b"handler detail")) { + Err(crate::Error::Internal { detail }) => assert_eq!(detail, "handler detail"), + other => panic!("expected an internal error, got {other:?}"), + } + } +} diff --git a/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs b/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs index 4f24ed34a..a5f9d253d 100644 --- a/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs +++ b/nodedb/src/control/update_from_join_orchestrator/expand_staged_update_from_join.rs @@ -7,7 +7,7 @@ use nodedb_types::TenantId; -use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::bridge::envelope::PhysicalPlan; use crate::control::maintenance::clone_materializer::{dispatch_local, read_all_source_rows}; use crate::control::state::SharedState; use crate::control::target_identity::{ @@ -187,15 +187,10 @@ async fn resolve_update_rows( task.txn_id, ) .await?; - if resolve_resp.status != Status::Ok { - return Err(crate::Error::Dispatch { - detail: format!( - "in-transaction UPDATE ... FROM resolve failed: {:?}", - resolve_resp.error_code - ), - }); - } - decode_resolved_update_rows(&resolve_resp.payload) + // A refused resolve keeps its Data-Plane code. + let payload = + crate::control::server::shared::response_payload::payload_or_typed_error(resolve_resp)?; + decode_resolved_update_rows(&payload) } /// Decode the RESOLVE pass payload (a msgpack `Vec`; diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify.rs index 8f2be245b..e34573e08 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify.rs @@ -191,7 +191,11 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { Error::InvalidLimitValue { clause, value } => { NodeDbError::invalid_limit_value(*clause, value.clone()) } - Error::RetryableSchemaChanged { .. } => NodeDbError::plan_error(e.to_string()), + // Same contract as a write conflict, which callers already retry. + Error::RetryableSchemaChanged { descriptor } => NodeDbError::write_conflict( + descriptor.clone(), + "schema changed during execution; retry the statement".to_owned(), + ), Error::RetryableLeaderChange { group_id, log_index, @@ -309,7 +313,7 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { NodeDbError::bad_request("session terminated: idle timeout exceeded".to_owned()) } Error::SessionTokenExpired => { - NodeDbError::bad_request("session terminated: OIDC token expired".to_owned()) + NodeDbError::auth_expired("session closed: OIDC token expired") } Error::SessionKilledByAdmin => { NodeDbError::bad_request("session terminated by administrator".to_owned()) diff --git a/nodedb/tests/crash_harness/pgwire.rs b/nodedb/tests/crash_harness/pgwire.rs index 271a90a03..a0a7f9a4d 100644 --- a/nodedb/tests/crash_harness/pgwire.rs +++ b/nodedb/tests/crash_harness/pgwire.rs @@ -11,7 +11,7 @@ use std::time::{Duration, Instant}; use super::CrashHarness; /// Bounded retry budget for the `Error::RetryableSchemaChanged` condition -/// (rendered over pgwire as `XX000: schema changed during execution +/// (rendered over pgwire as `40001: schema changed during execution /// (); please retry`). /// /// The server already retries this condition server-side for ~750ms @@ -32,9 +32,9 @@ const SCHEMA_CHANGE_RETRY_BACKOFF: Duration = Duration::from_millis(150); /// Substring of `Error::RetryableSchemaChanged`'s Display text /// (`#[error("schema changed during execution ({descriptor}); please retry")]` -/// in `nodedb/src/error/types.rs`). The message is the durable signal: a code -/// alone would blanket-retry unrelated internal errors, because the class this -/// condition carries depends on the mapper in front of it. +/// in `nodedb/src/error/types.rs`). The message is the durable signal: the +/// code alone would also retry every other serialization failure, which +/// shares `40001`. /// The server was still reporting `RetryableSchemaChanged` when the /// client-side retry budget ran out. /// From b42e8ab62b9c4910c6cade42762462f76bb624d7 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 21:55:01 +0800 Subject: [PATCH 53/64] feat(errors): cross a shard's typed error, not just its Data-Plane code Generalize VShardRefusal to carry any ClusterError as a ShardErrorWire mirror instead of only a Data-Plane verdict. WrongOwner, Raft redirects, Codec, and Transport errors now cross the wire typed and rebuild on the other side; an error with no wire mirror falls back to RemoteUntyped, keeping its message instead of collapsing to a closed stream. --- nodedb-cluster/src/error.rs | 5 + .../src/raft_loop/handle_rpc/plan_dispatch.rs | 60 +++- nodedb-cluster/src/rpc_codec/discriminants.rs | 5 +- nodedb-cluster/src/rpc_codec/mod.rs | 2 + nodedb-cluster/src/rpc_codec/raft_rpc.rs | 4 +- .../src/rpc_codec/shard_error/convert.rs | 257 ++++++++++++++++++ .../src/rpc_codec/shard_error/mod.rs | 15 + .../src/rpc_codec/shard_error/raft.rs | 110 ++++++++ .../src/rpc_codec/shard_error/wire.rs | 121 +++++++++ nodedb-cluster/src/rpc_codec/vshard.rs | 101 ++++++- 10 files changed, 648 insertions(+), 32 deletions(-) create mode 100644 nodedb-cluster/src/rpc_codec/shard_error/convert.rs create mode 100644 nodedb-cluster/src/rpc_codec/shard_error/mod.rs create mode 100644 nodedb-cluster/src/rpc_codec/shard_error/raft.rs create mode 100644 nodedb-cluster/src/rpc_codec/shard_error/wire.rs diff --git a/nodedb-cluster/src/error.rs b/nodedb-cluster/src/error.rs index 7d490cd46..5131cf065 100644 --- a/nodedb-cluster/src/error.rs +++ b/nodedb-cluster/src/error.rs @@ -228,4 +228,9 @@ pub enum ClusterError { #[error("timeseries gather error: {0}")] TsGather(#[from] crate::distributed_timeseries::TsGatherError), + + /// A remote node answered with an error whose type has no wire mirror. + /// `detail` is that error's message. + #[error("remote error: {detail}")] + RemoteUntyped { detail: String }, } diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs index 5571f7535..9d6058d43 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs @@ -52,22 +52,17 @@ impl RaftLoop { } // VShardEnvelope — dispatch to registered handler (Event Plane, etc.). - // A typed Data-Plane verdict answers as a `VShardRefusal` frame, so the - // caller rebuilds the same code. Any other handler error closes the stream. + // Every handler error answers as a typed `VShardRefusal` frame, so the + // caller rebuilds the same `ClusterError` instead of seeing a closed + // stream. pub(super) async fn handle_vshard_envelope_rpc(&self, bytes: Vec) -> Result { - if let Some(ref handler) = self.vshard_handler { - match handler(bytes).await { - Ok(response_bytes) => Ok(RaftRpc::VShardEnvelope(response_bytes)), - Err(ClusterError::DataPlane { code }) => { - Ok(RaftRpc::VShardRefusal(VShardRefusal { code })) - } - Err(other) => Err(other), - } - } else { - Err(ClusterError::Transport { + let result = match self.vshard_handler { + Some(ref handler) => handler(bytes).await, + None => Err(ClusterError::Transport { detail: "VShardEnvelope handler not configured".into(), - }) - } + }), + }; + Ok(vshard_answer(result)) } // Streaming physical-plan execution (L4) — delegate to the PlanExecutor's @@ -82,6 +77,15 @@ impl RaftLoop { } } +/// The frame that answers a VShardEnvelope request: the handler's response +/// envelope, or its typed error as a refusal. +fn vshard_answer(result: Result>) -> RaftRpc { + match result { + Ok(response_bytes) => RaftRpc::VShardEnvelope(response_bytes), + Err(error) => RaftRpc::VShardRefusal(VShardRefusal::from(error)), + } +} + /// Propose a forwarded entry to its target group on this node. /// /// Answers `not leader` with the known leader as a hint when this node does @@ -133,6 +137,34 @@ mod tests { } } + /// A handler error such as `WrongOwner` answers as a typed refusal the + /// caller rebuilds, not as a closed stream. + #[test] + fn a_handler_error_answers_as_a_typed_refusal() { + let answer = vshard_answer(Err(ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: None, + })); + match answer { + RaftRpc::VShardRefusal(refusal) => assert!(matches!( + ClusterError::from(refusal.error), + ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: None + } + )), + other => panic!("expected a refusal frame, got {other:?}"), + } + } + + #[test] + fn a_handler_response_answers_as_an_envelope() { + match vshard_answer(Ok(vec![1, 2, 3])) { + RaftRpc::VShardEnvelope(bytes) => assert_eq!(bytes, vec![1, 2, 3]), + other => panic!("expected a response envelope, got {other:?}"), + } + } + #[test] fn sequencer_target_is_proposed_to_the_sequencer_group_on_its_leader() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/nodedb-cluster/src/rpc_codec/discriminants.rs b/nodedb-cluster/src/rpc_codec/discriminants.rs index 446a55818..1682fcff9 100644 --- a/nodedb-cluster/src/rpc_codec/discriminants.rs +++ b/nodedb-cluster/src/rpc_codec/discriminants.rs @@ -151,9 +151,8 @@ pub const RPC_AUTH_LEASE_RENEW_RESP: u8 = 50; /// `RPC_AUTH_BARRIER_RESP`. pub const RPC_AUTH_BARRIER_REQ: u8 = 51; pub const RPC_AUTH_BARRIER_RESP: u8 = 52; -/// Answer to an `RPC_VSHARD_ENVELOPE` request whose handler returned a typed -/// Data-Plane verdict. It carries the verdict code in place of a response -/// envelope. +/// Answer to an `RPC_VSHARD_ENVELOPE` request whose handler failed. It +/// carries the handler's typed error in place of a response envelope. pub const RPC_VSHARD_REFUSAL: u8 = 53; // VShardMessageType discriminants for distributed array ops (u16, range 80-89). diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index 0c0a88c7f..b38419f8c 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -24,6 +24,7 @@ pub mod raft_msgs; pub mod raft_rpc; pub mod read_index; pub mod reservation; +pub mod shard_error; pub mod shuffle; pub mod surrogate; pub mod vshard; @@ -60,6 +61,7 @@ pub use read_index::{ReadIndexOutcome, ReadIndexRequest, ReadIndexResponse}; pub use reservation::{ ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, }; +pub use shard_error::{RaftErrorWire, ShardErrorWire}; pub use shuffle::{ JoinKeyPair, PartNodeEntry, ShuffleAggregateConsumeRequest, ShuffleAggregateConsumeResponse, ShuffleConsumeRequest, ShuffleConsumeResponse, ShuffleProduceRequest, ShuffleProduceResponse, diff --git a/nodedb-cluster/src/rpc_codec/raft_rpc.rs b/nodedb-cluster/src/rpc_codec/raft_rpc.rs index 7550d845a..b01149021 100644 --- a/nodedb-cluster/src/rpc_codec/raft_rpc.rs +++ b/nodedb-cluster/src/rpc_codec/raft_rpc.rs @@ -158,8 +158,8 @@ pub enum RaftRpc { AuthLeaseRenewResponse(AuthLeaseRenewResponse), AuthBarrierRequest(AuthBarrierRequest), AuthBarrierResponse(AuthBarrierResponse), - // Answer to a `VShardEnvelope` request whose handler refused with a - // typed Data-Plane verdict. + // Answer to a `VShardEnvelope` request whose handler failed. It carries + // the handler's typed error. VShardRefusal(VShardRefusal), } diff --git a/nodedb-cluster/src/rpc_codec/shard_error/convert.rs b/nodedb-cluster/src/rpc_codec/shard_error/convert.rs new file mode 100644 index 000000000..a5e8aff17 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/shard_error/convert.rs @@ -0,0 +1,257 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Conversion between `ClusterError` and its wire mirror. +//! +//! Both matches are exhaustive with no catch-all, so a new `ClusterError` +//! variant fails to compile here until it has a wire form. + +use super::wire::ShardErrorWire; +use crate::error::ClusterError; + +impl From for ShardErrorWire { + fn from(error: ClusterError) -> Self { + match error { + ClusterError::Raft(error) => Self::Raft { + error: error.into(), + }, + ClusterError::VShardNotMapped { vshard_id } => Self::VShardNotMapped { vshard_id }, + ClusterError::GroupNotFound { group_id } => Self::GroupNotFound { group_id }, + ClusterError::LearnerNotCaughtUp { + group_id, + node_id, + match_index, + commit_index, + } => Self::LearnerNotCaughtUp { + group_id, + node_id, + match_index, + commit_index, + }, + ClusterError::MigrationInProgress { vshard_id } => { + Self::MigrationInProgress { vshard_id } + } + ClusterError::MigrationPauseBudgetExceeded { + estimated_us, + budget_us, + } => Self::MigrationPauseBudgetExceeded { + estimated_us, + budget_us, + }, + ClusterError::NodeUnreachable { node_id } => Self::NodeUnreachable { node_id }, + ClusterError::GhostNotFound { node_id, shard_id } => { + Self::GhostNotFound { node_id, shard_id } + } + ClusterError::Transport { detail } => Self::Transport { detail }, + ClusterError::ShardTimeout { + vshard_id, + elapsed_ms, + } => Self::ShardTimeout { + vshard_id, + elapsed_ms, + }, + ClusterError::StreamTerminal { error, detail } => Self::StreamTerminal { + error: *error, + detail, + }, + ClusterError::Storage { detail } => Self::Storage { detail }, + ClusterError::DataPlane { code } => Self::DataPlane { code }, + ClusterError::Codec { detail } => Self::Codec { detail }, + ClusterError::UnsupportedWireVersion { + got, + supported_min, + supported_max, + } => Self::UnsupportedWireVersion { + got, + supported_min, + supported_max, + }, + ClusterError::CircuitOpen { node_id, failures } => { + Self::CircuitOpen { node_id, failures } + } + ClusterError::JoinGroupDisappeared { group_id } => { + Self::JoinGroupDisappeared { group_id } + } + ClusterError::JoinCommitTimeout { + group_id, + log_index, + } => Self::JoinCommitTimeout { + group_id, + log_index, + }, + ClusterError::ReadIndexNotLeader { group_id } => Self::ReadIndexNotLeader { group_id }, + ClusterError::ReadIndexTimeout { + group_id, + waited_ms, + } => Self::ReadIndexTimeout { + group_id, + waited_ms, + }, + ClusterError::Config { detail } => Self::Config { detail }, + ClusterError::WrongOwner { + vshard_id, + expected_owner_node, + } => Self::WrongOwner { + vshard_id, + expected_owner_node, + }, + ClusterError::SnapshotCrcMismatch { + group_id, + stored, + computed, + } => Self::SnapshotCrcMismatch { + group_id, + stored, + computed, + }, + ClusterError::SnapshotOffsetRegression { + group_id, + expected, + actual, + } => Self::SnapshotOffsetRegression { + group_id, + expected, + actual, + }, + ClusterError::PartialSnapshotCorrupt { group_id, detail } => { + Self::PartialSnapshotCorrupt { group_id, detail } + } + ClusterError::PartialSnapshotCleanupFailed { group_id, detail } => { + Self::PartialSnapshotCleanupFailed { group_id, detail } + } + ClusterError::SnapshotApplyFailed { group_id, detail } => { + Self::SnapshotApplyFailed { group_id, detail } + } + ClusterError::RemoteUntyped { detail } => Self::Untyped { detail }, + // Coordinator-side error families. Their message crosses. + other @ (ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::Calvin(_) + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_)) => Self::Untyped { + detail: other.to_string(), + }, + } + } +} + +impl From for ClusterError { + fn from(wire: ShardErrorWire) -> Self { + match wire { + ShardErrorWire::Raft { error } => Self::Raft(error.into()), + ShardErrorWire::VShardNotMapped { vshard_id } => Self::VShardNotMapped { vshard_id }, + ShardErrorWire::GroupNotFound { group_id } => Self::GroupNotFound { group_id }, + ShardErrorWire::LearnerNotCaughtUp { + group_id, + node_id, + match_index, + commit_index, + } => Self::LearnerNotCaughtUp { + group_id, + node_id, + match_index, + commit_index, + }, + ShardErrorWire::MigrationInProgress { vshard_id } => { + Self::MigrationInProgress { vshard_id } + } + ShardErrorWire::MigrationPauseBudgetExceeded { + estimated_us, + budget_us, + } => Self::MigrationPauseBudgetExceeded { + estimated_us, + budget_us, + }, + ShardErrorWire::NodeUnreachable { node_id } => Self::NodeUnreachable { node_id }, + ShardErrorWire::GhostNotFound { node_id, shard_id } => { + Self::GhostNotFound { node_id, shard_id } + } + ShardErrorWire::Transport { detail } => Self::Transport { detail }, + ShardErrorWire::ShardTimeout { + vshard_id, + elapsed_ms, + } => Self::ShardTimeout { + vshard_id, + elapsed_ms, + }, + ShardErrorWire::StreamTerminal { error, detail } => Self::StreamTerminal { + error: Box::new(error), + detail, + }, + ShardErrorWire::Storage { detail } => Self::Storage { detail }, + ShardErrorWire::DataPlane { code } => Self::DataPlane { code }, + ShardErrorWire::Codec { detail } => Self::Codec { detail }, + ShardErrorWire::UnsupportedWireVersion { + got, + supported_min, + supported_max, + } => Self::UnsupportedWireVersion { + got, + supported_min, + supported_max, + }, + ShardErrorWire::CircuitOpen { node_id, failures } => { + Self::CircuitOpen { node_id, failures } + } + ShardErrorWire::JoinGroupDisappeared { group_id } => { + Self::JoinGroupDisappeared { group_id } + } + ShardErrorWire::JoinCommitTimeout { + group_id, + log_index, + } => Self::JoinCommitTimeout { + group_id, + log_index, + }, + ShardErrorWire::ReadIndexNotLeader { group_id } => { + Self::ReadIndexNotLeader { group_id } + } + ShardErrorWire::ReadIndexTimeout { + group_id, + waited_ms, + } => Self::ReadIndexTimeout { + group_id, + waited_ms, + }, + ShardErrorWire::Config { detail } => Self::Config { detail }, + ShardErrorWire::WrongOwner { + vshard_id, + expected_owner_node, + } => Self::WrongOwner { + vshard_id, + expected_owner_node, + }, + ShardErrorWire::SnapshotCrcMismatch { + group_id, + stored, + computed, + } => Self::SnapshotCrcMismatch { + group_id, + stored, + computed, + }, + ShardErrorWire::SnapshotOffsetRegression { + group_id, + expected, + actual, + } => Self::SnapshotOffsetRegression { + group_id, + expected, + actual, + }, + ShardErrorWire::PartialSnapshotCorrupt { group_id, detail } => { + Self::PartialSnapshotCorrupt { group_id, detail } + } + ShardErrorWire::PartialSnapshotCleanupFailed { group_id, detail } => { + Self::PartialSnapshotCleanupFailed { group_id, detail } + } + ShardErrorWire::SnapshotApplyFailed { group_id, detail } => { + Self::SnapshotApplyFailed { group_id, detail } + } + ShardErrorWire::Untyped { detail } => Self::RemoteUntyped { detail }, + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/shard_error/mod.rs b/nodedb-cluster/src/rpc_codec/shard_error/mod.rs new file mode 100644 index 000000000..31a3d7e1f --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/shard_error/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Wire mirror of the typed error a shard-side handler answers with. +//! +//! A `VShardEnvelope` handler that fails answers with a +//! [`VShardRefusal`](super::VShardRefusal) frame carrying this mirror. The +//! caller rebuilds the same `ClusterError`, so its retry and reroute logic +//! sees the shard's own error instead of a closed stream. + +pub mod convert; +pub mod raft; +pub mod wire; + +pub use raft::RaftErrorWire; +pub use wire::ShardErrorWire; diff --git a/nodedb-cluster/src/rpc_codec/shard_error/raft.rs b/nodedb-cluster/src/rpc_codec/shard_error/raft.rs new file mode 100644 index 000000000..f4d4512d6 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/shard_error/raft.rs @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Wire mirror of `nodedb_raft::RaftError`. Variant order is the wire ABI: +//! append only. + +use nodedb_raft::RaftError; + +/// A `RaftError` carried across a node hop. `NotLeader` keeps its leader +/// hint, so the caller can chase the redirect. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum RaftErrorWire { + NotLeader { + leader_hint: Option, + }, + LogCompacted { + requested: u64, + first_available: u64, + }, + CompactionAheadOfApplied { + requested: u64, + last_applied: u64, + }, + ProposalRejected { + reason: String, + }, + InvalidTransferTarget { + target: u64, + }, + LeadershipTransferInProgress, + GroupNotFound { + group_id: u64, + }, + Transport { + detail: String, + }, + Storage { + detail: String, + }, + Serialization { + detail: String, + }, + SnapshotFormat { + detail: String, + }, + Shutdown, +} + +impl From for RaftErrorWire { + fn from(error: RaftError) -> Self { + match error { + RaftError::NotLeader { leader_hint } => Self::NotLeader { leader_hint }, + RaftError::LogCompacted { + requested, + first_available, + } => Self::LogCompacted { + requested, + first_available, + }, + RaftError::CompactionAheadOfApplied { + requested, + last_applied, + } => Self::CompactionAheadOfApplied { + requested, + last_applied, + }, + RaftError::ProposalRejected { reason } => Self::ProposalRejected { reason }, + RaftError::InvalidTransferTarget { target } => Self::InvalidTransferTarget { target }, + RaftError::LeadershipTransferInProgress => Self::LeadershipTransferInProgress, + RaftError::GroupNotFound { group_id } => Self::GroupNotFound { group_id }, + RaftError::Transport { detail } => Self::Transport { detail }, + RaftError::Storage { detail } => Self::Storage { detail }, + RaftError::Serialization { detail } => Self::Serialization { detail }, + RaftError::SnapshotFormat { detail } => Self::SnapshotFormat { detail }, + RaftError::Shutdown => Self::Shutdown, + } + } +} + +impl From for RaftError { + fn from(wire: RaftErrorWire) -> Self { + match wire { + RaftErrorWire::NotLeader { leader_hint } => Self::NotLeader { leader_hint }, + RaftErrorWire::LogCompacted { + requested, + first_available, + } => Self::LogCompacted { + requested, + first_available, + }, + RaftErrorWire::CompactionAheadOfApplied { + requested, + last_applied, + } => Self::CompactionAheadOfApplied { + requested, + last_applied, + }, + RaftErrorWire::ProposalRejected { reason } => Self::ProposalRejected { reason }, + RaftErrorWire::InvalidTransferTarget { target } => { + Self::InvalidTransferTarget { target } + } + RaftErrorWire::LeadershipTransferInProgress => Self::LeadershipTransferInProgress, + RaftErrorWire::GroupNotFound { group_id } => Self::GroupNotFound { group_id }, + RaftErrorWire::Transport { detail } => Self::Transport { detail }, + RaftErrorWire::Storage { detail } => Self::Storage { detail }, + RaftErrorWire::Serialization { detail } => Self::Serialization { detail }, + RaftErrorWire::SnapshotFormat { detail } => Self::SnapshotFormat { detail }, + RaftErrorWire::Shutdown => Self::Shutdown, + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/shard_error/wire.rs b/nodedb-cluster/src/rpc_codec/shard_error/wire.rs new file mode 100644 index 000000000..6241c9064 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/shard_error/wire.rs @@ -0,0 +1,121 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The shard error wire enum. Variant order is the wire ABI: append only. + +use super::raft::RaftErrorWire; +use crate::rpc_codec::data_plane_error::DataPlaneErrorCode; +use crate::rpc_codec::execute::TypedClusterError; + +/// A `ClusterError` carried across a node hop. +/// +/// One variant per `ClusterError` variant whose fields cross the wire. The +/// coordinator-side error families (gather, barrier, mirror, Calvin, +/// migration) cross as [`Self::Untyped`] with their message. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum ShardErrorWire { + Raft { + error: RaftErrorWire, + }, + VShardNotMapped { + vshard_id: u32, + }, + GroupNotFound { + group_id: u64, + }, + LearnerNotCaughtUp { + group_id: u64, + node_id: u64, + match_index: u64, + commit_index: u64, + }, + MigrationInProgress { + vshard_id: u32, + }, + MigrationPauseBudgetExceeded { + estimated_us: u64, + budget_us: u64, + }, + NodeUnreachable { + node_id: u64, + }, + GhostNotFound { + node_id: String, + shard_id: u32, + }, + Transport { + detail: String, + }, + ShardTimeout { + vshard_id: u32, + elapsed_ms: u64, + }, + StreamTerminal { + error: TypedClusterError, + detail: String, + }, + Storage { + detail: String, + }, + DataPlane { + code: DataPlaneErrorCode, + }, + Codec { + detail: String, + }, + UnsupportedWireVersion { + got: u8, + supported_min: u8, + supported_max: u8, + }, + CircuitOpen { + node_id: u64, + failures: u32, + }, + JoinGroupDisappeared { + group_id: u64, + }, + JoinCommitTimeout { + group_id: u64, + log_index: u64, + }, + ReadIndexNotLeader { + group_id: u64, + }, + ReadIndexTimeout { + group_id: u64, + waited_ms: u64, + }, + Config { + detail: String, + }, + WrongOwner { + vshard_id: u32, + expected_owner_node: Option, + }, + SnapshotCrcMismatch { + group_id: u64, + stored: u32, + computed: u32, + }, + SnapshotOffsetRegression { + group_id: u64, + expected: u64, + actual: u64, + }, + PartialSnapshotCorrupt { + group_id: u64, + detail: String, + }, + PartialSnapshotCleanupFailed { + group_id: u64, + detail: String, + }, + SnapshotApplyFailed { + group_id: u64, + detail: String, + }, + /// An error whose type has no wire mirror. `detail` is its message. + Untyped { + detail: String, + }, +} diff --git a/nodedb-cluster/src/rpc_codec/vshard.rs b/nodedb-cluster/src/rpc_codec/vshard.rs index ba249c5f2..4c03a0d1e 100644 --- a/nodedb-cluster/src/rpc_codec/vshard.rs +++ b/nodedb-cluster/src/rpc_codec/vshard.rs @@ -7,19 +7,28 @@ //! the handler. The envelope bytes are passed through raw (already serialized //! in their own binary format). //! -//! A handler that refuses with a typed Data-Plane verdict answers with a -//! [`VShardRefusal`] frame instead of a response envelope. +//! A handler that fails answers with a [`VShardRefusal`] frame instead of a +//! response envelope. The frame carries the handler's typed error. -use super::data_plane_error::DataPlaneErrorCode; use super::discriminants::{RPC_VSHARD_ENVELOPE, RPC_VSHARD_REFUSAL}; use super::header::write_frame; use super::raft_rpc::RaftRpc; +use super::shard_error::ShardErrorWire; use crate::error::{ClusterError, Result}; -/// A typed Data-Plane verdict that answers a VShardEnvelope request. -#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +/// The typed error that answers a VShardEnvelope request. The caller +/// rebuilds it with `ClusterError::from`. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub struct VShardRefusal { - pub code: DataPlaneErrorCode, + pub error: ShardErrorWire, +} + +impl From for VShardRefusal { + fn from(error: ClusterError) -> Self { + Self { + error: error.into(), + } + } } pub(super) fn encode_vshard_envelope(bytes: &[u8], out: &mut Vec) -> Result<()> { @@ -54,19 +63,85 @@ pub(super) fn decode_vshard_refusal(payload: &[u8]) -> Result { mod tests { use super::*; use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::DataPlaneErrorCode; use crate::rpc_codec::{decode, encode}; - #[test] - fn a_refusal_survives_the_wire() { - let code = DataPlaneErrorCode::Unsupported { - detail: "not on this engine".into(), - }; + /// Encode a shard error as a refusal frame, decode it, and rebuild it. + fn round_trip(error: ClusterError) -> ClusterError { let epoch = ClusterEpochState::default(); - let rpc = RaftRpc::VShardRefusal(VShardRefusal { code: code.clone() }); + let rpc = RaftRpc::VShardRefusal(VShardRefusal::from(error)); let encoded = encode(&rpc, &epoch).expect("encode"); match decode(&encoded, &epoch).expect("decode") { - RaftRpc::VShardRefusal(refusal) => assert_eq!(refusal.code, code), + RaftRpc::VShardRefusal(refusal) => ClusterError::from(refusal.error), other => panic!("decoded the wrong variant: {other:?}"), } } + + #[test] + fn a_data_plane_refusal_survives_the_wire() { + let code = DataPlaneErrorCode::Unsupported { + detail: "not on this engine".into(), + }; + match round_trip(ClusterError::DataPlane { code: code.clone() }) { + ClusterError::DataPlane { code: rebuilt } => assert_eq!(rebuilt, code), + other => panic!("expected the typed refusal, got {other:?}"), + } + } + + /// `WrongOwner` crosses typed, so the coordinator's reroute retry sees it. + #[test] + fn wrong_owner_survives_the_wire() { + let error = ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: Some(3), + }; + assert!(matches!( + round_trip(error), + ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: Some(3) + } + )); + } + + #[test] + fn a_raft_redirect_keeps_its_leader_hint() { + let error = ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: Some(5), + }); + assert!(matches!( + round_trip(error), + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint: Some(5) + }) + )); + } + + #[test] + fn a_codec_error_survives_the_wire() { + let error = ClusterError::Codec { + detail: "bad request body".into(), + }; + match round_trip(error) { + ClusterError::Codec { detail } => assert_eq!(detail, "bad request body"), + other => panic!("expected the codec error, got {other:?}"), + } + } + + /// An error with no wire mirror keeps its message. + #[test] + fn an_untyped_error_keeps_its_message() { + let error = + ClusterError::BspBarrier(crate::distributed_graph::BspBarrierError::Incomplete { + algorithm: "pagerank".into(), + iteration: 3, + acked: 1, + expected: 2, + }); + let message = error.to_string(); + match round_trip(error) { + ClusterError::RemoteUntyped { detail } => assert_eq!(detail, message), + other => panic!("expected the untyped error, got {other:?}"), + } + } } From 88069c9ad400cb74b5f60a6b09f9676707573cd3 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Sun, 27 Sep 2026 21:55:23 +0800 Subject: [PATCH 54/64] feat(errors): unify HTTP status mapping and widen typed error crossing Route every gateway error's HTTP status through the pgwire SQLSTATE table (GatewayErrorMap::to_http / sqlstate_to_http), so HTTP, native, and pgwire answer one class for one error; drop the now-redundant remote-code-to-HTTP table. ApiError::from derives its status the same way, and shape_error_to_api and the '3D000' database-not-found case follow the same table instead of their own if/else. Array cluster execution rebuilds a shard's typed error instead of only its Data-Plane code: WrongOwner reroutes to NotLeader/NoLeader, a local dispatch or channel timeout crosses as the typed DeadlineExceeded verdict, and a cross-shard write reports its target's refusal reason instead of a generic 'unexpected RPC response type'. A commit abort on a dispatch or DDL-propose error keeps its own class via the new SystemTxnError::CommitFailed, and a Calvin cancel/timeout now renders the deadline class instead of a bare internal error. The rate-gate DDL functions propagate a dispatch error or coded refusal with its own class instead of silently reading it as an absent key or zero usage, and add BACKUP_TENANT_MISMATCH / BACKUP_KEY_MISMATCH to the numeric-code-to-SQLSTATE table. --- .../control/array_sync/raft_apply/common.rs | 68 ++- .../cluster/array_cluster_exec/dispatch.rs | 7 +- .../control/cluster/array_cluster_helpers.rs | 34 ++ .../cluster/array_executor/executor.rs | 60 ++- .../control/cluster/array_executor/refusal.rs | 63 ++- nodedb/src/control/event_action_error.rs | 37 +- .../control/gateway/error_map/class_parity.rs | 490 ++++++++++++++++++ nodedb/src/control/gateway/error_map/http.rs | 102 ++-- .../control/gateway/error_map/remote_code.rs | 79 +-- .../gateway/error_map/sqlstate_status.rs | 4 +- nodedb/src/control/server/http/auth.rs | 105 +++- nodedb/src/control/server/http/routes/crdt.rs | 2 +- .../src/control/server/http/routes/query.rs | 10 +- .../http/routes/query/materialized/encode.rs | 5 +- .../server/http/routes/result_shape.rs | 27 +- .../control/server/pgwire/types/error_map.rs | 4 + .../server/shared/ddl/neutral/rate_gate.rs | 267 ++++++++-- nodedb/src/control/system_txn/run.rs | 44 +- nodedb/src/event/cross_shard/dispatcher.rs | 20 +- 19 files changed, 1156 insertions(+), 272 deletions(-) diff --git a/nodedb/src/control/array_sync/raft_apply/common.rs b/nodedb/src/control/array_sync/raft_apply/common.rs index e309c1e25..30e7be3d9 100644 --- a/nodedb/src/control/array_sync/raft_apply/common.rs +++ b/nodedb/src/control/array_sync/raft_apply/common.rs @@ -17,7 +17,7 @@ use crate::control::server::dispatch_utils::{ ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; use crate::control::state::SharedState; -use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; /// Identifies a committed Raft entry within the apply loop. /// @@ -235,15 +235,16 @@ pub(super) async fn ensure_array_open( Err(poisoned) => poisoned.into_inner().dispatch(open_request), }; - if let Err(e) = dispatch_result { - return Err(crate::Error::Internal { - detail: format!("ensure_array_open: dispatch failed: {e}"), - }); - } + // A dispatch refusal, such as a capacity limit, keeps its own class. + dispatch_result?; - await_data_plane(async move { open_rx.recv().await.ok_or(()) }, "OpenArray") - .await - .map(|_| ()) + await_data_plane( + async move { open_rx.recv().await.ok_or(()) }, + open_request_id, + "OpenArray", + ) + .await + .map(|_| ()) } /// Build a `Request` for an array apply/open with default deadline / priority. @@ -282,22 +283,24 @@ pub(super) fn build_array_request( } } -/// Await a Data Plane response. An error status becomes [`apply_refusal`]. -/// A timeout or a closed channel becomes `crate::Error::Internal` with a -/// contextual `op_label`. +/// How long [`await_data_plane`] waits for the Data Plane's response. +const DATA_PLANE_AWAIT_TIMEOUT: Duration = Duration::from_secs(30); + +/// Await the Data Plane response to request `request_id`. An error status +/// becomes [`apply_refusal`]. A timeout is `crate::Error::DeadlineExceeded` +/// (`57014`). A closed channel is `crate::Error::Internal` with `op_label`. pub(super) async fn await_data_plane( rx: impl std::future::Future>, + request_id: RequestId, op_label: &str, ) -> ProposeResult { - match tokio::time::timeout(Duration::from_secs(30), rx).await { + match tokio::time::timeout(DATA_PLANE_AWAIT_TIMEOUT, rx).await { Ok(Ok(resp)) if resp.status == Status::Ok => Ok(AppliedWrite::from_response(&resp)), Ok(Ok(resp)) => Err(apply_refusal(op_label, &resp)), Ok(Err(_)) => Err(crate::Error::Internal { detail: format!("{op_label}: response channel closed"), }), - Err(_) => Err(crate::Error::Internal { - detail: format!("{op_label}: deadline exceeded"), - }), + Err(_) => Err(crate::Error::DeadlineExceeded { request_id }), } } @@ -318,7 +321,7 @@ pub(super) fn apply_refusal(op_label: &str, response: &Response) -> crate::Error mod tests { use super::*; use crate::bridge::envelope::{ErrorCode, Payload}; - use crate::types::{Lsn, RequestId}; + use crate::types::Lsn; fn refusal(code: Option) -> Response { Response { @@ -365,10 +368,39 @@ mod tests { detail: "not on this engine".into(), }; let response = refusal(Some(code.clone())); - let result = await_data_plane(async move { Ok::<_, ()>(response) }, "OpenArray").await; + let result = await_data_plane( + async move { Ok::<_, ()>(response) }, + RequestId::new(1), + "OpenArray", + ) + .await; match result { Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), other => panic!("expected the typed refusal, got {other:?}"), } } + + /// A local timeout is the typed deadline error, `57014` on pgwire. + #[tokio::test(start_paused = true)] + async fn a_local_timeout_is_a_typed_deadline() { + let result = await_data_plane( + std::future::pending::>(), + RequestId::new(7), + "OpenArray", + ) + .await; + let error = match result { + Err(error) => error, + Ok(_) => panic!("a pending response must time out"), + }; + assert!( + matches!( + &error, + crate::Error::DeadlineExceeded { request_id } if *request_id == RequestId::new(7) + ), + "expected a typed deadline, got {error:?}" + ); + let (_, state, _) = crate::control::server::pgwire::types::error_to_sqlstate(&error); + assert_eq!(state, nodedb_types::error::sqlstate::QUERY_CANCELED); + } } diff --git a/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs b/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs index ba84f2ffe..2efa24a78 100644 --- a/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs +++ b/nodedb/src/control/cluster/array_cluster_exec/dispatch.rs @@ -145,10 +145,11 @@ impl NexarArrayDispatch { detail: "array shard response: failed to decode VShardEnvelope".into(), } }), - // The shard's Data Plane refused with a typed verdict. It keeps - // its code, the same error the local short-circuit returns. + // The shard's handler failed with a typed error. It is rebuilt as + // the error the local short-circuit returns, so `WrongOwner` + // reaches the fan-out's reroute retry. RaftRpc::VShardRefusal(refusal) => { - Err(nodedb_cluster::error::ClusterError::DataPlane { code: refusal.code }) + Err(nodedb_cluster::error::ClusterError::from(refusal.error)) } other => Err(nodedb_cluster::error::ClusterError::Transport { detail: format!( diff --git a/nodedb/src/control/cluster/array_cluster_helpers.rs b/nodedb/src/control/cluster/array_cluster_helpers.rs index 80c83f233..a982fd85d 100644 --- a/nodedb/src/control/cluster/array_cluster_helpers.rs +++ b/nodedb/src/control/cluster/array_cluster_helpers.rs @@ -105,6 +105,21 @@ pub(super) fn cluster_err(e: nodedb_cluster::error::ClusterError) -> Error { // A shard's Data-Plane verdict keeps its code, so the statement // renders the SQLSTATE a single-node execution renders. nodedb_cluster::error::ClusterError::DataPlane { code } => Error::DataPlane(code.into()), + // The shard still refused after the fan-out's reroute retry. The + // vShard's owner is moving, so the client retries the statement. + nodedb_cluster::error::ClusterError::WrongOwner { + vshard_id, + expected_owner_node, + } => match expected_owner_node { + Some(leader_node) => Error::NotLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + leader_node, + leader_addr: String::new(), + }, + None => Error::NoLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + }, + }, other => Error::Internal { detail: format!("array cluster: {other}"), }, @@ -153,4 +168,23 @@ mod tests { other => panic!("expected the typed verdict, got {other:?}"), } } + + /// A shard that still refused after the reroute retry answers the + /// retryable leader class, never `Internal`. + #[test] + fn a_persistent_wrong_owner_is_a_leader_error() { + let known = nodedb_cluster::error::ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: Some(3), + }; + assert!(matches!( + cluster_err(known), + Error::NotLeader { leader_node: 3, .. } + )); + let unknown = nodedb_cluster::error::ClusterError::WrongOwner { + vshard_id: 7, + expected_owner_node: None, + }; + assert!(matches!(cluster_err(unknown), Error::NoLeader { .. })); + } } diff --git a/nodedb/src/control/cluster/array_executor/executor.rs b/nodedb/src/control/cluster/array_executor/executor.rs index 46d0a381b..4ea186d7c 100644 --- a/nodedb/src/control/cluster/array_executor/executor.rs +++ b/nodedb/src/control/cluster/array_executor/executor.rs @@ -7,8 +7,10 @@ use std::time::{Duration, Instant}; use nodedb_array::types::ArrayId; use nodedb_cluster::error::{ClusterError, Result}; +use nodedb_cluster::rpc_codec::DataPlaneErrorCode; -use crate::bridge::envelope::{Priority, Request}; +use super::refusal::execution_error; +use crate::bridge::envelope::{Priority, Request, Response}; use crate::control::state::SharedState; use crate::event::types::EventSource; use crate::types::{ReadConsistency, RequestId, TraceId, TxnId, VShardId}; @@ -47,7 +49,7 @@ impl DataPlaneArrayExecutor { local_vshard_id: VShardId, plan: PhysicalPlan, txn_id: Option, - ) -> Result { + ) -> Result { let request_id = self.state.next_request_id(); let request = local_request(request_id, array_id, local_vshard_id, plan, txn_id); @@ -58,23 +60,28 @@ impl DataPlaneArrayExecutor { Err(poisoned) => poisoned.into_inner().dispatch(request), }; + // A dispatch refusal, such as a capacity limit, keeps its own class. if let Err(e) = dispatch_result { - return Err(ClusterError::Storage { - detail: format!("array executor dispatch: {e}"), - }); + return Err(execution_error("array executor dispatch", e)); } - match tokio::time::timeout(LOCAL_DISPATCH_TIMEOUT, async { rx.recv().await.ok_or(()) }) - .await - { - Ok(Ok(resp)) => Ok(resp), - Ok(Err(_)) => Err(ClusterError::Storage { - detail: "array executor: response channel closed".into(), - }), - Err(_) => Err(ClusterError::Storage { - detail: "array executor: local dispatch timed out".into(), - }), - } + await_local_response(rx.recv()).await + } +} + +/// Await the local Data Plane's response. A timeout is the typed +/// `DeadlineExceeded` verdict, which the coordinator renders as `57014`. +async fn await_local_response( + rx: impl std::future::Future>, +) -> Result { + match tokio::time::timeout(LOCAL_DISPATCH_TIMEOUT, rx).await { + Ok(Some(resp)) => Ok(resp), + Ok(None) => Err(ClusterError::Storage { + detail: "array executor: response channel closed".into(), + }), + Err(_) => Err(ClusterError::DataPlane { + code: DataPlaneErrorCode::DeadlineExceeded, + }), } } @@ -152,6 +159,27 @@ mod tests { assert_eq!(request.vshard_id, vshard_id); } + /// A local timeout crosses as the typed deadline verdict, and the + /// coordinator renders it as `57014`. + #[tokio::test(start_paused = true)] + async fn a_local_timeout_is_a_typed_deadline() { + let error = await_local_response(std::future::pending::>()) + .await + .expect_err("a pending response must time out"); + assert!( + matches!( + &error, + ClusterError::DataPlane { + code: DataPlaneErrorCode::DeadlineExceeded + } + ), + "expected a typed deadline, got {error:?}" + ); + let rebuilt = crate::control::cluster::array_cluster_helpers::cluster_err(error); + let (_, state, _) = crate::control::server::pgwire::types::error_to_sqlstate(&rebuilt); + assert_eq!(state, nodedb_types::error::sqlstate::QUERY_CANCELED); + } + #[test] fn nonzero_vshard_is_preserved_for_read_and_write_requests() { let array_id = ArrayId::new(TenantId::new(41), "measurements"); diff --git a/nodedb/src/control/cluster/array_executor/refusal.rs b/nodedb/src/control/cluster/array_executor/refusal.rs index 1844772c5..027616950 100644 --- a/nodedb/src/control/cluster/array_executor/refusal.rs +++ b/nodedb/src/control/cluster/array_executor/refusal.rs @@ -5,9 +5,11 @@ //! A coded refusal crosses as `ClusterError::DataPlane`, so the coordinator //! rebuilds `crate::Error::DataPlane(code)` and renders the SQLSTATE a //! single-node execution renders. Only a refusal with no code is a storage -//! error. +//! error. A local-execution error keeps its class where the cluster wire has +//! one. use nodedb_cluster::error::ClusterError; +use nodedb_cluster::rpc_codec::DataPlaneErrorCode; use crate::bridge::envelope::Response; @@ -23,12 +25,36 @@ pub(super) fn refusal_error(context: &str, response: &Response) -> ClusterError } } -/// The cluster error for a local-execution error. A Data-Plane verdict keeps -/// its code. Every other error is a storage error with `context` before its -/// message. +/// The cluster error for a local-execution error. +/// +/// - A Data-Plane verdict keeps its code. +/// - A deadline and a capacity refusal cross as their Data-Plane verdicts. +/// - A missing leader crosses as `WrongOwner`, so the coordinator re-reads +/// its routing and retries. +/// - Every other error is a storage error with `context` before its message. pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterError { match error { crate::Error::DataPlane(code) => ClusterError::DataPlane { code: code.into() }, + crate::Error::DeadlineExceeded { .. } => ClusterError::DataPlane { + code: DataPlaneErrorCode::DeadlineExceeded, + }, + capacity @ crate::Error::DispatchCapacity { .. } => ClusterError::DataPlane { + code: DataPlaneErrorCode::DispatchCapacity { + reason: capacity.to_string(), + }, + }, + crate::Error::NotLeader { + vshard_id, + leader_node, + .. + } => ClusterError::WrongOwner { + vshard_id: vshard_id.as_u32(), + expected_owner_node: (leader_node != 0).then_some(leader_node), + }, + crate::Error::NoLeader { vshard_id } => ClusterError::WrongOwner { + vshard_id: vshard_id.as_u32(), + expected_owner_node: None, + }, other => ClusterError::Storage { detail: format!("{context}: {other}"), }, @@ -82,6 +108,35 @@ mod tests { } } + #[test] + fn a_local_deadline_crosses_as_the_deadline_verdict() { + let error = crate::Error::DeadlineExceeded { + request_id: RequestId::new(1), + }; + assert!(matches!( + execution_error("array put", error), + ClusterError::DataPlane { + code: DataPlaneErrorCode::DeadlineExceeded + } + )); + } + + #[test] + fn a_missing_leader_crosses_as_wrong_owner() { + let error = crate::Error::NotLeader { + vshard_id: crate::types::VShardId::new(9), + leader_node: 4, + leader_addr: "10.0.0.4:9000".into(), + }; + assert!(matches!( + execution_error("array put raft propose", error), + ClusterError::WrongOwner { + vshard_id: 9, + expected_owner_node: Some(4) + } + )); + } + #[test] fn an_execution_verdict_keeps_its_code() { let error = crate::Error::DataPlane(unsupported()); diff --git a/nodedb/src/control/event_action_error.rs b/nodedb/src/control/event_action_error.rs index 943e1b81a..7e801f648 100644 --- a/nodedb/src/control/event_action_error.rs +++ b/nodedb/src/control/event_action_error.rs @@ -75,9 +75,9 @@ impl From for crate::Error { TriggerActionError::Plan { source } | TriggerActionError::LeaseAdmission { source } => { source } - TriggerActionError::Transaction { source } => crate::Error::Internal { - detail: source.to_string(), - }, + // The transaction error keeps its class: its statement or commit + // error, or the Data-Plane verdict that aborted the commit. + TriggerActionError::Transaction { source } => source.into(), } } } @@ -164,6 +164,37 @@ mod tests { } } + /// A commit abort keeps the Data-Plane verdict that decided it. + #[test] + fn a_failed_transaction_keeps_its_abort_code() { + let error = TriggerActionError::Transaction { + source: SystemTxnError::Commit { + detail: "serialization failure against a concurrent write".to_owned(), + code: Some(Box::new(crate::bridge::envelope::ErrorCode::ConflictRetry)), + }, + }; + match crate::Error::from(error) { + crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::ConflictRetry) => {} + other => panic!("expected the abort code to survive, got {other:?}"), + } + } + + /// A commit that failed to dispatch keeps the dispatch error. + #[test] + fn a_failed_commit_dispatch_keeps_its_error() { + let error = TriggerActionError::Transaction { + source: SystemTxnError::CommitFailed { + source: crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(1), + }, + }, + }; + assert!(matches!( + crate::Error::from(error), + crate::Error::DeadlineExceeded { .. } + )); + } + #[test] fn a_render_failure_reports_as_a_bad_request() { let error = TriggerActionError::Rejected { diff --git a/nodedb/src/control/gateway/error_map/class_parity.rs b/nodedb/src/control/gateway/error_map/class_parity.rs index d11d0ff20..3c73f3621 100644 --- a/nodedb/src/control/gateway/error_map/class_parity.rs +++ b/nodedb/src/control/gateway/error_map/class_parity.rs @@ -375,3 +375,493 @@ fn expired_session_token_is_invalid_authorization_everywhere() { status ); } + +/// The number of `crate::Error` variants [`error_variant_index`] numbers. +const ERROR_VARIANT_COUNT: usize = 108; + +/// A dense index per `crate::Error` variant. Exhaustive, so a new variant +/// fails to compile here until it gets an index, and +/// [`every_error_variant_has_a_sample`] then fails until +/// [`error_samples`] carries it. +fn error_variant_index(err: &crate::Error) -> usize { + use crate::Error as E; + match err { + E::RejectedConstraint { .. } => 0, + E::TxnOverlayMemoryExceeded { .. } => 1, + E::RejectedAuthz { .. } => 2, + E::OffsetRegression { .. } => 3, + E::DeadlineExceeded { .. } => 4, + E::ConflictRetry { .. } => 5, + E::CalvinSerializationConflict => 6, + E::CalvinParticipantError => 7, + E::RejectedPrevalidation { .. } => 8, + E::RetryableRefusal { .. } => 9, + E::AppendOnlyViolation { .. } => 10, + E::BalanceViolation { .. } => 11, + E::MaterializedSumTargetNotFound { .. } => 12, + E::MaterializedSumResolutionMissing { .. } => 13, + E::PeriodLocked { .. } => 14, + E::PeriodLockMisconfigured { .. } => 15, + E::RetentionViolation { .. } => 16, + E::LegalHoldActive { .. } => 17, + E::StateTransitionViolation { .. } => 18, + E::TransitionCheckViolation { .. } => 19, + E::TypeGuardViolation { .. } => 20, + E::TypeMismatch { .. } => 21, + E::InsufficientBalance { .. } => 22, + E::RateExceeded { .. } => 23, + E::CollectionNotFound { .. } => 24, + E::DocumentNotFound { .. } => 25, + E::CollectionDeactivated { .. } => 26, + E::VShardAdmissionCapacityExceeded { .. } => 27, + E::CrdtAdmissionRetriesExhausted { .. } => 28, + E::CrdtAdmissionInvalidPlan { .. } => 29, + E::CrdtAdmissionCallerFence => 30, + E::CrdtApplyRequiresAdmission => 31, + E::CrdtApplyForbiddenInTransaction => 32, + E::NotInTransactionBlock { .. } => 33, + E::CrdtAdmissionTimeout { .. } => 34, + E::NoLeader { .. } => 35, + E::NotLeader { .. } => 36, + E::FanOutExceeded { .. } => 37, + E::CrossCollectionNotColocated { .. } => 38, + E::SourceFrozen { .. } => 39, + E::CloneWriteRequiresMaterialize { .. } => 40, + E::BadRequest { .. } => 41, + E::BackupTenantMismatch { .. } => 42, + E::BackupKeyMismatch => 43, + E::QuotaOvercommit { .. } => 44, + E::PlanError { .. } => 45, + E::FeatureNotSupported { .. } => 46, + E::UndefinedFunction { .. } => 47, + E::UndefinedObject { .. } => 48, + E::ObjectNotInPrerequisiteState { .. } => 49, + E::UndefinedColumn { .. } => 50, + E::AmbiguousColumn { .. } => 51, + E::UnknownStrictField { .. } => 52, + E::DivisionByZero => 53, + E::DataException { .. } => 54, + E::InvalidLimitValue { .. } => 55, + E::RetryableSchemaChanged { .. } => 56, + E::RetryableLeaderChange { .. } => 57, + E::GroupQuorumUnavailable { .. } => 58, + E::GroupMarksUnavailable { .. } => 59, + E::MetadataLeaderUnavailable => 60, + E::AuthorizationStateBehind { .. } => 61, + E::ExecutionLimitExceeded { .. } => 62, + E::LimitExceeded { .. } => 63, + E::Wal(_) => 64, + E::Dispatch { .. } => 65, + E::DispatchCapacity { .. } => 66, + E::Storage { .. } => 67, + E::ColdStorage { .. } => 68, + E::Serialization { .. } => 69, + E::Codec { .. } => 70, + E::SegmentCorrupted { .. } => 71, + E::MemoryExhausted { .. } => 72, + E::Backpressure { .. } => 73, + E::Crdt(_) => 74, + E::Io(_) => 75, + E::Config { .. } => 76, + E::Encryption { .. } => 77, + E::Bridge { .. } => 78, + E::VersionCompat { .. } => 79, + E::Internal { .. } => 80, + E::Shaping(_) => 81, + E::RemoteTyped { .. } => 82, + E::DescriptorVersionAnomaly { .. } => 83, + E::CollectionPurgeRowMissing { .. } => 84, + E::CatalogIntegrityViolation { .. } => 85, + E::DataPlane(_) => 86, + E::Promql(_) => 87, + E::DependentObjectsExist { .. } => 88, + E::CascadeCycle { .. } => 89, + E::CrossShardInExplicitTransaction => 90, + E::SequencerUnavailable => 91, + E::SessionCapExceeded { .. } => 92, + E::SessionIdleTimeout => 93, + E::SessionTokenExpired => 94, + E::SessionKilledByAdmin => 95, + E::SessionUserDropped => 96, + E::OidcProviderTenantUnbound => 97, + E::OidcProviderTenantUnavailable { .. } => 98, + E::ExternalRoleUndefined { .. } => 99, + E::OidcNoDefaultDatabase { .. } => 100, + E::TenantVectorDimExceeded { .. } => 101, + E::TenantGraphDepthExceeded { .. } => 102, + E::RoleInheritanceCycle { .. } => 103, + E::RoleInheritanceDepthExceeded { .. } => 104, + E::OllpExhausted { .. } => 105, + E::MirrorReadOnly { .. } => 106, + E::StaleReadNotLeader { .. } => 107, + } +} + +/// One sample per `crate::Error` variant. +fn error_samples() -> Vec { + use crate::Error as E; + use crate::types::{DatabaseId, RequestId, TenantId, VShardId}; + + let text = || "detail".to_owned(); + let collection = || "c".to_owned(); + vec![ + E::RejectedConstraint { + collection: collection(), + constraint: "unique".into(), + detail: text(), + }, + E::TxnOverlayMemoryExceeded { limit: 1 << 20 }, + E::RejectedAuthz { + tenant_id: TenantId::new(1), + resource: text(), + }, + E::OffsetRegression { + stream: "s".into(), + group: "g".into(), + partition_id: 0, + current_lsn: 2, + current_sequence: 2, + attempted_lsn: 1, + attempted_sequence: 1, + }, + E::DeadlineExceeded { + request_id: RequestId::new(1), + }, + E::ConflictRetry { + collection: collection(), + document_id: "d".into(), + }, + E::CalvinSerializationConflict, + E::CalvinParticipantError, + E::RejectedPrevalidation { + constraint: "check".into(), + reason: text(), + }, + E::RetryableRefusal { reason: text() }, + E::AppendOnlyViolation { + collection: collection(), + detail: text(), + }, + E::BalanceViolation { + collection: collection(), + detail: text(), + }, + E::MaterializedSumTargetNotFound { + target_collection: "t".into(), + join_column: "k".into(), + join_value: "1".into(), + }, + E::MaterializedSumResolutionMissing { + target_collection: "t".into(), + join_column: "k".into(), + join_value: "1".into(), + }, + E::PeriodLocked { + collection: collection(), + detail: text(), + }, + E::PeriodLockMisconfigured { + collection: collection(), + ref_table: "periods".into(), + status_column: "status".into(), + row_identity: "p1".into(), + }, + E::RetentionViolation { + collection: collection(), + detail: text(), + }, + E::LegalHoldActive { + collection: collection(), + detail: text(), + }, + E::StateTransitionViolation { + collection: collection(), + detail: text(), + }, + E::TransitionCheckViolation { + collection: collection(), + detail: text(), + }, + E::TypeGuardViolation { + collection: collection(), + detail: text(), + }, + E::TypeMismatch { + collection: collection(), + key: "k".into(), + detail: text(), + }, + E::InsufficientBalance { + collection: collection(), + key: "k".into(), + detail: text(), + }, + E::RateExceeded { + gate: "g".into(), + detail: text(), + retry_after_ms: 10, + }, + E::CollectionNotFound { + tenant_id: TenantId::new(1), + collection: collection(), + }, + E::DocumentNotFound { + collection: collection(), + document_id: "d".into(), + }, + E::CollectionDeactivated { + tenant_id: TenantId::new(1), + collection: collection(), + retention_expires_at_ns: 1, + }, + E::VShardAdmissionCapacityExceeded { + vshard_id: VShardId::new(1), + capacity: 4, + }, + E::CrdtAdmissionRetriesExhausted { + vshard_id: VShardId::new(1), + attempts: 3, + }, + E::CrdtAdmissionInvalidPlan { reason: "empty" }, + E::CrdtAdmissionCallerFence, + E::CrdtApplyRequiresAdmission, + E::CrdtApplyForbiddenInTransaction, + E::NotInTransactionBlock { + statement: "VACUUM".into(), + }, + E::CrdtAdmissionTimeout { + vshard_id: VShardId::new(1), + timeout_ms: 10, + }, + E::NoLeader { + vshard_id: VShardId::new(1), + }, + E::NotLeader { + vshard_id: VShardId::new(1), + leader_node: 2, + leader_addr: "10.0.0.1:9000".into(), + }, + E::FanOutExceeded { + shards_touched: 9, + limit: 8, + }, + E::CrossCollectionNotColocated { + op: "insert-select", + source_collection: "a".into(), + target_collection: "b".into(), + }, + E::SourceFrozen { + database_id: DatabaseId::new(7), + }, + E::CloneWriteRequiresMaterialize { + collection: collection(), + engine: "kv".into(), + database: "db".into(), + reason: "shadowed", + }, + E::BadRequest { detail: text() }, + E::BackupTenantMismatch { + expected: 1, + actual: 2, + }, + E::BackupKeyMismatch, + E::QuotaOvercommit { + field: "max_storage".into(), + detail: text(), + }, + E::PlanError { detail: text() }, + E::FeatureNotSupported { detail: text() }, + E::UndefinedFunction { name: "f".into() }, + E::UndefinedObject { + kind: "sequence", + name: "s".into(), + }, + E::ObjectNotInPrerequisiteState { + object: "s".into(), + detail: text(), + }, + E::UndefinedColumn { column: "x".into() }, + E::AmbiguousColumn { + column: "id".into(), + }, + E::UnknownStrictField { + collection: collection(), + column: "x".into(), + }, + E::DivisionByZero, + E::DataException { detail: text() }, + E::InvalidLimitValue { + clause: "LIMIT", + value: "-1".into(), + }, + E::RetryableSchemaChanged { + descriptor: "orders".into(), + }, + E::RetryableLeaderChange { + group_id: 1, + log_index: 2, + }, + E::GroupQuorumUnavailable { + group_id: 1, + voters: vec![1, 2, 3], + unreachable: vec![2, 3], + }, + E::GroupMarksUnavailable { + group_id: 1, + refused_by: vec![2], + }, + E::MetadataLeaderUnavailable, + E::AuthorizationStateBehind { detail: text() }, + E::ExecutionLimitExceeded { detail: text() }, + E::LimitExceeded { + limit_name: "max_rows", + value: 10, + max: 5, + }, + E::Wal(nodedb_wal::WalError::Sealed), + E::Dispatch { detail: text() }, + E::DispatchCapacity { + scope: crate::DispatchCapacityScope::QueueFull { + core_id: 0, + capacity: 4, + }, + }, + E::Storage { + engine: "kv".into(), + detail: text(), + }, + E::ColdStorage { detail: text() }, + E::Serialization { + format: "msgpack".into(), + detail: text(), + }, + E::Codec { detail: text() }, + E::SegmentCorrupted { detail: text() }, + E::MemoryExhausted { + engine: "kv".into(), + }, + E::Backpressure { + engine: nodedb_mem::EngineId::Vector, + }, + E::Crdt(nodedb_crdt::CrdtError::ConstraintViolation { + constraint: "unique".into(), + collection: collection(), + detail: text(), + }), + E::Io(std::io::Error::other("disk")), + E::Config { detail: text() }, + E::Encryption { detail: text() }, + E::Bridge { detail: text() }, + E::VersionCompat { detail: text() }, + E::Internal { detail: text() }, + E::Shaping(Box::new(nodedb_types::NodeDbError::bad_request(text()))), + E::RemoteTyped { + code: nodedb_types::error::ErrorCode::WRITE_CONFLICT, + message: text(), + }, + E::DescriptorVersionAnomaly { + descriptor: "orders".into(), + carried: 5, + prior: 2, + }, + E::CollectionPurgeRowMissing { + database_id: 1, + tenant_id: 1, + name: collection(), + }, + E::CatalogIntegrityViolation { + entry_kind: "PutCollection".into(), + detail: text(), + }, + E::DataPlane(ErrorCode::NotFound), + E::Promql(crate::control::promql::PromqlError::UnexpectedEof), + E::DependentObjectsExist { + tenant_id: 1, + root_kind: "collection", + root_name: collection(), + dependent_count: 1, + dependents: vec![("view".into(), "v".into())], + }, + E::CascadeCycle { + tenant_id: 1, + root: collection(), + depth: 64, + }, + E::CrossShardInExplicitTransaction, + E::SequencerUnavailable, + E::SessionCapExceeded { cap: 8 }, + E::SessionIdleTimeout, + E::SessionTokenExpired, + E::SessionKilledByAdmin, + E::SessionUserDropped, + E::OidcProviderTenantUnbound, + E::OidcProviderTenantUnavailable { tenant_id: 1 }, + E::ExternalRoleUndefined { + subject: "alice".into(), + role: "auditor".into(), + tenant_id: 1, + }, + E::OidcNoDefaultDatabase { + sub: "alice".into(), + }, + E::TenantVectorDimExceeded { + dim: 4096, + limit: 1024, + }, + E::TenantGraphDepthExceeded { + depth: 20, + limit: 10, + }, + E::RoleInheritanceCycle { + child: "a".into(), + parent: "b".into(), + }, + E::RoleInheritanceDepthExceeded { depth: 9, limit: 8 }, + E::OllpExhausted { + retries: 3, + cause: crate::OllpExhaustedCause::PredicateDrift, + }, + E::MirrorReadOnly { + database: "db".into(), + }, + E::StaleReadNotLeader { + database: "db".into(), + source_cluster: "src".into(), + detail: text(), + }, + ] +} + +#[test] +fn every_error_variant_has_a_sample() { + let mut seen = [false; ERROR_VARIANT_COUNT]; + for err in error_samples() { + seen[error_variant_index(&err)] = true; + } + let missing: Vec = (0..ERROR_VARIANT_COUNT).filter(|i| !seen[*i]).collect(); + assert!( + missing.is_empty(), + "error variants with no sample: {missing:?}" + ); +} + +/// Every `crate::Error` variant answers the HTTP status its pgwire SQLSTATE +/// class has. Only an internal or system error reads as a 500. +#[test] +fn every_error_variant_has_the_http_status_of_its_sqlstate() { + for err in error_samples() { + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + assert_eq!( + status, + GatewayErrorMap::sqlstate_to_http(pg_state), + "{err:?} is {pg_state} on pgwire" + ); + let server_fault = matches!(class(pg_state), "XX" | "58"); + assert_eq!( + status == 500, + server_fault, + "{err:?} is {pg_state} on pgwire but HTTP {status}" + ); + } +} diff --git a/nodedb/src/control/gateway/error_map/http.rs b/nodedb/src/control/gateway/error_map/http.rs index a1ad0c5e8..0f9e4316f 100644 --- a/nodedb/src/control/gateway/error_map/http.rs +++ b/nodedb/src/control/gateway/error_map/http.rs @@ -3,71 +3,33 @@ //! HTTP error shape: `(status_code, message)`. use super::gateway_map::GatewayErrorMap; -use super::remote_code::remote_code_to_http_status; use super::sqlstate_status::sqlstate_to_http_status; use crate::Error; impl GatewayErrorMap { /// Map a SQLSTATE into an HTTP status, for an error that reaches HTTP as - /// a SQLSTATE, such as a DDL error. Each class takes the status - /// [`Self::to_http`] gives the gateway errors of that class. + /// a SQLSTATE, such as a DDL error. [`Self::to_http`] reads the same + /// table, so a DDL error and a gateway error of one class answer one + /// status. pub fn sqlstate_to_http(sqlstate: &str) -> u16 { sqlstate_to_http_status(sqlstate) } /// Map a gateway error into `(http_status_code, message)` for HTTP. /// - /// Uses standard HTTP status semantics: - /// - 400 Bad Request for client-side errors (bad SQL, not found) - /// - 401 Unauthorized for an expired session token - /// - 403 Forbidden for authz errors - /// - 409 Conflict for write-conflict / constraint violations - /// - 429 Too Many Requests for a rate-gate refusal - /// - 501 Not Implemented for an unsupported feature - /// - 503 Service Unavailable for routing/leader errors and dispatch overload - /// - 504 Gateway Timeout for deadline exceeded - /// - 500 Internal Server Error as the default fallback + /// The status follows the SQLSTATE pgwire renders for the error, through + /// the one SQLSTATE status table. One error answers one class on pgwire, + /// native and HTTP. Only an `XX000` or `58` class error reads as a 500. + /// A Data-Plane verdict answers with its public message. pub fn to_http(err: &Error) -> (u16, String) { - match err { - Error::NotLeader { leader_addr, .. } => ( - 503, - format!("cluster in leader election; leader hint: {leader_addr}"), - ), - Error::DeadlineExceeded { .. } => (504, err.to_string()), - // The retryable serialization class, the status a write conflict - // takes on every path. - Error::RetryableSchemaChanged { .. } => (409, err.to_string()), - Error::CollectionNotFound { collection, .. } => { - (404, format!("collection \"{collection}\" does not exist")) - } - Error::RejectedAuthz { .. } => (403, err.to_string()), - Error::SessionTokenExpired => (401, err.to_string()), - Error::BadRequest { detail } => (400, detail.clone()), - Error::PlanError { detail } => (400, detail.clone()), - Error::RejectedConstraint { detail, .. } => (409, detail.clone()), - Error::RateExceeded { .. } => (429, err.to_string()), - Error::NoLeader { .. } => (503, err.to_string()), - Error::DispatchCapacity { .. } => (503, err.to_string()), - Error::Serialization { .. } | Error::Codec { .. } => (500, err.to_string()), - Error::Internal { .. } => (500, err.to_string()), - // 501 Not Implemented: a valid op this server does not support, - // such as a cross-collection write whose collections are not - // co-resident (fail-closed safety floor). - Error::CrossCollectionNotColocated { .. } | Error::FeatureNotSupported { .. } => { - (501, err.to_string()) - } - Error::RemoteTyped { code, message } => { - (remote_code_to_http_status(*code), message.clone()) - } - Error::DataPlane(_) => { - let public = crate::error_classify::classify(err); - ( - remote_code_to_http_status(public.code()), - public.message().to_owned(), - ) - } - _ => (500, err.to_string()), - } + let (_severity, state, message) = + crate::control::server::pgwire::types::error_to_sqlstate(err); + let message = if let Error::DataPlane(_) = err { + crate::error_classify::classify(err).message().to_owned() + } else { + message + }; + (sqlstate_to_http_status(state), message) } } @@ -106,6 +68,40 @@ mod tests { assert_eq!(status, 500); } + #[test] + fn http_data_plane_not_found() { + let err = Error::DataPlane(crate::bridge::envelope::ErrorCode::NotFound); + assert_eq!(GatewayErrorMap::to_http(&err).0, 404); + } + + #[test] + fn http_conflict_retry() { + let err = Error::ConflictRetry { + collection: "orders".into(), + document_id: "o1".into(), + }; + assert_eq!(GatewayErrorMap::to_http(&err).0, 409); + } + + #[test] + fn http_backup_key_mismatch_is_invalid_authorization() { + assert_eq!(GatewayErrorMap::to_http(&Error::BackupKeyMismatch).0, 401); + } + + /// A remote rendering of an error keeps the status of the local one. + #[test] + fn http_backup_key_mismatch_keeps_its_status_across_nodes() { + use nodedb_types::error::ErrorCode; + let remote = Error::RemoteTyped { + code: ErrorCode::BACKUP_KEY_MISMATCH, + message: "wrong backup KEK".into(), + }; + assert_eq!( + GatewayErrorMap::to_http(&remote).0, + GatewayErrorMap::to_http(&Error::BackupKeyMismatch).0 + ); + } + #[test] fn to_http_remote_typed_is_wired_to_helper() { use nodedb_types::error::ErrorCode; diff --git a/nodedb/src/control/gateway/error_map/remote_code.rs b/nodedb/src/control/gateway/error_map/remote_code.rs index 8f9d94cb2..e983aa33f 100644 --- a/nodedb/src/control/gateway/error_map/remote_code.rs +++ b/nodedb/src/control/gateway/error_map/remote_code.rs @@ -1,57 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Numeric-`ErrorCode` fallbacks shared by the HTTP and RESP surfaces. +//! Numeric-`ErrorCode` fallback for the RESP surface. //! //! A remote peer on a newer build can mint a code this build does not know, -//! so both helpers degrade to the generic shape rather than misclassifying. - -/// Map a numeric `ErrorCode` to an HTTP status, mirroring the local variant -/// arms in `to_http` for the same condition (e.g. `CONSTRAINT_VIOLATION` -/// mirrors `RejectedConstraint`'s 409). -/// -/// Serves a `RemoteTyped` error and every Data-Plane verdict. Each code a -/// Data-Plane code classifies to has its own status here, so a typed -/// refusal never reads as a server fault. -pub(super) fn remote_code_to_http_status(code: nodedb_types::error::ErrorCode) -> u16 { - use nodedb_types::error::ErrorCode as Ec; - match code { - Ec::NOT_LEADER - | Ec::NO_LEADER - | Ec::SERVER_OVERLOAD - | Ec::MEMORY_EXHAUSTED - | Ec::COLLECTION_DRAINING => 503, - Ec::DEADLINE_EXCEEDED => 504, - Ec::COLLECTION_NOT_FOUND | Ec::DOCUMENT_NOT_FOUND => 404, - Ec::AUTHORIZATION_DENIED => 403, - Ec::AUTH_EXPIRED => 401, - Ec::BAD_REQUEST - | Ec::PLAN_ERROR - | Ec::TYPE_MISMATCH - | Ec::OVERFLOW - | Ec::DATA_EXCEPTION - | Ec::DIVISION_BY_ZERO - | Ec::UNDEFINED_COLUMN - | Ec::UNDEFINED_FUNCTION - | Ec::FAN_OUT_EXCEEDED - | Ec::PROGRAM_LIMIT_EXCEEDED => 400, - Ec::CONSTRAINT_VIOLATION - | Ec::WRITE_CONFLICT - | Ec::PREVALIDATION_REJECTED - | Ec::APPEND_ONLY_VIOLATION - | Ec::BALANCE_VIOLATION - | Ec::PERIOD_LOCKED - | Ec::PERIOD_LOCK_MISCONFIGURED - | Ec::STATE_TRANSITION_VIOLATION - | Ec::TRANSITION_CHECK_VIOLATION - | Ec::TYPE_GUARD_VIOLATION - | Ec::RETENTION_VIOLATION - | Ec::LEGAL_HOLD_ACTIVE - | Ec::INSUFFICIENT_BALANCE => 409, - Ec::RATE_EXCEEDED => 429, - Ec::SQL_NOT_ENABLED => 501, - _ => 500, - } -} +//! so the helper degrades to the generic shape rather than misclassifying. +//! HTTP needs no helper: `to_http` renders a remote code through its SQLSTATE. /// Map a numeric `ErrorCode` from a `RemoteTyped` error to a RESP error /// prefix, mirroring the local variant arms in `to_resp`. @@ -72,29 +25,6 @@ pub(super) fn remote_code_to_resp_prefix(code: nodedb_types::error::ErrorCode) - mod tests { use super::*; - #[test] - fn remote_http_status_maps_known_code() { - use nodedb_types::error::ErrorCode; - // Mirrors `RejectedConstraint`'s 409 in `to_http`. - assert_eq!( - remote_code_to_http_status(ErrorCode::CONSTRAINT_VIOLATION), - 409 - ); - assert_eq!( - remote_code_to_http_status(ErrorCode::AUTHORIZATION_DENIED), - 403 - ); - } - - #[test] - fn remote_http_status_unmapped_code_falls_back_to_500() { - use nodedb_types::error::ErrorCode; - // A code with no explicit arm (e.g. one a newer remote node minted - // that this build doesn't recognize) must degrade to the generic - // 500 fallback, not silently misreport a specific status. - assert_eq!(remote_code_to_http_status(ErrorCode(65000)), 500); - } - #[test] fn remote_resp_prefix_maps_known_code() { use nodedb_types::error::ErrorCode; @@ -115,8 +45,7 @@ mod tests { #[test] fn remote_resp_prefix_unmapped_code_falls_back_to_err() { use nodedb_types::error::ErrorCode; - // Same degrade path as the HTTP fallback: an unrecognized remote code - // must still surface as the generic `ERR` prefix. + // An unrecognized remote code surfaces as the generic `ERR` prefix. assert_eq!(remote_code_to_resp_prefix(ErrorCode(65000)), "ERR"); } } diff --git a/nodedb/src/control/gateway/error_map/sqlstate_status.rs b/nodedb/src/control/gateway/error_map/sqlstate_status.rs index 5a181f25e..5108909d2 100644 --- a/nodedb/src/control/gateway/error_map/sqlstate_status.rs +++ b/nodedb/src/control/gateway/error_map/sqlstate_status.rs @@ -3,8 +3,8 @@ //! SQLSTATE to HTTP status, for an error that reaches HTTP as a SQLSTATE. //! //! A DDL error carries a SQLSTATE and a numeric code. Several DDL SQLSTATEs -//! share one code, so the status follows the SQLSTATE. Each class takes the -//! status `to_http` gives the gateway errors of that class. +//! share one code, so the status follows the SQLSTATE. `to_http` reads this +//! table for every gateway error, so it is the one status table. use nodedb_types::error::sqlstate; diff --git a/nodedb/src/control/server/http/auth.rs b/nodedb/src/control/server/http/auth.rs index dc2aa268c..d45ace268 100644 --- a/nodedb/src/control/server/http/auth.rs +++ b/nodedb/src/control/server/http/auth.rs @@ -432,22 +432,19 @@ impl FromRequestParts for ResolvedAuth { } } +/// The status and message come from the one gateway mapping, +/// [`GatewayErrorMap::to_http`](crate::control::gateway::GatewayErrorMap::to_http). +/// A rate refusal also carries its `Retry-After` hint. impl From for ApiError { fn from(e: crate::Error) -> Self { - match &e { - crate::Error::RejectedAuthz { .. } => Self::Forbidden(e.to_string()), - crate::Error::RateExceeded { retry_after_ms, .. } => Self::RateLimited { - message: e.to_string(), + let (status, message) = crate::control::gateway::GatewayErrorMap::to_http(&e); + if let crate::Error::RateExceeded { retry_after_ms, .. } = &e { + return Self::RateLimited { + message, retry_after_secs: retry_after_ms.div_ceil(1000).max(1), - }, - crate::Error::BadRequest { .. } - | crate::Error::PlanError { .. } - | crate::Error::Config { .. } => Self::BadRequest(e.to_string()), - crate::Error::CollectionNotFound { .. } | crate::Error::DocumentNotFound { .. } => { - Self::BadRequest(e.to_string()) - } - _ => Self::Internal(e.to_string()), + }; } + Self::HttpStatus(status, message) } } @@ -546,6 +543,90 @@ mod tests { assert_authorization_fields_unchanged(&context, &result); } + fn api_status(error: crate::Error) -> StatusCode { + ApiError::from(error).into_response().status() + } + + /// Every error takes the status the gateway mapping gives it. + #[test] + fn api_errors_take_the_gateway_status() { + use crate::types::{RequestId, VShardId}; + + let cases = [ + ( + crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::NotFound), + StatusCode::NOT_FOUND, + ), + ( + crate::Error::DeadlineExceeded { + request_id: RequestId::new(1), + }, + StatusCode::GATEWAY_TIMEOUT, + ), + ( + crate::Error::NotLeader { + vshard_id: VShardId::new(1), + leader_node: 2, + leader_addr: "10.0.0.1:9000".into(), + }, + StatusCode::SERVICE_UNAVAILABLE, + ), + ( + crate::Error::ConflictRetry { + collection: "orders".into(), + document_id: "o1".into(), + }, + StatusCode::CONFLICT, + ), + ( + crate::Error::CollectionNotFound { + tenant_id: TenantId::new(1), + collection: "orders".into(), + }, + StatusCode::NOT_FOUND, + ), + ( + crate::Error::RejectedAuthz { + tenant_id: TenantId::new(1), + resource: "orders".into(), + }, + StatusCode::FORBIDDEN, + ), + ( + crate::Error::BadRequest { + detail: "bad".into(), + }, + StatusCode::BAD_REQUEST, + ), + ]; + for (error, expected) in cases { + let label = format!("{error:?}"); + let gateway = crate::control::gateway::GatewayErrorMap::to_http(&error).0; + let status = api_status(error); + assert_eq!(status, expected, "{label}"); + assert_eq!(status.as_u16(), gateway, "{label}"); + } + } + + /// A rate refusal keeps its 429 and carries its `Retry-After` hint. + #[test] + fn a_rate_refusal_carries_retry_after() { + let response = ApiError::from(crate::Error::RateExceeded { + gate: "write".into(), + detail: "over budget".into(), + retry_after_ms: 1500, + }) + .into_response(); + assert_eq!(response.status(), StatusCode::TOO_MANY_REQUESTS); + assert_eq!( + response + .headers() + .get("Retry-After") + .and_then(|value| value.to_str().ok()), + Some("2") + ); + } + #[test] fn x_on_deny_malformed_value_is_ignored() { let context = auth_context(); diff --git a/nodedb/src/control/server/http/routes/crdt.rs b/nodedb/src/control/server/http/routes/crdt.rs index 730edb7c8..f45ed92bf 100644 --- a/nodedb/src/control/server/http/routes/crdt.rs +++ b/nodedb/src/control/server/http/routes/crdt.rs @@ -104,7 +104,7 @@ pub async fn crdt_apply( identity.tenant_id, body.doc_id.as_bytes(), ) - .map_err(|e| ApiError::Internal(e.to_string()))?; + .map_err(ApiError::from)?; let plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new( diff --git a/nodedb/src/control/server/http/routes/query.rs b/nodedb/src/control/server/http/routes/query.rs index 7a5e51513..083164a18 100644 --- a/nodedb/src/control/server/http/routes/query.rs +++ b/nodedb/src/control/server/http/routes/query.rs @@ -50,9 +50,13 @@ pub(crate) fn resolve_database_id( match catalog.get_database_id_by_name(&db_name) { Ok(Some(id)) => Ok(id), - Ok(None) => Err(ApiError::BadRequest(format!( - "3D000 database '{db_name}' does not exist" - ))), + // The status of `3D000` from the one gateway status table. + Ok(None) => Err(ApiError::HttpStatus( + crate::control::gateway::GatewayErrorMap::sqlstate_to_http( + nodedb_types::error::sqlstate::INVALID_CATALOG_NAME, + ), + format!("3D000 database '{db_name}' does not exist"), + )), Err(e) => Err(ApiError::Internal(format!("catalog lookup failed: {e}"))), } } diff --git a/nodedb/src/control/server/http/routes/query/materialized/encode.rs b/nodedb/src/control/server/http/routes/query/materialized/encode.rs index 138bfcfba..ae6c3721e 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/encode.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/encode.rs @@ -18,9 +18,10 @@ pub(super) fn ddl_error_to_api(error: crate::control::server::shared::ddl::DdlEr } } +/// Map a gateway error to the HTTP error the client reads, through the one +/// `crate::Error` to `ApiError` conversion. pub(super) fn gateway_error(error: crate::Error) -> ApiError { - let (status, msg) = crate::control::gateway::GatewayErrorMap::to_http(&error); - ApiError::HttpStatus(status, msg) + ApiError::from(error) } /// Map a Data-Plane refusal to the HTTP error the client reads. A typed diff --git a/nodedb/src/control/server/http/routes/result_shape.rs b/nodedb/src/control/server/http/routes/result_shape.rs index 28ee3e84f..e217ff02d 100644 --- a/nodedb/src/control/server/http/routes/result_shape.rs +++ b/nodedb/src/control/server/http/routes/result_shape.rs @@ -16,30 +16,21 @@ use crate::control::server::response_shape::cell::row_to_wire_json; use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; use crate::control::server::response_shape::request::MaterializedShapeRequest; use nodedb_types::NodeDbError; -use nodedb_types::error::ErrorCode; use super::super::auth::ApiError; /// Map a shaping error to the HTTP error the client reads, keeping its -/// numeric code. A statement-level refusal the shaper raises per row — an -/// unknown sequence, `currval` before `nextval`, a bad accessor argument, -/// division by zero — is the caller's error and answers `400`; anything -/// else is the server's and answers `500`. +/// numeric code. The status follows the SQLSTATE the code renders on +/// pgwire, through the one gateway status table. A per-row refusal, such as +/// an unknown sequence or a division by zero, answers its client class. +/// An internal error answers `500`. pub(super) fn shape_error_to_api(e: NodeDbError) -> ApiError { let code = e.code(); - let status = if matches!( - code, - ErrorCode::UNDEFINED_OBJECT - | ErrorCode::OBJECT_NOT_READY - | ErrorCode::PLAN_ERROR - | ErrorCode::DIVISION_BY_ZERO - | ErrorCode::DATA_EXCEPTION - | ErrorCode::BAD_REQUEST - ) { - StatusCode::BAD_REQUEST - } else { - StatusCode::INTERNAL_SERVER_ERROR - }; + let state = crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate(code); + let status = StatusCode::from_u16(crate::control::gateway::GatewayErrorMap::sqlstate_to_http( + state, + )) + .unwrap_or(StatusCode::INTERNAL_SERVER_ERROR); ApiError::Coded { status, message: e.message().to_string(), diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 9e9798904..86750d872 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -353,6 +353,10 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> Ec::NOT_LEADER => sqlstate::DATABASE_DROPPED, // Mirrors the `CloneWriteRequiresMaterialize` arm. Ec::CLONE_WRITE_REQUIRES_MATERIALIZE => sqlstate::CLONE_WRITE_REQUIRES_MATERIALIZE.0, + // Mirrors the `BackupTenantMismatch` arm. + Ec::BACKUP_TENANT_MISMATCH => sqlstate::BACKUP_TENANT_MISMATCH, + // Mirrors the `BackupKeyMismatch` arm. + Ec::BACKUP_KEY_MISMATCH => sqlstate::BACKUP_KEY_MISMATCH, // The codes below mirror the Data-Plane code table // (`error_code_to_sqlstate`) for the public code each Data-Plane code // classifies to, so a verdict that crossed a node as a numeric code diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index 6b79f5a8e..0b9de8806 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -14,7 +14,7 @@ //! `SELECT RATE_RESET(gate_name, key)` //! — Deletes the counter key (admin cooldown clear). -use crate::bridge::envelope::{PhysicalPlan, Status}; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::response_payload::payload_or_typed_error; use crate::control::state::SharedState; @@ -71,7 +71,7 @@ pub async fn rate_check( ), key: rate_key.as_bytes().to_vec(), }); - match crate::control::server::dispatch_utils::dispatch_to_data_plane( + let result = crate::control::server::dispatch_utils::dispatch_to_data_plane( state, tenant_id, crate::types::DatabaseId::DEFAULT, @@ -79,16 +79,8 @@ pub async fn rate_check( check, TraceId::ZERO, ) - .await - { - Ok(resp) if resp.status == Status::Ok => { - let text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - // ttl_ms == -2 means key does not exist. - !text.contains("-2") - } - _ => false, - } + .await; + ttl_from("RATE_CHECK", result)?.is_some() }; let actual_ttl = if key_exists { 0 } else { ttl_ms }; @@ -125,11 +117,14 @@ pub async fn rate_check( let current: i64 = sonic_rs::from_str::(&payload_text) .ok() .and_then(|v| v.get("value")?.as_i64()) - .unwrap_or(1); + .ok_or(ddl_err( + "XX000", + format!("RATE_CHECK: counter '{rate_key}' increment answered no integer value"), + ))?; if current > max_count { // Read TTL to compute retry_after_ms. - let ttl_remaining = read_ttl_ms(state, tenant_id, vshard, &rate_key).await; + let ttl_remaining = read_ttl_ms("RATE_CHECK", state, tenant_id, vshard, &rate_key).await?; Err(ddl_err( "53300", format!( @@ -179,7 +174,7 @@ pub async fn rate_remaining( surrogate_ceiling: None, }); - let current = match crate::control::server::dispatch_utils::dispatch_to_data_plane( + let result = crate::control::server::dispatch_utils::dispatch_to_data_plane( state, tenant_id, crate::types::DatabaseId::DEFAULT, @@ -187,26 +182,11 @@ pub async fn rate_remaining( plan, TraceId::ZERO, ) - .await - { - Ok(resp) if resp.status == Status::Ok && !resp.payload.is_empty() => { - // `KV_INCR` stores the counter as a raw body: its decimal text. - std::str::from_utf8(&resp.payload) - .ok() - .and_then(|text| text.parse::().ok()) - .ok_or(ddl_err( - "XX000", - format!( - "RATE_REMAINING: counter '{rate_key}' does not hold decimal text; \ - reset the gate with RATE_RESET" - ), - ))? - } - _ => 0, // Key doesn't exist yet — no usage. - }; + .await; + let current = counter_from(&rate_key, result)?; let ttl_remaining = if current > 0 { - read_ttl_ms(state, tenant_id, vshard, &rate_key).await + read_ttl_ms("RATE_REMAINING", state, tenant_id, vshard, &rate_key).await? } else { 0 }; @@ -306,19 +286,81 @@ fn counter_write_payload( .map_err(|e| DdlError::from_error_in_context(context, &e)) } -/// Read TTL remaining for a KV key (in milliseconds). +/// The `ttl_ms` a `GetTtl` read reports for a key that does not exist. +const TTL_ABSENT: i64 = -2; + +/// The payload of a counter read, or `None` when the key is absent. A +/// `NotFound` verdict is a result, not an error. Every other refusal or +/// dispatch error keeps its SQLSTATE and code, with `context` before the +/// message. +fn read_payload( + context: &str, + result: crate::Result, +) -> Result>, DdlError> { + match result.and_then(payload_or_typed_error) { + Ok(payload) => Ok(Some(payload)), + Err(crate::Error::DataPlane(ErrorCode::NotFound)) => Ok(None), + Err(e) => Err(DdlError::from_error_in_context(context, &e)), + } +} + +/// The TTL a `GetTtl` read reports, or `None` when the key is absent. +fn ttl_from( + context: &str, + result: crate::Result, +) -> Result, DdlError> { + let Some(payload) = read_payload(context, result)? else { + return Ok(None); + }; + let text = crate::data::executor::response_codec::decode_payload_to_json(&payload); + let ttl_ms = sonic_rs::from_str::(&text) + .ok() + .and_then(|v| v.get("ttl_ms")?.as_i64()) + .ok_or(ddl_err( + "XX000", + format!("{context}: TTL read answered no integer ttl_ms: {text}"), + ))?; + Ok((ttl_ms != TTL_ABSENT).then_some(ttl_ms)) +} + +/// The counter a `Get` read reports. An absent key has no usage, so it +/// reads as `0`. +fn counter_from( + rate_key: &str, + result: crate::Result, +) -> Result { + match read_payload("RATE_REMAINING", result)? { + None => Ok(0), + Some(payload) if payload.is_empty() => Ok(0), + // `KV_INCR` stores the counter as a raw body: its decimal text. + Some(payload) => std::str::from_utf8(&payload) + .ok() + .and_then(|text| text.parse::().ok()) + .ok_or(ddl_err( + "XX000", + format!( + "RATE_REMAINING: counter '{rate_key}' does not hold decimal text; \ + reset the gate with RATE_RESET" + ), + )), + } +} + +/// Read TTL remaining for a KV key (in milliseconds). An absent key, or a +/// key with no expiry, reads as `0`. `context` names the calling function. async fn read_ttl_ms( + context: &str, state: &SharedState, tenant_id: crate::types::TenantId, vshard: VShardId, key: &str, -) -> u64 { +) -> Result { let plan = PhysicalPlan::Kv(KvOp::GetTtl { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, RATE_COLLECTION), key: key.as_bytes().to_vec(), }); - match crate::control::server::dispatch_utils::dispatch_to_data_plane( + let result = crate::control::server::dispatch_utils::dispatch_to_data_plane( state, tenant_id, crate::types::DatabaseId::DEFAULT, @@ -326,19 +368,17 @@ async fn read_ttl_ms( plan, TraceId::ZERO, ) - .await - { - Ok(resp) if resp.status == Status::Ok => { - let payload_text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - sonic_rs::from_str::(&payload_text) - .ok() - .and_then(|v| v.get("ttl_ms")?.as_i64()) - .map(|ttl| if ttl > 0 { ttl as u64 } else { 0 }) - .unwrap_or(0) - } - _ => 0, - } + .await; + remaining_ttl_ms(context, result) +} + +/// The milliseconds a `GetTtl` read leaves on a key. An absent key, or a key +/// with no expiry (`-1`), reads as `0`. +fn remaining_ttl_ms( + context: &str, + result: crate::Result, +) -> Result { + Ok(ttl_from(context, result)?.map_or(0, |ttl| u64::try_from(ttl).unwrap_or(0))) } fn unquote(s: &str) -> String { @@ -380,7 +420,7 @@ mod tests { use nodedb_types::error::sqlstate; use super::*; - use crate::bridge::envelope::{ErrorCode, Payload, Response}; + use crate::bridge::envelope::{Payload, Response, Status}; use crate::types::{Lsn, RequestId}; fn refusal(code: Option) -> Response { @@ -430,4 +470,131 @@ mod tests { .expect_err("a refused counter write fails the call"); assert_eq!(err.sqlstate, sqlstate::INTERNAL_ERROR, "{err:?}"); } + + fn answer(payload: Vec) -> Response { + Response { + status: Status::Ok, + payload: Payload::from_vec(payload), + error_code: None, + ..refusal(None) + } + } + + /// The msgpack map `{"ttl_ms": ttl}` a `GetTtl` read answers with. `ttl` + /// fits a msgpack fixint. + fn ttl_payload(ttl: i8) -> Vec { + let mut bytes = vec![0x81, 0xa6]; + bytes.extend_from_slice(b"ttl_ms"); + bytes.push(ttl.to_ne_bytes()[0]); + bytes + } + + fn deadline() -> crate::Error { + crate::Error::DeadlineExceeded { + request_id: RequestId::new(1), + } + } + + fn not_found() -> Response { + refusal(Some(ErrorCode::NotFound)) + } + + /// The existence check propagates a dispatch error and a coded refusal + /// with their own class, instead of reading them as "no key". + #[test] + fn the_existence_check_propagates_errors() { + let err = ttl_from("RATE_CHECK", Err(deadline())).expect_err("a failed read fails"); + assert_eq!(err.sqlstate, sqlstate::QUERY_CANCELED, "{err:?}"); + assert!(err.message.starts_with("RATE_CHECK: "), "{}", err.message); + + let refused = refusal(Some(ErrorCode::Unsupported { + detail: "not on this engine".into(), + })); + let err = ttl_from("RATE_CHECK", Ok(refused)).expect_err("a refused read fails"); + assert_eq!(err.sqlstate, sqlstate::FEATURE_NOT_SUPPORTED, "{err:?}"); + } + + /// An absent key reads as absent, whether the read answers `-2` or a + /// `NotFound` verdict. A live key reads as present. + #[test] + fn the_existence_check_reads_absence_as_a_result() { + assert_eq!( + ttl_from("RATE_CHECK", Ok(answer(ttl_payload(-2)))).expect("read succeeds"), + None + ); + assert_eq!( + ttl_from("RATE_CHECK", Ok(not_found())).expect("read succeeds"), + None + ); + assert_eq!( + ttl_from("RATE_CHECK", Ok(answer(ttl_payload(30)))).expect("read succeeds"), + Some(30) + ); + assert_eq!( + ttl_from("RATE_CHECK", Ok(answer(ttl_payload(-1)))).expect("read succeeds"), + Some(-1) + ); + } + + /// A TTL read that answers no `ttl_ms` is an internal error, never a guess. + #[test] + fn a_ttl_read_with_no_ttl_is_internal() { + let err = ttl_from("RATE_CHECK", Ok(answer(Vec::new()))).expect_err("no ttl_ms fails"); + assert_eq!(err.sqlstate, sqlstate::INTERNAL_ERROR, "{err:?}"); + } + + /// The counter read propagates a dispatch error instead of reading it as + /// no usage. + #[test] + fn the_counter_read_propagates_errors() { + let err = counter_from("_rate:g:k", Err(deadline())).expect_err("a failed read fails"); + assert_eq!(err.sqlstate, sqlstate::QUERY_CANCELED, "{err:?}"); + assert!( + err.message.starts_with("RATE_REMAINING: "), + "{}", + err.message + ); + } + + /// An absent counter reads as zero usage. A stored counter reads as its + /// decimal value. + #[test] + fn the_counter_read_reads_absence_as_zero() { + assert_eq!( + counter_from("_rate:g:k", Ok(answer(Vec::new()))).expect("read succeeds"), + 0 + ); + assert_eq!( + counter_from("_rate:g:k", Ok(not_found())).expect("read succeeds"), + 0 + ); + assert_eq!( + counter_from("_rate:g:k", Ok(answer(b"7".to_vec()))).expect("read succeeds"), + 7 + ); + } + + /// The TTL read behind `retry after` propagates a dispatch error instead + /// of reading it as no time left. + #[test] + fn the_ttl_read_propagates_errors() { + let err = + remaining_ttl_ms("RATE_REMAINING", Err(deadline())).expect_err("a failed read fails"); + assert_eq!(err.sqlstate, sqlstate::QUERY_CANCELED, "{err:?}"); + } + + /// An absent key and a key with no expiry leave no time. A live key + /// leaves its TTL. + #[test] + fn the_ttl_read_reads_absence_as_zero() { + assert_eq!( + remaining_ttl_ms("RATE_REMAINING", Ok(not_found())).expect("read succeeds"), + 0 + ); + for (ttl, expected) in [(-2, 0), (-1, 0), (30, 30)] { + let remaining = remaining_ttl_ms("RATE_REMAINING", Ok(answer(ttl_payload(ttl)))) + .expect("read succeeds"); + assert_eq!(remaining, expected, "ttl_ms {ttl}"); + } + } } diff --git a/nodedb/src/control/system_txn/run.rs b/nodedb/src/control/system_txn/run.rs index 534bcd9b9..d2e2446a3 100644 --- a/nodedb/src/control/system_txn/run.rs +++ b/nodedb/src/control/system_txn/run.rs @@ -47,12 +47,22 @@ pub enum SystemTxnError { detail: String, code: Option>, }, + + /// COMMIT aborted on a dispatch or DDL-propose error. The transaction + /// applied nothing. `source` keeps its own class. + #[error("system transaction aborted at commit: {source}")] + CommitFailed { + #[source] + source: crate::Error, + }, } impl From for crate::Error { fn from(error: SystemTxnError) -> Self { match error { - SystemTxnError::Begin { source } | SystemTxnError::Statement { source, .. } => source, + SystemTxnError::Begin { source } + | SystemTxnError::Statement { source, .. } + | SystemTxnError::CommitFailed { source } => source, SystemTxnError::Commit { code: Some(code), .. } => crate::Error::DataPlane(*code), @@ -200,10 +210,27 @@ pub async fn run_statements_atomically( match commit::run_commit(scope.sessions(), scope.session_id(), identity, state, &dp).await { CommitOutcome::Committed => Ok(()), - CommitOutcome::Aborted { reason } => Err(SystemTxnError::Commit { + CommitOutcome::Aborted { reason } => Err(commit_abort_error(reason)), + } +} + +/// The error a commit abort answers with. A dispatch or DDL-propose error +/// keeps its own class. Every other abort carries its Data-Plane verdict +/// when one decided it. +fn commit_abort_error(reason: AbortReason) -> SystemTxnError { + match reason { + AbortReason::Dispatch(source) | AbortReason::DdlPropose(source) => { + SystemTxnError::CommitFailed { source } + } + reason @ (AbortReason::Serialization + | AbortReason::NoTransaction + | AbortReason::BatchRejected { .. } + | AbortReason::CalvinCancelled + | AbortReason::CalvinTimeout + | AbortReason::SchemaChanged { .. }) => SystemTxnError::Commit { detail: describe(&reason), code: abort_code(&reason).map(Box::new), - }), + }, } } @@ -279,11 +306,12 @@ fn abort_code(reason: &AbortReason) -> Option { Some(crate::bridge::envelope::ErrorCode::ConflictRetry) } - AbortReason::NoTransaction - | AbortReason::CalvinCancelled - | AbortReason::CalvinTimeout - | AbortReason::Dispatch(_) - | AbortReason::DdlPropose(_) => None, + // The cross-shard coordinator ran out of time or was cancelled at + // its deadline: the deadline class, `57014`. + AbortReason::CalvinCancelled | AbortReason::CalvinTimeout => { + Some(crate::bridge::envelope::ErrorCode::DeadlineExceeded) + } + AbortReason::NoTransaction | AbortReason::Dispatch(_) | AbortReason::DdlPropose(_) => None, } } diff --git a/nodedb/src/event/cross_shard/dispatcher.rs b/nodedb/src/event/cross_shard/dispatcher.rs index e9829ae52..c4ab312ea 100644 --- a/nodedb/src/event/cross_shard/dispatcher.rs +++ b/nodedb/src/event/cross_shard/dispatcher.rs @@ -188,10 +188,22 @@ async fn send_write( detail: format!("transport: {e}"), })?; - let RaftRpc::VShardEnvelope(response_bytes) = response_rpc else { - return Err(crate::Error::Dispatch { - detail: "unexpected RPC response type".to_string(), - }); + let response_bytes = match response_rpc { + RaftRpc::VShardEnvelope(response_bytes) => response_bytes, + // The target's handler failed. Its typed error names why. + RaftRpc::VShardRefusal(refusal) => { + return Err(crate::Error::Dispatch { + detail: format!( + "target refused: {}", + nodedb_cluster::error::ClusterError::from(refusal.error) + ), + }); + } + _ => { + return Err(crate::Error::Dispatch { + detail: "unexpected RPC response type".to_string(), + }); + } }; let response_env = From 0ed67450a1b2dcd247231ba666c6f12e2f6c0e26 Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Mon, 28 Sep 2026 00:33:08 +0800 Subject: [PATCH 55/64] feat(errors): exhaust every typed error match instead of a wildcard arm Replace the last catch-all arms in the error-crossing paths with exhaustive matches over crate::Error and bridge::envelope::ErrorCode, so a new variant fails to compile here instead of silently falling into Internal or the wrong compensation/retry class. Covers the cluster propose path (new propose_error module), CRDT delta compensation hints (new compensation module), sync refusal/retry classification, gateway error mapping, array cluster execution, backup restore, DDL neutral handlers, and the recursive-value/point/transaction executor handlers. Split error_classify.rs into a directory: public.rs keeps the Error-to-NodeDbError table, unclassified.rs keeps the is-unclassified-failure predicate. Add numeric_sqlstate.rs so a numeric ErrorCode crossing from a remote node maps back to the same SQLSTATE class its local variant would have chosen. Extend nodedb-cluster's wire/circuit-breaker/vshard/data_propose types and nodedb-types' sqlstate table with the codes this now threads through. --- nodedb-cluster/src/circuit_breaker.rs | 47 ++- .../src/distributed_array/scatter.rs | 44 +- nodedb-cluster/src/error.rs | 11 + nodedb-cluster/src/rpc_codec/data_propose.rs | 52 ++- .../src/rpc_codec/shard_error/convert.rs | 8 + .../src/rpc_codec/shard_error/wire.rs | 5 + nodedb-cluster/src/rpc_codec/vshard.rs | 23 ++ nodedb-types/src/error/sqlstate.rs | 13 + nodedb/src/bridge/envelope/error_code.rs | 110 ++++- .../backup/restore/orchestrate/restore.rs | 22 +- .../backup/restore/redo_reissue/commit.rs | 20 +- .../control/cluster/array_cluster_helpers.rs | 80 +++- .../control/cluster/array_executor/refusal.rs | 176 +++++++- .../control/cluster/data_plane_error_wire.rs | 120 +++++- .../control/cluster/metadata_applier/wedge.rs | 104 ++++- nodedb/src/control/cluster/read_index.rs | 41 +- nodedb/src/control/cluster/start_raft/mod.rs | 1 + .../cluster/start_raft/propose_error.rs | 123 ++++++ .../cluster/start_raft/proposer_wiring.rs | 23 +- nodedb/src/control/gateway/dispatch_remote.rs | 59 ++- nodedb/src/control/gateway/dispatcher.rs | 20 +- .../control/gateway/error_map/class_parity.rs | 120 +++++- nodedb/src/control/gateway/error_map/mod.rs | 2 +- nodedb/src/control/gateway/error_map/resp.rs | 112 ++++- .../src/control/metadata_proposer/handle.rs | 88 +++- nodedb/src/control/planner/plan_error_map.rs | 90 +++- .../src/control/security/role_assignment.rs | 11 +- nodedb/src/control/sequence/error_map.rs | 16 +- .../server/dispatch_utils/write_abort.rs | 62 ++- .../server/pgwire/handler/session_explain.rs | 15 +- .../control/server/pgwire/types/error_map.rs | 175 ++++---- nodedb/src/control/server/pgwire/types/mod.rs | 1 + .../server/pgwire/types/numeric_sqlstate.rs | 101 +++++ .../shared/ddl/neutral/column_default.rs | 41 +- .../ddl/neutral/consumer_group/commit.rs | 3 +- .../shared/ddl/neutral/database/drop.rs | 10 +- .../ddl/neutral/database/materialize.rs | 7 +- .../server/shared/ddl/neutral/read_gate.rs | 15 +- .../control/server/shared/ddl/neutral/rls.rs | 3 +- .../shared/ddl/neutral/router/dispatch.rs | 23 +- .../shared/ddl/neutral/topic/publish.rs | 20 +- .../ddl/neutral/version_history/dispatch.rs | 7 +- .../src/control/server/shared/ddl/result.rs | 20 + nodedb/src/control/server/shared/retry.rs | 108 ++++- .../sync/async_dispatch/delta/compensation.rs | 207 ++++++++++ .../server/sync/async_dispatch/delta/mod.rs | 1 + .../sync/async_dispatch/delta/outcome.rs | 85 ++-- nodedb/src/control/server/sync/refusal.rs | 391 ++++++++++++++++-- .../executor/handlers/control/reindex/csr.rs | 47 ++- .../handlers/point/apply_put/stored_body.rs | 9 +- .../handlers/point/apply_put/types.rs | 59 ++- .../handlers/point/update/post_image.rs | 9 +- .../executor/handlers/recursive_value/eval.rs | 42 +- .../handlers/recursive_value/handler.rs | 6 + .../handlers/transaction/stage_write/body.rs | 18 +- .../transaction/stage_write/stage_upsert.rs | 11 +- nodedb/src/engine/sparse/inverted/errors.rs | 43 ++ nodedb/src/error_classify/mod.rs | 10 + .../public.rs} | 140 ++++--- nodedb/src/error_classify/unclassified.rs | 147 +++++++ nodedb/src/error_from.rs | 119 +++++- 61 files changed, 3053 insertions(+), 443 deletions(-) create mode 100644 nodedb/src/control/cluster/start_raft/propose_error.rs create mode 100644 nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs create mode 100644 nodedb/src/control/server/sync/async_dispatch/delta/compensation.rs create mode 100644 nodedb/src/error_classify/mod.rs rename nodedb/src/{error_classify.rs => error_classify/public.rs} (81%) create mode 100644 nodedb/src/error_classify/unclassified.rs diff --git a/nodedb-cluster/src/circuit_breaker.rs b/nodedb-cluster/src/circuit_breaker.rs index bbffd3f1f..a38d24208 100644 --- a/nodedb-cluster/src/circuit_breaker.rs +++ b/nodedb-cluster/src/circuit_breaker.rs @@ -251,10 +251,51 @@ impl RetryPolicy { /// Determine if an error is retryable. /// - /// Only transport errors (connection failures, timeouts) are retried. - /// Codec errors, circuit-open errors, and application errors are not. + /// Only transport errors (connection failures) are retried. Codec errors, + /// circuit-open errors, shard timeouts, and application errors are not. pub fn is_retryable(err: &ClusterError) -> bool { - matches!(err, ClusterError::Transport { .. }) + match err { + ClusterError::Transport { .. } => true, + // A timed-out send may still have reached the peer, so resending + // it can apply the request twice. + ClusterError::ShardTimeout { .. } => false, + ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::StreamTerminal { .. } + | ClusterError::Storage { .. } + | ClusterError::DataPlane { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. } + | ClusterError::ShardExecution { .. } => false, + } } } diff --git a/nodedb-cluster/src/distributed_array/scatter.rs b/nodedb-cluster/src/distributed_array/scatter.rs index fef1f4453..fbde79528 100644 --- a/nodedb-cluster/src/distributed_array/scatter.rs +++ b/nodedb-cluster/src/distributed_array/scatter.rs @@ -82,18 +82,52 @@ use super::rpc::ShardRpcDispatch; /// single retry in `call_with_wrong_owner_retry` already re-reads the live /// table). Counting it as a liveness failure would open the shared breaker and /// then fast-fail healthy shards' slice/put/agg/delete with `CircuitOpen`. -/// `WrongOwner` and a typed Data-Plane verdict are excluded. Every genuine -/// transport/timeout/unreachable error still counts, mirroring +/// `WrongOwner` and a typed verdict from a shard that answered are excluded. +/// Every genuine transport/timeout/unreachable error still counts, mirroring /// `RetryPolicy::is_retryable`'s conservative policy. fn counts_against_breaker(err: &ClusterError) -> bool { match err { ClusterError::WrongOwner { .. } => false, - // A typed Data-Plane verdict comes from a healthy shard that answered. - ClusterError::DataPlane { .. } => false, + // A typed verdict comes from a healthy shard that answered. + ClusterError::DataPlane { .. } + | ClusterError::ShardExecution { .. } + | ClusterError::StreamTerminal { .. } => false, // An unresponsive peer is exactly what the breaker exists to shed // load from, so a shard timeout counts like any other liveness failure. ClusterError::ShardTimeout { .. } => true, - _ => true, + ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. } => true, } } diff --git a/nodedb-cluster/src/error.rs b/nodedb-cluster/src/error.rs index 5131cf065..2aa8b43b7 100644 --- a/nodedb-cluster/src/error.rs +++ b/nodedb-cluster/src/error.rs @@ -233,4 +233,15 @@ pub enum ClusterError { /// `detail` is that error's message. #[error("remote error: {detail}")] RemoteUntyped { detail: String }, + + /// A shard's local execution failed with a classified error. + /// + /// `error` is the typed wire form of that error, so the coordinator + /// rebuilds the error and renders the SQLSTATE a single-node execution + /// renders. `detail` is the message with the shard's context, for logs. + #[error("shard execution error: {detail}")] + ShardExecution { + error: Box, + detail: String, + }, } diff --git a/nodedb-cluster/src/rpc_codec/data_propose.rs b/nodedb-cluster/src/rpc_codec/data_propose.rs index e22527ce0..f1fcb4831 100644 --- a/nodedb-cluster/src/rpc_codec/data_propose.rs +++ b/nodedb-cluster/src/rpc_codec/data_propose.rs @@ -76,7 +76,57 @@ impl DataProposeResponse { ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) => { (ForwardedProposeRefusal::LeadershipTransferInProgress, None) } - _ => (ForwardedProposeRefusal::Failed, None), + // A refusal with no retry contract. The forwarding node reads it + // as a transport error carrying the leader's message. + ClusterError::Raft( + nodedb_raft::RaftError::LogCompacted { .. } + | nodedb_raft::RaftError::CompactionAheadOfApplied { .. } + | nodedb_raft::RaftError::ProposalRejected { .. } + | nodedb_raft::RaftError::InvalidTransferTarget { .. } + | nodedb_raft::RaftError::GroupNotFound { .. } + | nodedb_raft::RaftError::Transport { .. } + | nodedb_raft::RaftError::Storage { .. } + | nodedb_raft::RaftError::Serialization { .. } + | nodedb_raft::RaftError::SnapshotFormat { .. } + | nodedb_raft::RaftError::Shutdown, + ) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::ShardTimeout { .. } + | ClusterError::StreamTerminal { .. } + | ClusterError::Storage { .. } + | ClusterError::DataPlane { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. } + | ClusterError::ShardExecution { .. } => (ForwardedProposeRefusal::Failed, None), }; Self { success: false, diff --git a/nodedb-cluster/src/rpc_codec/shard_error/convert.rs b/nodedb-cluster/src/rpc_codec/shard_error/convert.rs index a5e8aff17..eda394f3b 100644 --- a/nodedb-cluster/src/rpc_codec/shard_error/convert.rs +++ b/nodedb-cluster/src/rpc_codec/shard_error/convert.rs @@ -122,6 +122,10 @@ impl From for ShardErrorWire { Self::SnapshotApplyFailed { group_id, detail } } ClusterError::RemoteUntyped { detail } => Self::Untyped { detail }, + ClusterError::ShardExecution { error, detail } => Self::ShardExecution { + error: *error, + detail, + }, // Coordinator-side error families. Their message crosses. other @ (ClusterError::MigrationCheckpoint(_) | ClusterError::MigrationRecovery(_) @@ -252,6 +256,10 @@ impl From for ClusterError { Self::SnapshotApplyFailed { group_id, detail } } ShardErrorWire::Untyped { detail } => Self::RemoteUntyped { detail }, + ShardErrorWire::ShardExecution { error, detail } => Self::ShardExecution { + error: Box::new(error), + detail, + }, } } } diff --git a/nodedb-cluster/src/rpc_codec/shard_error/wire.rs b/nodedb-cluster/src/rpc_codec/shard_error/wire.rs index 6241c9064..6c585d80a 100644 --- a/nodedb-cluster/src/rpc_codec/shard_error/wire.rs +++ b/nodedb-cluster/src/rpc_codec/shard_error/wire.rs @@ -118,4 +118,9 @@ pub enum ShardErrorWire { Untyped { detail: String, }, + /// A shard's classified local-execution error, in its typed wire form. + ShardExecution { + error: TypedClusterError, + detail: String, + }, } diff --git a/nodedb-cluster/src/rpc_codec/vshard.rs b/nodedb-cluster/src/rpc_codec/vshard.rs index 4c03a0d1e..45c4bd98b 100644 --- a/nodedb-cluster/src/rpc_codec/vshard.rs +++ b/nodedb-cluster/src/rpc_codec/vshard.rs @@ -144,4 +144,27 @@ mod tests { other => panic!("expected the untyped error, got {other:?}"), } } + + /// A shard's typed execution error crosses the wire with its typed form. + #[test] + fn a_shard_execution_error_survives_the_wire() { + let typed = crate::rpc_codec::TypedClusterError::Internal { + code: 2000, + message: "permission denied on orders".into(), + }; + let error = ClusterError::ShardExecution { + error: Box::new(typed), + detail: "array put: permission denied on orders".into(), + }; + match round_trip(error) { + ClusterError::ShardExecution { error, detail } => { + assert!(matches!( + *error, + crate::rpc_codec::TypedClusterError::Internal { code: 2000, .. } + )); + assert_eq!(detail, "array put: permission denied on orders"); + } + other => panic!("expected the shard execution error, got {other:?}"), + } + } } diff --git a/nodedb-types/src/error/sqlstate.rs b/nodedb-types/src/error/sqlstate.rs index afb953658..2845ea99e 100644 --- a/nodedb-types/src/error/sqlstate.rs +++ b/nodedb-types/src/error/sqlstate.rs @@ -131,6 +131,16 @@ pub const INVALID_AUTHORIZATION: &str = "28000"; /// block. pub const ACTIVE_SQL_TRANSACTION: &str = "25001"; +/// `25006` — `read_only_sql_transaction`: a write targeted a read-only +/// database, such as an unpromoted mirror. +pub const READ_ONLY_SQL_TRANSACTION: &str = "25006"; + +// ── Class 2B — Dependent Privilege Descriptors Still Exist ─────────────────── + +/// `2BP01` — `dependent_objects_still_exist`: a DROP names an object that +/// other objects depend on. +pub const DEPENDENT_OBJECTS_STILL_EXIST: &str = "2BP01"; + // ── Class 3D — Invalid Catalog Name ────────────────────────────────────────── /// `3D000` — `invalid_catalog_name` (the selected database does not exist) @@ -381,6 +391,9 @@ mod tests { LEGAL_HOLD_ACTIVE, TYPE_GUARD_VIOLATION, INVALID_AUTHORIZATION, + ACTIVE_SQL_TRANSACTION, + READ_ONLY_SQL_TRANSACTION, + DEPENDENT_OBJECTS_STILL_EXIST, SERIALIZATION_FAILURE, INSUFFICIENT_PRIVILEGE, SYNTAX_ERROR, diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index 3493782b5..ababf8ba6 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -305,8 +305,114 @@ impl From for ErrorCode { // Already a Data-Plane verdict: hand back the same code rather // than re-wrapping it as `Internal` and losing its SQLSTATE. crate::Error::DataPlane(code) => code, - other => Self::Internal { - detail: other.to_string(), + // Class `22`, the class the Control Plane gives both. + e @ (crate::Error::OffsetRegression { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::InvalidLimitValue { .. }) => Self::DataException { + detail: e.to_string(), + }, + // Class `40`: the client retries the statement. + e @ (crate::Error::CalvinParticipantError + | crate::Error::RetryableSchemaChanged { .. }) => Self::RetryableRefusal { + reason: e.to_string(), + }, + crate::Error::CrdtAdmissionRetriesExhausted { .. } => Self::ConflictRetry, + // Retryable refusals whose class (`55P03`) no Data-Plane code has. + // The retry contract survives: nothing was applied. + e @ (crate::Error::NoLeader { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::StaleReadNotLeader { .. }) => Self::RetryableRefusal { + reason: e.to_string(), + }, + // Class `57`: the client retries once the leader settles. + e @ crate::Error::NotLeader { .. } => Self::DispatchCapacity { + reason: e.to_string(), + }, + crate::Error::CrdtAdmissionTimeout { .. } => Self::DeadlineExceeded, + e @ crate::Error::VShardAdmissionCapacityExceeded { .. } => Self::RateExceeded { + gate: e.to_string(), + retry_after_ms: 0, + }, + // Class `53`: a configured resource ceiling. + crate::Error::QuotaOvercommit { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } => Self::ResourcesExhausted, + // Class `28` has no Data-Plane code. The nearest is the access + // refusal, which keeps it a client error the client cannot retry. + e @ (crate::Error::BackupKeyMismatch | crate::Error::SessionTokenExpired) => { + Self::RejectedAuthz { + resource: e.to_string(), + } + } + // Client errors of class `42`, and client errors whose class + // (`25`, `2B`, `55`) no Data-Plane code has. `BadRequest` is the + // class their public code has. + e @ (crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::DependentObjectsExist { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Promql(_) + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. }) => Self::BadRequest { + detail: e.to_string(), + }, + // Retry exhaustion takes the code of its cause. + crate::Error::OllpExhausted { cause, .. } => match cause { + crate::OllpExhaustedCause::PredicateDrift => Self::ConflictRetry, + crate::OllpExhaustedCause::PreAdmission(inner) => Self::from(*inner), + crate::OllpExhaustedCause::AdmissionRefused { detail } => Self::RateExceeded { + gate: detail, + retry_after_ms: 0, + }, + }, + // Server-side faults and system defects. `Shaping` and + // `RemoteTyped` carry a public numeric code that has no + // Data-Plane twin, and neither is raised on the Data Plane. + e @ (crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::CascadeCycle { .. }) => Self::Internal { + detail: e.to_string(), }, } } diff --git a/nodedb/src/control/backup/restore/orchestrate/restore.rs b/nodedb/src/control/backup/restore/orchestrate/restore.rs index 18d076ed5..5055fbaad 100644 --- a/nodedb/src/control/backup/restore/orchestrate/restore.rs +++ b/nodedb/src/control/backup/restore/orchestrate/restore.rs @@ -109,17 +109,21 @@ pub async fn restore_tenant( // applier's own register hook and a later boot seed are both // idempotent with it. A registration failure fails the restore. for coll in &restored_collections { - // A Data-Plane verdict keeps its code. + // A classified error keeps its class. Only a machinery failure + // gains the restore context. dispatch_register_from_stored(state, coll) .await - .map_err(|e| match e { - Error::DataPlane(_) => e, - other => Error::Internal { - detail: format!( - "restore: Data Plane registration of collection '{}' failed: {other}", - coll.name - ), - }, + .map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + Error::Internal { + detail: format!( + "restore: Data Plane registration of collection '{}' failed: {e}", + coll.name + ), + } + } else { + e + } })?; } } diff --git a/nodedb/src/control/backup/restore/redo_reissue/commit.rs b/nodedb/src/control/backup/restore/redo_reissue/commit.rs index 3021a300d..bce611c46 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/commit.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/commit.rs @@ -144,15 +144,17 @@ pub(super) async fn commit_collection( let mut records = 0usize; for batch in batch_units(units) { let payload = batch_payload(&collection, batch); - // A Data-Plane verdict keeps its code. - commit_record(state, target, &payload) - .await - .map_err(|e| match e { - crate::Error::DataPlane(_) => e, - other => crate::Error::Internal { - detail: format!("restore: re-issuing rows of '{collection}' failed: {other}"), - }, - })?; + // A classified error keeps its class. Only a machinery failure gains + // the restore context. + commit_record(state, target, &payload).await.map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + crate::Error::Internal { + detail: format!("restore: re-issuing rows of '{collection}' failed: {e}"), + } + } else { + e + } + })?; records += 1; } Ok(records) diff --git a/nodedb/src/control/cluster/array_cluster_helpers.rs b/nodedb/src/control/cluster/array_cluster_helpers.rs index a982fd85d..482467674 100644 --- a/nodedb/src/control/cluster/array_cluster_helpers.rs +++ b/nodedb/src/control/cluster/array_cluster_helpers.rs @@ -5,6 +5,7 @@ //! fast path. use nodedb_cluster::distributed_array::merge::ArrayAggPartial; +use nodedb_cluster::error::ClusterError; use nodedb_cluster::wire::VShardMessageType; use crate::Error; @@ -94,20 +95,25 @@ pub(super) fn finalize_agg_partials( rows } -pub(super) fn cluster_err(e: nodedb_cluster::error::ClusterError) -> Error { +pub(super) fn cluster_err(e: ClusterError) -> Error { match e { // A shard did not answer within its timeout: surface as a deterministic // deadline rather than an opaque internal error, matching the // `TypedClusterError::DeadlineExceeded` mapping used elsewhere. - nodedb_cluster::error::ClusterError::ShardTimeout { .. } => Error::DeadlineExceeded { + ClusterError::ShardTimeout { .. } => Error::DeadlineExceeded { request_id: crate::types::RequestId::new(0), }, // A shard's Data-Plane verdict keeps its code, so the statement // renders the SQLSTATE a single-node execution renders. - nodedb_cluster::error::ClusterError::DataPlane { code } => Error::DataPlane(code.into()), + ClusterError::DataPlane { code } => Error::DataPlane(code.into()), + // A shard's typed execution error is rebuilt, so the statement + // renders the SQLSTATE a single-node execution renders. + ClusterError::ShardExecution { error, .. } | ClusterError::StreamTerminal { error, .. } => { + Error::from(*error) + } // The shard still refused after the fan-out's reroute retry. The // vShard's owner is moving, so the client retries the statement. - nodedb_cluster::error::ClusterError::WrongOwner { + ClusterError::WrongOwner { vshard_id, expected_owner_node, } => match expected_owner_node { @@ -120,7 +126,44 @@ pub(super) fn cluster_err(e: nodedb_cluster::error::ClusterError) -> Error { vshard_id: crate::types::VShardId::new(vshard_id), }, }, - other => Error::Internal { + // The vShard is moving to another node. It has no serving owner + // until the cut-over, so the client retries the statement. + ClusterError::MigrationInProgress { vshard_id } => Error::NoLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + }, + // Cluster machinery faults. The client can act on none of them. + other @ (ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. }) => Error::Internal { detail: format!("array cluster: {other}"), }, } @@ -187,4 +230,31 @@ mod tests { }; assert!(matches!(cluster_err(unknown), Error::NoLeader { .. })); } + + /// A vShard mid-migration has no serving owner, so the statement answers + /// the retryable no-leader class, never `Internal`. + #[test] + fn a_migrating_vshard_is_a_no_leader_error() { + let wire = ClusterError::MigrationInProgress { vshard_id: 5 }; + match cluster_err(wire) { + Error::NoLeader { vshard_id } => assert_eq!(vshard_id.as_u32(), 5), + other => panic!("expected NoLeader, got {other:?}"), + } + } + + /// A typed terminal error is rebuilt, never flattened to `Internal`. + #[test] + fn a_typed_terminal_error_is_rebuilt() { + let typed = nodedb_cluster::rpc_codec::TypedClusterError::DataPlane { + code: ErrorCode::DivisionByZero.into(), + }; + let wire = ClusterError::StreamTerminal { + error: Box::new(typed), + detail: "division by zero".into(), + }; + match cluster_err(wire) { + Error::DataPlane(code) => assert_eq!(code, ErrorCode::DivisionByZero), + other => panic!("expected the typed verdict, got {other:?}"), + } + } } diff --git a/nodedb/src/control/cluster/array_executor/refusal.rs b/nodedb/src/control/cluster/array_executor/refusal.rs index 027616950..c34d041c5 100644 --- a/nodedb/src/control/cluster/array_executor/refusal.rs +++ b/nodedb/src/control/cluster/array_executor/refusal.rs @@ -5,8 +5,8 @@ //! A coded refusal crosses as `ClusterError::DataPlane`, so the coordinator //! rebuilds `crate::Error::DataPlane(code)` and renders the SQLSTATE a //! single-node execution renders. Only a refusal with no code is a storage -//! error. A local-execution error keeps its class where the cluster wire has -//! one. +//! error. A local-execution error crosses in its typed wire form and keeps +//! its class. use nodedb_cluster::error::ClusterError; use nodedb_cluster::rpc_codec::DataPlaneErrorCode; @@ -31,7 +31,8 @@ pub(super) fn refusal_error(context: &str, response: &Response) -> ClusterError /// - A deadline and a capacity refusal cross as their Data-Plane verdicts. /// - A missing leader crosses as `WrongOwner`, so the coordinator re-reads /// its routing and retries. -/// - Every other error is a storage error with `context` before its message. +/// - Every other error crosses as `ShardExecution` with its typed wire form. +/// `context` goes before its message in the log detail. pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterError { match error { crate::Error::DataPlane(code) => ClusterError::DataPlane { code: code.into() }, @@ -55,9 +56,119 @@ pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterErro vshard_id: vshard_id.as_u32(), expected_owner_node: None, }, - other => ClusterError::Storage { - detail: format!("{context}: {other}"), - }, + // Every other error crosses in its typed wire form, so the + // coordinator rebuilds it and renders its own SQLSTATE. + other @ (crate::Error::RejectedConstraint { .. } + | crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::RejectedAuthz { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RejectedPrevalidation { .. } + | crate::Error::RetryableRefusal { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::TypeMismatch { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::RateExceeded { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::SourceFrozen { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BadRequest { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::Backpressure { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::OllpExhausted { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::StaleReadNotLeader { .. }) => { + let detail = format!("{context}: {other}"); + ClusterError::ShardExecution { + error: Box::new( + crate::control::cluster::data_plane_error_wire::execution_error_to_typed(other), + ), + detail, + } + } } } @@ -147,4 +258,57 @@ mod tests { other => panic!("expected the typed refusal, got {other:?}"), } } + + /// A classified error with no Data-Plane twin crosses in its typed wire + /// form, and the coordinator renders the SQLSTATE a single-node + /// execution renders. + #[test] + fn a_classified_error_keeps_its_sqlstate_at_the_coordinator() { + use crate::control::cluster::array_cluster_helpers::cluster_err; + use crate::control::server::pgwire::types::error_to_sqlstate; + use nodedb_cluster::rpc_codec::ShardErrorWire; + + let local = || crate::Error::RejectedAuthz { + tenant_id: crate::types::TenantId::new(1), + resource: "grid".into(), + }; + let shard = execution_error("array put", local()); + match &shard { + ClusterError::ShardExecution { detail, .. } => { + assert!(detail.starts_with("array put: "), "{detail}"); + } + other => panic!("expected a typed shard execution error, got {other:?}"), + } + let received = ClusterError::from(ShardErrorWire::from(shard)); + let rebuilt = cluster_err(received); + assert_eq!(error_to_sqlstate(&rebuilt).1, error_to_sqlstate(&local()).1); + } + + /// Every variant keeps its SQLSTATE class through the array shard hop, + /// except those whose class no public numeric code carries. + #[test] + fn every_variant_keeps_its_class_through_the_array_hop() { + use crate::control::cluster::array_cluster_helpers::cluster_err; + use crate::control::gateway::error_map::class_parity::{ + error_samples, error_variant_index, + }; + use crate::control::server::pgwire::types::error_to_sqlstate; + use nodedb_cluster::rpc_codec::ShardErrorWire; + + let gaps = [7, 32, 33, 88, 90]; + for (err, twin) in error_samples().into_iter().zip(error_samples()) { + if gaps.contains(&error_variant_index(&err)) { + continue; + } + let (_, local, _) = error_to_sqlstate(&err); + let wire = ShardErrorWire::from(execution_error("array put", twin)); + let rebuilt = cluster_err(ClusterError::from(wire)); + let (_, remote, _) = error_to_sqlstate(&rebuilt); + assert_eq!( + remote.get(..2), + local.get(..2), + "{err:?} is {local} locally but {remote} at the coordinator" + ); + } + } } diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 9c5ce4f8d..002260bfa 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -51,14 +51,124 @@ pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { reason: capacity.to_string(), }, }, - other => { - let message = other.to_string(); - let code = u32::from(nodedb_types::error::NodeDbError::from(other).code().0); - TypedClusterError::Internal { code, message } - } + // Every other error crosses as its public numeric code and message. + // The coordinator renders the SQLSTATE that code maps to. + other @ (crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::RejectedAuthz { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RejectedPrevalidation { .. } + | crate::Error::RetryableRefusal { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::TypeMismatch { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::RateExceeded { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::NoLeader { .. } + | crate::Error::NotLeader { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::SourceFrozen { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BadRequest { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::Backpressure { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::OllpExhausted { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::StaleReadNotLeader { .. }) => numeric_typed(other), } } +/// The wire error for a local error with no typed wire carrier: its public +/// numeric code from `NodeDbError::from(err).code()`, and its message. The +/// coordinator rebuilds it as `Error::RemoteTyped`. +pub(crate) fn numeric_typed(err: crate::Error) -> TypedClusterError { + let message = err.to_string(); + let code = u32::from(nodedb_types::error::NodeDbError::from(err).code().0); + TypedClusterError::Internal { code, message } +} + /// Widen a pointer-width count to the wire's fixed `u64`. fn to_wire_count(value: usize) -> u64 { value as u64 diff --git a/nodedb/src/control/cluster/metadata_applier/wedge.rs b/nodedb/src/control/cluster/metadata_applier/wedge.rs index f27ca69a5..6e40355d3 100644 --- a/nodedb/src/control/cluster/metadata_applier/wedge.rs +++ b/nodedb/src/control/cluster/metadata_applier/wedge.rs @@ -62,7 +62,109 @@ pub fn classify(error: &crate::Error) -> ApplyFailureClass { crate::Error::BadRequest { .. } | crate::Error::TypeMismatch { .. } => { ApplyFailureClass::Permanent } - _ => ApplyFailureClass::Transient, + // Not provably a pure function of the entry and persisted state. + crate::Error::RejectedConstraint { .. } + | crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::RejectedAuthz { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::DeadlineExceeded { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RejectedPrevalidation { .. } + | crate::Error::RetryableRefusal { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::RateExceeded { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::NoLeader { .. } + | crate::Error::NotLeader { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::SourceFrozen { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::DispatchCapacity { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::Backpressure { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::DataPlane(_) + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::OllpExhausted { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::StaleReadNotLeader { .. } => ApplyFailureClass::Transient, } } diff --git a/nodedb/src/control/cluster/read_index.rs b/nodedb/src/control/cluster/read_index.rs index c4e91dd60..28a93ed20 100644 --- a/nodedb/src/control/cluster/read_index.rs +++ b/nodedb/src/control/cluster/read_index.rs @@ -85,8 +85,45 @@ fn refusal_of(error: ClusterError) -> ReadIndexRefusal { match error { ClusterError::ReadIndexTimeout { waited_ms, .. } => ReadIndexRefusal::Timeout { waited_ms }, // Not hosted here, not leading, leadership lost mid-probe, or the - // leader unreachable: the caller asks again later. - _ => ReadIndexRefusal::NotLeader, + // leader unreachable: the caller asks again later. Every other error + // also leaves this node unable to prove its leadership now. + ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::ShardTimeout { .. } + | ClusterError::StreamTerminal { .. } + | ClusterError::Storage { .. } + | ClusterError::DataPlane { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. } + | ClusterError::ShardExecution { .. } => ReadIndexRefusal::NotLeader, } } diff --git a/nodedb/src/control/cluster/start_raft/mod.rs b/nodedb/src/control/cluster/start_raft/mod.rs index da76d588b..b2fc86371 100644 --- a/nodedb/src/control/cluster/start_raft/mod.rs +++ b/nodedb/src/control/cluster/start_raft/mod.rs @@ -22,6 +22,7 @@ mod group_setup; mod hooks; mod loop_build; mod observability; +mod propose_error; mod proposer_wiring; pub use core::start_raft; diff --git a/nodedb/src/control/cluster/start_raft/propose_error.rs b/nodedb/src/control/cluster/start_raft/propose_error.rs new file mode 100644 index 000000000..ce7aa5673 --- /dev/null +++ b/nodedb/src/control/cluster/start_raft/propose_error.rs @@ -0,0 +1,123 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The error an async Raft propose returns to its statement. + +use nodedb_cluster::ClusterError; +use nodedb_raft::RaftError; + +use crate::types::VShardId; + +/// The error an async propose returns for a cluster error. +/// +/// A group with no leader to take the proposal right now accepts the same +/// proposal once it has one, so the proposal is retried: +/// [`crate::Error::NoLeader`]. That covers an election, a leadership transfer +/// in flight, a leader that stepped down after this node or a forwarding node +/// chose it, and a vShard whose owner is moving. A forwarded refusal arrives +/// here with its typed Raft error (`DataProposeResponse::refusal_error`). A +/// typed verdict keeps its class. Every other failure is final here. +pub(super) fn async_propose_error(vshard_id: u32, error: ClusterError) -> crate::Error { + match error { + ClusterError::Raft( + RaftError::LeadershipTransferInProgress | RaftError::NotLeader { .. }, + ) + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::WrongOwner { .. } => crate::Error::NoLeader { + vshard_id: VShardId::new(vshard_id), + }, + // The forward did not answer before its timeout. The statement's + // deadline class, the same class the array fan-out gives it. + ClusterError::ShardTimeout { .. } => crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + }, + ClusterError::DataPlane { code } => crate::Error::DataPlane(code.into()), + ClusterError::ShardExecution { error, .. } | ClusterError::StreamTerminal { error, .. } => { + crate::Error::from(*error) + } + other @ (ClusterError::Raft( + RaftError::LogCompacted { .. } + | RaftError::CompactionAheadOfApplied { .. } + | RaftError::ProposalRejected { .. } + | RaftError::InvalidTransferTarget { .. } + | RaftError::GroupNotFound { .. } + | RaftError::Transport { .. } + | RaftError::Storage { .. } + | RaftError::Serialization { .. } + | RaftError::SnapshotFormat { .. } + | RaftError::Shutdown, + ) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. }) => crate::Error::Internal { + detail: format!("raft propose (async): {other}"), + }, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_missing_leader_is_retryable() { + let error = ClusterError::Raft(RaftError::NotLeader { leader_hint: None }); + assert!(matches!( + async_propose_error(3, error), + crate::Error::NoLeader { .. } + )); + } + + /// A moving vShard has no owner to take the proposal until the + /// cut-over, so the statement answers the retryable no-leader class. + #[test] + fn a_moving_vshard_is_retryable() { + let error = ClusterError::WrongOwner { + vshard_id: 3, + expected_owner_node: None, + }; + assert!(matches!( + async_propose_error(3, error), + crate::Error::NoLeader { .. } + )); + } + + #[test] + fn a_forward_timeout_is_a_deadline() { + let error = ClusterError::ShardTimeout { + vshard_id: 3, + elapsed_ms: 50, + }; + assert!(matches!( + async_propose_error(3, error), + crate::Error::DeadlineExceeded { .. } + )); + } +} diff --git a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs index 64219b99b..d0023522a 100644 --- a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs +++ b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs @@ -15,6 +15,7 @@ use crate::control::distributed_applier::{ApplyBatch, ProposeTracker, run_apply_ use crate::control::state::SharedState; use super::loop_build::RaftLoopType; +use super::propose_error::async_propose_error; /// Install the sync `raft_proposer` / `raft_compactor` / /// `raft_applied_index_sink`, the async `async_raft_proposer`, and spawn the @@ -295,28 +296,6 @@ fn apply_progress(state: Option<&SharedState>, group_id: u64) -> String { } } -/// The error an async propose that reached no leader returns. -/// -/// A group with no leader to take the proposal right now accepts the same -/// proposal once it has one, so the proposal is retried: -/// [`crate::Error::NoLeader`]. That covers an election, a leadership transfer -/// in flight, and a leader that stepped down after this node or a forwarding -/// node chose it; a forwarded refusal arrives here with its typed Raft error -/// (`DataProposeResponse::refusal_error`). Every other failure is final here. -fn async_propose_error(vshard_id: u32, error: nodedb_cluster::ClusterError) -> crate::Error { - match error { - nodedb_cluster::ClusterError::Raft( - nodedb_raft::RaftError::LeadershipTransferInProgress - | nodedb_raft::RaftError::NotLeader { .. }, - ) => crate::Error::NoLeader { - vshard_id: crate::types::VShardId::new(vshard_id), - }, - other => crate::Error::Internal { - detail: format!("raft propose (async): {other}"), - }, - } -} - /// The error a proposal returns once the caller's statement deadline passed. /// /// The proposer carries no request id, so the error names request 0. diff --git a/nodedb/src/control/gateway/dispatch_remote.rs b/nodedb/src/control/gateway/dispatch_remote.rs index 99aeaa999..bdf0d253e 100644 --- a/nodedb/src/control/gateway/dispatch_remote.rs +++ b/nodedb/src/control/gateway/dispatch_remote.rs @@ -11,6 +11,7 @@ use std::sync::Arc; use futures::StreamExt; +use nodedb_cluster::ClusterError; use nodedb_cluster::rpc_codec::{ExecuteRequest, RaftRpc}; use tracing::debug; @@ -345,20 +346,58 @@ pub(super) async fn dispatch_remote_stream( Ok(Box::pin(head.chain(rest))) } -/// Map a pre-row [`nodedb_cluster::ClusterError`] from a streaming dispatch to a -/// retryable internal [`Error`]. +/// Map a pre-row [`nodedb_cluster::ClusterError`] from a streaming dispatch to +/// an [`Error`]. /// -/// A `StreamTerminal` carrying a typed `NotLeader` / `DescriptorMismatch` maps -/// through the same [`map_typed_cluster_error`] used by the one-shot path so the -/// gateway retry loop handles it identically. Any other cluster error becomes a -/// transport-style `NotLeader` (leader_node = 0) so the next attempt re-resolves -/// routing rather than re-entrenching an unreachable node. -fn map_stream_cluster_error(err: nodedb_cluster::ClusterError, vshard_id: u64) -> Error { +/// A typed error (`StreamTerminal`, `ShardExecution`) maps through the same +/// [`map_typed_cluster_error`] used by the one-shot path, so the gateway retry +/// loop handles it identically. A Data-Plane verdict keeps its code. Any other +/// cluster error becomes a transport-style `NotLeader` (leader_node = 0), so +/// the next attempt re-resolves routing rather than re-entrenching an +/// unreachable node. +fn map_stream_cluster_error(err: ClusterError, vshard_id: u64) -> Error { match err { - nodedb_cluster::ClusterError::StreamTerminal { error, .. } => { + ClusterError::StreamTerminal { error, .. } | ClusterError::ShardExecution { error, .. } => { map_typed_cluster_error(*error, vshard_id) } - other => Error::NotLeader { + // A verdict from a shard that answered. Retrying it on another route + // repeats it, so it keeps its SQLSTATE. + ClusterError::DataPlane { code } => Error::DataPlane(code.into()), + other @ (ClusterError::Raft(_) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::ShardTimeout { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. }) => Error::NotLeader { vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: format!("stream dispatch error: {other}"), diff --git a/nodedb/src/control/gateway/dispatcher.rs b/nodedb/src/control/gateway/dispatcher.rs index fbc989a24..3996c5eee 100644 --- a/nodedb/src/control/gateway/dispatcher.rs +++ b/nodedb/src/control/gateway/dispatcher.rs @@ -467,7 +467,10 @@ pub(super) fn map_typed_cluster_error(err: TypedClusterError, vshard_id: u64) -> constraint, detail, }, - TypedClusterError::Internal { message, .. } => Error::Internal { detail: message }, + // A numeric class crosses as `Error::RemoteTyped`, so the client sees + // the SQLSTATE the executing node gave it. Only a code of 0 (no class) + // decodes as `Error::Internal`. + internal @ TypedClusterError::Internal { .. } => Error::from(internal), } } @@ -515,6 +518,21 @@ mod tests { } } + /// A remote error with a numeric class keeps it, never `Internal`. + #[test] + fn map_internal_keeps_its_numeric_class() { + let err = TypedClusterError::Internal { + code: u32::from(nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED.0), + message: "permission denied on orders".into(), + }; + match map_typed_cluster_error(err, 0) { + Error::RemoteTyped { code, .. } => { + assert_eq!(code, nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED); + } + other => panic!("expected RemoteTyped, got {other:?}"), + } + } + #[test] fn map_deadline_exceeded() { let err = TypedClusterError::DeadlineExceeded { elapsed_ms: 100 }; diff --git a/nodedb/src/control/gateway/error_map/class_parity.rs b/nodedb/src/control/gateway/error_map/class_parity.rs index 3c73f3621..c39b58c62 100644 --- a/nodedb/src/control/gateway/error_map/class_parity.rs +++ b/nodedb/src/control/gateway/error_map/class_parity.rs @@ -383,7 +383,7 @@ const ERROR_VARIANT_COUNT: usize = 108; /// fails to compile here until it gets an index, and /// [`every_error_variant_has_a_sample`] then fails until /// [`error_samples`] carries it. -fn error_variant_index(err: &crate::Error) -> usize { +pub(crate) fn error_variant_index(err: &crate::Error) -> usize { use crate::Error as E; match err { E::RejectedConstraint { .. } => 0, @@ -498,7 +498,7 @@ fn error_variant_index(err: &crate::Error) -> usize { } /// One sample per `crate::Error` variant. -fn error_samples() -> Vec { +pub(crate) fn error_samples() -> Vec { use crate::Error as E; use crate::types::{DatabaseId, RequestId, TenantId, VShardId}; @@ -865,3 +865,119 @@ fn every_error_variant_has_the_http_status_of_its_sqlstate() { ); } } + +/// Variants whose pgwire SQLSTATE class has no public numeric code: `25` +/// (`CrdtApplyForbiddenInTransaction`, `NotInTransactionBlock`, +/// `CrossShardInExplicitTransaction`), `2B` (`DependentObjectsExist`), and +/// `40000` (`CalvinParticipantError`, whose code is deliberately not a write +/// conflict). The numeric wire form cannot carry their class. +const NUMERIC_CLASS_GAPS: [usize; 5] = [7, 32, 33, 88, 90]; + +/// Every `crate::Error` variant renders the SQLSTATE class it renders locally +/// after it crosses a node hop, through both wire encoders and the decoder. +#[test] +fn every_error_variant_keeps_its_class_across_a_node_hop() { + use nodedb_cluster::rpc_codec::TypedClusterError; + + use crate::control::cluster::data_plane_error_wire::execution_error_to_typed; + + let encoders: [(&str, fn(crate::Error) -> TypedClusterError); 2] = [ + ("execution_error_to_typed", execution_error_to_typed), + ("From", TypedClusterError::from), + ]; + for (name, encode) in encoders { + for (err, twin) in error_samples().into_iter().zip(error_samples()) { + if NUMERIC_CLASS_GAPS.contains(&error_variant_index(&err)) { + continue; + } + let (_, local, _) = error_to_sqlstate(&err); + let rebuilt = crate::Error::from(encode(twin)); + let (_, remote, _) = error_to_sqlstate(&rebuilt); + assert_eq!( + class(remote), + class(local), + "{name}: {err:?} is {local} locally but {remote} after the hop as {rebuilt:?}" + ); + } + } +} + +/// The SQLSTATE each Control-Plane variant renders where it has a class of +/// its own, pinned by variant index. +fn classified_sqlstates() -> Vec<(usize, &'static str)> { + vec![ + (3, sqlstate::INVALID_PARAMETER_VALUE), + (27, sqlstate::TOO_MANY_CONNECTIONS), + (28, sqlstate::SERIALIZATION_FAILURE), + (29, sqlstate::SYNTAX_ERROR), + (30, sqlstate::SYNTAX_ERROR), + (31, sqlstate::SYNTAX_ERROR), + (32, sqlstate::ACTIVE_SQL_TRANSACTION), + (34, sqlstate::QUERY_CANCELED), + (44, sqlstate::QUOTA_OVERCOMMIT), + (62, sqlstate::SYNTAX_ERROR), + (63, sqlstate::SYNTAX_ERROR), + (87, sqlstate::SYNTAX_ERROR), + (88, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), + (90, sqlstate::ACTIVE_SQL_TRANSACTION), + (91, sqlstate::SYNTAX_ERROR), + (92, sqlstate::SYNTAX_ERROR), + (93, sqlstate::SYNTAX_ERROR), + (95, sqlstate::SYNTAX_ERROR), + (96, sqlstate::SYNTAX_ERROR), + (97, sqlstate::SYNTAX_ERROR), + (98, sqlstate::SYNTAX_ERROR), + (99, sqlstate::SYNTAX_ERROR), + (100, sqlstate::SYNTAX_ERROR), + (101, sqlstate::QUOTA_EXCEEDED), + (102, sqlstate::QUOTA_EXCEEDED), + (103, sqlstate::SYNTAX_ERROR), + (104, sqlstate::SYNTAX_ERROR), + (106, sqlstate::READ_ONLY_SQL_TRANSACTION), + (107, sqlstate::STALE_READ_NOT_LEADER), + ] +} + +/// A client-facing Control-Plane variant renders its own SQLSTATE, never the +/// internal-error default. +#[test] +fn client_facing_variants_render_their_own_sqlstate() { + let expected = classified_sqlstates(); + let mut seen = 0; + for err in error_samples() { + let index = error_variant_index(&err); + if let Some((_, state)) = expected.iter().find(|(i, _)| *i == index) { + assert_eq!(error_to_sqlstate(&err).1, *state, "{err:?}"); + seen += 1; + } + } + assert_eq!(seen, expected.len(), "a pinned variant has no sample"); +} + +/// A variant with a dedicated public code renders that code's class on the +/// numeric table too, so native and remote renderings agree with pgwire. +#[test] +fn dedicated_codes_render_the_class_of_their_variant() { + use nodedb_types::error::ErrorCode as Ec; + + assert_eq!( + numeric_code_to_sqlstate(Ec::QUOTA_OVERCOMMIT), + sqlstate::QUOTA_OVERCOMMIT + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TENANT_VECTOR_DIM_EXCEEDED), + sqlstate::QUOTA_EXCEEDED + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TENANT_GRAPH_DEPTH_EXCEEDED), + sqlstate::QUOTA_EXCEEDED + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::MIRROR_READ_ONLY), + sqlstate::READ_ONLY_SQL_TRANSACTION + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::STALE_READ_NOT_LEADER), + sqlstate::STALE_READ_NOT_LEADER + ); +} diff --git a/nodedb/src/control/gateway/error_map/mod.rs b/nodedb/src/control/gateway/error_map/mod.rs index 640544fe6..a704ab5c8 100644 --- a/nodedb/src/control/gateway/error_map/mod.rs +++ b/nodedb/src/control/gateway/error_map/mod.rs @@ -7,7 +7,7 @@ //! to its SQLSTATE / HTTP / RESP / native codes is a one-file edit. #[cfg(test)] -mod class_parity; +pub(crate) mod class_parity; mod gateway_map; mod http; mod native; diff --git a/nodedb/src/control/gateway/error_map/resp.rs b/nodedb/src/control/gateway/error_map/resp.rs index 5eb4ae095..dd2318e1f 100644 --- a/nodedb/src/control/gateway/error_map/resp.rs +++ b/nodedb/src/control/gateway/error_map/resp.rs @@ -46,7 +46,107 @@ impl GatewayErrorMap { public.message() ) } - _ => format!("ERR {err}"), + // Every other variant takes the prefix of its public code, the + // same prefix a remote rendering of it gets. + Error::TxnOverlayMemoryExceeded { .. } + | Error::OffsetRegression { .. } + | Error::ConflictRetry { .. } + | Error::CalvinSerializationConflict + | Error::CalvinParticipantError + | Error::RejectedPrevalidation { .. } + | Error::RetryableRefusal { .. } + | Error::AppendOnlyViolation { .. } + | Error::BalanceViolation { .. } + | Error::MaterializedSumTargetNotFound { .. } + | Error::MaterializedSumResolutionMissing { .. } + | Error::PeriodLocked { .. } + | Error::PeriodLockMisconfigured { .. } + | Error::RetentionViolation { .. } + | Error::LegalHoldActive { .. } + | Error::StateTransitionViolation { .. } + | Error::TransitionCheckViolation { .. } + | Error::TypeGuardViolation { .. } + | Error::InsufficientBalance { .. } + | Error::RateExceeded { .. } + | Error::DocumentNotFound { .. } + | Error::CollectionDeactivated { .. } + | Error::VShardAdmissionCapacityExceeded { .. } + | Error::CrdtAdmissionRetriesExhausted { .. } + | Error::CrdtAdmissionInvalidPlan { .. } + | Error::CrdtAdmissionCallerFence + | Error::CrdtApplyRequiresAdmission + | Error::CrdtApplyForbiddenInTransaction + | Error::NotInTransactionBlock { .. } + | Error::CrdtAdmissionTimeout { .. } + | Error::NoLeader { .. } + | Error::FanOutExceeded { .. } + | Error::CrossCollectionNotColocated { .. } + | Error::SourceFrozen { .. } + | Error::CloneWriteRequiresMaterialize { .. } + | Error::BackupTenantMismatch { .. } + | Error::BackupKeyMismatch + | Error::QuotaOvercommit { .. } + | Error::FeatureNotSupported { .. } + | Error::UndefinedFunction { .. } + | Error::UndefinedObject { .. } + | Error::ObjectNotInPrerequisiteState { .. } + | Error::UndefinedColumn { .. } + | Error::AmbiguousColumn { .. } + | Error::UnknownStrictField { .. } + | Error::DivisionByZero + | Error::DataException { .. } + | Error::InvalidLimitValue { .. } + | Error::RetryableLeaderChange { .. } + | Error::GroupQuorumUnavailable { .. } + | Error::GroupMarksUnavailable { .. } + | Error::MetadataLeaderUnavailable + | Error::AuthorizationStateBehind { .. } + | Error::ExecutionLimitExceeded { .. } + | Error::LimitExceeded { .. } + | Error::Wal(_) + | Error::Dispatch { .. } + | Error::Storage { .. } + | Error::ColdStorage { .. } + | Error::Serialization { .. } + | Error::Codec { .. } + | Error::SegmentCorrupted { .. } + | Error::MemoryExhausted { .. } + | Error::Backpressure { .. } + | Error::Crdt(_) + | Error::Io(_) + | Error::Config { .. } + | Error::Encryption { .. } + | Error::Bridge { .. } + | Error::VersionCompat { .. } + | Error::Internal { .. } + | Error::Shaping(_) + | Error::DescriptorVersionAnomaly { .. } + | Error::CollectionPurgeRowMissing { .. } + | Error::CatalogIntegrityViolation { .. } + | Error::Promql(_) + | Error::DependentObjectsExist { .. } + | Error::CascadeCycle { .. } + | Error::CrossShardInExplicitTransaction + | Error::SequencerUnavailable + | Error::SessionCapExceeded { .. } + | Error::SessionIdleTimeout + | Error::SessionTokenExpired + | Error::SessionKilledByAdmin + | Error::SessionUserDropped + | Error::OidcProviderTenantUnbound + | Error::OidcProviderTenantUnavailable { .. } + | Error::ExternalRoleUndefined { .. } + | Error::OidcNoDefaultDatabase { .. } + | Error::TenantVectorDimExceeded { .. } + | Error::TenantGraphDepthExceeded { .. } + | Error::RoleInheritanceCycle { .. } + | Error::RoleInheritanceDepthExceeded { .. } + | Error::OllpExhausted { .. } + | Error::MirrorReadOnly { .. } + | Error::StaleReadNotLeader { .. } => format!( + "{} {err}", + remote_code_to_resp_prefix(crate::error_classify::classify(err).code()) + ), } } } @@ -134,4 +234,14 @@ mod tests { let msg = GatewayErrorMap::to_resp(&err); assert_eq!(msg, "CONSTRAINT unique key clash"); } + + /// A variant with no arm of its own takes the prefix of its public code. + #[test] + fn resp_prefix_follows_the_public_code() { + let err = Error::CrdtAdmissionTimeout { + vshard_id: crate::types::VShardId::new(1), + timeout_ms: 10, + }; + assert!(GatewayErrorMap::to_resp(&err).starts_with("TIMEOUT ")); + } } diff --git a/nodedb/src/control/metadata_proposer/handle.rs b/nodedb/src/control/metadata_proposer/handle.rs index 608c3d33d..48bf80222 100644 --- a/nodedb/src/control/metadata_proposer/handle.rs +++ b/nodedb/src/control/metadata_proposer/handle.rs @@ -4,6 +4,9 @@ use std::sync::{Arc, Weak}; +use nodedb_cluster::ClusterError; +use nodedb_raft::RaftError; + use crate::error::Error; /// Type-erased handle for proposing to the metadata raft group. @@ -70,18 +73,77 @@ impl MetadataRaftHandle for RaftLoopProposerHandle { tokio::runtime::Handle::current() .block_on(raft_loop.propose_to_metadata_group_via_leader(bytes)) }) - .map_err(|e| match e { - // An election in progress is transient, not a failure of this - // proposal. Keep it typed rather than flattening it into a generic - // config error, so callers can wait the election out instead of - // failing the statement — a node that has just restarted answers - // every metadata proposal this way for a moment. - nodedb_cluster::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { - leader_hint: None, - }) => Error::MetadataLeaderUnavailable, - other => Error::Config { - detail: format!("metadata propose: {other}"), - }, - }) + .map_err(metadata_propose_error) + } +} + +/// The error a metadata proposal returns for a cluster error. +fn metadata_propose_error(error: ClusterError) -> Error { + match error { + // An election in progress is transient, not a failure of this + // proposal. Keep it typed rather than flattening it into a generic + // config error, so callers can wait the election out instead of + // failing the statement — a node that has just restarted answers + // every metadata proposal this way for a moment. + ClusterError::Raft(RaftError::NotLeader { leader_hint: None }) => { + Error::MetadataLeaderUnavailable + } + // A typed verdict keeps its class. + ClusterError::DataPlane { code } => Error::DataPlane(code.into()), + ClusterError::ShardExecution { error, .. } | ClusterError::StreamTerminal { error, .. } => { + Error::from(*error) + } + other @ (ClusterError::Raft( + RaftError::NotLeader { + leader_hint: Some(_), + } + | RaftError::LogCompacted { .. } + | RaftError::CompactionAheadOfApplied { .. } + | RaftError::ProposalRejected { .. } + | RaftError::InvalidTransferTarget { .. } + | RaftError::LeadershipTransferInProgress + | RaftError::GroupNotFound { .. } + | RaftError::Transport { .. } + | RaftError::Storage { .. } + | RaftError::Serialization { .. } + | RaftError::SnapshotFormat { .. } + | RaftError::Shutdown, + ) + | ClusterError::VShardNotMapped { .. } + | ClusterError::GroupNotFound { .. } + | ClusterError::LearnerNotCaughtUp { .. } + | ClusterError::MigrationInProgress { .. } + | ClusterError::MigrationPauseBudgetExceeded { .. } + | ClusterError::NodeUnreachable { .. } + | ClusterError::GhostNotFound { .. } + | ClusterError::Transport { .. } + | ClusterError::ShardTimeout { .. } + | ClusterError::Storage { .. } + | ClusterError::Codec { .. } + | ClusterError::UnsupportedWireVersion { .. } + | ClusterError::CircuitOpen { .. } + | ClusterError::JoinGroupDisappeared { .. } + | ClusterError::JoinCommitTimeout { .. } + | ClusterError::ReadIndexNotLeader { .. } + | ClusterError::ReadIndexTimeout { .. } + | ClusterError::Config { .. } + | ClusterError::MigrationCheckpoint(_) + | ClusterError::MigrationRecovery(_) + | ClusterError::WrongOwner { .. } + | ClusterError::Calvin(_) + | ClusterError::SnapshotCrcMismatch { .. } + | ClusterError::SnapshotOffsetRegression { .. } + | ClusterError::PartialSnapshotCorrupt { .. } + | ClusterError::PartialSnapshotCleanupFailed { .. } + | ClusterError::SnapshotApplyFailed { .. } + | ClusterError::Mirror(_) + | ClusterError::BspBarrier(_) + | ClusterError::VectorGather(_) + | ClusterError::SpatialGather(_) + | ClusterError::Bm25Gather(_) + | ClusterError::TsGather(_) + | ClusterError::RemoteUntyped { .. }) => Error::Config { + detail: format!("metadata propose: {other}"), + }, } } diff --git a/nodedb/src/control/planner/plan_error_map.rs b/nodedb/src/control/planner/plan_error_map.rs index 796b2fc3d..29f7dccfd 100644 --- a/nodedb/src/control/planner/plan_error_map.rs +++ b/nodedb/src/control/planner/plan_error_map.rs @@ -37,7 +37,7 @@ pub(crate) fn map_plan_error( } // A per-row sequence accessor and a search function outside its // search plan are refusals, not syntax errors, so they keep SQLSTATE - // `0A000` rather than the `42601` the fallback gives. + // `0A000` rather than the syntax class `42601`. nodedb_sql::SqlError::SequencePerRowUnsupported { .. } | nodedb_sql::SqlError::SearchFunctionOutsideSearch { .. } => { crate::Error::FeatureNotSupported { @@ -66,8 +66,94 @@ pub(crate) fn map_plan_error( // A target/expression count mismatch is a syntax error in PostgreSQL, // so it renders 42601 through `BadRequest`. nodedb_sql::SqlError::Arity { detail } => crate::Error::BadRequest { detail }, - other => crate::Error::PlanError { + // Refusals of a constraint or clause NodeDB does not implement. The + // DDL router renders both as `0A000`, so the planner path does too. + nodedb_sql::SqlError::UnsupportedConstraint { .. } + | nodedb_sql::SqlError::ConflictingEngineClause { .. } => { + crate::Error::FeatureNotSupported { + detail: error.to_string(), + } + } + // A value out of range for its type is a data exception (class `22`), + // the class PostgreSQL and the DDL DEFAULT gate give it. + nodedb_sql::SqlError::ConstantOverflow { .. } + | nodedb_sql::SqlError::IntegerOutOfRange { .. } + | nodedb_sql::SqlError::FloatOutOfRange { .. } => crate::Error::DataException { + detail: error.to_string(), + }, + // The executor's recursion cap: the program-limit class (`54000`) the + // Data-Plane verdict for the same condition renders. + nodedb_sql::SqlError::RecursionDepthExceeded { + cte_name, + max_depth, + } => crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::RecursionDepthExceeded { + cte_name, + max_depth, + }), + // Statement errors the client must fix: the syntax class `42601`. + other @ (nodedb_sql::SqlError::Parse { .. } + | nodedb_sql::SqlError::TypeMismatch { .. } + | nodedb_sql::SqlError::Unsupported { .. } + | nodedb_sql::SqlError::UnevaluableDefault { .. } + | nodedb_sql::SqlError::SetvalInColumnDefault { .. } + | nodedb_sql::SqlError::InvalidFunction { .. } + | nodedb_sql::SqlError::InvalidWindowFrame { .. } + | nodedb_sql::SqlError::MissingField { .. } + | nodedb_sql::SqlError::InsertColumnArityMismatch { .. } + | nodedb_sql::SqlError::PositionalKvInsertUnsupported { .. } + | nodedb_sql::SqlError::InvalidIdentifier { .. } + | nodedb_sql::SqlError::ReservedIdentifier { .. } + | nodedb_sql::SqlError::InvalidRecursiveSetOp { .. } + | nodedb_sql::SqlError::InvalidRecursiveSelfRef { .. } + | nodedb_sql::SqlError::RecursiveColumnMismatch { .. } + | nodedb_sql::SqlError::DuplicateRecursiveColumn { .. }) => crate::Error::PlanError { detail: other.to_string(), }, } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::types::TenantId; + + #[test] + fn a_value_out_of_range_is_a_data_exception() { + let error = nodedb_sql::SqlError::IntegerOutOfRange { + column: "qty".into(), + value: 1 << 40, + declared_type: "integer", + }; + match map_plan_error(error, TenantId::new(1)) { + crate::Error::DataException { detail } => assert!(detail.contains("out of range")), + other => panic!("expected a data exception, got {other:?}"), + } + } + + #[test] + fn an_unsupported_constraint_is_feature_not_supported() { + let error = nodedb_sql::SqlError::UnsupportedConstraint { + feature: "EXCLUDE".into(), + hint: "use a unique index".into(), + }; + assert!(matches!( + map_plan_error(error, TenantId::new(1)), + crate::Error::FeatureNotSupported { .. } + )); + } + + #[test] + fn a_recursion_cap_is_a_program_limit() { + let error = nodedb_sql::SqlError::RecursionDepthExceeded { + cte_name: "walk".into(), + max_depth: 100, + }; + assert!(matches!( + map_plan_error(error, TenantId::new(1)), + crate::Error::DataPlane(crate::bridge::envelope::ErrorCode::RecursionDepthExceeded { + max_depth: 100, + .. + }) + )); + } +} diff --git a/nodedb/src/control/security/role_assignment.rs b/nodedb/src/control/security/role_assignment.rs index c09bead8f..1a2c32e21 100644 --- a/nodedb/src/control/security/role_assignment.rs +++ b/nodedb/src/control/security/role_assignment.rs @@ -58,9 +58,14 @@ impl From for crate::Error { fn from(refusal: RoleRefusal) -> Self { match refusal { RoleRefusal::Undefined { name } => crate::Error::UndefinedObject { kind: "role", name }, - other => crate::Error::BadRequest { - detail: other.to_string(), - }, + // A refused DROP of a role still in use: a client error. No + // `crate::Error` variant carries `2BP01` without a tenant and a + // CASCADE hint that does not apply to roles. + other @ (RoleRefusal::HeldByUsers { .. } | RoleRefusal::InheritedBy { .. }) => { + crate::Error::BadRequest { + detail: other.to_string(), + } + } } } } diff --git a/nodedb/src/control/sequence/error_map.rs b/nodedb/src/control/sequence/error_map.rs index 5fd09090f..74cf7a332 100644 --- a/nodedb/src/control/sequence/error_map.rs +++ b/nodedb/src/control/sequence/error_map.rs @@ -21,7 +21,13 @@ pub(crate) fn undefined_sequence(name: &str) -> SqlError { pub(crate) fn map_sequence_error(name: &str, error: SequenceError) -> SqlError { match error { SequenceError::NotFound { .. } => undefined_sequence(name), - other => SqlError::ObjectNotInPrerequisiteState { + other @ (SequenceError::Exhausted { .. } + | SequenceError::NotYetCalled { .. } + | SequenceError::OutOfRange { .. } + | SequenceError::AlreadyExists { .. } + | SequenceError::InvalidDefinition { .. } + | SequenceError::FormatParse { .. } + | SequenceError::InvalidResetScope { .. }) => SqlError::ObjectNotInPrerequisiteState { object: name.to_string(), detail: other.to_string(), }, @@ -40,7 +46,13 @@ pub(crate) fn sequence_error_to_error(name: &str, error: SequenceError) -> crate kind: "sequence", name: name.to_string(), }, - other => crate::Error::ObjectNotInPrerequisiteState { + other @ (SequenceError::Exhausted { .. } + | SequenceError::NotYetCalled { .. } + | SequenceError::OutOfRange { .. } + | SequenceError::AlreadyExists { .. } + | SequenceError::InvalidDefinition { .. } + | SequenceError::FormatParse { .. } + | SequenceError::InvalidResetScope { .. }) => crate::Error::ObjectNotInPrerequisiteState { object: name.to_string(), detail: other.to_string(), }, diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index e6ccb3c17..05ba610aa 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -40,19 +40,55 @@ use crate::bridge::envelope::ErrorCode; /// which a committed-redo apply answers with after it rolled a failed /// install back. pub(crate) fn refusal_is_final(code: &ErrorCode) -> bool { - write_definitely_not_applied(code) - && !matches!( - code, - ErrorCode::RetryableRefusal { .. } - | ErrorCode::SyncNotApplied { .. } - | ErrorCode::RateExceeded { .. } - | ErrorCode::CollectionDraining { .. } - | ErrorCode::DispatchCapacity { .. } - | ErrorCode::ExpiredBeforeExecution - | ErrorCode::ConflictRetry - | ErrorCode::OllpRetryRequired - | ErrorCode::TxnOverlayMemoryExceeded { .. } - ) + write_definitely_not_applied(code) && !is_transient_verdict(code) +} + +/// Whether `code` depends on this node's momentary load or on a transient +/// precondition, so a redelivery of the same entry can apply it. +fn is_transient_verdict(code: &ErrorCode) -> bool { + match code { + ErrorCode::RetryableRefusal { .. } + | ErrorCode::SyncNotApplied { .. } + | ErrorCode::RateExceeded { .. } + | ErrorCode::CollectionDraining { .. } + | ErrorCode::DispatchCapacity { .. } + | ErrorCode::ExpiredBeforeExecution + | ErrorCode::ConflictRetry + | ErrorCode::OllpRetryRequired + | ErrorCode::TxnOverlayMemoryExceeded { .. } => true, + ErrorCode::DeadlineExceeded + | ErrorCode::RejectedConstraint { .. } + | ErrorCode::RejectedPrevalidation { .. } + | ErrorCode::SyncRejected { .. } + | ErrorCode::NotFound + | ErrorCode::RejectedAuthz { .. } + | ErrorCode::CrdtFrontierMismatch { .. } + | ErrorCode::FanOutExceeded + | ErrorCode::ResourcesExhausted + | ErrorCode::RejectedDanglingEdge { .. } + | ErrorCode::DuplicateWrite + | ErrorCode::AppendOnlyViolation { .. } + | ErrorCode::BalanceViolation { .. } + | ErrorCode::PeriodLocked { .. } + | ErrorCode::PeriodLockMisconfigured { .. } + | ErrorCode::RetentionViolation { .. } + | ErrorCode::LegalHoldActive { .. } + | ErrorCode::StateTransitionViolation { .. } + | ErrorCode::TransitionCheckViolation { .. } + | ErrorCode::TypeGuardViolation { .. } + | ErrorCode::TypeMismatch { .. } + | ErrorCode::CounterFault { .. } + | ErrorCode::InsufficientBalance { .. } + | ErrorCode::RecursionDepthExceeded { .. } + | ErrorCode::UndefinedColumn { .. } + | ErrorCode::Internal { .. } + | ErrorCode::Unsupported { .. } + | ErrorCode::RollbackFailed { .. } + | ErrorCode::DivisionByZero + | ErrorCode::UndefinedFunction { .. } + | ErrorCode::DataException { .. } + | ErrorCode::BadRequest { .. } => false, + } } /// Whether a committed proposal's apply `error` is a final refusal: the diff --git a/nodedb/src/control/server/pgwire/handler/session_explain.rs b/nodedb/src/control/server/pgwire/handler/session_explain.rs index 0eee63bdf..68988a315 100644 --- a/nodedb/src/control/server/pgwire/handler/session_explain.rs +++ b/nodedb/src/control/server/pgwire/handler/session_explain.rs @@ -48,15 +48,18 @@ impl NodeDbPgHandler { ))]); } Some(Err(error)) => { - let sqlstate = match error { - nodedb_sql::SqlError::UnsupportedConstraint { .. } - | nodedb_sql::SqlError::ConflictingEngineClause { .. } => "0A000", - _ => "42601", - }; + // The SQLSTATE the planner path renders for the same error. + let message = error.to_string(); + let (_, sqlstate, _) = crate::control::server::pgwire::types::error_to_sqlstate( + &crate::control::planner::plan_error_map::map_plan_error( + error, + identity.tenant_id, + ), + ); return Err(PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), sqlstate.to_owned(), - error.to_string(), + message, )))); } None => {} diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 86750d872..9a0c4bfb5 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -9,6 +9,8 @@ use crate::OllpExhaustedCause; use crate::bridge::envelope::{ErrorCode, Status}; use crate::control::server::response_shape::types::DmlFoldError; +pub(crate) use super::numeric_sqlstate::numeric_code_to_sqlstate; + /// Create a pgwire ErrorResponse with a SQLSTATE code. pub fn sqlstate_error(code: &str, message: &str) -> PgWireError { PgWireError::UserError(Box::new(ErrorInfo::new( @@ -291,92 +293,93 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str numeric_code_to_sqlstate(e.code()), e.message().to_string(), ), - _ => ("ERROR", sqlstate::INTERNAL_ERROR, err.to_string()), - } -} - -/// Map a numeric `ErrorCode` received from a remote node back to a SQLSTATE. -/// Local errors map by variant identity above; a remote error arrives as a bare -/// numeric code, so this recovers the classification. Each bucket mirrors the -/// sqlstate the corresponding local variant arm chooses above for the same -/// numeric code, so a constraint violation (say) maps to the same SQLSTATE -/// whether it happened locally or on a remote node. Unmapped/unknown codes -/// fall back to INTERNAL_ERROR — the behaviour before codes were preserved. -pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> &'static str { - use nodedb_types::error::ErrorCode as Ec; - match code { - // Mirrors the `RejectedConstraint` arm. - Ec::CONSTRAINT_VIOLATION => sqlstate::UNIQUE_VIOLATION, - // Mirrors the `ConflictRetry` / `CalvinSerializationConflict` / - // `SourceFrozen` / `RetryableSchemaChanged` arms, and `OllpExhausted` - // when it exhausted on drift. - Ec::WRITE_CONFLICT => sqlstate::SERIALIZATION_FAILURE, - // Mirrors the `DeadlineExceeded` arm. - Ec::DEADLINE_EXCEEDED => sqlstate::QUERY_CANCELED, - // Mirrors the `CollectionNotFound` / `CollectionDeactivated` arms. - Ec::COLLECTION_NOT_FOUND | Ec::COLLECTION_DEACTIVATED => sqlstate::UNDEFINED_TABLE, - // Mirrors the `DocumentNotFound` arm. - Ec::DOCUMENT_NOT_FOUND => sqlstate::NO_DATA, - // Mirrors the `BadRequest` / `PlanError` arms. - Ec::BAD_REQUEST | Ec::PLAN_ERROR => sqlstate::SYNTAX_ERROR, - // Mirrors the `UndefinedFunction` arm. - Ec::UNDEFINED_FUNCTION => sqlstate::UNDEFINED_FUNCTION, - // Mirrors the `UndefinedObject` arm. - Ec::UNDEFINED_OBJECT => sqlstate::UNDEFINED_OBJECT, - // Mirrors the `ObjectNotInPrerequisiteState` arm. - Ec::OBJECT_NOT_READY => sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, - // Mirrors the `UndefinedColumn` arm. - Ec::UNDEFINED_COLUMN => sqlstate::UNDEFINED_COLUMN, - // Mirrors the `AmbiguousColumn` arm. - Ec::AMBIGUOUS_COLUMN => sqlstate::AMBIGUOUS_COLUMN, - // Mirrors the `DivisionByZero` arm. - Ec::DIVISION_BY_ZERO => sqlstate::DIVISION_BY_ZERO, - // Mirrors the `DataException` arm. - Ec::DATA_EXCEPTION => sqlstate::DATA_EXCEPTION, - // Mirrors the `InvalidLimitValue` arm. - Ec::INVALID_LIMIT_VALUE => sqlstate::INVALID_LIMIT_VALUE, - // Mirrors the `FanOutExceeded` arm. - Ec::FAN_OUT_EXCEEDED => sqlstate::STATEMENT_TOO_COMPLEX, - // Mirrors the `RejectedAuthz` arm. - Ec::AUTHORIZATION_DENIED => sqlstate::INSUFFICIENT_PRIVILEGE, - // Mirrors the `SessionTokenExpired` arm. - Ec::AUTH_EXPIRED => sqlstate::INVALID_AUTHORIZATION, - // Mirrors the `RateExceeded` arm. - Ec::RATE_EXCEEDED => sqlstate::TOO_MANY_CONNECTIONS, - // Mirrors the `MemoryExhausted` / `Backpressure` arms. - Ec::MEMORY_EXHAUSTED => sqlstate::OUT_OF_MEMORY, - // Mirrors the `DispatchCapacity` arm. - Ec::SERVER_OVERLOAD => sqlstate::SERVER_OVERLOAD, - // Mirrors the `NoLeader` arm. - Ec::NO_LEADER => sqlstate::LOCK_NOT_AVAILABLE, - // Mirrors the `NotLeader` arm. - Ec::NOT_LEADER => sqlstate::DATABASE_DROPPED, - // Mirrors the `CloneWriteRequiresMaterialize` arm. - Ec::CLONE_WRITE_REQUIRES_MATERIALIZE => sqlstate::CLONE_WRITE_REQUIRES_MATERIALIZE.0, - // Mirrors the `BackupTenantMismatch` arm. - Ec::BACKUP_TENANT_MISMATCH => sqlstate::BACKUP_TENANT_MISMATCH, - // Mirrors the `BackupKeyMismatch` arm. - Ec::BACKUP_KEY_MISMATCH => sqlstate::BACKUP_KEY_MISMATCH, - // The codes below mirror the Data-Plane code table - // (`error_code_to_sqlstate`) for the public code each Data-Plane code - // classifies to, so a verdict that crossed a node as a numeric code - // renders in the class it has locally. - Ec::PREVALIDATION_REJECTED | Ec::INSUFFICIENT_BALANCE => sqlstate::CHECK_VIOLATION, - Ec::APPEND_ONLY_VIOLATION => sqlstate::APPEND_ONLY_VIOLATION, - Ec::BALANCE_VIOLATION => sqlstate::BALANCE_VIOLATION, - Ec::PERIOD_LOCKED => sqlstate::PERIOD_LOCKED, - Ec::PERIOD_LOCK_MISCONFIGURED => sqlstate::PERIOD_LOCK_MISCONFIGURED, - Ec::STATE_TRANSITION_VIOLATION => sqlstate::STATE_TRANSITION_VIOLATION, - Ec::TRANSITION_CHECK_VIOLATION => sqlstate::TRANSITION_CHECK_VIOLATION, - Ec::RETENTION_VIOLATION => sqlstate::RETENTION_VIOLATION, - Ec::LEGAL_HOLD_ACTIVE => sqlstate::LEGAL_HOLD_ACTIVE, - Ec::TYPE_GUARD_VIOLATION => sqlstate::TYPE_GUARD_VIOLATION, - Ec::TYPE_MISMATCH => sqlstate::CANNOT_COERCE, - Ec::OVERFLOW => sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, - Ec::COLLECTION_DRAINING => sqlstate::CANNOT_CONNECT_NOW, - Ec::SQL_NOT_ENABLED => sqlstate::FEATURE_NOT_SUPPORTED, - Ec::PROGRAM_LIMIT_EXCEEDED => sqlstate::PROGRAM_LIMIT_EXCEEDED, - _ => sqlstate::INTERNAL_ERROR, + // The DDL path renders a regressed consumer offset as an invalid + // parameter value, so the typed error takes that class too. + crate::Error::OffsetRegression { .. } => { + ("ERROR", sqlstate::INVALID_PARAMETER_VALUE, err.to_string()) + } + // A full admission queue is a rate refusal, its public code's class. + crate::Error::VShardAdmissionCapacityExceeded { .. } => { + ("ERROR", sqlstate::TOO_MANY_CONNECTIONS, err.to_string()) + } + // The CRDT frontier kept moving. The client retries the write. + crate::Error::CrdtAdmissionRetriesExhausted { .. } => { + ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) + } + crate::Error::CrdtAdmissionTimeout { .. } => { + ("ERROR", sqlstate::QUERY_CANCELED, err.to_string()) + } + // Statements refused inside an explicit transaction block share + // the class of `NotInTransactionBlock`. + crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::CrossShardInExplicitTransaction => { + ("ERROR", sqlstate::ACTIVE_SQL_TRANSACTION, err.to_string()) + } + // Each variant here has the public code `BAD_REQUEST`, so pgwire + // renders the class that code renders on native and across nodes. + crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Promql(_) + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } => { + ("ERROR", sqlstate::SYNTAX_ERROR, err.to_string()) + } + crate::Error::DependentObjectsExist { .. } => ( + "ERROR", + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + err.to_string(), + ), + crate::Error::QuotaOvercommit { .. } => { + ("ERROR", sqlstate::QUOTA_OVERCOMMIT, err.to_string()) + } + crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } => { + ("ERROR", sqlstate::QUOTA_EXCEEDED, err.to_string()) + } + crate::Error::MirrorReadOnly { .. } => ( + "ERROR", + sqlstate::READ_ONLY_SQL_TRANSACTION, + err.to_string(), + ), + // The client redirects the strong read to the source cluster. + crate::Error::StaleReadNotLeader { .. } => { + ("ERROR", sqlstate::STALE_READ_NOT_LEADER, err.to_string()) + } + // Server-side faults and system defects. The client can act on none + // of them, and their public codes are internal classes. + crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CascadeCycle { .. } => ("ERROR", sqlstate::INTERNAL_ERROR, err.to_string()), } } diff --git a/nodedb/src/control/server/pgwire/types/mod.rs b/nodedb/src/control/server/pgwire/types/mod.rs index b15b76920..7e03142c1 100644 --- a/nodedb/src/control/server/pgwire/types/mod.rs +++ b/nodedb/src/control/server/pgwire/types/mod.rs @@ -7,6 +7,7 @@ pub mod error_map; pub mod field; +pub mod numeric_sqlstate; pub mod parse; pub mod privilege; diff --git a/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs b/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs new file mode 100644 index 000000000..2f77ac300 --- /dev/null +++ b/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs @@ -0,0 +1,101 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Public numeric `ErrorCode` to PostgreSQL SQLSTATE mapping. + +use nodedb_types::error::sqlstate; + +/// Map a numeric `ErrorCode` received from a remote node back to a SQLSTATE. +/// Local errors map by variant identity in `error_to_sqlstate`. A remote +/// error arrives as a bare numeric code, so this recovers the class. Each +/// bucket mirrors the SQLSTATE the local variant arm chooses for the same +/// numeric code, so a constraint violation (say) maps to the same SQLSTATE +/// whether it happened locally or on a remote node. `ErrorCode` is an open +/// numeric newtype, so an unmapped or unknown code renders `INTERNAL_ERROR`. +pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> &'static str { + use nodedb_types::error::ErrorCode as Ec; + match code { + // Mirrors the `RejectedConstraint` arm. + Ec::CONSTRAINT_VIOLATION => sqlstate::UNIQUE_VIOLATION, + // Mirrors the `ConflictRetry` / `CalvinSerializationConflict` / + // `SourceFrozen` / `RetryableSchemaChanged` arms, and `OllpExhausted` + // when it exhausted on drift. + Ec::WRITE_CONFLICT => sqlstate::SERIALIZATION_FAILURE, + // Mirrors the `DeadlineExceeded` arm. + Ec::DEADLINE_EXCEEDED => sqlstate::QUERY_CANCELED, + // Mirrors the `CollectionNotFound` / `CollectionDeactivated` arms. + Ec::COLLECTION_NOT_FOUND | Ec::COLLECTION_DEACTIVATED => sqlstate::UNDEFINED_TABLE, + // Mirrors the `DocumentNotFound` arm. + Ec::DOCUMENT_NOT_FOUND => sqlstate::NO_DATA, + // Mirrors the `BadRequest` / `PlanError` arms. + Ec::BAD_REQUEST | Ec::PLAN_ERROR => sqlstate::SYNTAX_ERROR, + // Mirrors the `UndefinedFunction` arm. + Ec::UNDEFINED_FUNCTION => sqlstate::UNDEFINED_FUNCTION, + // Mirrors the `UndefinedObject` arm. + Ec::UNDEFINED_OBJECT => sqlstate::UNDEFINED_OBJECT, + // Mirrors the `ObjectNotInPrerequisiteState` arm. + Ec::OBJECT_NOT_READY => sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, + // Mirrors the `UndefinedColumn` arm. + Ec::UNDEFINED_COLUMN => sqlstate::UNDEFINED_COLUMN, + // Mirrors the `AmbiguousColumn` arm. + Ec::AMBIGUOUS_COLUMN => sqlstate::AMBIGUOUS_COLUMN, + // Mirrors the `DivisionByZero` arm. + Ec::DIVISION_BY_ZERO => sqlstate::DIVISION_BY_ZERO, + // Mirrors the `DataException` arm. + Ec::DATA_EXCEPTION => sqlstate::DATA_EXCEPTION, + // Mirrors the `InvalidLimitValue` arm. + Ec::INVALID_LIMIT_VALUE => sqlstate::INVALID_LIMIT_VALUE, + // Mirrors the `FanOutExceeded` arm. + Ec::FAN_OUT_EXCEEDED => sqlstate::STATEMENT_TOO_COMPLEX, + // Mirrors the `RejectedAuthz` arm. + Ec::AUTHORIZATION_DENIED => sqlstate::INSUFFICIENT_PRIVILEGE, + // Mirrors the `SessionTokenExpired` arm. + Ec::AUTH_EXPIRED => sqlstate::INVALID_AUTHORIZATION, + // Mirrors the `RateExceeded` arm. + Ec::RATE_EXCEEDED => sqlstate::TOO_MANY_CONNECTIONS, + // Mirrors the `MemoryExhausted` / `Backpressure` arms. + Ec::MEMORY_EXHAUSTED => sqlstate::OUT_OF_MEMORY, + // Mirrors the `DispatchCapacity` arm. + Ec::SERVER_OVERLOAD => sqlstate::SERVER_OVERLOAD, + // Mirrors the `NoLeader` arm. + Ec::NO_LEADER => sqlstate::LOCK_NOT_AVAILABLE, + // Mirrors the `NotLeader` arm. + Ec::NOT_LEADER => sqlstate::DATABASE_DROPPED, + // Mirrors the `CloneWriteRequiresMaterialize` arm. + Ec::CLONE_WRITE_REQUIRES_MATERIALIZE => sqlstate::CLONE_WRITE_REQUIRES_MATERIALIZE.0, + // Mirrors the `BackupTenantMismatch` arm. + Ec::BACKUP_TENANT_MISMATCH => sqlstate::BACKUP_TENANT_MISMATCH, + // Mirrors the `BackupKeyMismatch` arm. + Ec::BACKUP_KEY_MISMATCH => sqlstate::BACKUP_KEY_MISMATCH, + // Mirrors the `QuotaOvercommit` arm. + Ec::QUOTA_OVERCOMMIT => sqlstate::QUOTA_OVERCOMMIT, + // Mirrors the `TenantVectorDimExceeded` / `TenantGraphDepthExceeded` + // arms. + Ec::TENANT_VECTOR_DIM_EXCEEDED | Ec::TENANT_GRAPH_DEPTH_EXCEEDED => { + sqlstate::QUOTA_EXCEEDED + } + // Mirrors the `MirrorReadOnly` arm. + Ec::MIRROR_READ_ONLY => sqlstate::READ_ONLY_SQL_TRANSACTION, + // Mirrors the `StaleReadNotLeader` arm. + Ec::STALE_READ_NOT_LEADER => sqlstate::STALE_READ_NOT_LEADER, + // The codes below mirror the Data-Plane code table + // (`error_code_to_sqlstate`) for the public code each Data-Plane code + // classifies to, so a verdict that crossed a node as a numeric code + // renders in the class it has locally. + Ec::PREVALIDATION_REJECTED | Ec::INSUFFICIENT_BALANCE => sqlstate::CHECK_VIOLATION, + Ec::APPEND_ONLY_VIOLATION => sqlstate::APPEND_ONLY_VIOLATION, + Ec::BALANCE_VIOLATION => sqlstate::BALANCE_VIOLATION, + Ec::PERIOD_LOCKED => sqlstate::PERIOD_LOCKED, + Ec::PERIOD_LOCK_MISCONFIGURED => sqlstate::PERIOD_LOCK_MISCONFIGURED, + Ec::STATE_TRANSITION_VIOLATION => sqlstate::STATE_TRANSITION_VIOLATION, + Ec::TRANSITION_CHECK_VIOLATION => sqlstate::TRANSITION_CHECK_VIOLATION, + Ec::RETENTION_VIOLATION => sqlstate::RETENTION_VIOLATION, + Ec::LEGAL_HOLD_ACTIVE => sqlstate::LEGAL_HOLD_ACTIVE, + Ec::TYPE_GUARD_VIOLATION => sqlstate::TYPE_GUARD_VIOLATION, + Ec::TYPE_MISMATCH => sqlstate::CANNOT_COERCE, + Ec::OVERFLOW => sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, + Ec::COLLECTION_DRAINING => sqlstate::CANNOT_CONNECT_NOW, + Ec::SQL_NOT_ENABLED => sqlstate::FEATURE_NOT_SUPPORTED, + Ec::PROGRAM_LIMIT_EXCEEDED => sqlstate::PROGRAM_LIMIT_EXCEEDED, + _ => sqlstate::INTERNAL_ERROR, + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/column_default.rs b/nodedb/src/control/server/shared/ddl/neutral/column_default.rs index d7c07eb6f..179145088 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/column_default.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/column_default.rs @@ -105,8 +105,8 @@ pub(super) fn validate_column_default( /// The expression is classified and parsed, never evaluated, so a /// `DEFAULT nextval('s')` column never advances its sequence at DDL time. /// -/// An unregistered function name raises SQLSTATE `42883`; every other -/// rejection raises SQLSTATE `42601`. +/// An unregistered function name raises SQLSTATE `42883`, a type error +/// `42804`, a value out of range `22003`, and a statement error `42601`. pub(super) fn validate_clause_expr(clause: &str, owner: &str, expr: &str) -> Result<(), DdlError> { nodedb_sql::planner::defaults::validate_default_expr(expr, owner) .map_err(|error| clause_error(clause, owner, &error)) @@ -117,10 +117,41 @@ fn clause_error(clause: &str, owner: &str, error: &SqlError) -> DdlError { let sqlstate = match error { SqlError::UndefinedFunction { .. } => sqlstate::UNDEFINED_FUNCTION, SqlError::TypeMismatch { .. } => sqlstate::DATATYPE_MISMATCH, - SqlError::IntegerOutOfRange { .. } | SqlError::FloatOutOfRange { .. } => { - sqlstate::NUMERIC_VALUE_OUT_OF_RANGE + SqlError::IntegerOutOfRange { .. } + | SqlError::FloatOutOfRange { .. } + | SqlError::ConstantOverflow { .. } => sqlstate::NUMERIC_VALUE_OUT_OF_RANGE, + SqlError::DivisionByZero => sqlstate::DIVISION_BY_ZERO, + SqlError::DataException { .. } => sqlstate::DATA_EXCEPTION, + SqlError::InvalidLimitValue { .. } => sqlstate::INVALID_LIMIT_VALUE, + SqlError::UnknownTable { .. } | SqlError::CollectionDeactivated { .. } => { + sqlstate::UNDEFINED_TABLE } - _ => sqlstate::SYNTAX_ERROR, + SqlError::UnknownColumn { .. } => sqlstate::UNDEFINED_COLUMN, + SqlError::AmbiguousColumn { .. } => sqlstate::AMBIGUOUS_COLUMN, + SqlError::UndefinedObject { .. } => sqlstate::UNDEFINED_OBJECT, + SqlError::ObjectNotInPrerequisiteState { .. } => sqlstate::OBJECT_NOT_IN_PREREQUISITE_STATE, + SqlError::SequencePerRowUnsupported { .. } + | SqlError::SearchFunctionOutsideSearch { .. } + | SqlError::UnsupportedConstraint { .. } + | SqlError::ConflictingEngineClause { .. } => sqlstate::FEATURE_NOT_SUPPORTED, + SqlError::RetryableSchemaChanged { .. } => sqlstate::SERIALIZATION_FAILURE, + SqlError::RecursionDepthExceeded { .. } => sqlstate::PROGRAM_LIMIT_EXCEEDED, + SqlError::Parse { .. } + | SqlError::Arity { .. } + | SqlError::Unsupported { .. } + | SqlError::UnevaluableDefault { .. } + | SqlError::SetvalInColumnDefault { .. } + | SqlError::InvalidFunction { .. } + | SqlError::InvalidWindowFrame { .. } + | SqlError::MissingField { .. } + | SqlError::InsertColumnArityMismatch { .. } + | SqlError::PositionalKvInsertUnsupported { .. } + | SqlError::InvalidIdentifier { .. } + | SqlError::ReservedIdentifier { .. } + | SqlError::InvalidRecursiveSetOp { .. } + | SqlError::InvalidRecursiveSelfRef { .. } + | SqlError::RecursiveColumnMismatch { .. } + | SqlError::DuplicateRecursiveColumn { .. } => sqlstate::SYNTAX_ERROR, }; DdlError::new( sqlstate, diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs index d496380c3..824c27de6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs @@ -157,7 +157,8 @@ pub async fn commit_offset( ) .map_err(|e| match e { crate::Error::OffsetRegression { .. } => DdlError::new("22023", e.to_string()), - _ => DdlError::new("XX000", format!("offset commit: {e}")), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error_in_context("offset commit", &other), })?; return Ok(status("COMMIT OFFSET")); diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs index 3a7ee3b98..f05865413 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs @@ -164,12 +164,14 @@ pub fn drop_database( // Gated until per-engine row copy lands — surface `0A000` // (`feature_not_supported`) so clients know not to retry. crate::Error::BadRequest { detail } => ddl_err("0A000", detail), - other => ddl_err( - "XX000", - format!( - "force materialization of dependent clone {} failed: {other}", + // Any other error keeps the class the SQLSTATE table + // gives it. + other => DdlError::from_error_in_context( + &format!( + "force materialization of dependent clone {} failed", dep_id.as_u64() ), + &other, ), }, )?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs b/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs index 1396cf711..2e5a4d862 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs @@ -55,9 +55,10 @@ pub fn alter_database_materialize( // per-engine bulk-copy implementation to land). force_materialize_blocking(db_id, state, catalog, Some(&handle)).map_err(|e| match e { crate::Error::BadRequest { detail } => ddl_err("0A000", detail), - other => ddl_err( - "XX000", - format!("clone materialization of '{name}' failed: {other}"), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error_in_context( + &format!("clone materialization of '{name}' failed"), + &other, ), })?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs index ded823621..8cdcb2d25 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs @@ -164,12 +164,15 @@ impl<'a> CollectionReadGate<'a> { &self.state.rls, self.scope.auth(), ) - .map_err(|error| { - let sqlstate = match &error { - crate::Error::RejectedAuthz { .. } => INSUFFICIENT_PRIVILEGE, - _ => FEATURE_NOT_SUPPORTED, - }; - gate_err(sqlstate, error.to_string()) + .map_err(|error| match &error { + crate::Error::RejectedAuthz { .. } => { + gate_err(INSUFFICIENT_PRIVILEGE, error.to_string()) + } + // The injection pass refuses a plan shape it cannot cover. A + // hand-built read of that shape is a feature this door lacks. + crate::Error::PlanError { .. } => gate_err(FEATURE_NOT_SUPPORTED, error.to_string()), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error(other), }) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/rls.rs b/nodedb/src/control/server/shared/ddl/neutral/rls.rs index 7cf67644d..68a31ccad 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rls.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rls.rs @@ -80,7 +80,8 @@ fn compile_rls_predicate( crate::Error::CollectionNotFound { .. } => { DdlError::new("42P01", format!("collection '{collection}' does not exist")) } - other => DdlError::new("XX000", format!("catalog read: {other}")), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error_in_context("catalog read", &other), })?; let compiled = compile_policy_predicate(predicate_str, &columns) .map_err(|e| DdlError::new("42601", e.to_string()))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs index bbc6b2254..059873586 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs @@ -69,10 +69,10 @@ pub async fn try_dispatch( return Some(r); } - // Parse errors surface as a typed `DdlError` here: `UnsupportedConstraint` - // maps to `0A000` (feature_not_supported), every other parse error to - // `42601` (syntax error), with the parser's own `Display` text as the - // message. This is the sole parse-error gate for the DDL router; the + // Parse errors surface as a typed `DdlError` here, with the SQLSTATE the + // planner path gives the same `SqlError` (`UnsupportedConstraint` is + // `0A000`, a parse error `42601`) and the parser's own `Display` text as + // the message. This is the sole parse-error gate for the DDL router; the // GRAPH / MATCH / SHOW GRAPH STATS prefixed inputs that previously carried // their own parse-error reproduction are subsumed by this arm. // @@ -86,14 +86,13 @@ pub async fn try_dispatch( let stmt = match nodedb_sql::ddl_ast::parse(sql) { Some(Ok(stmt)) => stmt, Some(Err(e)) => { - // UnsupportedConstraint / ConflictingEngineClause → 0A000 (feature_not_supported). - // All other parse errors → 42601 (syntax error). - let sqlstate = match &e { - nodedb_sql::SqlError::UnsupportedConstraint { .. } - | nodedb_sql::SqlError::ConflictingEngineClause { .. } => "0A000", - _ => "42601", - }; - return Some(Err(DdlError::new(sqlstate, e.to_string()))); + // The SQLSTATE the planner path renders for the same error, so a + // parse refusal answers one class wherever it is raised. + let message = e.to_string(); + let (_, sqlstate, _) = crate::control::server::pgwire::types::error_to_sqlstate( + &crate::control::planner::plan_error_map::map_plan_error(e, identity.tenant_id), + ); + return Some(Err(DdlError::new(sqlstate, message))); } None => { // Bulk import: `COPY FROM STDIN [WITH (...)]`. The diff --git a/nodedb/src/control/server/shared/ddl/neutral/topic/publish.rs b/nodedb/src/control/server/shared/ddl/neutral/topic/publish.rs index dc6cd4f27..43a40c9a9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/topic/publish.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/topic/publish.rs @@ -32,18 +32,14 @@ pub async fn handle_publish( ) -> Result, DdlError> { match dispatch_sql_in_database(state, identity, database_id, sql).await { Ok(Some(_)) => Ok(status("PUBLISH")), - Err(e) => { - let sqlstate = match &e { - crate::Error::CollectionNotFound { .. } => "42704", - crate::Error::BadRequest { .. } => "42601", - crate::Error::Dispatch { .. } => "58000", - crate::Error::DispatchCapacity { .. } => { - nodedb_types::error::sqlstate::SERVER_OVERLOAD - } - _ => "XX000", - }; - Err(DdlError::new(sqlstate.to_string(), e.to_string())) - } + Err(e) => Err(match &e { + // The named topic does not exist. + crate::Error::CollectionNotFound { .. } => DdlError::new("42704", e.to_string()), + crate::Error::BadRequest { .. } => DdlError::new("42601", e.to_string()), + crate::Error::Dispatch { .. } => DdlError::new("58000", e.to_string()), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error(other), + }), Ok(None) => Err(DdlError::new( "42601", "expected PUBLISH TO ''", diff --git a/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs index a900f72c1..fd503d104 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/version_history/dispatch.rs @@ -87,15 +87,14 @@ pub(super) async fn dispatch_authorized_read( /// into [`crate::Error::RejectedAuthz`], so the client-visible SQLSTATE stays /// the same with clone interception running ahead of it. Everything else the gate /// can raise (a clone read shape with no sound rewrite, a catalog read failure) -/// is an internal-error class the client cannot act on by SQLSTATE alone, so it -/// carries its own message under `XX000`. Both SQLSTATEs have exactly one -/// `ErrorCode` meaning, so `DdlError::new` derives the right code for each. +/// keeps the SQLSTATE, code and details the SQLSTATE table gives it, with the +/// read named before its message. fn gate_error(error: crate::Error) -> DdlError { match error { crate::Error::RejectedAuthz { resource, .. } => { DdlError::new("42501", format!("permission denied: {resource}")) } - other => DdlError::new("XX000", format!("version-history read: {other}")), + other => DdlError::from_error_in_context("version-history read", &other), } } diff --git a/nodedb/src/control/server/shared/ddl/result.rs b/nodedb/src/control/server/shared/ddl/result.rs index 7c4bceab5..885568303 100644 --- a/nodedb/src/control/server/shared/ddl/result.rs +++ b/nodedb/src/control/server/shared/ddl/result.rs @@ -215,6 +215,9 @@ pub fn code_for_sqlstate(sqlstate_str: &str) -> ErrorCode { sqlstate::UNDEFINED_COLUMN => ErrorCode::UNDEFINED_COLUMN, sqlstate::AMBIGUOUS_COLUMN => ErrorCode::AMBIGUOUS_COLUMN, sqlstate::DATA_EXCEPTION => ErrorCode::DATA_EXCEPTION, + sqlstate::DIVISION_BY_ZERO => ErrorCode::DIVISION_BY_ZERO, + sqlstate::INVALID_LIMIT_VALUE => ErrorCode::INVALID_LIMIT_VALUE, + sqlstate::PROGRAM_LIMIT_EXCEEDED => ErrorCode::PROGRAM_LIMIT_EXCEEDED, // A malformed request and a plan that cannot be built both render as // `42601`; both are non-retriable client errors, so one code covers // both without losing anything a client acts on. @@ -276,6 +279,23 @@ mod tests { assert_eq!(code_for_sqlstate("02000"), ErrorCode::NOT_FOUND); } + /// A statement-class SQLSTATE derives its own code, never `INTERNAL`. + #[test] + fn statement_class_sqlstates_derive_their_code() { + assert_eq!( + code_for_sqlstate(sqlstate::DIVISION_BY_ZERO), + ErrorCode::DIVISION_BY_ZERO + ); + assert_eq!( + code_for_sqlstate(sqlstate::INVALID_LIMIT_VALUE), + ErrorCode::INVALID_LIMIT_VALUE + ); + assert_eq!( + code_for_sqlstate(sqlstate::PROGRAM_LIMIT_EXCEEDED), + ErrorCode::PROGRAM_LIMIT_EXCEEDED + ); + } + #[test] fn unknown_sqlstate_falls_back_to_internal() { assert_eq!(code_for_sqlstate("99999"), ErrorCode::INTERNAL); diff --git a/nodedb/src/control/server/shared/retry.rs b/nodedb/src/control/server/shared/retry.rs index 5aca7c897..4bd7178b9 100644 --- a/nodedb/src/control/server/shared/retry.rs +++ b/nodedb/src/control/server/shared/retry.rs @@ -63,7 +63,113 @@ impl RetryableSchemaChange for Error { // re-run statements whose failure is real. match self { Error::RetryableSchemaChanged { descriptor } => Some(descriptor.as_str()), - _ => None, + Error::RejectedConstraint { .. } + | Error::TxnOverlayMemoryExceeded { .. } + | Error::RejectedAuthz { .. } + | Error::OffsetRegression { .. } + | Error::DeadlineExceeded { .. } + | Error::ConflictRetry { .. } + | Error::CalvinSerializationConflict + | Error::CalvinParticipantError + | Error::RejectedPrevalidation { .. } + | Error::RetryableRefusal { .. } + | Error::AppendOnlyViolation { .. } + | Error::BalanceViolation { .. } + | Error::MaterializedSumTargetNotFound { .. } + | Error::MaterializedSumResolutionMissing { .. } + | Error::PeriodLocked { .. } + | Error::PeriodLockMisconfigured { .. } + | Error::RetentionViolation { .. } + | Error::LegalHoldActive { .. } + | Error::StateTransitionViolation { .. } + | Error::TransitionCheckViolation { .. } + | Error::TypeGuardViolation { .. } + | Error::TypeMismatch { .. } + | Error::InsufficientBalance { .. } + | Error::RateExceeded { .. } + | Error::CollectionNotFound { .. } + | Error::DocumentNotFound { .. } + | Error::CollectionDeactivated { .. } + | Error::VShardAdmissionCapacityExceeded { .. } + | Error::CrdtAdmissionRetriesExhausted { .. } + | Error::CrdtAdmissionInvalidPlan { .. } + | Error::CrdtAdmissionCallerFence + | Error::CrdtApplyRequiresAdmission + | Error::CrdtApplyForbiddenInTransaction + | Error::NotInTransactionBlock { .. } + | Error::CrdtAdmissionTimeout { .. } + | Error::NoLeader { .. } + | Error::NotLeader { .. } + | Error::FanOutExceeded { .. } + | Error::CrossCollectionNotColocated { .. } + | Error::SourceFrozen { .. } + | Error::CloneWriteRequiresMaterialize { .. } + | Error::BadRequest { .. } + | Error::BackupTenantMismatch { .. } + | Error::BackupKeyMismatch + | Error::QuotaOvercommit { .. } + | Error::PlanError { .. } + | Error::FeatureNotSupported { .. } + | Error::UndefinedFunction { .. } + | Error::UndefinedObject { .. } + | Error::ObjectNotInPrerequisiteState { .. } + | Error::UndefinedColumn { .. } + | Error::AmbiguousColumn { .. } + | Error::UnknownStrictField { .. } + | Error::DivisionByZero + | Error::DataException { .. } + | Error::InvalidLimitValue { .. } + | Error::RetryableLeaderChange { .. } + | Error::GroupQuorumUnavailable { .. } + | Error::GroupMarksUnavailable { .. } + | Error::MetadataLeaderUnavailable + | Error::AuthorizationStateBehind { .. } + | Error::ExecutionLimitExceeded { .. } + | Error::LimitExceeded { .. } + | Error::Wal(_) + | Error::Dispatch { .. } + | Error::DispatchCapacity { .. } + | Error::Storage { .. } + | Error::ColdStorage { .. } + | Error::Serialization { .. } + | Error::Codec { .. } + | Error::SegmentCorrupted { .. } + | Error::MemoryExhausted { .. } + | Error::Backpressure { .. } + | Error::Crdt(_) + | Error::Io(_) + | Error::Config { .. } + | Error::Encryption { .. } + | Error::Bridge { .. } + | Error::VersionCompat { .. } + | Error::Internal { .. } + | Error::Shaping(_) + | Error::RemoteTyped { .. } + | Error::DescriptorVersionAnomaly { .. } + | Error::CollectionPurgeRowMissing { .. } + | Error::CatalogIntegrityViolation { .. } + | Error::DataPlane(_) + | Error::Promql(_) + | Error::DependentObjectsExist { .. } + | Error::CascadeCycle { .. } + | Error::CrossShardInExplicitTransaction + | Error::SequencerUnavailable + | Error::SessionCapExceeded { .. } + | Error::SessionIdleTimeout + | Error::SessionTokenExpired + | Error::SessionKilledByAdmin + | Error::SessionUserDropped + | Error::OidcProviderTenantUnbound + | Error::OidcProviderTenantUnavailable { .. } + | Error::ExternalRoleUndefined { .. } + | Error::OidcNoDefaultDatabase { .. } + | Error::TenantVectorDimExceeded { .. } + | Error::TenantGraphDepthExceeded { .. } + | Error::RoleInheritanceCycle { .. } + | Error::RoleInheritanceDepthExceeded { .. } + | Error::OllpExhausted { .. } + | Error::MirrorReadOnly { .. } + | Error::StaleReadNotLeader { .. } => None, } } } diff --git a/nodedb/src/control/server/sync/async_dispatch/delta/compensation.rs b/nodedb/src/control/server/sync/async_dispatch/delta/compensation.rs new file mode 100644 index 000000000..e9dfe31aa --- /dev/null +++ b/nodedb/src/control/server/sync/async_dispatch/delta/compensation.rs @@ -0,0 +1,207 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The compensation hint a rejected CRDT delta carries to the edge. + +use crate::bridge::envelope::ErrorCode; + +use super::super::super::wire::CompensationHint; + +/// Classify a dispatch failure into the hint the edge compensates against. +/// +/// A failure that never judged the write is answered with a gap ack before +/// this runs. It still maps here to [`CompensationHint::Retry`], the class it +/// has, so no path turns it into a compensation. +pub(super) fn compensation_hint_for_dispatch_error(e: &crate::Error) -> CompensationHint { + match e { + crate::Error::DataPlane(code) => compensation_hint_for_code(code), + crate::Error::RejectedConstraint { + constraint, detail, .. + } => CompensationHint::Custom { + constraint: constraint.clone(), + detail: detail.clone(), + }, + crate::Error::RejectedPrevalidation { constraint, reason } => CompensationHint::Custom { + constraint: constraint.clone(), + detail: reason.clone(), + }, + crate::Error::RejectedAuthz { .. } => CompensationHint::PermissionDenied, + crate::Error::RateExceeded { retry_after_ms, .. } => CompensationHint::RateLimited { + retry_after_ms: *retry_after_ms, + }, + crate::Error::OllpExhausted { cause, .. } => match cause { + crate::OllpExhaustedCause::PreAdmission(inner) => { + compensation_hint_for_dispatch_error(inner) + } + crate::OllpExhaustedCause::PredicateDrift + | crate::OllpExhaustedCause::AdmissionRefused { .. } => { + CompensationHint::Retry { retry_after_ms: 0 } + } + }, + // Never judged the write: re-push it. + crate::Error::DeadlineExceeded { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RetryableRefusal { .. } + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::NoLeader { .. } + | crate::Error::NotLeader { .. } + | crate::Error::SourceFrozen { .. } + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::DispatchCapacity { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::Backpressure { .. } + | crate::Error::SequencerUnavailable + | crate::Error::StaleReadNotLeader { .. } => CompensationHint::Retry { retry_after_ms: 0 }, + // Refused on its merits, or a fault the same bytes reproduce. + other @ (crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::TypeMismatch { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BadRequest { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::MirrorReadOnly { .. }) => CompensationHint::Custom { + constraint: "apply_failed".into(), + detail: other.to_string(), + }, + } +} + +/// [`compensation_hint_for_dispatch_error`] for a Data-Plane verdict. +fn compensation_hint_for_code(code: &ErrorCode) -> CompensationHint { + match code { + ErrorCode::RejectedConstraint { constraint, detail } => CompensationHint::Custom { + constraint: constraint.clone(), + detail: detail.clone(), + }, + ErrorCode::RejectedPrevalidation { reason } => CompensationHint::Custom { + constraint: "prevalidation".into(), + detail: reason.clone(), + }, + ErrorCode::RejectedAuthz { .. } => CompensationHint::PermissionDenied, + ErrorCode::RateExceeded { retry_after_ms, .. } => CompensationHint::RateLimited { + retry_after_ms: *retry_after_ms, + }, + // Never judged the write: re-push it. + ErrorCode::DeadlineExceeded + | ErrorCode::ExpiredBeforeExecution + | ErrorCode::ResourcesExhausted + | ErrorCode::DispatchCapacity { .. } + | ErrorCode::ConflictRetry + | ErrorCode::OllpRetryRequired + | ErrorCode::CrdtFrontierMismatch { .. } + | ErrorCode::CollectionDraining { .. } + | ErrorCode::RetryableRefusal { .. } => CompensationHint::Retry { retry_after_ms: 0 }, + other @ (ErrorCode::SyncRejected { .. } + | ErrorCode::SyncNotApplied { .. } + | ErrorCode::NotFound + | ErrorCode::FanOutExceeded + | ErrorCode::RejectedDanglingEdge { .. } + | ErrorCode::DuplicateWrite + | ErrorCode::AppendOnlyViolation { .. } + | ErrorCode::BalanceViolation { .. } + | ErrorCode::PeriodLocked { .. } + | ErrorCode::PeriodLockMisconfigured { .. } + | ErrorCode::RetentionViolation { .. } + | ErrorCode::LegalHoldActive { .. } + | ErrorCode::StateTransitionViolation { .. } + | ErrorCode::TransitionCheckViolation { .. } + | ErrorCode::TypeGuardViolation { .. } + | ErrorCode::TypeMismatch { .. } + | ErrorCode::CounterFault { .. } + | ErrorCode::InsufficientBalance { .. } + | ErrorCode::RecursionDepthExceeded { .. } + | ErrorCode::UndefinedColumn { .. } + | ErrorCode::Internal { .. } + | ErrorCode::Unsupported { .. } + | ErrorCode::RollbackFailed { .. } + | ErrorCode::TxnOverlayMemoryExceeded { .. } + | ErrorCode::DivisionByZero + | ErrorCode::UndefinedFunction { .. } + | ErrorCode::DataException { .. } + | ErrorCode::BadRequest { .. }) => CompensationHint::Custom { + constraint: "apply_failed".into(), + detail: format!("{other:?}"), + }, + } +} diff --git a/nodedb/src/control/server/sync/async_dispatch/delta/mod.rs b/nodedb/src/control/server/sync/async_dispatch/delta/mod.rs index 9d412c484..483b0b90a 100644 --- a/nodedb/src/control/server/sync/async_dispatch/delta/mod.rs +++ b/nodedb/src/control/server/sync/async_dispatch/delta/mod.rs @@ -4,6 +4,7 @@ mod apply; mod authorize; +mod compensation; mod outcome; mod peer_identity; mod signature; diff --git a/nodedb/src/control/server/sync/async_dispatch/delta/outcome.rs b/nodedb/src/control/server/sync/async_dispatch/delta/outcome.rs index aa5b50871..9a1a2421c 100644 --- a/nodedb/src/control/server/sync/async_dispatch/delta/outcome.rs +++ b/nodedb/src/control/server/sync/async_dispatch/delta/outcome.rs @@ -28,10 +28,11 @@ use tracing::warn; use nodedb_types::sync::violation::ViolationType; use nodedb_types::sync::wire::{AckStatus, SyncAckResult, SyncOutcome}; -use super::super::super::refusal::retryable_refusal_reason; +use super::super::super::refusal::ack_status_for_dispatch_error; use super::super::super::wire::{ CompensationHint, DeltaAckMsg, DeltaPushMsg, DeltaRejectMsg, SyncFrame, SyncMessageType, }; +use super::compensation::compensation_hint_for_dispatch_error; /// Build the client frame for a completed dispatch. /// @@ -153,11 +154,17 @@ fn frame_for_dispatch_error(delta_msg: &DeltaPushMsg, error: &crate::Error) -> O { return reject_frame(delta_msg, violation); } - if let Some(reason) = retryable_refusal_reason(error) { + let hint = compensation_hint_for_dispatch_error(error); + // A rate refusal carries its delay in the hint, which a gap ack cannot. + // Every other failure that never judged the write is refused retryably, + // through the classifier every engine ack uses. + if !matches!(hint, CompensationHint::RateLimited { .. }) + && let AckStatus::Gap { expected } = ack_status_for_dispatch_error(error, delta_msg.seq) + { warn!( collection = %delta_msg.collection, doc = %delta_msg.document_id, - reason, + error = %error, "sync: delta refused retryably before apply; client should re-push at this seq" ); let ack = DeltaAckMsg { @@ -165,14 +172,11 @@ fn frame_for_dispatch_error(delta_msg: &DeltaPushMsg, error: &crate::Error) -> O lsn: 0, clock_skew_warning_ms: None, applied_seq: delta_msg.seq.saturating_sub(1), - status: AckStatus::Gap { - expected: delta_msg.seq, - }, + status: AckStatus::Gap { expected }, }; return SyncFrame::try_encode(SyncMessageType::DeltaAck, &ack); } - let hint = compensation_hint_for_dispatch_error(error); warn!( collection = %delta_msg.collection, doc = %delta_msg.document_id, @@ -188,50 +192,6 @@ fn frame_for_dispatch_error(delta_msg: &DeltaPushMsg, error: &crate::Error) -> O SyncFrame::try_encode(SyncMessageType::DeltaReject, &reject) } -/// Classify a dispatch failure into the hint the edge compensates against. -pub(super) fn compensation_hint_for_dispatch_error(e: &crate::Error) -> CompensationHint { - use crate::bridge::envelope::ErrorCode; - - match e { - crate::Error::DataPlane(code) => match code { - ErrorCode::RejectedConstraint { constraint, detail } => CompensationHint::Custom { - constraint: constraint.clone(), - detail: detail.clone(), - }, - ErrorCode::RejectedPrevalidation { reason } => CompensationHint::Custom { - constraint: "prevalidation".into(), - detail: reason.clone(), - }, - ErrorCode::RejectedAuthz { .. } => CompensationHint::PermissionDenied, - ErrorCode::RateExceeded { retry_after_ms, .. } => CompensationHint::RateLimited { - retry_after_ms: *retry_after_ms, - }, - other => CompensationHint::Custom { - constraint: "apply_failed".into(), - detail: format!("{other:?}"), - }, - }, - crate::Error::RejectedConstraint { - constraint, detail, .. - } => CompensationHint::Custom { - constraint: constraint.clone(), - detail: detail.clone(), - }, - crate::Error::RejectedPrevalidation { constraint, reason } => CompensationHint::Custom { - constraint: constraint.clone(), - detail: reason.clone(), - }, - crate::Error::RejectedAuthz { .. } => CompensationHint::PermissionDenied, - crate::Error::RateExceeded { retry_after_ms, .. } => CompensationHint::RateLimited { - retry_after_ms: *retry_after_ms, - }, - other => CompensationHint::Custom { - constraint: "apply_failed".into(), - detail: other.to_string(), - }, - } -} - #[cfg(test)] mod tests { use super::*; @@ -441,4 +401,27 @@ mod tests { CompensationHint::PermissionDenied ); } + + /// A dispatch that timed out never judged the delta, so the sender + /// re-pushes it instead of compensating. + #[test] + fn a_timed_out_dispatch_reaches_the_client_as_a_retryable_ack() { + let error = crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(1), + }; + let frame = frame_for_dispatch(&delta(), &provisional(), Err(error)).expect("frame"); + let ack = decode_ack(&frame); + assert_eq!(ack.status, AckStatus::Gap { expected: 5 }); + } + + /// A rate refusal keeps its delay: it rejects with the rate hint. + #[test] + fn a_rate_refusal_keeps_its_delay_hint() { + let error = crate::Error::DataPlane(ErrorCode::RateExceeded { + gate: "writes".into(), + retry_after_ms: 1500, + }); + let frame = frame_for_dispatch(&delta(), &provisional(), Err(error)).expect("frame"); + assert_eq!(frame.msg_type, SyncMessageType::DeltaReject); + } } diff --git a/nodedb/src/control/server/sync/refusal.rs b/nodedb/src/control/server/sync/refusal.rs index ed75eb617..ddab2846d 100644 --- a/nodedb/src/control/server/sync/refusal.rs +++ b/nodedb/src/control/server/sync/refusal.rs @@ -15,16 +15,164 @@ use nodedb_types::sync::wire::AckStatus; +use crate::bridge::envelope::ErrorCode; + /// The reason text when `error` means "nothing applied, re-send the same frame". /// /// Matched on the typed error only — never by substring-matching the human /// message, which is how a rewording silently turns a retry into a loss. pub(super) fn retryable_refusal_reason(error: &crate::Error) -> Option<&str> { - use crate::bridge::envelope::ErrorCode; match error { crate::Error::RetryableRefusal { reason } => Some(reason), - crate::Error::DataPlane(ErrorCode::RetryableRefusal { reason }) => Some(reason), - _ => None, + crate::Error::DataPlane(code) => match code { + ErrorCode::RetryableRefusal { reason } => Some(reason), + ErrorCode::DeadlineExceeded + | ErrorCode::RejectedConstraint { .. } + | ErrorCode::RejectedPrevalidation { .. } + | ErrorCode::SyncRejected { .. } + | ErrorCode::SyncNotApplied { .. } + | ErrorCode::NotFound + | ErrorCode::RejectedAuthz { .. } + | ErrorCode::ConflictRetry + | ErrorCode::CrdtFrontierMismatch { .. } + | ErrorCode::FanOutExceeded + | ErrorCode::ResourcesExhausted + | ErrorCode::RejectedDanglingEdge { .. } + | ErrorCode::DuplicateWrite + | ErrorCode::AppendOnlyViolation { .. } + | ErrorCode::BalanceViolation { .. } + | ErrorCode::PeriodLocked { .. } + | ErrorCode::PeriodLockMisconfigured { .. } + | ErrorCode::RetentionViolation { .. } + | ErrorCode::LegalHoldActive { .. } + | ErrorCode::StateTransitionViolation { .. } + | ErrorCode::TransitionCheckViolation { .. } + | ErrorCode::TypeGuardViolation { .. } + | ErrorCode::TypeMismatch { .. } + | ErrorCode::CounterFault { .. } + | ErrorCode::InsufficientBalance { .. } + | ErrorCode::RateExceeded { .. } + | ErrorCode::CollectionDraining { .. } + | ErrorCode::RecursionDepthExceeded { .. } + | ErrorCode::UndefinedColumn { .. } + | ErrorCode::Internal { .. } + | ErrorCode::Unsupported { .. } + | ErrorCode::RollbackFailed { .. } + | ErrorCode::OllpRetryRequired + | ErrorCode::TxnOverlayMemoryExceeded { .. } + | ErrorCode::DivisionByZero + | ErrorCode::UndefinedFunction { .. } + | ErrorCode::DataException { .. } + | ErrorCode::DispatchCapacity { .. } + | ErrorCode::ExpiredBeforeExecution + | ErrorCode::BadRequest { .. } => None, + }, + crate::Error::RejectedConstraint { .. } + | crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::RejectedAuthz { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::DeadlineExceeded { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RejectedPrevalidation { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::TypeMismatch { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::RateExceeded { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::NoLeader { .. } + | crate::Error::NotLeader { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::SourceFrozen { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BadRequest { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::DispatchCapacity { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::Backpressure { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SequencerUnavailable + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::OllpExhausted { .. } + | crate::Error::MirrorReadOnly { .. } + | crate::Error::StaleReadNotLeader { .. } => None, } } @@ -32,31 +180,180 @@ pub(super) fn retryable_refusal_reason(error: &crate::Error) -> Option<&str> { /// refused on its merits. /// /// These are the failures where the cluster never judged the write at all — it -/// timed out, the leader moved, the sequencer was absent, memory pressure -/// shed it, or the dispatcher refused it at capacity. Nothing about the write +/// timed out, the leader moved, a quorum or the sequencer was absent, memory or +/// rate pressure shed it, the dispatcher refused it at capacity, or a +/// concurrent change aborted it before it applied. Nothing about the write /// itself is wrong, so the same bytes are expected to land once the condition /// clears. fn is_indeterminate(error: &crate::Error) -> bool { - use crate::bridge::envelope::ErrorCode; - matches!( - error, + match error { crate::Error::DeadlineExceeded { .. } - | crate::Error::CrdtAdmissionTimeout { .. } - | crate::Error::NoLeader { .. } - | crate::Error::NotLeader { .. } - | crate::Error::StaleReadNotLeader { .. } - | crate::Error::SequencerUnavailable - | crate::Error::Backpressure { .. } - | crate::Error::DispatchCapacity { .. } - | crate::Error::ConflictRetry { .. } - | crate::Error::DataPlane( - ErrorCode::DeadlineExceeded - | ErrorCode::ExpiredBeforeExecution - | ErrorCode::ResourcesExhausted - | ErrorCode::DispatchCapacity { .. } - | ErrorCode::ConflictRetry - ) - ) + | crate::Error::CrdtAdmissionTimeout { .. } + | crate::Error::NoLeader { .. } + | crate::Error::NotLeader { .. } + | crate::Error::StaleReadNotLeader { .. } + | crate::Error::SequencerUnavailable + | crate::Error::Backpressure { .. } + | crate::Error::DispatchCapacity { .. } + | crate::Error::ConflictRetry { .. } + | crate::Error::RetryableRefusal { .. } + // A concurrent change aborted the write before it applied: the + // retryable class `40`. + | crate::Error::RetryableSchemaChanged { .. } + | crate::Error::CrdtAdmissionRetriesExhausted { .. } + | crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::SourceFrozen { .. } + // No leader or no quorum took the write: the retryable leader class. + | crate::Error::RetryableLeaderChange { .. } + | crate::Error::GroupQuorumUnavailable { .. } + | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::MetadataLeaderUnavailable + | crate::Error::AuthorizationStateBehind { .. } + // Shed by a resource or rate gate: the class `53`. + | crate::Error::VShardAdmissionCapacityExceeded { .. } + | crate::Error::MemoryExhausted { .. } + | crate::Error::RateExceeded { .. } => true, + crate::Error::DataPlane(code) => is_indeterminate_code(code), + crate::Error::OllpExhausted { cause, .. } => match cause { + crate::OllpExhaustedCause::PredicateDrift + | crate::OllpExhaustedCause::AdmissionRefused { .. } => true, + crate::OllpExhaustedCause::PreAdmission(inner) => is_indeterminate(inner), + }, + // Refused on its merits, or a fault the same bytes reproduce. + crate::Error::RejectedConstraint { .. } + | crate::Error::TxnOverlayMemoryExceeded { .. } + | crate::Error::RejectedAuthz { .. } + | crate::Error::OffsetRegression { .. } + | crate::Error::RejectedPrevalidation { .. } + | crate::Error::AppendOnlyViolation { .. } + | crate::Error::BalanceViolation { .. } + | crate::Error::MaterializedSumTargetNotFound { .. } + | crate::Error::MaterializedSumResolutionMissing { .. } + | crate::Error::PeriodLocked { .. } + | crate::Error::PeriodLockMisconfigured { .. } + | crate::Error::RetentionViolation { .. } + | crate::Error::LegalHoldActive { .. } + | crate::Error::StateTransitionViolation { .. } + | crate::Error::TransitionCheckViolation { .. } + | crate::Error::TypeGuardViolation { .. } + | crate::Error::TypeMismatch { .. } + | crate::Error::InsufficientBalance { .. } + | crate::Error::CollectionNotFound { .. } + | crate::Error::DocumentNotFound { .. } + | crate::Error::CollectionDeactivated { .. } + | crate::Error::CrdtAdmissionInvalidPlan { .. } + | crate::Error::CrdtAdmissionCallerFence + | crate::Error::CrdtApplyRequiresAdmission + | crate::Error::CrdtApplyForbiddenInTransaction + | crate::Error::NotInTransactionBlock { .. } + | crate::Error::FanOutExceeded { .. } + | crate::Error::CrossCollectionNotColocated { .. } + | crate::Error::CloneWriteRequiresMaterialize { .. } + | crate::Error::BadRequest { .. } + | crate::Error::BackupTenantMismatch { .. } + | crate::Error::BackupKeyMismatch + | crate::Error::QuotaOvercommit { .. } + | crate::Error::PlanError { .. } + | crate::Error::FeatureNotSupported { .. } + | crate::Error::UndefinedFunction { .. } + | crate::Error::UndefinedObject { .. } + | crate::Error::ObjectNotInPrerequisiteState { .. } + | crate::Error::UndefinedColumn { .. } + | crate::Error::AmbiguousColumn { .. } + | crate::Error::UnknownStrictField { .. } + | crate::Error::DivisionByZero + | crate::Error::DataException { .. } + | crate::Error::InvalidLimitValue { .. } + | crate::Error::ExecutionLimitExceeded { .. } + | crate::Error::LimitExceeded { .. } + | crate::Error::Wal(_) + | crate::Error::Dispatch { .. } + | crate::Error::Storage { .. } + | crate::Error::ColdStorage { .. } + | crate::Error::Serialization { .. } + | crate::Error::Codec { .. } + | crate::Error::SegmentCorrupted { .. } + | crate::Error::Crdt(_) + | crate::Error::Io(_) + | crate::Error::Config { .. } + | crate::Error::Encryption { .. } + | crate::Error::Bridge { .. } + | crate::Error::VersionCompat { .. } + | crate::Error::Internal { .. } + | crate::Error::Shaping(_) + | crate::Error::RemoteTyped { .. } + | crate::Error::DescriptorVersionAnomaly { .. } + | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CatalogIntegrityViolation { .. } + | crate::Error::Promql(_) + | crate::Error::DependentObjectsExist { .. } + | crate::Error::CascadeCycle { .. } + | crate::Error::CrossShardInExplicitTransaction + | crate::Error::SessionCapExceeded { .. } + | crate::Error::SessionIdleTimeout + | crate::Error::SessionTokenExpired + | crate::Error::SessionKilledByAdmin + | crate::Error::SessionUserDropped + | crate::Error::OidcProviderTenantUnbound + | crate::Error::OidcProviderTenantUnavailable { .. } + | crate::Error::ExternalRoleUndefined { .. } + | crate::Error::OidcNoDefaultDatabase { .. } + | crate::Error::TenantVectorDimExceeded { .. } + | crate::Error::TenantGraphDepthExceeded { .. } + | crate::Error::RoleInheritanceCycle { .. } + | crate::Error::RoleInheritanceDepthExceeded { .. } + | crate::Error::MirrorReadOnly { .. } => false, + } +} + +/// [`is_indeterminate`] for a Data-Plane verdict. +fn is_indeterminate_code(code: &ErrorCode) -> bool { + match code { + ErrorCode::DeadlineExceeded + | ErrorCode::ExpiredBeforeExecution + | ErrorCode::ResourcesExhausted + | ErrorCode::DispatchCapacity { .. } + | ErrorCode::ConflictRetry + | ErrorCode::RetryableRefusal { .. } + | ErrorCode::OllpRetryRequired + | ErrorCode::CrdtFrontierMismatch { .. } + | ErrorCode::CollectionDraining { .. } + | ErrorCode::RateExceeded { .. } => true, + // A sync hold is decided by the session that owns the stream before + // it reaches this classifier. Every other code is a verdict. + ErrorCode::RejectedConstraint { .. } + | ErrorCode::RejectedPrevalidation { .. } + | ErrorCode::SyncRejected { .. } + | ErrorCode::SyncNotApplied { .. } + | ErrorCode::NotFound + | ErrorCode::RejectedAuthz { .. } + | ErrorCode::FanOutExceeded + | ErrorCode::RejectedDanglingEdge { .. } + | ErrorCode::DuplicateWrite + | ErrorCode::AppendOnlyViolation { .. } + | ErrorCode::BalanceViolation { .. } + | ErrorCode::PeriodLocked { .. } + | ErrorCode::PeriodLockMisconfigured { .. } + | ErrorCode::RetentionViolation { .. } + | ErrorCode::LegalHoldActive { .. } + | ErrorCode::StateTransitionViolation { .. } + | ErrorCode::TransitionCheckViolation { .. } + | ErrorCode::TypeGuardViolation { .. } + | ErrorCode::TypeMismatch { .. } + | ErrorCode::CounterFault { .. } + | ErrorCode::InsufficientBalance { .. } + | ErrorCode::RecursionDepthExceeded { .. } + | ErrorCode::UndefinedColumn { .. } + | ErrorCode::Internal { .. } + | ErrorCode::Unsupported { .. } + | ErrorCode::RollbackFailed { .. } + | ErrorCode::TxnOverlayMemoryExceeded { .. } + | ErrorCode::DivisionByZero + | ErrorCode::UndefinedFunction { .. } + | ErrorCode::DataException { .. } + | ErrorCode::BadRequest { .. } => false, + } } /// The [`AckStatus`] an engine ack must carry when its dispatch failed. @@ -214,4 +511,50 @@ mod tests { "a failed dispatch must never be reported as applied" ); } + + /// A write aborted by a concurrent change, or not taken for want of a + /// quorum or under a rate gate, never got a verdict, so it is retried. + #[test] + fn retry_class_failures_are_retryable() { + let errors = [ + crate::Error::CalvinSerializationConflict, + crate::Error::GroupQuorumUnavailable { + group_id: 1, + voters: vec![1, 2, 3], + unreachable: vec![2, 3], + }, + crate::Error::RateExceeded { + gate: "sync".into(), + detail: "over budget".into(), + retry_after_ms: 10, + }, + crate::Error::DataPlane(ErrorCode::OllpRetryRequired), + crate::Error::OllpExhausted { + retries: 3, + cause: crate::OllpExhaustedCause::PredicateDrift, + }, + ]; + for error in errors { + assert_eq!( + ack_status_for_dispatch_error(&error, 5), + AckStatus::Gap { expected: 5 }, + "{error:?}" + ); + } + } + + /// Retry exhaustion before admission takes the verdict of its cause. + #[test] + fn pre_admission_exhaustion_follows_its_cause() { + let error = crate::Error::OllpExhausted { + retries: 3, + cause: crate::OllpExhaustedCause::PreAdmission(Box::new(crate::Error::BadRequest { + detail: "bad key".into(), + })), + }; + assert!(matches!( + ack_status_for_dispatch_error(&error, 5), + AckStatus::Rejected { .. } + )); + } } diff --git a/nodedb/src/data/executor/handlers/control/reindex/csr.rs b/nodedb/src/data/executor/handlers/control/reindex/csr.rs index b316ee6df..c5ec2ebe5 100644 --- a/nodedb/src/data/executor/handlers/control/reindex/csr.rs +++ b/nodedb/src/data/executor/handlers/control/reindex/csr.rs @@ -32,9 +32,35 @@ pub const CSR_REBUILD_JOURNAL_MAX_BYTES: usize = 64 << 20; /// Map a graph-engine error into the crate error. pub(super) fn graph_err(e: nodedb_graph::GraphError) -> crate::Error { - crate::Error::Storage { - engine: "graph".to_string(), - detail: e.to_string(), + use nodedb_graph::GraphError; + match e { + // The engine's memory budget, the same class a vector or FTS budget + // refusal has. + GraphError::MemoryBudget(_) => crate::Error::MemoryExhausted { + engine: "graph".to_string(), + }, + // A rebuild already holds the partition's journal: the index is busy. + GraphError::RebuildInProgress => crate::Error::ObjectNotInPrerequisiteState { + object: "graph CSR index".to_string(), + detail: e.to_string(), + }, + other @ (GraphError::LabelOverflow { .. } + | GraphError::NodeOverflow { .. } + | GraphError::WithdrawRefused { .. } + | GraphError::RebuildSuperseded + | GraphError::RebuildJournalOverflow { .. } + | GraphError::RebuildReplayDiverged { .. } + | GraphError::RebuildSnapshotInvalid { .. }) => crate::Error::Storage { + engine: "graph".to_string(), + detail: other.to_string(), + }, + // `GraphError` is `#[non_exhaustive]` and lives in another crate, so + // the compiler requires this arm. A variant this build cannot name is + // a storage fault. + other => crate::Error::Storage { + engine: "graph".to_string(), + detail: other.to_string(), + }, } } @@ -145,3 +171,18 @@ impl CoreLoop { ) } } + +#[cfg(test)] +mod tests { + use super::*; + + /// A REINDEX refused because a rebuild is running reports a busy index, + /// not a storage fault. + #[test] + fn a_running_rebuild_is_a_busy_index() { + assert!(matches!( + graph_err(nodedb_graph::GraphError::RebuildInProgress), + crate::Error::ObjectNotInPrerequisiteState { .. } + )); + } +} diff --git a/nodedb/src/data/executor/handlers/point/apply_put/stored_body.rs b/nodedb/src/data/executor/handlers/point/apply_put/stored_body.rs index 63c8a0e11..e6fb32f3a 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/stored_body.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/stored_body.rs @@ -125,14 +125,7 @@ impl CoreLoop { ) } else { strict_format::bytes_to_binary_tuple(&value, schema, collection) - } - .map_err(|e| match e { - crate::Error::UnknownStrictField { .. } => e, - other => crate::Error::Serialization { - format: "binary_tuple".into(), - detail: other.to_string(), - }, - })?; + }?; Ok(StoredBody { value, stored }) } diff --git a/nodedb/src/data/executor/handlers/point/apply_put/types.rs b/nodedb/src/data/executor/handlers/point/apply_put/types.rs index 51e046b11..0672a11b1 100644 --- a/nodedb/src/data/executor/handlers/point/apply_put/types.rs +++ b/nodedb/src/data/executor/handlers/point/apply_put/types.rs @@ -140,9 +140,60 @@ pub(in crate::data::executor) fn map_enforcement_error(e: ErrorCode) -> crate::E collection, detail: "collection has an active legal hold: DELETE rejected".to_string(), }, - other => crate::Error::Storage { - engine: "enforcement".into(), - detail: format!("unexpected enforcement error: {other:?}"), - }, + // Every other verdict keeps its code, so the statement renders the + // SQLSTATE the transactional path renders for it. + other @ (ErrorCode::DeadlineExceeded + | ErrorCode::RejectedConstraint { .. } + | ErrorCode::RejectedPrevalidation { .. } + | ErrorCode::RetryableRefusal { .. } + | ErrorCode::SyncRejected { .. } + | ErrorCode::SyncNotApplied { .. } + | ErrorCode::NotFound + | ErrorCode::RejectedAuthz { .. } + | ErrorCode::ConflictRetry + | ErrorCode::CrdtFrontierMismatch { .. } + | ErrorCode::FanOutExceeded + | ErrorCode::ResourcesExhausted + | ErrorCode::RejectedDanglingEdge { .. } + | ErrorCode::DuplicateWrite + | ErrorCode::BalanceViolation { .. } + | ErrorCode::TypeGuardViolation { .. } + | ErrorCode::TypeMismatch { .. } + | ErrorCode::CounterFault { .. } + | ErrorCode::InsufficientBalance { .. } + | ErrorCode::RateExceeded { .. } + | ErrorCode::CollectionDraining { .. } + | ErrorCode::RecursionDepthExceeded { .. } + | ErrorCode::UndefinedColumn { .. } + | ErrorCode::Internal { .. } + | ErrorCode::Unsupported { .. } + | ErrorCode::RollbackFailed { .. } + | ErrorCode::OllpRetryRequired + | ErrorCode::TxnOverlayMemoryExceeded { .. } + | ErrorCode::DivisionByZero + | ErrorCode::UndefinedFunction { .. } + | ErrorCode::DataException { .. } + | ErrorCode::DispatchCapacity { .. } + | ErrorCode::ExpiredBeforeExecution + | ErrorCode::BadRequest { .. }) => crate::Error::DataPlane(other), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A verdict with no Control-Plane twin keeps its code instead of + /// becoming a storage error. + #[test] + fn an_unmapped_verdict_keeps_its_code() { + let code = ErrorCode::TypeGuardViolation { + collection: "orders".into(), + detail: "amount must be positive".into(), + }; + match map_enforcement_error(code.clone()) { + crate::Error::DataPlane(kept) => assert_eq!(kept, code), + other => panic!("expected the typed verdict, got {other:?}"), + } } } diff --git a/nodedb/src/data/executor/handlers/point/update/post_image.rs b/nodedb/src/data/executor/handlers/point/update/post_image.rs index f8e33afc0..d277e3759 100644 --- a/nodedb/src/data/executor/handlers/point/update/post_image.rs +++ b/nodedb/src/data/executor/handlers/point/update/post_image.rs @@ -103,14 +103,7 @@ impl CoreLoop { } else { strict_format::value_to_binary_tuple(&ndb_val, schema, collection) }; - result.map_err(|e| match e { - crate::Error::UnknownStrictField { column, .. } => { - ErrorCode::UndefinedColumn { column } - } - other => ErrorCode::Internal { - detail: format!("strict re-encode: {other}"), - }, - }) + result.map_err(ErrorCode::from) } /// Apply the assignments and recompute generated columns, touching no diff --git a/nodedb/src/data/executor/handlers/recursive_value/eval.rs b/nodedb/src/data/executor/handlers/recursive_value/eval.rs index 0ac0473c2..b10e5c0a9 100644 --- a/nodedb/src/data/executor/handlers/recursive_value/eval.rs +++ b/nodedb/src/data/executor/handlers/recursive_value/eval.rs @@ -38,8 +38,16 @@ impl From for ErrorCode { match err { EvalError::UndefinedColumn { column } => ErrorCode::UndefinedColumn { column }, EvalError::DivisionByZero => ErrorCode::DivisionByZero, - other => ErrorCode::Unsupported { - detail: other.to_string(), + EvalError::Unsupported { detail } => ErrorCode::Unsupported { detail }, + // A result out of range is a data exception (class `22`), not an + // unsupported feature. + overflow @ EvalError::Overflow { .. } => ErrorCode::DataException { + detail: overflow.to_string(), + }, + // A non-boolean condition is a datatype error in the statement + // (class `42`), not an unsupported feature. + condition @ EvalError::NonBooleanCondition => ErrorCode::BadRequest { + detail: condition.to_string(), }, } } @@ -306,3 +314,33 @@ fn finite(result: f64, op: &'static str) -> EvalResult { Err(EvalError::Overflow { op }) } } + +#[cfg(test)] +mod tests { + use super::*; + + /// Each evaluation failure takes the Data-Plane code of its own class. + #[test] + fn eval_errors_take_the_code_of_their_class() { + assert!(matches!( + ErrorCode::from(EvalError::Overflow { op: "addition" }), + ErrorCode::DataException { .. } + )); + assert!(matches!( + ErrorCode::from(EvalError::NonBooleanCondition), + ErrorCode::BadRequest { .. } + )); + assert_eq!( + ErrorCode::from(EvalError::Unsupported { + detail: "lateral".into() + }), + ErrorCode::Unsupported { + detail: "lateral".into() + } + ); + assert_eq!( + ErrorCode::from(EvalError::DivisionByZero), + ErrorCode::DivisionByZero + ); + } +} diff --git a/nodedb/src/data/executor/handlers/recursive_value/handler.rs b/nodedb/src/data/executor/handlers/recursive_value/handler.rs index 08b8c8b41..0c170d1e3 100644 --- a/nodedb/src/data/executor/handlers/recursive_value/handler.rs +++ b/nodedb/src/data/executor/handlers/recursive_value/handler.rs @@ -143,6 +143,12 @@ fn locate_error(cte_name: &str, arm: &str, err: super::eval::EvalError) -> Error ErrorCode::Unsupported { detail } => ErrorCode::Unsupported { detail: format!("WITH RECURSIVE '{cte_name}' ({arm}): {detail}"), }, + ErrorCode::DataException { detail } => ErrorCode::DataException { + detail: format!("WITH RECURSIVE '{cte_name}' ({arm}): {detail}"), + }, + ErrorCode::BadRequest { detail } => ErrorCode::BadRequest { + detail: format!("WITH RECURSIVE '{cte_name}' ({arm}): {detail}"), + }, other => other, } } diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/body.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/body.rs index 979a46b07..d8344b4b6 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/body.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/body.rs @@ -124,14 +124,7 @@ impl CoreLoop { ) } else { strict_format::bytes_to_binary_tuple(&encoded_input, schema, collection) - } - .map_err(|e| match e { - crate::Error::UnknownStrictField { .. } => e, - other => crate::Error::Serialization { - format: "binary_tuple".into(), - detail: other.to_string(), - }, - })?; + }?; Ok(stored) } else { Ok(value) @@ -243,14 +236,7 @@ impl CoreLoop { ) } else { strict_format::value_to_binary_tuple(&ndb_val, schema, collection) - } - .map_err(|e| match e { - crate::Error::UnknownStrictField { .. } => e, - other => crate::Error::Serialization { - format: "binary_tuple".into(), - detail: other.to_string(), - }, - })?; + }?; Ok(bytes) } None => Ok(doc_format::encode_to_msgpack(&doc)), diff --git a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs index 404fa20cb..aede4a8a9 100644 --- a/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs +++ b/nodedb/src/data/executor/handlers/transaction/stage_write/stage_upsert.rs @@ -124,7 +124,7 @@ impl CoreLoop { }; if let Some(ref schema) = strict_schema { - let result = if bitemporal && schema.bitemporal { + if bitemporal && schema.bitemporal { strict_format::value_to_binary_tuple_bitemporal( &merged, schema, @@ -135,14 +135,7 @@ impl CoreLoop { ) } else { strict_format::value_to_binary_tuple(&merged, schema, ctx.collection) - }; - result.map_err(|e| match e { - crate::Error::UnknownStrictField { .. } => e, - other => crate::Error::Serialization { - format: "binary_tuple".into(), - detail: other.to_string(), - }, - }) + } } else { nodedb_types::value_to_msgpack(&merged).map_err(|e| crate::Error::Serialization { format: "msgpack".into(), diff --git a/nodedb/src/engine/sparse/inverted/errors.rs b/nodedb/src/engine/sparse/inverted/errors.rs index 06b76de1a..77698501f 100644 --- a/nodedb/src/engine/sparse/inverted/errors.rs +++ b/nodedb/src/engine/sparse/inverted/errors.rs @@ -13,6 +13,26 @@ pub(super) fn fts_index_err(e: nodedb_fts::FtsIndexError) -> crate detail: q.to_string(), }, FtsIndexError::Backend(inner) => inner, + // A document term past the segment format's cap is the caller's + // input, not a storage fault. + FtsIndexError::TermTooLong { len, max } => crate::Error::LimitExceeded { + limit_name: "fts_term_length", + value: len as u64, + max: max as u64, + }, + // The engine's memory budget, the same class a vector budget refusal has. + FtsIndexError::BudgetExhausted(_) => crate::Error::MemoryExhausted { + engine: "fts".into(), + }, + other @ (FtsIndexError::SurrogateOutOfRange { .. } | FtsIndexError::Segment(_)) => { + crate::Error::Storage { + engine: "inverted".into(), + detail: other.to_string(), + } + } + // `FtsIndexError` is `#[non_exhaustive]` and lives in another crate, + // so the compiler requires this arm. A variant this build cannot name + // is a storage fault. other => crate::Error::Storage { engine: "inverted".into(), detail: other.to_string(), @@ -34,3 +54,26 @@ pub(super) fn inverted_err(ctx: &str, e: impl std::fmt::Display) -> crate::Error pub(super) fn into_result_err(e: crate::Error) -> crate::Error { e } + +#[cfg(test)] +mod tests { + use super::*; + + /// An over-long term is the caller's input, not a storage fault. + #[test] + fn an_over_long_term_is_a_limit_refusal() { + let term: nodedb_fts::FtsIndexError = + nodedb_fts::FtsIndexError::TermTooLong { + len: 70_000, + max: 65_535, + }; + assert!(matches!( + fts_index_err(term), + crate::Error::LimitExceeded { + limit_name: "fts_term_length", + value: 70_000, + max: 65_535, + } + )); + } +} diff --git a/nodedb/src/error_classify/mod.rs b/nodedb/src/error_classify/mod.rs new file mode 100644 index 000000000..905bef551 --- /dev/null +++ b/nodedb/src/error_classify/mod.rs @@ -0,0 +1,10 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Internal [`crate::Error`] classification: the public mapping table and +//! the unclassified-failure predicate. + +mod public; +mod unclassified; + +pub(crate) use public::classify; +pub(crate) use unclassified::is_unclassified_failure; diff --git a/nodedb/src/error_classify.rs b/nodedb/src/error_classify/public.rs similarity index 81% rename from nodedb/src/error_classify.rs rename to nodedb/src/error_classify/public.rs index e34573e08..ac8b66095 100644 --- a/nodedb/src/error_classify.rs +++ b/nodedb/src/error_classify/public.rs @@ -33,7 +33,9 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { NodeDbError::backup_tenant_mismatch(*expected, *actual) } Error::BackupKeyMismatch => NodeDbError::backup_key_mismatch(), - Error::OffsetRegression { .. } => NodeDbError::bad_request(e.to_string()), + // Class `22`, the class of the `22023` SQLSTATE the SQL surfaces + // render for a regressed offset. + Error::OffsetRegression { .. } => NodeDbError::data_exception(e.to_string()), Error::DeadlineExceeded { .. } => NodeDbError::deadline_exceeded(), // Same contract as a write conflict, which callers already retry. Error::RetryableRefusal { reason } => NodeDbError::write_conflict("crdt", reason.clone()), @@ -205,9 +207,17 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { Error::MetadataLeaderUnavailable => NodeDbError::dispatch( "metadata raft group has no elected leader yet; retry exhausted".to_string(), ), - Error::AuthorizationStateBehind { .. } => NodeDbError::cluster(e.to_string()), - Error::GroupQuorumUnavailable { .. } => NodeDbError::cluster(e.to_string()), - Error::GroupMarksUnavailable { .. } => NodeDbError::cluster(e.to_string()), + // The code renders `55P03`, the SQLSTATE pgwire gives the variant, so + // the class survives a node hop as a bare numeric code. + Error::AuthorizationStateBehind { .. } => NodeDbError::from_wire( + nodedb_types::error::ErrorCode::STALE_READ_NOT_LEADER, + e.to_string(), + ), + // The code renders `55P03`, the SQLSTATE pgwire gives both variants. + // Nothing was applied, and a retry succeeds once a majority answers. + Error::GroupQuorumUnavailable { .. } | Error::GroupMarksUnavailable { .. } => { + NodeDbError::from_wire(nodedb_types::error::ErrorCode::NO_LEADER, e.to_string()) + } Error::ExecutionLimitExceeded { detail } => NodeDbError::bad_request(detail), Error::LimitExceeded { limit_name, @@ -273,9 +283,21 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { this node is running in embedded/local mode" .to_owned(), ), - Error::OllpExhausted { retries, cause } => NodeDbError::bad_request(format!( - "optimistic retry gave up after {retries} attempts: {cause}" - )), + // The code follows the cause, the same way pgwire picks the SQLSTATE, + // so the class survives a node hop as a bare numeric code. + Error::OllpExhausted { retries, cause } => { + let message = format!("optimistic retry gave up after {retries} attempts: {cause}"); + let code = match cause { + crate::OllpExhaustedCause::PredicateDrift => { + nodedb_types::error::ErrorCode::WRITE_CONFLICT + } + crate::OllpExhaustedCause::PreAdmission(inner) => classify(inner).code(), + crate::OllpExhaustedCause::AdmissionRefused { .. } => { + nodedb_types::error::ErrorCode::RATE_EXCEEDED + } + }; + NodeDbError::from_wire(code, message) + } Error::SessionCapExceeded { cap } => NodeDbError::bad_request(format!( "session cap ({cap}) exceeded — rejecting new login" )), @@ -346,34 +368,6 @@ pub(crate) fn classify(e: &Error) -> NodeDbError { } } -/// True when the error carries neither a client-matchable classification from -/// [`classify`] nor a retry contract a caller matches by variant. Only these -/// may be re-wrapped in a transport error. -pub(crate) fn is_unclassified_failure(e: &Error) -> bool { - matches!( - e, - Error::Wal(_) - | Error::Dispatch { .. } - | Error::Storage { .. } - | Error::ColdStorage { .. } - | Error::Serialization { .. } - | Error::Codec { .. } - | Error::SegmentCorrupted { .. } - | Error::Crdt(_) - | Error::Io(_) - | Error::Config { .. } - | Error::Encryption { .. } - | Error::Bridge { .. } - | Error::VersionCompat { .. } - | Error::Internal { .. } - | Error::DescriptorVersionAnomaly { .. } - | Error::CatalogIntegrityViolation { .. } - | Error::CollectionPurgeRowMissing { .. } - | Error::MaterializedSumResolutionMissing { .. } - | Error::CascadeCycle { .. } - ) -} - #[cfg(test)] mod tests { use nodedb_types::error::ErrorCode; @@ -412,23 +406,65 @@ mod tests { assert!(err.to_string().contains("secret_vault")); } - /// Verdicts the state machine reached must never be re-wrapped as - /// transport failures; machinery failures may be. + /// Retry exhaustion carries the code of its cause: drift is a write + /// conflict, a refused gate is a rate refusal, and a pre-admission + /// failure keeps the class of the error that caused it. #[test] - fn only_machinery_failures_are_unclassified() { - assert!(!super::is_unclassified_failure( - &Error::RejectedConstraint { - collection: "docs".to_owned(), - constraint: "unique".to_owned(), - detail: "duplicate key value 'dup'".to_owned(), - } - )); - assert!(!super::is_unclassified_failure(&Error::RejectedAuthz { - tenant_id: TenantId::new(1), - resource: "docs".to_owned(), - })); - assert!(super::is_unclassified_failure(&Error::Internal { - detail: "apply error".to_owned(), - })); + fn ollp_exhaustion_takes_the_code_of_its_cause() { + let drift = Error::OllpExhausted { + retries: 3, + cause: crate::OllpExhaustedCause::PredicateDrift, + }; + assert_eq!(classify(&drift).code(), ErrorCode::WRITE_CONFLICT); + + let refused = Error::OllpExhausted { + retries: 3, + cause: crate::OllpExhaustedCause::AdmissionRefused { + detail: "circuit open".to_owned(), + }, + }; + assert_eq!(classify(&refused).code(), ErrorCode::RATE_EXCEEDED); + + let pre_admission = Error::OllpExhausted { + retries: 3, + cause: crate::OllpExhaustedCause::PreAdmission(Box::new(Error::CollectionNotFound { + tenant_id: TenantId::new(1), + collection: "orders".to_owned(), + })), + }; + let public = classify(&pre_admission); + assert_eq!(public.code(), ErrorCode::COLLECTION_NOT_FOUND); + assert!(public.message().contains("orders")); + } + + /// A retryable cluster refusal carries the code whose SQLSTATE class is + /// the one pgwire renders, so the class survives a node hop. + #[test] + fn retryable_cluster_refusals_carry_the_leader_class_code() { + let quorum = Error::GroupQuorumUnavailable { + group_id: 1, + voters: vec![1, 2, 3], + unreachable: vec![2, 3], + }; + assert_eq!(classify(&quorum).code(), ErrorCode::NO_LEADER); + let marks = Error::GroupMarksUnavailable { + group_id: 1, + refused_by: vec![2], + }; + assert_eq!(classify(&marks).code(), ErrorCode::NO_LEADER); + let behind = Error::AuthorizationStateBehind { + detail: "roles".to_owned(), + }; + assert_eq!(classify(&behind).code(), ErrorCode::STALE_READ_NOT_LEADER); + let regression = Error::OffsetRegression { + stream: "s".to_owned(), + group: "g".to_owned(), + partition_id: 0, + current_lsn: 2, + current_sequence: 2, + attempted_lsn: 1, + attempted_sequence: 1, + }; + assert_eq!(classify(®ression).code(), ErrorCode::DATA_EXCEPTION); } } diff --git a/nodedb/src/error_classify/unclassified.rs b/nodedb/src/error_classify/unclassified.rs new file mode 100644 index 000000000..ac7ed47ca --- /dev/null +++ b/nodedb/src/error_classify/unclassified.rs @@ -0,0 +1,147 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which internal errors carry no classification a caller acts on. + +use crate::error::Error; + +/// True when the error carries neither a client-matchable classification from +/// [`super::classify`] nor a retry contract a caller matches by variant. Only these +/// may be re-wrapped in a transport error. +pub(crate) fn is_unclassified_failure(e: &Error) -> bool { + match e { + Error::Wal(_) + | Error::Dispatch { .. } + | Error::Storage { .. } + | Error::ColdStorage { .. } + | Error::Serialization { .. } + | Error::Codec { .. } + | Error::SegmentCorrupted { .. } + | Error::Crdt(_) + | Error::Io(_) + | Error::Config { .. } + | Error::Encryption { .. } + | Error::Bridge { .. } + | Error::VersionCompat { .. } + | Error::Internal { .. } + | Error::DescriptorVersionAnomaly { .. } + | Error::CatalogIntegrityViolation { .. } + | Error::CollectionPurgeRowMissing { .. } + | Error::MaterializedSumResolutionMissing { .. } + | Error::CascadeCycle { .. } => true, + // A client-matchable class or a retry contract a caller matches by + // variant. + Error::RejectedConstraint { .. } + | Error::TxnOverlayMemoryExceeded { .. } + | Error::RejectedAuthz { .. } + | Error::OffsetRegression { .. } + | Error::DeadlineExceeded { .. } + | Error::ConflictRetry { .. } + | Error::CalvinSerializationConflict + | Error::CalvinParticipantError + | Error::RejectedPrevalidation { .. } + | Error::RetryableRefusal { .. } + | Error::AppendOnlyViolation { .. } + | Error::BalanceViolation { .. } + | Error::MaterializedSumTargetNotFound { .. } + | Error::PeriodLocked { .. } + | Error::PeriodLockMisconfigured { .. } + | Error::RetentionViolation { .. } + | Error::LegalHoldActive { .. } + | Error::StateTransitionViolation { .. } + | Error::TransitionCheckViolation { .. } + | Error::TypeGuardViolation { .. } + | Error::TypeMismatch { .. } + | Error::InsufficientBalance { .. } + | Error::RateExceeded { .. } + | Error::CollectionNotFound { .. } + | Error::DocumentNotFound { .. } + | Error::CollectionDeactivated { .. } + | Error::VShardAdmissionCapacityExceeded { .. } + | Error::CrdtAdmissionRetriesExhausted { .. } + | Error::CrdtAdmissionInvalidPlan { .. } + | Error::CrdtAdmissionCallerFence + | Error::CrdtApplyRequiresAdmission + | Error::CrdtApplyForbiddenInTransaction + | Error::NotInTransactionBlock { .. } + | Error::CrdtAdmissionTimeout { .. } + | Error::NoLeader { .. } + | Error::NotLeader { .. } + | Error::FanOutExceeded { .. } + | Error::CrossCollectionNotColocated { .. } + | Error::SourceFrozen { .. } + | Error::CloneWriteRequiresMaterialize { .. } + | Error::BadRequest { .. } + | Error::BackupTenantMismatch { .. } + | Error::BackupKeyMismatch + | Error::QuotaOvercommit { .. } + | Error::PlanError { .. } + | Error::FeatureNotSupported { .. } + | Error::UndefinedFunction { .. } + | Error::UndefinedObject { .. } + | Error::ObjectNotInPrerequisiteState { .. } + | Error::UndefinedColumn { .. } + | Error::AmbiguousColumn { .. } + | Error::UnknownStrictField { .. } + | Error::DivisionByZero + | Error::DataException { .. } + | Error::InvalidLimitValue { .. } + | Error::RetryableSchemaChanged { .. } + | Error::RetryableLeaderChange { .. } + | Error::GroupQuorumUnavailable { .. } + | Error::GroupMarksUnavailable { .. } + | Error::MetadataLeaderUnavailable + | Error::AuthorizationStateBehind { .. } + | Error::ExecutionLimitExceeded { .. } + | Error::LimitExceeded { .. } + | Error::DispatchCapacity { .. } + | Error::MemoryExhausted { .. } + | Error::Backpressure { .. } + | Error::Shaping(_) + | Error::RemoteTyped { .. } + | Error::DataPlane(_) + | Error::Promql(_) + | Error::DependentObjectsExist { .. } + | Error::CrossShardInExplicitTransaction + | Error::SequencerUnavailable + | Error::SessionCapExceeded { .. } + | Error::SessionIdleTimeout + | Error::SessionTokenExpired + | Error::SessionKilledByAdmin + | Error::SessionUserDropped + | Error::OidcProviderTenantUnbound + | Error::OidcProviderTenantUnavailable { .. } + | Error::ExternalRoleUndefined { .. } + | Error::OidcNoDefaultDatabase { .. } + | Error::TenantVectorDimExceeded { .. } + | Error::TenantGraphDepthExceeded { .. } + | Error::RoleInheritanceCycle { .. } + | Error::RoleInheritanceDepthExceeded { .. } + | Error::OllpExhausted { .. } + | Error::MirrorReadOnly { .. } + | Error::StaleReadNotLeader { .. } => false, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::types::TenantId; + + /// Verdicts the state machine reached must never be re-wrapped as + /// transport failures; machinery failures may be. + #[test] + fn only_machinery_failures_are_unclassified() { + assert!(!is_unclassified_failure(&Error::RejectedConstraint { + collection: "docs".to_owned(), + constraint: "unique".to_owned(), + detail: "duplicate key value 'dup'".to_owned(), + })); + assert!(!is_unclassified_failure(&Error::RejectedAuthz { + tenant_id: TenantId::new(1), + resource: "docs".to_owned(), + })); + assert!(is_unclassified_failure(&Error::Internal { + detail: "apply error".to_owned(), + })); + } +} diff --git a/nodedb/src/error_from.rs b/nodedb/src/error_from.rs index b9a7b1420..0b8c24a3a 100644 --- a/nodedb/src/error_from.rs +++ b/nodedb/src/error_from.rs @@ -151,8 +151,12 @@ impl From for Error { | Ve::CheckpointPlaintextKeyRequired | Ve::CheckpointEncryptionError { .. } | Ve::CheckpointSerializationError { .. } - | Ve::CheckpointDeserializationError { .. } => Self::SegmentCorrupted { detail }, - // Unknown vector errors fail-stop. + | Ve::CheckpointDeserializationError { .. } + | Ve::VectorUnavailable { .. } + | Ve::VectorDecodeFailed { .. } => Self::SegmentCorrupted { detail }, + // `VectorError` is `#[non_exhaustive]` and lives in another crate, + // so the compiler requires this arm. A variant this build cannot + // name fails stop. _ => Self::SegmentCorrupted { detail }, } } @@ -315,11 +319,112 @@ impl From for nodedb_cluster::rpc_codec::TypedClusterError { constraint, detail, }, - other => { - // Preserve classification across multi-hop forwarding. - let message = other.to_string(); - let code = u32::from(NodeDbError::from(other).code().0); - TypedClusterError::Internal { code, message } + // Every other error crosses as its public numeric code, so a + // multi-hop forward keeps its class. + other @ (Error::TxnOverlayMemoryExceeded { .. } + | Error::RejectedAuthz { .. } + | Error::OffsetRegression { .. } + | Error::ConflictRetry { .. } + | Error::CalvinSerializationConflict + | Error::CalvinParticipantError + | Error::RejectedPrevalidation { .. } + | Error::RetryableRefusal { .. } + | Error::AppendOnlyViolation { .. } + | Error::BalanceViolation { .. } + | Error::MaterializedSumTargetNotFound { .. } + | Error::MaterializedSumResolutionMissing { .. } + | Error::PeriodLocked { .. } + | Error::PeriodLockMisconfigured { .. } + | Error::RetentionViolation { .. } + | Error::LegalHoldActive { .. } + | Error::StateTransitionViolation { .. } + | Error::TransitionCheckViolation { .. } + | Error::TypeGuardViolation { .. } + | Error::TypeMismatch { .. } + | Error::InsufficientBalance { .. } + | Error::RateExceeded { .. } + | Error::CollectionNotFound { .. } + | Error::DocumentNotFound { .. } + | Error::CollectionDeactivated { .. } + | Error::VShardAdmissionCapacityExceeded { .. } + | Error::CrdtAdmissionRetriesExhausted { .. } + | Error::CrdtAdmissionInvalidPlan { .. } + | Error::CrdtAdmissionCallerFence + | Error::CrdtApplyRequiresAdmission + | Error::CrdtApplyForbiddenInTransaction + | Error::NotInTransactionBlock { .. } + | Error::CrdtAdmissionTimeout { .. } + | Error::NoLeader { .. } + | Error::FanOutExceeded { .. } + | Error::CrossCollectionNotColocated { .. } + | Error::SourceFrozen { .. } + | Error::CloneWriteRequiresMaterialize { .. } + | Error::BadRequest { .. } + | Error::BackupTenantMismatch { .. } + | Error::BackupKeyMismatch + | Error::QuotaOvercommit { .. } + | Error::PlanError { .. } + | Error::FeatureNotSupported { .. } + | Error::UndefinedFunction { .. } + | Error::UndefinedObject { .. } + | Error::ObjectNotInPrerequisiteState { .. } + | Error::UndefinedColumn { .. } + | Error::AmbiguousColumn { .. } + | Error::UnknownStrictField { .. } + | Error::DivisionByZero + | Error::DataException { .. } + | Error::InvalidLimitValue { .. } + | Error::RetryableSchemaChanged { .. } + | Error::RetryableLeaderChange { .. } + | Error::GroupQuorumUnavailable { .. } + | Error::GroupMarksUnavailable { .. } + | Error::MetadataLeaderUnavailable + | Error::AuthorizationStateBehind { .. } + | Error::ExecutionLimitExceeded { .. } + | Error::LimitExceeded { .. } + | Error::Wal(_) + | Error::Dispatch { .. } + | Error::DispatchCapacity { .. } + | Error::Storage { .. } + | Error::ColdStorage { .. } + | Error::Serialization { .. } + | Error::Codec { .. } + | Error::SegmentCorrupted { .. } + | Error::MemoryExhausted { .. } + | Error::Backpressure { .. } + | Error::Crdt(_) + | Error::Io(_) + | Error::Config { .. } + | Error::Encryption { .. } + | Error::Bridge { .. } + | Error::VersionCompat { .. } + | Error::Internal { .. } + | Error::Shaping(_) + | Error::DescriptorVersionAnomaly { .. } + | Error::CollectionPurgeRowMissing { .. } + | Error::CatalogIntegrityViolation { .. } + | Error::Promql(_) + | Error::DependentObjectsExist { .. } + | Error::CascadeCycle { .. } + | Error::CrossShardInExplicitTransaction + | Error::SequencerUnavailable + | Error::SessionCapExceeded { .. } + | Error::SessionIdleTimeout + | Error::SessionTokenExpired + | Error::SessionKilledByAdmin + | Error::SessionUserDropped + | Error::OidcProviderTenantUnbound + | Error::OidcProviderTenantUnavailable { .. } + | Error::ExternalRoleUndefined { .. } + | Error::OidcNoDefaultDatabase { .. } + | Error::TenantVectorDimExceeded { .. } + | Error::TenantGraphDepthExceeded { .. } + | Error::RoleInheritanceCycle { .. } + | Error::RoleInheritanceDepthExceeded { .. } + | Error::OllpExhausted { .. } + | Error::MirrorReadOnly { .. } + | Error::StaleReadNotLeader { .. }) => { + crate::control::cluster::data_plane_error_wire::numeric_typed(other) } } } From 60c9b8e3549dcc7dee4cb6c3aea4800b548a7d1a Mon Sep 17 00:00:00 2001 From: Farhan Syah Date: Mon, 28 Sep 2026 03:00:33 +0800 Subject: [PATCH 56/64] feat(errors): replace generic DDL error strings with typed propagation Give DdlError first-class helpers for typed sources: from_error_in_context carries a crate::Error's own SQLSTATE/code behind a message prefix, internal marks a fault with no typed source (codec, storage, broken invariant) as XX000, and in_context prefixes a built DdlError's message without disturbing its class. Replace every neutral DDL handler's ddl_err(format!(...)) call site with one of these, so a catalog or propose failure keeps its real class (deadlock retry, privilege denial, not-found) instead of collapsing to a generic internal error. Add three error codes this now threads through: TRANSACTION_ROLLBACK for a Calvin participant abort with no read-set validated (retriable, unlike a write conflict), ACTIVE_SQL_TRANSACTION for a statement that cannot run inside an explicit transaction block (folds in NotInTransactionBlock, CrdtApplyForbiddenInTransaction, and CrossShardInExplicitTransaction), and DEPENDENT_OBJECTS_EXIST for a drop or revoke blocked by dependents. Give DROP ROLE a typed RoleInUse error carrying RoleDependents (Users or ChildRoles) in place of a formatted BadRequest, so it classifies as DEPENDENT_OBJECTS_EXIST/2BP01 like other dependent-object refusals. Extract the DROP USER reassignment path's OwnerKind enum into its own owner_kind module, shared unchanged by reassign_owned.rs. --- nodedb-types/src/error/code.rs | 8 + nodedb-types/src/error/code_table.rs | 6 + .../src/error/ctors/read_query_auth.rs | 14 ++ nodedb-types/src/error/ctors/write_path.rs | 26 +++ nodedb-types/src/error/details.rs | 12 ++ nodedb-types/src/error/msgpack/constants.rs | 6 + .../error/msgpack/decode/from_messagepack.rs | 52 ++++++ nodedb-types/src/error/msgpack/encode.rs | 9 + nodedb-types/src/error/types.rs | 23 +++ nodedb/src/bridge/envelope/error_code.rs | 1 + .../control/cluster/array_executor/refusal.rs | 1 + .../control/cluster/data_plane_error_wire.rs | 1 + .../control/cluster/metadata_applier/wedge.rs | 1 + .../control/gateway/error_map/class_parity.rs | 106 ++++++++++-- nodedb/src/control/gateway/error_map/resp.rs | 1 + nodedb/src/control/security/role.rs | 14 +- .../src/control/security/role_assignment.rs | 71 +++++++- .../http/routes/query/materialized/encode.rs | 5 +- .../server/native/dispatch/conversion.rs | 76 ++++++++- .../server/native/dispatch/direct_ops.rs | 2 +- .../src/control/server/native/dispatch/mod.rs | 5 +- .../src/control/server/native/dispatch/sql.rs | 6 +- .../server/native/dispatch/sql_loop.rs | 2 +- .../server/native/dispatch/streaming.rs | 2 +- .../server/native/dispatch/transaction.rs | 74 +++++++-- .../src/control/server/native/session/auth.rs | 22 ++- .../src/control/server/pgwire/connection.rs | 3 +- .../pgwire/ddl/database/use_database.rs | 4 +- .../control/server/pgwire/factory/startup.rs | 6 +- .../server/pgwire/handler/copy_handler.rs | 20 +-- .../server/pgwire/handler/cursor_query.rs | 10 +- .../control/server/pgwire/handler/facet.rs | 2 +- .../pgwire/handler/routing/calvin_dispatch.rs | 2 +- .../handler/routing/dispatch_loop/run.rs | 2 +- .../handler/routing/execute_dml_hooks.rs | 8 +- .../server/pgwire/handler/routing/planning.rs | 4 +- .../server/pgwire/handler/session_cmds.rs | 2 +- .../server/pgwire/handler/stream_response.rs | 41 ++--- .../server/pgwire/handler/tenant_session.rs | 4 +- .../control/server/pgwire/types/error_map.rs | 46 +++++- nodedb/src/control/server/pgwire/types/mod.rs | 2 +- .../server/pgwire/types/numeric_sqlstate.rs | 7 + .../src/control/server/shared/ddl/catalog.rs | 2 +- .../control/server/shared/ddl/engine_apply.rs | 7 +- .../server/shared/ddl/index_registry.rs | 12 +- .../server/shared/ddl/neutral/alert/create.rs | 2 +- .../shared/ddl/neutral/alert/replicate.rs | 8 +- .../shared/ddl/neutral/apikey/create.rs | 4 +- .../shared/ddl/neutral/apikey/manage.rs | 4 +- .../server/shared/ddl/neutral/apikey/parse.rs | 4 +- .../server/shared/ddl/neutral/auth_user.rs | 6 +- .../server/shared/ddl/neutral/blacklist.rs | 6 +- .../ddl/neutral/change_stream/create.rs | 2 +- .../shared/ddl/neutral/change_stream/drop.rs | 4 +- .../server/shared/ddl/neutral/cluster/raft.rs | 4 +- .../ddl/neutral/cluster/rebalance_cmd.rs | 2 +- .../neutral/collection/alter/add_column.rs | 2 +- .../collection/alter/materialized_sum.rs | 16 +- .../ddl/neutral/collection/alter/ownership.rs | 4 +- .../neutral/collection/alter/strict_schema.rs | 6 +- .../ddl/neutral/collection/alter/support.rs | 13 +- .../neutral/collection/alter/vector_model.rs | 5 +- .../ddl/neutral/collection/copy_from/entry.rs | 6 +- .../ddl/neutral/collection/copy_to/entry.rs | 16 +- .../ddl/neutral/collection/copy_to/format.rs | 10 +- .../ddl/neutral/collection/create/build.rs | 6 +- .../collection/dml/indexed_vector_fields.rs | 6 +- .../ddl/neutral/collection/dml/insert.rs | 15 +- .../neutral/collection/dml/parse/dispatch.rs | 45 ++--- .../ddl/neutral/collection/dml/triggers.rs | 15 +- .../shared/ddl/neutral/collection/drop.rs | 16 +- .../ddl/neutral/collection/index/build.rs | 32 ++-- .../ddl/neutral/collection/index/commit.rs | 6 +- .../ddl/neutral/collection/index/create.rs | 2 +- .../ddl/neutral/collection/index/drop.rs | 2 +- .../ddl/neutral/collection/index/teardown.rs | 44 ++--- .../ddl/neutral/collection/index_fanout.rs | 39 ++--- .../ddl/neutral/collection/show_indexes.rs | 2 +- .../shared/ddl/neutral/collection/undrop.rs | 6 +- .../ddl/neutral/collection/vector_metadata.rs | 4 +- .../shared/ddl/neutral/conflict_policy.rs | 13 +- .../shared/ddl/neutral/constraint/handlers.rs | 16 +- .../shared/ddl/neutral/constraint/show.rs | 2 +- .../ddl/neutral/consumer_group/commit.rs | 2 +- .../ddl/neutral/consumer_group/create.rs | 2 +- .../ddl/neutral/consumer_group/identity.rs | 2 +- .../ddl/neutral/consumer_group/replicate.rs | 6 +- .../ddl/neutral/continuous_agg/create.rs | 4 +- .../shared/ddl/neutral/continuous_agg/drop.rs | 2 +- .../shared/ddl/neutral/continuous_agg/show.rs | 2 +- .../shared/ddl/neutral/convert/driver.rs | 9 +- .../server/shared/ddl/neutral/crdt_ops.rs | 4 +- .../server/shared/ddl/neutral/custom_type.rs | 10 +- .../shared/ddl/neutral/database/alter.rs | 22 +-- .../ddl/neutral/database/backup_restore.rs | 2 +- .../shared/ddl/neutral/database/clone.rs | 95 +++++------ .../shared/ddl/neutral/database/create.rs | 60 ++++++- .../shared/ddl/neutral/database/drop.rs | 22 +-- .../ddl/neutral/database/materialize.rs | 2 +- .../ddl/neutral/database/mirror/create.rs | 6 +- .../ddl/neutral/database/mirror/promote.rs | 24 +-- .../ddl/neutral/database/mirror/show.rs | 2 +- .../shared/ddl/neutral/database/show.rs | 4 +- .../ddl/neutral/database/show_lineage.rs | 12 +- .../shared/ddl/neutral/database/show_quota.rs | 4 +- .../shared/ddl/neutral/database/show_usage.rs | 4 +- .../shared/ddl/neutral/deferred_effects.rs | 2 +- .../shared/ddl/neutral/dsl/crdt_merge.rs | 4 +- .../shared/ddl/neutral/dsl/sparse_index.rs | 4 +- .../shared/ddl/neutral/dsl/text_index.rs | 4 +- .../shared/ddl/neutral/dsl/vector_index.rs | 6 +- .../shared/ddl/neutral/estimate_count.rs | 2 +- .../server/shared/ddl/neutral/explain_ddl.rs | 2 +- .../server/shared/ddl/neutral/field_def.rs | 8 +- .../shared/ddl/neutral/function/alter.rs | 4 +- .../ddl/neutral/function/create/handler.rs | 2 +- .../shared/ddl/neutral/function/drop.rs | 6 +- .../ddl/neutral/function/wasm_aggregate.rs | 7 +- .../ddl/neutral/function/wasm_create.rs | 2 +- .../ddl/neutral/grant/database_permission.rs | 12 +- .../shared/ddl/neutral/grant/permission.rs | 10 +- .../server/shared/ddl/neutral/grant/role.rs | 4 +- .../shared/ddl/neutral/graph_ops/algo.rs | 6 +- .../shared/ddl/neutral/graph_ops/edge.rs | 49 +++--- .../ddl/neutral/graph_ops/edge_parse.rs | 2 +- .../shared/ddl/neutral/graph_ops/edge_rls.rs | 8 +- .../ddl/neutral/graph_ops/edge_stage.rs | 26 +-- .../shared/ddl/neutral/graph_ops/stats.rs | 5 +- .../shared/ddl/neutral/graph_ops/support.rs | 13 +- .../shared/ddl/neutral/graph_ops/traverse.rs | 6 +- .../shared/ddl/neutral/inspect_audit.rs | 6 +- .../shared/ddl/neutral/kv_atomic/dispatch.rs | 22 +-- .../shared/ddl/neutral/kv_atomic/handlers.rs | 8 +- .../shared/ddl/neutral/kv_sorted_index/ddl.rs | 8 +- .../ddl/neutral/kv_sorted_index/dispatch.rs | 10 +- .../server/shared/ddl/neutral/last_value.rs | 8 +- .../shared/ddl/neutral/maintenance/analyze.rs | 6 +- .../ddl/neutral/maintenance/auto_analyze.rs | 6 +- .../shared/ddl/neutral/maintenance/reindex.rs | 2 +- .../ddl/neutral/maintenance/vector_index.rs | 8 +- .../neutral/maintenance/vector_index_set.rs | 4 +- .../server/shared/ddl/neutral/match_ops.rs | 8 +- .../ddl/neutral/materialized_view/create.rs | 8 +- .../ddl/neutral/materialized_view/drop.rs | 6 +- .../ddl/neutral/materialized_view/show.rs | 6 +- .../control/server/shared/ddl/neutral/oidc.rs | 24 +-- .../server/shared/ddl/neutral/org_ddl.rs | 4 +- .../server/shared/ddl/neutral/period_lock.rs | 8 +- .../shared/ddl/neutral/permission_tree.rs | 10 +- .../shared/ddl/neutral/procedure/call.rs | 2 +- .../ddl/neutral/procedure/create/handler.rs | 2 +- .../shared/ddl/neutral/procedure/drop.rs | 6 +- .../neutral/query_functions/balance_as_of.rs | 10 +- .../convert_currency_lookup.rs | 2 +- .../ddl/neutral/query_functions/helpers.rs | 2 +- .../query_functions/temporal_lookup.rs | 2 +- .../query_functions/verify_audit_chain.rs | 2 +- .../neutral/query_functions/verify_balance.rs | 6 +- .../query_functions/verify_hash_chain.rs | 4 +- .../server/shared/ddl/neutral/quota_ddl.rs | 4 +- .../server/shared/ddl/neutral/rate_gate.rs | 30 +++- .../server/shared/ddl/neutral/read_gate.rs | 6 +- .../shared/ddl/neutral/redaction/create.rs | 6 +- .../shared/ddl/neutral/redaction/drop_show.rs | 4 +- .../server/shared/ddl/neutral/replicate.rs | 4 +- .../ddl/neutral/retention_policy/create.rs | 15 +- .../ddl/neutral/retention_policy/replicate.rs | 8 +- .../control/server/shared/ddl/neutral/rls.rs | 10 +- .../control/server/shared/ddl/neutral/role.rs | 12 +- .../ddl/neutral/router/string_engine_ops.rs | 9 +- .../ddl/neutral/router/string_versioning.rs | 5 +- .../ddl/neutral/router/typed_collection.rs | 4 +- .../ddl/neutral/router/typed_database.rs | 9 +- .../shared/ddl/neutral/schedule/create.rs | 2 +- .../shared/ddl/neutral/schedule/drop.rs | 6 +- .../shared/ddl/neutral/scope_ddl/define.rs | 2 +- .../shared/ddl/neutral/scope_ddl/grant.rs | 8 +- .../server/shared/ddl/neutral/sequence.rs | 71 +++++++- .../shared/ddl/neutral/service_account.rs | 14 +- .../server/shared/ddl/neutral/spatial.rs | 4 +- .../ddl/neutral/synonym_group/create.rs | 6 +- .../shared/ddl/neutral/synonym_group/drop.rs | 4 +- .../server/shared/ddl/neutral/tenant/alter.rs | 6 +- .../shared/ddl/neutral/tenant/alter_quota.rs | 10 +- .../shared/ddl/neutral/tenant/create.rs | 12 +- .../server/shared/ddl/neutral/tenant/drop.rs | 8 +- .../ddl/neutral/tenant/move_tenant/entry.rs | 16 +- .../neutral/tenant/move_tenant/recovery.rs | 4 +- .../ddl/neutral/tenant/show_in_database.rs | 6 +- .../shared/ddl/neutral/tenant/support.rs | 6 +- .../shared/ddl/neutral/timeseries/alter.rs | 4 +- .../shared/ddl/neutral/timeseries/create.rs | 4 +- .../server/shared/ddl/neutral/topic/create.rs | 2 +- .../server/shared/ddl/neutral/topic/drop.rs | 9 +- .../shared/ddl/neutral/topic/replicate.rs | 4 +- .../shared/ddl/neutral/topic_subscribe.rs | 2 +- .../server/shared/ddl/neutral/transfer.rs | 6 +- .../shared/ddl/neutral/tree_ops/children.rs | 2 +- .../ddl/neutral/tree_ops/create_index.rs | 45 +++-- .../server/shared/ddl/neutral/tree_ops/sum.rs | 4 +- .../shared/ddl/neutral/trigger/create.rs | 2 +- .../server/shared/ddl/neutral/trigger/drop.rs | 6 +- .../shared/ddl/neutral/typeguard/handlers.rs | 20 +-- .../shared/ddl/neutral/typeguard/validate.rs | 4 +- .../server/shared/ddl/neutral/user/alter.rs | 10 +- .../server/shared/ddl/neutral/user/create.rs | 6 +- .../server/shared/ddl/neutral/user/drop.rs | 6 +- .../server/shared/ddl/neutral/user/mod.rs | 1 + .../shared/ddl/neutral/user/owner_kind.rs | 63 +++++++ .../shared/ddl/neutral/user/reassign_owned.rs | 149 +++++++---------- .../shared/ddl/neutral/user/tenant_purge.rs | 22 ++- .../shared/ddl/neutral/vector_replicate.rs | 8 +- .../ddl/neutral/version_history/at_version.rs | 2 +- .../ddl/neutral/version_history/checkpoint.rs | 6 +- .../ddl/neutral/version_history/compact.rs | 4 +- .../ddl/neutral/version_history/replicate.rs | 6 +- .../ddl/neutral/version_history/restore.rs | 2 +- .../neutral/version_history/show_versions.rs | 2 +- .../shared/ddl/neutral/weighted_pick.rs | 23 ++- nodedb/src/control/server/shared/ddl/owner.rs | 12 +- .../src/control/server/shared/ddl/result.rs | 155 +++++++++++++++++- nodedb/src/control/server/shared/retry.rs | 1 + .../control/server/shared/session/outcome.rs | 2 +- .../sync/async_dispatch/delta/compensation.rs | 1 + nodedb/src/control/server/sync/refusal.rs | 2 + nodedb/src/error/types.rs | 8 + nodedb/src/error_classify/public.rs | 36 ++-- nodedb/src/error_classify/unclassified.rs | 1 + nodedb/src/error_from.rs | 1 + 229 files changed, 1747 insertions(+), 1028 deletions(-) create mode 100644 nodedb/src/control/server/shared/ddl/neutral/user/owner_kind.rs diff --git a/nodedb-types/src/error/code.rs b/nodedb-types/src/error/code.rs index c9e3d8b4c..b18de5d66 100644 --- a/nodedb-types/src/error/code.rs +++ b/nodedb-types/src/error/code.rs @@ -32,6 +32,11 @@ impl ErrorCode { pub const INSUFFICIENT_BALANCE: Self = Self(1022); pub const RATE_EXCEEDED: Self = Self(1023); pub const TYPE_GUARD_VIOLATION: Self = Self(1024); + /// A transaction rolled back for a reason other than a serialization + /// conflict, such as a participant error. The client retries it. + pub const TRANSACTION_ROLLBACK: Self = Self(1030); + /// The statement cannot run inside an explicit transaction block. + pub const ACTIVE_SQL_TRANSACTION: Self = Self(1031); // Read path (1100–1199) pub const COLLECTION_NOT_FOUND: Self = Self(1100); @@ -52,6 +57,9 @@ impl ErrorCode { /// A requested value/record does not exist. Generic: use for lookups /// that don't fit `DOCUMENT_NOT_FOUND`'s collection/id shape. pub const NOT_FOUND: Self = Self(1114); + /// A drop or revoke refused because other objects still depend on the + /// named object. + pub const DEPENDENT_OBJECTS_EXIST: Self = Self(1115); // Query (1200–1299) pub const PLAN_ERROR: Self = Self(1200); diff --git a/nodedb-types/src/error/code_table.rs b/nodedb-types/src/error/code_table.rs index eb0a2f8df..8caf50c95 100644 --- a/nodedb-types/src/error/code_table.rs +++ b/nodedb-types/src/error/code_table.rs @@ -71,6 +71,8 @@ error_code_table! { OVERFLOW => Overflow { collection: String::new() }, INSUFFICIENT_BALANCE => InsufficientBalance { collection: String::new() }, RATE_EXCEEDED => RateExceeded { gate: String::new() }, + TRANSACTION_ROLLBACK => TransactionRollback { detail: message.to_owned() }, + ACTIVE_SQL_TRANSACTION => ActiveSqlTransaction { detail: message.to_owned() }, // Read path. COLLECTION_NOT_FOUND => CollectionNotFound { collection: String::new() }, @@ -79,6 +81,7 @@ error_code_table! { ALREADY_EXISTS => AlreadyExists { object: String::new() }, OBJECT_NOT_READY => ObjectNotReady { object: String::new() }, NOT_FOUND => NotFound { detail: message.to_owned() }, + DEPENDENT_OBJECTS_EXIST => DependentObjectsExist { object: String::new() }, DOCUMENT_NOT_FOUND => DocumentNotFound { collection: String::new(), document_id: String::new() }, COLLECTION_DRAINING => CollectionDraining { collection: String::new() }, COLLECTION_DEACTIVATED => CollectionDeactivated { @@ -240,6 +243,9 @@ mod tests { ErrorCode::COLLECTION_DEACTIVATED, ErrorCode::ARRAY, ErrorCode::PROGRAM_LIMIT_EXCEEDED, + ErrorCode::TRANSACTION_ROLLBACK, + ErrorCode::ACTIVE_SQL_TRANSACTION, + ErrorCode::DEPENDENT_OBJECTS_EXIST, ErrorCode::QUOTA_OVERCOMMIT, ErrorCode::CLONE_DEPTH_EXCEEDED, ErrorCode::CLONE_WRITE_REQUIRES_MATERIALIZE, diff --git a/nodedb-types/src/error/ctors/read_query_auth.rs b/nodedb-types/src/error/ctors/read_query_auth.rs index b564d0740..8c47c5b25 100644 --- a/nodedb-types/src/error/ctors/read_query_auth.rs +++ b/nodedb-types/src/error/ctors/read_query_auth.rs @@ -147,6 +147,20 @@ impl NodeDbError { } } + /// A drop or revoke refused because other objects still depend on + /// `object`. SQLSTATE `2BP01` (`dependent_objects_still_exist`). + /// `message` is the full message and names the dependents. + pub fn dependent_objects_exist(object: impl Into, message: impl Into) -> Self { + Self { + code: ErrorCode::DEPENDENT_OBJECTS_EXIST, + message: message.into(), + details: ErrorDetails::DependentObjectsExist { + object: object.into(), + }, + cause: None, + } + } + /// An object exists but a prerequisite step has not run, such as `currval` /// before this session called `nextval`. Renders as SQLSTATE `55000` /// (`object_not_in_prerequisite_state`). diff --git a/nodedb-types/src/error/ctors/write_path.rs b/nodedb-types/src/error/ctors/write_path.rs index 156c5c4ce..3fa68fac7 100644 --- a/nodedb-types/src/error/ctors/write_path.rs +++ b/nodedb-types/src/error/ctors/write_path.rs @@ -257,4 +257,30 @@ impl NodeDbError { cause: None, } } + + /// A transaction rolled back for a reason other than a serialization + /// conflict. SQLSTATE `40000` (`transaction_rollback`). The client + /// retries it. `detail` is the full message. + pub fn transaction_rollback(detail: impl Into) -> Self { + let detail = detail.into(); + Self { + code: ErrorCode::TRANSACTION_ROLLBACK, + message: detail.clone(), + details: ErrorDetails::TransactionRollback { detail }, + cause: None, + } + } + + /// The statement cannot run inside an explicit transaction block. + /// SQLSTATE `25001` (`active_sql_transaction`). `detail` is the full + /// message. + pub fn active_sql_transaction(detail: impl Into) -> Self { + let detail = detail.into(); + Self { + code: ErrorCode::ACTIVE_SQL_TRANSACTION, + message: detail.clone(), + details: ErrorDetails::ActiveSqlTransaction { detail }, + cause: None, + } + } } diff --git a/nodedb-types/src/error/details.rs b/nodedb-types/src/error/details.rs index 314d997b5..13b0f819f 100644 --- a/nodedb-types/src/error/details.rs +++ b/nodedb-types/src/error/details.rs @@ -62,6 +62,14 @@ pub enum ErrorDetails { InsufficientBalance { collection: String }, #[serde(rename = "rate_exceeded")] RateExceeded { gate: String }, + /// A transaction rolled back for a reason other than a serialization + /// conflict. `detail` names the reason. The client retries it. + #[serde(rename = "transaction_rollback")] + TransactionRollback { detail: String }, + /// The statement cannot run inside an explicit transaction block. + /// `detail` names the statement or the refused operation. + #[serde(rename = "active_sql_transaction")] + ActiveSqlTransaction { detail: String }, // Read path #[serde(rename = "collection_not_found")] @@ -89,6 +97,10 @@ pub enum ErrorDetails { /// collection/document shape of `DocumentNotFound`. #[serde(rename = "not_found")] NotFound { detail: String }, + /// A drop or revoke refused because other objects still depend on + /// `object`. The message lists the dependents. + #[serde(rename = "dependent_objects_exist")] + DependentObjectsExist { object: String }, #[serde(rename = "collection_draining")] CollectionDraining { collection: String }, #[serde(rename = "collection_deactivated")] diff --git a/nodedb-types/src/error/msgpack/constants.rs b/nodedb-types/src/error/msgpack/constants.rs index c2fef997a..fa3c97cb3 100644 --- a/nodedb-types/src/error/msgpack/constants.rs +++ b/nodedb-types/src/error/msgpack/constants.rs @@ -87,6 +87,9 @@ // | 81 | PeriodLockMisconfigured | // | 82 | DataException | // | 83 | ProgramLimitExceeded | +// | 84 | TransactionRollback | +// | 85 | ActiveSqlTransaction | +// | 86 | DependentObjectsExist | pub(super) const TAG_CONSTRAINT_VIOLATION: u16 = 1; pub(super) const TAG_WRITE_CONFLICT: u16 = 2; @@ -171,3 +174,6 @@ pub(super) const TAG_AMBIGUOUS_COLUMN: u16 = 80; pub(super) const TAG_PERIOD_LOCK_MISCONFIGURED: u16 = 81; pub(super) const TAG_DATA_EXCEPTION: u16 = 82; pub(super) const TAG_PROGRAM_LIMIT_EXCEEDED: u16 = 83; +pub(super) const TAG_TRANSACTION_ROLLBACK: u16 = 84; +pub(super) const TAG_ACTIVE_SQL_TRANSACTION: u16 = 85; +pub(super) const TAG_DEPENDENT_OBJECTS_EXIST: u16 = 86; diff --git a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs index 1789a6ddf..3d6db8325 100644 --- a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs +++ b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs @@ -96,6 +96,14 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { let (gate,) = read1_str(reader, field_count)?; Ok(ErrorDetails::RateExceeded { gate }) } + TAG_TRANSACTION_ROLLBACK => { + let (detail,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::TransactionRollback { detail }) + } + TAG_ACTIVE_SQL_TRANSACTION => { + let (detail,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::ActiveSqlTransaction { detail }) + } TAG_COLLECTION_NOT_FOUND => { let (collection,) = read1_str(reader, field_count)?; Ok(ErrorDetails::CollectionNotFound { collection }) @@ -395,6 +403,10 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { let (detail,) = read1_str(reader, field_count)?; Ok(ErrorDetails::NotFound { detail }) } + TAG_DEPENDENT_OBJECTS_EXIST => { + let (object,) = read1_str(reader, field_count)?; + Ok(ErrorDetails::DependentObjectsExist { object }) + } TAG_CANNOT_DROP_DEFAULT_DATABASE => { skip_fields(reader, field_count)?; Ok(ErrorDetails::CannotDropDefaultDatabase) @@ -524,6 +536,46 @@ mod tests { assert_eq!(roundtrip(&v), v); } + #[test] + fn transaction_rollback_roundtrip() { + let v = ErrorDetails::TransactionRollback { + detail: "a participant vShard returned an error".into(), + }; + assert_eq!(roundtrip(&v), v); + } + + #[test] + fn active_sql_transaction_roundtrip() { + let v = ErrorDetails::ActiveSqlTransaction { + detail: "VACUUM cannot run inside a transaction block".into(), + }; + assert_eq!(roundtrip(&v), v); + } + + #[test] + fn dependent_objects_exist_roundtrip() { + let v = ErrorDetails::DependentObjectsExist { + object: "role \"auditor\"".into(), + }; + assert_eq!(roundtrip(&v), v); + } + + /// The details of each transaction and dependency code decode to the + /// same variant, which answers the same numeric code. + #[test] + fn transaction_and_dependency_details_keep_their_code() { + use crate::error::NodeDbError; + for e in [ + NodeDbError::transaction_rollback("participant failed"), + NodeDbError::active_sql_transaction("VACUUM"), + NodeDbError::dependent_objects_exist("role \"r\"", "held by users: alice"), + ] { + let back = roundtrip(e.details()); + assert_eq!(&back, e.details()); + assert_eq!(back.code(), e.code()); + } + } + #[test] fn bridge_enriched_roundtrip() { let v = ErrorDetails::Bridge { diff --git a/nodedb-types/src/error/msgpack/encode.rs b/nodedb-types/src/error/msgpack/encode.rs index d43c5f64e..be3e2ab9c 100644 --- a/nodedb-types/src/error/msgpack/encode.rs +++ b/nodedb-types/src/error/msgpack/encode.rs @@ -136,6 +136,12 @@ impl ToMessagePack for ErrorDetails { write1(writer, TAG_INSUFFICIENT_BALANCE, collection) } ErrorDetails::RateExceeded { gate } => write1(writer, TAG_RATE_EXCEEDED, gate), + ErrorDetails::TransactionRollback { detail } => { + write1(writer, TAG_TRANSACTION_ROLLBACK, detail) + } + ErrorDetails::ActiveSqlTransaction { detail } => { + write1(writer, TAG_ACTIVE_SQL_TRANSACTION, detail) + } ErrorDetails::CollectionNotFound { collection } => { write1(writer, TAG_COLLECTION_NOT_FOUND, collection) } @@ -329,6 +335,9 @@ impl ToMessagePack for ErrorDetails { ErrorDetails::AlreadyExists { object } => write1(writer, TAG_ALREADY_EXISTS, object), ErrorDetails::ObjectNotReady { object } => write1(writer, TAG_OBJECT_NOT_READY, object), ErrorDetails::NotFound { detail } => write1(writer, TAG_NOT_FOUND, detail), + ErrorDetails::DependentObjectsExist { object } => { + write1(writer, TAG_DEPENDENT_OBJECTS_EXIST, object) + } ErrorDetails::CannotDropDefaultDatabase => { write_unit(writer, TAG_CANNOT_DROP_DEFAULT_DATABASE) } diff --git a/nodedb-types/src/error/types.rs b/nodedb-types/src/error/types.rs index 9837849c9..74de7a7f4 100644 --- a/nodedb-types/src/error/types.rs +++ b/nodedb-types/src/error/types.rs @@ -68,6 +68,7 @@ impl NodeDbError { matches!( self.details, ErrorDetails::WriteConflict { .. } + | ErrorDetails::TransactionRollback { .. } | ErrorDetails::DeadlineExceeded | ErrorDetails::NoLeader | ErrorDetails::NotLeader { .. } @@ -104,6 +105,8 @@ impl NodeDbError { | ErrorDetails::DivisionByZero | ErrorDetails::DataException { .. } | ErrorDetails::ProgramLimitExceeded { .. } + | ErrorDetails::ActiveSqlTransaction { .. } + | ErrorDetails::DependentObjectsExist { .. } | ErrorDetails::InvalidLimitValue { .. } | ErrorDetails::BackupTenantMismatch { .. } | ErrorDetails::BackupKeyMismatch @@ -236,6 +239,26 @@ mod tests { assert!(!NodeDbError::internal("oops").is_client_error()); } + /// A participant rollback is retriable. A statement refused inside a + /// transaction block and a drop refused by dependents are client errors. + #[test] + fn transaction_and_dependency_codes_classify() { + let rollback = NodeDbError::transaction_rollback("participant failed"); + assert!(rollback.is_retriable()); + assert!(!rollback.is_client_error()); + assert_eq!(rollback.code(), ErrorCode::TRANSACTION_ROLLBACK); + + let in_block = NodeDbError::active_sql_transaction("VACUUM"); + assert!(in_block.is_client_error()); + assert!(!in_block.is_retriable()); + assert_eq!(in_block.code(), ErrorCode::ACTIVE_SQL_TRANSACTION); + + let dependents = NodeDbError::dependent_objects_exist("role \"r\"", "held by users"); + assert!(dependents.is_client_error()); + assert!(!dependents.is_retriable()); + assert_eq!(dependents.code(), ErrorCode::DEPENDENT_OBJECTS_EXIST); + } + #[test] fn json_serialization() { let e = NodeDbError::collection_not_found("users"); diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index ababf8ba6..ae83444ed 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -359,6 +359,7 @@ impl From for ErrorCode { | crate::Error::ObjectNotInPrerequisiteState { .. } | crate::Error::MirrorReadOnly { .. } | crate::Error::DependentObjectsExist { .. } + | crate::Error::RoleInUse { .. } | crate::Error::UndefinedObject { .. } | crate::Error::AmbiguousColumn { .. } | crate::Error::ExecutionLimitExceeded { .. } diff --git a/nodedb/src/control/cluster/array_executor/refusal.rs b/nodedb/src/control/cluster/array_executor/refusal.rs index c34d041c5..bd05b43a4 100644 --- a/nodedb/src/control/cluster/array_executor/refusal.rs +++ b/nodedb/src/control/cluster/array_executor/refusal.rs @@ -142,6 +142,7 @@ pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterErro | crate::Error::CatalogIntegrityViolation { .. } | crate::Error::Promql(_) | crate::Error::DependentObjectsExist { .. } + | crate::Error::RoleInUse { .. } | crate::Error::CascadeCycle { .. } | crate::Error::CrossShardInExplicitTransaction | crate::Error::SequencerUnavailable diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 002260bfa..405a71fc2 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -138,6 +138,7 @@ pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { | crate::Error::CatalogIntegrityViolation { .. } | crate::Error::Promql(_) | crate::Error::DependentObjectsExist { .. } + | crate::Error::RoleInUse { .. } | crate::Error::CascadeCycle { .. } | crate::Error::CrossShardInExplicitTransaction | crate::Error::SequencerUnavailable diff --git a/nodedb/src/control/cluster/metadata_applier/wedge.rs b/nodedb/src/control/cluster/metadata_applier/wedge.rs index 6e40355d3..ee5ec2fd0 100644 --- a/nodedb/src/control/cluster/metadata_applier/wedge.rs +++ b/nodedb/src/control/cluster/metadata_applier/wedge.rs @@ -146,6 +146,7 @@ pub fn classify(error: &crate::Error) -> ApplyFailureClass { | crate::Error::DataPlane(_) | crate::Error::Promql(_) | crate::Error::DependentObjectsExist { .. } + | crate::Error::RoleInUse { .. } | crate::Error::CascadeCycle { .. } | crate::Error::CrossShardInExplicitTransaction | crate::Error::SequencerUnavailable diff --git a/nodedb/src/control/gateway/error_map/class_parity.rs b/nodedb/src/control/gateway/error_map/class_parity.rs index c39b58c62..2f2c98894 100644 --- a/nodedb/src/control/gateway/error_map/class_parity.rs +++ b/nodedb/src/control/gateway/error_map/class_parity.rs @@ -377,7 +377,7 @@ fn expired_session_token_is_invalid_authorization_everywhere() { } /// The number of `crate::Error` variants [`error_variant_index`] numbers. -const ERROR_VARIANT_COUNT: usize = 108; +const ERROR_VARIANT_COUNT: usize = 109; /// A dense index per `crate::Error` variant. Exhaustive, so a new variant /// fails to compile here until it gets an index, and @@ -494,6 +494,7 @@ pub(crate) fn error_variant_index(err: &crate::Error) -> usize { E::OllpExhausted { .. } => 105, E::MirrorReadOnly { .. } => 106, E::StaleReadNotLeader { .. } => 107, + E::RoleInUse { .. } => 108, } } @@ -829,6 +830,12 @@ pub(crate) fn error_samples() -> Vec { source_cluster: "src".into(), detail: text(), }, + E::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::Users(vec![ + "bob".into(), + ]), + }, ] } @@ -866,13 +873,6 @@ fn every_error_variant_has_the_http_status_of_its_sqlstate() { } } -/// Variants whose pgwire SQLSTATE class has no public numeric code: `25` -/// (`CrdtApplyForbiddenInTransaction`, `NotInTransactionBlock`, -/// `CrossShardInExplicitTransaction`), `2B` (`DependentObjectsExist`), and -/// `40000` (`CalvinParticipantError`, whose code is deliberately not a write -/// conflict). The numeric wire form cannot carry their class. -const NUMERIC_CLASS_GAPS: [usize; 5] = [7, 32, 33, 88, 90]; - /// Every `crate::Error` variant renders the SQLSTATE class it renders locally /// after it crosses a node hop, through both wire encoders and the decoder. #[test] @@ -887,9 +887,6 @@ fn every_error_variant_keeps_its_class_across_a_node_hop() { ]; for (name, encode) in encoders { for (err, twin) in error_samples().into_iter().zip(error_samples()) { - if NUMERIC_CLASS_GAPS.contains(&error_variant_index(&err)) { - continue; - } let (_, local, _) = error_to_sqlstate(&err); let rebuilt = crate::Error::from(encode(twin)); let (_, remote, _) = error_to_sqlstate(&rebuilt); @@ -935,6 +932,7 @@ fn classified_sqlstates() -> Vec<(usize, &'static str)> { (104, sqlstate::SYNTAX_ERROR), (106, sqlstate::READ_ONLY_SQL_TRANSACTION), (107, sqlstate::STALE_READ_NOT_LEADER), + (108, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), ] } @@ -980,4 +978,90 @@ fn dedicated_codes_render_the_class_of_their_variant() { numeric_code_to_sqlstate(Ec::STALE_READ_NOT_LEADER), sqlstate::STALE_READ_NOT_LEADER ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TRANSACTION_ROLLBACK), + sqlstate::TRANSACTION_ROLLBACK + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::ACTIVE_SQL_TRANSACTION), + sqlstate::ACTIVE_SQL_TRANSACTION + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::DEPENDENT_OBJECTS_EXIST), + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST + ); +} + +/// The transaction-state and dependency variants render their exact +/// SQLSTATE after a node hop through both encoders, and carry the public +/// code of that class. +#[test] +fn transaction_and_dependency_variants_keep_their_sqlstate_across_a_hop() { + use nodedb_cluster::rpc_codec::TypedClusterError; + use nodedb_types::error::ErrorCode as Ec; + + use crate::control::cluster::data_plane_error_wire::execution_error_to_typed; + + let cases: [(fn() -> crate::Error, &str, Ec); 6] = [ + ( + || crate::Error::CalvinParticipantError, + sqlstate::TRANSACTION_ROLLBACK, + Ec::TRANSACTION_ROLLBACK, + ), + ( + || crate::Error::NotInTransactionBlock { + statement: "VACUUM".into(), + }, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::CrdtApplyForbiddenInTransaction, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::CrossShardInExplicitTransaction, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::DependentObjectsExist { + tenant_id: 1, + root_kind: "collection", + root_name: "c".into(), + dependent_count: 1, + dependents: vec![("view".into(), "v".into())], + }, + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + Ec::DEPENDENT_OBJECTS_EXIST, + ), + ( + || crate::Error::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::ChildRoles( + vec!["junior".into()], + ), + }, + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + Ec::DEPENDENT_OBJECTS_EXIST, + ), + ]; + let encoders: [(&str, fn(crate::Error) -> TypedClusterError); 2] = [ + ("execution_error_to_typed", execution_error_to_typed), + ("From", TypedClusterError::from), + ]; + for (make, state, code) in cases { + let err = make(); + assert_eq!(error_to_sqlstate(&err).1, state, "{err:?} locally"); + assert_eq!(native_error_fields(&err).code, code, "{err:?} native code"); + for (name, encode) in encoders { + let rebuilt = crate::Error::from(encode(make())); + assert_eq!( + error_to_sqlstate(&rebuilt).1, + state, + "{name}: {err:?} after the hop as {rebuilt:?}" + ); + } + } } diff --git a/nodedb/src/control/gateway/error_map/resp.rs b/nodedb/src/control/gateway/error_map/resp.rs index dd2318e1f..5e85143ef 100644 --- a/nodedb/src/control/gateway/error_map/resp.rs +++ b/nodedb/src/control/gateway/error_map/resp.rs @@ -125,6 +125,7 @@ impl GatewayErrorMap { | Error::CatalogIntegrityViolation { .. } | Error::Promql(_) | Error::DependentObjectsExist { .. } + | Error::RoleInUse { .. } | Error::CascadeCycle { .. } | Error::CrossShardInExplicitTransaction | Error::SequencerUnavailable diff --git a/nodedb/src/control/security/role.rs b/nodedb/src/control/security/role.rs index c6bf16d08..0a21b09fe 100644 --- a/nodedb/src/control/security/role.rs +++ b/nodedb/src/control/security/role.rs @@ -193,10 +193,16 @@ impl RoleStore { let mut roles = self.roles.write(); // Check no other role inherits from this one. - let has_children = roles.values().any(|r| r.parent.as_deref() == Some(name)); - if has_children { - return Err(crate::Error::BadRequest { - detail: format!("cannot drop role '{name}': other roles inherit from it"), + let mut children: Vec = roles + .values() + .filter(|r| r.parent.as_deref() == Some(name)) + .map(|r| r.name.clone()) + .collect(); + if !children.is_empty() { + children.sort(); + return Err(crate::Error::RoleInUse { + role: name.to_string(), + dependents: super::role_assignment::RoleDependents::ChildRoles(children), }); } diff --git a/nodedb/src/control/security/role_assignment.rs b/nodedb/src/control/security/role_assignment.rs index 1a2c32e21..7b758f21f 100644 --- a/nodedb/src/control/security/role_assignment.rs +++ b/nodedb/src/control/security/role_assignment.rs @@ -20,6 +20,8 @@ //! node alike, and a replayed log never produces a user holding an //! undefined role. +use std::fmt; + use crate::control::security::catalog::{StoredRole, StoredUser}; use super::identity::Role; @@ -54,18 +56,38 @@ impl RoleRefusal { } } +/// What still depends on a custom role, so a DROP ROLE of it is refused. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum RoleDependents { + /// Users that hold the role. + Users(Vec), + /// Roles that inherit from the role. + ChildRoles(Vec), +} + +impl fmt::Display for RoleDependents { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Users(users) => write!(f, "users still hold it: {}", users.join(", ")), + Self::ChildRoles(children) => { + write!(f, "other roles inherit from it: {}", children.join(", ")) + } + } + } +} + impl From for crate::Error { fn from(refusal: RoleRefusal) -> Self { match refusal { RoleRefusal::Undefined { name } => crate::Error::UndefinedObject { kind: "role", name }, - // A refused DROP of a role still in use: a client error. No - // `crate::Error` variant carries `2BP01` without a tenant and a - // CASCADE hint that does not apply to roles. - other @ (RoleRefusal::HeldByUsers { .. } | RoleRefusal::InheritedBy { .. }) => { - crate::Error::BadRequest { - detail: other.to_string(), - } - } + RoleRefusal::HeldByUsers { name, users } => crate::Error::RoleInUse { + role: name, + dependents: RoleDependents::Users(users), + }, + RoleRefusal::InheritedBy { name, children } => crate::Error::RoleInUse { + role: name, + dependents: RoleDependents::ChildRoles(children), + }, } } } @@ -245,6 +267,39 @@ mod tests { ); } + /// A role still in use keeps the `2BP01` class as a `crate::Error`, and + /// its message names the role and its dependents with no tenant and no + /// CASCADE hint. + #[test] + fn a_role_in_use_is_a_dependent_objects_error() { + use crate::control::server::pgwire::types::error_to_sqlstate; + + for refusal in [ + RoleRefusal::HeldByUsers { + name: "analyst".into(), + users: vec!["bob".into()], + }, + RoleRefusal::InheritedBy { + name: "analyst".into(), + children: vec!["junior".into()], + }, + ] { + let message = refusal.to_string(); + let err = crate::Error::from(refusal); + let (_, state, rendered) = error_to_sqlstate(&err); + assert_eq!(state, "2BP01"); + assert_eq!(rendered, message); + assert!(!rendered.contains("tenant"), "{rendered}"); + assert!(!rendered.contains("CASCADE"), "{rendered}"); + let public = crate::error_classify::classify(&err); + assert_eq!( + public.code(), + nodedb_types::error::ErrorCode::DEPENDENT_OBJECTS_EXIST + ); + assert_eq!(public.message(), message); + } + } + #[test] fn every_parsed_built_in_name_is_built_in() { for name in [ diff --git a/nodedb/src/control/server/http/routes/query/materialized/encode.rs b/nodedb/src/control/server/http/routes/query/materialized/encode.rs index ae6c3721e..190b7b8e9 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/encode.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/encode.rs @@ -115,11 +115,10 @@ mod tests { assert_eq!(json["cause"]["error"], "not on this engine"); } - /// A plain `XX000` DDL error is a server fault, never a client error. + /// An internal DDL error is a server fault, never a client error. #[tokio::test] async fn xx000_ddl_error_is_500() { - let error = - crate::control::server::shared::ddl::DdlError::new("XX000", "catalog write failed"); + let error = crate::control::server::shared::ddl::DdlError::internal("catalog write failed"); let (status, _) = response_json(ddl_error_to_api(error)).await; assert_eq!(status, axum::http::StatusCode::INTERNAL_SERVER_ERROR); } diff --git a/nodedb/src/control/server/native/dispatch/conversion.rs b/nodedb/src/control/server/native/dispatch/conversion.rs index a3592317b..4d7286cba 100644 --- a/nodedb/src/control/server/native/dispatch/conversion.rs +++ b/nodedb/src/control/server/native/dispatch/conversion.rs @@ -100,6 +100,22 @@ pub(crate) fn native_error_fields(e: &crate::Error) -> NativeErrorFields { } } +/// Convert a Control-Plane error into a native error frame with `context` +/// before its message. The SQLSTATE and code stay the error's own. +pub(crate) fn error_to_native_in_context( + seq: u64, + context: &str, + e: &crate::Error, +) -> NativeResponse { + let fields = native_error_fields(e); + NativeResponse::error_with_code( + seq, + fields.sqlstate, + format!("{context}: {}", fields.message), + fields.code.0, + ) +} + /// Convert a Control-Plane error into a native error frame under a SQLSTATE /// the call site chooses. /// @@ -126,18 +142,26 @@ pub(crate) fn error_to_native_with_sqlstate( /// Convert a `NodeDbError` produced while shaping a response into a /// NativeResponse error frame. /// -/// The numeric code travels alongside the SQLSTATE: the error is already -/// classified here, and rendering only `XX000` would make the client rebuild -/// it as a generic internal failure. +/// The numeric code travels alongside the SQLSTATE its code maps to, the same +/// SQLSTATE pgwire's `shape_error_to_pg` renders. pub(crate) fn shape_error_to_native(seq: u64, e: &nodedb_types::NodeDbError) -> NativeResponse { - NativeResponse::error_with_code(seq, "XX000", e.message().to_string(), e.code().0) + NativeResponse::error_with_code( + seq, + crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate(e.code()), + e.message().to_string(), + e.code().0, + ) } /// Render a statement-tag fold refusal as a native error frame. Two tasks of /// one statement disagreeing on their verb is a planner bug, so it is an /// internal error, the same class pgwire's `dml_fold_error_to_pg` renders. pub(crate) fn dml_fold_error_to_native(seq: u64, e: &DmlFoldError) -> NativeResponse { - sqlstate_error(seq, "XX000", e.to_string()) + sqlstate_error( + seq, + nodedb_types::error::sqlstate::INTERNAL_ERROR, + e.to_string(), + ) } /// Render an error [`Response`] from the Data Plane as a native error frame. @@ -170,7 +194,11 @@ pub(crate) fn error_code_to_native( code: Option<&crate::bridge::envelope::ErrorCode>, ) -> NativeResponse { let Some(code) = code else { - return sqlstate_error(seq, "XX000", "unknown data plane error"); + return sqlstate_error( + seq, + nodedb_types::error::sqlstate::INTERNAL_ERROR, + "unknown data plane error", + ); }; let (_, sqlstate, message) = error_code_to_sqlstate(code); let public = nodedb_types::NodeDbError::from(crate::Error::DataPlane(code.clone())); @@ -394,6 +422,42 @@ mod tests { assert_eq!(error.code, "28000"); } + /// A shaping error answers the SQLSTATE its code maps to, the one pgwire + /// renders, never a bare internal error. + #[test] + fn a_shaping_error_keeps_its_sqlstate() { + let shaped = shape_error_to_native(1, &nodedb_types::NodeDbError::division_by_zero()); + let error = shaped.error.expect("error responses carry a payload"); + assert_eq!(error.code, "22012"); + assert_eq!( + error.ndb_code, + nodedb_types::error::ErrorCode::DIVISION_BY_ZERO.0 + ); + } + + /// A typed error behind a context prefix keeps its SQLSTATE and code. + #[test] + fn an_error_in_context_keeps_its_class() { + let missing = crate::Error::CollectionNotFound { + tenant_id: crate::types::TenantId::new(1), + collection: "orders".into(), + }; + let response = error_to_native_in_context(1, "database catalog lookup failed", &missing); + let error = response.error.expect("error responses carry a payload"); + assert_eq!(error.code, "42P01"); + assert_eq!( + error.ndb_code, + nodedb_types::error::ErrorCode::COLLECTION_NOT_FOUND.0 + ); + assert!( + error + .message + .starts_with("database catalog lookup failed: "), + "{}", + error.message + ); + } + /// One statement running out of time answers ONE SQLSTATE, whichever half /// of the race reported it: the Control-Plane timer, which raises /// `DeadlineExceeded` directly, or a shard refusing an already-expired diff --git a/nodedb/src/control/server/native/dispatch/direct_ops.rs b/nodedb/src/control/server/native/dispatch/direct_ops.rs index bff3d3c9e..5eece82c3 100644 --- a/nodedb/src/control/server/native/dispatch/direct_ops.rs +++ b/nodedb/src/control/server/native/dispatch/direct_ops.rs @@ -336,7 +336,7 @@ pub(crate) async fn handle_direct_op( None => { return sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "authorization returned no task capability", ); } diff --git a/nodedb/src/control/server/native/dispatch/mod.rs b/nodedb/src/control/server/native/dispatch/mod.rs index 0901ed537..14cd8c68b 100644 --- a/nodedb/src/control/server/native/dispatch/mod.rs +++ b/nodedb/src/control/server/native/dispatch/mod.rs @@ -31,8 +31,9 @@ pub(crate) use admission_op::admission_operation; pub(crate) use auth::{NativeAuthOutcome, handle_auth, handle_ping}; pub(crate) use conversion::{ apply_dml_outcome, ddl_result_to_native, dml_fold_error_to_native, error_code_to_native, - error_response_to_native, error_to_native, error_to_native_with_sqlstate, native_error_fields, - shape_error_to_native, to_native_columns_rows, + error_response_to_native, error_to_native, error_to_native_in_context, + error_to_native_with_sqlstate, native_error_fields, shape_error_to_native, + to_native_columns_rows, }; pub(crate) use ctx::DispatchCtx; pub(crate) use direct_ops::handle_direct_op; diff --git a/nodedb/src/control/server/native/dispatch/sql.rs b/nodedb/src/control/server/native/dispatch/sql.rs index bb3450bdc..d1f177a79 100644 --- a/nodedb/src/control/server/native/dispatch/sql.rs +++ b/nodedb/src/control/server/native/dispatch/sql.rs @@ -387,7 +387,7 @@ async fn execute_planned( let Some(scope) = lease_scope.take() else { return resp(sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: query lease scope missing before SQL stream dispatch", )); }; @@ -408,7 +408,7 @@ async fn execute_planned( let Some(lease_scope) = lease_scope.take() else { return resp(sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: query lease scope missing before materialized SQL dispatch", )); }; @@ -433,7 +433,7 @@ async fn execute_planned( // expects (a single, fully-resolved SQL string). // // Errors here surface as `42P02` (`undefined_parameter`) so the client -// gets a typed SQLSTATE rather than a generic `XX000` opaque failure. +// gets a typed SQLSTATE rather than an opaque internal error. /// Substitute `$N` placeholders in `sql` with canonical SQL literals. fn inline_params(sql: &str, params: &[Value]) -> String { diff --git a/nodedb/src/control/server/native/dispatch/sql_loop.rs b/nodedb/src/control/server/native/dispatch/sql_loop.rs index 0c8e63be7..f9c8b234e 100644 --- a/nodedb/src/control/server/native/dispatch/sql_loop.rs +++ b/nodedb/src/control/server/native/dispatch/sql_loop.rs @@ -191,7 +191,7 @@ pub(super) async fn run_dispatch_loop( { return resp(sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: failed to retain descriptor leases for buffered transaction tasks", )); } diff --git a/nodedb/src/control/server/native/dispatch/streaming.rs b/nodedb/src/control/server/native/dispatch/streaming.rs index 725a9b77e..d63bd3b1e 100644 --- a/nodedb/src/control/server/native/dispatch/streaming.rs +++ b/nodedb/src/control/server/native/dispatch/streaming.rs @@ -49,7 +49,7 @@ impl SqlOutcome { SqlOutcome::Response(r) => *r, SqlOutcome::Stream(s) => crate::control::server::native::sqlstate_code::sqlstate_error( s.seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: SQL stream produced on a non-streaming path", ), } diff --git a/nodedb/src/control/server/native/dispatch/transaction.rs b/nodedb/src/control/server/native/dispatch/transaction.rs index a0e5448b0..03dc13649 100644 --- a/nodedb/src/control/server/native/dispatch/transaction.rs +++ b/nodedb/src/control/server/native/dispatch/transaction.rs @@ -123,9 +123,9 @@ pub(crate) async fn handle_rollback(ctx: &DispatchCtx<'_>, seq: u64) -> NativeRe NativeResponse::status_row(seq, "ROLLBACK") } -/// Map a neutral commit abort reason to the native error frame native emitted -/// before extraction (batch/dispatch failures collapse to `40001`, batch -/// rejections carry the Data-Plane SQLSTATE). +/// Map a neutral commit abort reason to the native error frame. A batch +/// rejection carries the Data-Plane SQLSTATE. A dispatch or DDL-propose error +/// keeps the SQLSTATE and code of its typed error, as on pgwire. fn commit_abort_to_native(seq: u64, reason: &AbortReason) -> NativeResponse { // The numeric NodeDB code rides alongside the SQLSTATE wherever the abort // was classified: a UNIQUE violation that only surfaces at COMMIT is the @@ -170,16 +170,64 @@ fn commit_abort_to_native(seq: u64, reason: &AbortReason) -> NativeResponse { format!("could not serialize access due to concurrent schema change: {detail}"), nodedb_types::error::ErrorCode::WRITE_CONFLICT.0, ), - AbortReason::Dispatch(e) => ( - "40001", - format!("transaction commit failed: {e}"), - nodedb_types::error::ErrorCode::WRITE_CONFLICT.0, - ), - AbortReason::DdlPropose(e) => ( - "XX000", - format!("{e}"), - nodedb_types::error::ErrorCode::INTERNAL.0, - ), + AbortReason::Dispatch(e) => { + let fields = super::native_error_fields(e); + ( + fields.sqlstate, + format!("transaction commit failed: {}", fields.message), + fields.code.0, + ) + } + AbortReason::DdlPropose(e) => { + let fields = super::native_error_fields(e); + (fields.sqlstate, fields.message, fields.code.0) + } }; NativeResponse::error_with_code(seq, code, message, ndb_code) } + +#[cfg(test)] +mod tests { + use nodedb_types::error::ErrorCode as PublicCode; + + use super::*; + + fn frame(reason: &AbortReason) -> (String, String, u16) { + let payload = commit_abort_to_native(1, reason) + .error + .expect("an aborted commit answers an error frame"); + (payload.code, payload.message, payload.ndb_code) + } + + /// A buffered DDL refused at COMMIT keeps the class of its typed error, + /// the same SQLSTATE pgwire renders for it. + #[test] + fn a_ddl_propose_abort_keeps_its_class() { + let in_use = crate::Error::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::Users(vec![ + "bob".into(), + ]), + }; + let (code, _, ndb_code) = frame(&AbortReason::DdlPropose(in_use)); + assert_eq!(code, "2BP01"); + assert_eq!(ndb_code, PublicCode::DEPENDENT_OBJECTS_EXIST.0); + } + + /// A commit dispatch error keeps its class instead of reading as a + /// serialization failure. + #[test] + fn a_dispatch_abort_keeps_its_class() { + let missing = crate::Error::CollectionNotFound { + tenant_id: crate::types::TenantId::new(1), + collection: "orders".into(), + }; + let (code, message, ndb_code) = frame(&AbortReason::Dispatch(missing)); + assert_eq!(code, "42P01"); + assert_eq!(ndb_code, PublicCode::COLLECTION_NOT_FOUND.0); + assert!( + message.starts_with("transaction commit failed: "), + "{message}" + ); + } +} diff --git a/nodedb/src/control/server/native/session/auth.rs b/nodedb/src/control/server/native/session/auth.rs index a2ffdefde..1557077e9 100644 --- a/nodedb/src/control/server/native/session/auth.rs +++ b/nodedb/src/control/server/native/session/auth.rs @@ -12,7 +12,9 @@ use crate::control::server::shared::authorization::authorize_database; use super::NativeSession; use super::dispatch; -use crate::control::server::native::dispatch::error_to_native_with_sqlstate; +use crate::control::server::native::dispatch::{ + error_to_native_in_context, error_to_native_with_sqlstate, +}; use crate::control::server::native::sqlstate_code::sqlstate_error; impl NativeSession { @@ -88,8 +90,12 @@ impl NativeSession { "selected database does not exist", ); } - Err(_) => { - return sqlstate_error(seq, "XX000", "database catalog lookup failed"); + Err(e) => { + return error_to_native_in_context( + seq, + "database catalog lookup failed", + &e, + ); } }, None => identity @@ -105,8 +111,12 @@ impl NativeSession { Ok(None) => { return sqlstate_error(seq, "3D000", "selected database does not exist"); } - Err(_) => { - return sqlstate_error(seq, "XX000", "database catalog lookup failed"); + Err(e) => { + return error_to_native_in_context( + seq, + "database catalog lookup failed", + &e, + ); } } @@ -150,7 +160,7 @@ impl NativeSession { drop(scoped); return sqlstate_error( seq, - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: global admission permit missing during auth assembly", ); }; diff --git a/nodedb/src/control/server/pgwire/connection.rs b/nodedb/src/control/server/pgwire/connection.rs index 3c181b06b..6561d41a9 100644 --- a/nodedb/src/control/server/pgwire/connection.rs +++ b/nodedb/src/control/server/pgwire/connection.rs @@ -27,7 +27,6 @@ use super::connection_identity::PgConnectionContext; use super::factory::NodeDbPgHandlerFactory; const STARTUP_TIMEOUT: Duration = Duration::from_secs(60); -const INTERNAL_ERROR_CODE: &str = "XX000"; const INTERNAL_ERROR_MESSAGE: &str = "internal server error"; /// The observable outcome of a single connection loop. @@ -63,7 +62,7 @@ fn materialize_handlers(build: impl FnOnce() -> T) -> Result { fn fixed_panic_response() -> PgWireBackendMessage { let error = ErrorInfo::new( "FATAL".to_owned(), - INTERNAL_ERROR_CODE.to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), INTERNAL_ERROR_MESSAGE.to_owned(), ); PgWireBackendMessage::ErrorResponse(error.into()) diff --git a/nodedb/src/control/server/pgwire/ddl/database/use_database.rs b/nodedb/src/control/server/pgwire/ddl/database/use_database.rs index 3b335f122..e2964a436 100644 --- a/nodedb/src/control/server/pgwire/ddl/database/use_database.rs +++ b/nodedb/src/control/server/pgwire/ddl/database/use_database.rs @@ -19,7 +19,7 @@ use crate::control::server::shared::session::{ }; use crate::control::state::SharedState; -use super::super::super::types::sqlstate_error; +use super::super::super::types::{error_to_pg_in_context, sqlstate_error}; /// Handle `USE DATABASE `. /// @@ -38,7 +38,7 @@ pub async fn handle_use_database( // Verify the named database exists. let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| sqlstate_error("XX000", &format!("catalog lookup failed: {e}")))? + .map_err(|e| error_to_pg_in_context("catalog lookup failed", &e))? .ok_or_else(|| sqlstate_error("3D000", &format!("database '{name}' does not exist")))?; // Enforce `accessible_databases`: reject the switch if the identity does diff --git a/nodedb/src/control/server/pgwire/factory/startup.rs b/nodedb/src/control/server/pgwire/factory/startup.rs index f3e84e559..d7279c10e 100644 --- a/nodedb/src/control/server/pgwire/factory/startup.rs +++ b/nodedb/src/control/server/pgwire/factory/startup.rs @@ -22,7 +22,7 @@ use crate::control::server::session_auth::identity::stored_user_identity; use crate::control::state::SharedState; use super::super::handler::NodeDbPgHandler; -use super::super::types::sqlstate_error; +use super::super::types::{error_to_pg_in_context, sqlstate_error}; use super::provider::NodeDbParameterProvider; /// Enum dispatch for startup handler — avoids dyn trait object issues. @@ -68,7 +68,7 @@ fn bind_startup_database( .credentials .catalog() .get_database_id_by_name(&db_name) - .map_err(|e| sqlstate_error("XX000", &format!("catalog lookup failed: {e}")))? + .map_err(|e| error_to_pg_in_context("catalog lookup failed", &e))? .ok_or_else(|| sqlstate_error("3D000", &format!("database '{db_name}' does not exist")))?; handler.sessions.set_current_database(session_id, db_id); @@ -109,7 +109,7 @@ fn admit_connection( .set_admission_permit(handler.session_id, permit) { return Err(sqlstate_error( - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "internal error: connection session is not registered", )); } diff --git a/nodedb/src/control/server/pgwire/handler/copy_handler.rs b/nodedb/src/control/server/pgwire/handler/copy_handler.rs index 94a452b7e..9f01ab643 100644 --- a/nodedb/src/control/server/pgwire/handler/copy_handler.rs +++ b/nodedb/src/control/server/pgwire/handler/copy_handler.rs @@ -70,7 +70,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "FATAL".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "connection session metadata is unavailable".to_owned(), ))) })?, @@ -127,10 +127,10 @@ impl NodeDbPgHandler { // byte. The charge below is on the success path and so can // never be where a cap blocks anything. admit_backup_restore_quota(&self.state, request.scope(), tenant_id) - .map_err(internal)?; + .map_err(typed)?; let bytes = backup::backup_tenant(&self.state, tenant_id) .await - .map_err(internal)?; + .map_err(typed)?; // Metered here, on the success path, before the response is // built below — there is no `PhysicalPlan` for a whole-tenant // backup, so the collection dimension is a synthetic @@ -290,7 +290,7 @@ impl CopyHandler for NodeDbCopyHandler { self.state.auth_stores(), database_id, ); - admit_backup_restore_quota(&self.state, &scope, pending.tenant_id).map_err(internal)?; + admit_backup_restore_quota(&self.state, &scope, pending.tenant_id).map_err(typed)?; let stats = backup::restore_tenant( &self.state, @@ -300,7 +300,7 @@ impl CopyHandler for NodeDbCopyHandler { pending.force, ) .await - .map_err(internal)?; + .map_err(typed)?; // pgwire does not auto-send CommandComplete after `on_copy_done` // returns Ok — the trait contract leaves message construction to // the handler. Send a `RESTORE TENANT N ` tag so the @@ -343,11 +343,11 @@ fn sqlstate(code: &str, message: &str) -> PgWireError { ))) } -fn internal(e: crate::Error) -> PgWireError { - // Surface error string but never echo deserializer context — the - // restore orchestrator already scrubs envelope errors. We pass - // through everything else (RPC failures, dispatch errors). - sqlstate(ss::INTERNAL_ERROR, &e.to_string()) +/// Render a backup or restore error with its own SQLSTATE: a spent quota, +/// a tenant mismatch or a wrong key keeps its class. The restore +/// orchestrator already scrubs envelope errors before they reach here. +fn typed(e: crate::Error) -> PgWireError { + super::super::types::error_to_pg(&e) } #[cfg(test)] diff --git a/nodedb/src/control/server/pgwire/handler/cursor_query.rs b/nodedb/src/control/server/pgwire/handler/cursor_query.rs index ce4ed5225..b190690e7 100644 --- a/nodedb/src/control/server/pgwire/handler/cursor_query.rs +++ b/nodedb/src/control/server/pgwire/handler/cursor_query.rs @@ -3,7 +3,7 @@ //! `DECLARE CURSOR` materialisation: plan a SELECT, dispatch it to the //! Data Plane, and collect JSON-encoded rows for cursor storage. -use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; +use pgwire::error::{PgWireError, PgWireResult}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::retry::retry_on_schema_change; @@ -112,13 +112,7 @@ impl NodeDbPgHandler { TraceId::ZERO, ) .await - .map_err(|e| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - e.to_string(), - ))) - })?; + .map_err(|e| super::super::types::error_to_pg(&e))?; if !resp.payload.is_empty() { let json = diff --git a/nodedb/src/control/server/pgwire/handler/facet.rs b/nodedb/src/control/server/pgwire/handler/facet.rs index b17488135..d96c806d8 100644 --- a/nodedb/src/control/server/pgwire/handler/facet.rs +++ b/nodedb/src/control/server/pgwire/handler/facet.rs @@ -305,7 +305,7 @@ fn build_filter_bytes(filter_text: &str) -> PgWireResult> { zerompk::to_msgpack_vec(&filters).map_err(|e| { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), format!("filter serialization failed: {e}"), ))) }) diff --git a/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs index 9c694b685..58cd7b32b 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs @@ -196,7 +196,7 @@ impl NodeDbPgHandler { // invariant is ever broken by a future refactor. PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "internal: static Calvin path reached the OLLP dispatch branch".to_owned(), ))) })? diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs index 29246e846..c665af5cb 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs @@ -167,7 +167,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "ClusterArray authorization returned no capability".to_owned(), ))) })?; diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs index 36f7ae56a..a4fd1dfd8 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs @@ -129,7 +129,7 @@ impl NodeDbPgHandler { { return Err(PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "internal error: failed to retain descriptor leases for buffered transaction tasks" .to_owned(), )))); @@ -158,7 +158,11 @@ impl NodeDbPgHandler { Some(code) => { crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code) } - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), + None => ( + "ERROR", + nodedb_types::error::sqlstate::INTERNAL_ERROR, + "unknown data plane error".to_owned(), + ), }; Err(PgWireError::UserError(Box::new(ErrorInfo::new( severity.to_owned(), diff --git a/nodedb/src/control/server/pgwire/handler/routing/planning.rs b/nodedb/src/control/server/pgwire/handler/routing/planning.rs index c2c6584fc..3b566bac9 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/planning.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/planning.rs @@ -38,7 +38,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "FATAL".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "connection session metadata is unavailable".to_owned(), ))) })?, @@ -89,7 +89,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { StatementSetupError::protocol( "FATAL", - "XX000", + nodedb_types::error::sqlstate::INTERNAL_ERROR, "connection session metadata is unavailable", ) })?, diff --git a/nodedb/src/control/server/pgwire/handler/session_cmds.rs b/nodedb/src/control/server/pgwire/handler/session_cmds.rs index ce70d756d..d4010d887 100644 --- a/nodedb/src/control/server/pgwire/handler/session_cmds.rs +++ b/nodedb/src/control/server/pgwire/handler/session_cmds.rs @@ -202,7 +202,7 @@ impl NodeDbPgHandler { .ok_or_else(|| { PgWireError::UserError(Box::new(ErrorInfo::new( "FATAL".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "connection metadata is missing".to_owned(), ))) })?; diff --git a/nodedb/src/control/server/pgwire/handler/stream_response.rs b/nodedb/src/control/server/pgwire/handler/stream_response.rs index a63f7c5ff..f839ca70b 100644 --- a/nodedb/src/control/server/pgwire/handler/stream_response.rs +++ b/nodedb/src/control/server/pgwire/handler/stream_response.rs @@ -22,7 +22,7 @@ use crate::control::state::SharedState; use crate::data::executor::response_codec::{decode_payload_to_json, decode_payload_value}; use super::super::ddl_encode::col_type_to_field_with_format; -use super::super::types::{error_to_sqlstate, text_field}; +use super::super::types::{error_to_pg_in_context, error_to_sqlstate, text_field}; use super::shape_encode::{encode_shaped_row, shaped_query_response}; /// The per-request plumbing a lazily-streamed pgwire response owns for its @@ -115,7 +115,7 @@ pub(crate) fn streaming_multirow_response( encoder.encode_field(&item.to_string()).map_err(|e| { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), format!("failed to encode streamed row: {e}"), ))) })?; @@ -214,13 +214,8 @@ pub(crate) fn streaming_shaped_response( break; } - let value = decode_payload_value(&batch.payload).map_err(|e| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!("failed to decode streamed batch: {e}"), - ))) - })?; + let value = decode_payload_value(&batch.payload) + .map_err(|e| error_to_pg_in_context("failed to decode streamed batch", &e))?; // Resolved once before the first batch was pulled; this only // re-borrows it, so no batch can slip out ahead of the policy. // A streamed plan never carries Control-Plane computed columns: @@ -232,13 +227,7 @@ pub(crate) fn streaming_shaped_response( redaction.as_ref().map(|r| r.ctx(&state.redaction)), None, ) - .map_err(|e| { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!("failed to shape streamed batch: {e}"), - ))) - })?; + .map_err(|e| error_to_pg_in_context("failed to shape streamed batch", &e))?; for row in &shaped.rows { if emitted >= limit { break; @@ -313,16 +302,15 @@ pub(crate) async fn streaming_star_response( Ok(_) => { return single_pgwire_error(PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - "XX000".to_owned(), + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), "streamed batch payload was not a row array".to_owned(), )))); } Err(e) => { - return single_pgwire_error(PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!("failed to decode streamed batch: {e}"), - )))); + return single_pgwire_error(error_to_pg_in_context( + "failed to decode streamed batch", + &e, + )); } } } @@ -349,11 +337,10 @@ pub(crate) async fn streaming_star_response( ) { Ok(s) => s, Err(e) => { - return single_pgwire_error(PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "XX000".to_owned(), - format!("failed to shape streamed batch: {e}"), - )))); + return single_pgwire_error(error_to_pg_in_context( + "failed to shape streamed batch", + &e, + )); } }; // `SELECT *` derives its columns from the rows and has no client-requested diff --git a/nodedb/src/control/server/pgwire/handler/tenant_session.rs b/nodedb/src/control/server/pgwire/handler/tenant_session.rs index 88d2214b9..2af4e91eb 100644 --- a/nodedb/src/control/server/pgwire/handler/tenant_session.rs +++ b/nodedb/src/control/server/pgwire/handler/tenant_session.rs @@ -8,7 +8,7 @@ use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::{SessionId, TransactionState}; -use super::super::types::sqlstate_error; +use super::super::types::{error_to_pg_in_context, sqlstate_error}; use super::core::NodeDbPgHandler; impl NodeDbPgHandler { @@ -78,7 +78,7 @@ impl NodeDbPgHandler { let catalog = self.state.credentials.catalog(); let stored = catalog .find_tenant_by_name(value) - .map_err(|error| sqlstate_error("XX000", &format!("catalog read: {error}")))? + .map_err(|error| error_to_pg_in_context("catalog read", &error))? .ok_or_else(|| sqlstate_error("42704", &format!("tenant '{value}' not found")))?; crate::types::TenantId::new(stored.tenant_id) }; diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index 9a0c4bfb5..ebc617d15 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -24,7 +24,7 @@ pub fn sqlstate_error(code: &str, message: &str) -> PgWireError { /// Two tasks of one statement disagreeing on their verb is a planner bug, /// so it surfaces as an internal error. pub fn dml_fold_error_to_pg(e: &DmlFoldError) -> PgWireError { - sqlstate_error("XX000", &e.to_string()) + sqlstate_error(sqlstate::INTERNAL_ERROR, &e.to_string()) } /// Map a NodeDB `Error` to the pgwire error the client reads, through the @@ -38,6 +38,17 @@ pub fn error_to_pg(err: &crate::Error) -> PgWireError { ))) } +/// Map a NodeDB `Error` to the pgwire error the client reads, with `context` +/// before its message. The SQLSTATE stays the error's own. +pub fn error_to_pg_in_context(context: &str, err: &crate::Error) -> PgWireError { + let (severity, code, message) = error_to_sqlstate(err); + PgWireError::UserError(Box::new(ErrorInfo::new( + severity.to_owned(), + code.to_owned(), + format!("{context}: {message}"), + ))) +} + /// Map an error raised while shaping a response to the pgwire error the /// client reads, with the SQLSTATE its numeric code maps to. A per-row /// sequence accessor refusal (`42704`, `55000`) or a division by zero @@ -336,7 +347,7 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str | crate::Error::RoleInheritanceDepthExceeded { .. } => { ("ERROR", sqlstate::SYNTAX_ERROR, err.to_string()) } - crate::Error::DependentObjectsExist { .. } => ( + crate::Error::DependentObjectsExist { .. } | crate::Error::RoleInUse { .. } => ( "ERROR", sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, err.to_string(), @@ -403,8 +414,37 @@ pub fn response_status_to_sqlstate( if let Some(code) = error_code { Some(crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(code)) } else { - Some(("ERROR", "XX000", "unknown data plane error".into())) + Some(( + "ERROR", + sqlstate::INTERNAL_ERROR, + "unknown data plane error".into(), + )) + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A typed error behind a context prefix keeps its own SQLSTATE. + #[test] + fn an_error_in_context_keeps_its_sqlstate() { + let missing = crate::Error::CollectionNotFound { + tenant_id: crate::types::TenantId::new(1), + collection: "orders".into(), + }; + match error_to_pg_in_context("catalog read", &missing) { + PgWireError::UserError(info) => { + assert_eq!(info.code, sqlstate::UNDEFINED_TABLE); + assert!( + info.message.starts_with("catalog read: "), + "{}", + info.message + ); } + other => panic!("expected a user error, got {other:?}"), } } } diff --git a/nodedb/src/control/server/pgwire/types/mod.rs b/nodedb/src/control/server/pgwire/types/mod.rs index 7e03142c1..b7aa1ef45 100644 --- a/nodedb/src/control/server/pgwire/types/mod.rs +++ b/nodedb/src/control/server/pgwire/types/mod.rs @@ -12,7 +12,7 @@ pub mod parse; pub mod privilege; pub use error_map::{ - dml_fold_error_to_pg, error_to_pg, error_to_sqlstate, notice_warning, + dml_fold_error_to_pg, error_to_pg, error_to_pg_in_context, error_to_sqlstate, notice_warning, response_status_to_sqlstate, shape_error_to_pg, sqlstate_error, }; pub use field::{ diff --git a/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs b/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs index 2f77ac300..98e476d5f 100644 --- a/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs +++ b/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs @@ -20,6 +20,13 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> // `SourceFrozen` / `RetryableSchemaChanged` arms, and `OllpExhausted` // when it exhausted on drift. Ec::WRITE_CONFLICT => sqlstate::SERIALIZATION_FAILURE, + // Mirrors the `CalvinParticipantError` arm. + Ec::TRANSACTION_ROLLBACK => sqlstate::TRANSACTION_ROLLBACK, + // Mirrors the `NotInTransactionBlock` / `CrdtApplyForbiddenInTransaction` + // / `CrossShardInExplicitTransaction` arms. + Ec::ACTIVE_SQL_TRANSACTION => sqlstate::ACTIVE_SQL_TRANSACTION, + // Mirrors the `DependentObjectsExist` arm. + Ec::DEPENDENT_OBJECTS_EXIST => sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, // Mirrors the `DeadlineExceeded` arm. Ec::DEADLINE_EXCEEDED => sqlstate::QUERY_CANCELED, // Mirrors the `CollectionNotFound` / `CollectionDeactivated` arms. diff --git a/nodedb/src/control/server/shared/ddl/catalog.rs b/nodedb/src/control/server/shared/ddl/catalog.rs index 1d41aaf75..2eecf15bb 100644 --- a/nodedb/src/control/server/shared/ddl/catalog.rs +++ b/nodedb/src/control/server/shared/ddl/catalog.rs @@ -34,7 +34,7 @@ pub fn propose_and_apply( entry: &CatalogEntry, ) -> Result { let outcome = propose_catalog_entry(state, entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; apply_locally_if_needed(state, entry, outcome); Ok(outcome) } diff --git a/nodedb/src/control/server/shared/ddl/engine_apply.rs b/nodedb/src/control/server/shared/ddl/engine_apply.rs index baae6f168..6aaef0f89 100644 --- a/nodedb/src/control/server/shared/ddl/engine_apply.rs +++ b/nodedb/src/control/server/shared/ddl/engine_apply.rs @@ -95,9 +95,8 @@ pub(crate) async fn refuse_materialized_vector_index( &format!("{context}: vector index probe"), &crate::Error::DataPlane(code.clone()), )), - (_, None) => Err(DdlError::new( - "XX000", - format!("{context}: vector index probe failed with no error code"), - )), + (_, None) => Err(DdlError::internal(format!( + "{context}: vector index probe failed with no error code" + ))), } } diff --git a/nodedb/src/control/server/shared/ddl/index_registry.rs b/nodedb/src/control/server/shared/ddl/index_registry.rs index 7cbdb9c6c..473581153 100644 --- a/nodedb/src/control/server/shared/ddl/index_registry.rs +++ b/nodedb/src/control/server/shared/ddl/index_registry.rs @@ -15,10 +15,6 @@ use crate::types::{DatabaseId, TenantId}; use super::result::DdlError; -fn registry_err(message: String) -> DdlError { - DdlError::new("XX000", message) -} - /// The identity of one index, as its creating statement declared it. pub struct IndexRegistration<'a> { pub database_id: DatabaseId, @@ -45,13 +41,13 @@ pub fn propose_index_record( }; let entry = CatalogEntry::PutIndexRecord(Box::new(record.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| registry_err(format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { state .credentials .catalog() .put_index_record(&record) - .map_err(|e| registry_err(format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } Ok(()) } @@ -71,13 +67,13 @@ pub fn propose_delete_index_record( collection: collection.to_string(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| registry_err(format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { state .credentials .catalog() .delete_index_record(database_id.as_u64(), tenant_id.as_u64(), name) - .map_err(|e| registry_err(format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs index 04df9df7f..8cb1fe73b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs @@ -104,7 +104,7 @@ pub fn create_alert( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| err("XX000", "system clock error".to_string()))? + .map_err(|_| DdlError::internal("system clock error"))? .as_secs(); let def = AlertDef { diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs index a89a46b90..ddac16443 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs @@ -14,10 +14,6 @@ use crate::event::alert::types::AlertDef; use super::super::super::result::DdlError; use super::super::replicate::propose_and_apply; -fn err(sqlstate: &str, message: String) -> DdlError { - DdlError::new(sqlstate, message) -} - /// Propose the full alert record. CREATE and ALTER both re-put the row. /// /// The leader validates before proposing, so apply never rejects. @@ -28,7 +24,7 @@ pub(super) fn propose_put(state: &SharedState, def: &AlertDef) -> Result<(), Ddl .credentials .catalog() .put_alert_rule(def) - .map_err(|e| err("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; post_apply::put(def, state); Ok(()) }) @@ -52,7 +48,7 @@ pub(super) fn propose_delete( .credentials .catalog() .delete_alert_rule(database_id, tenant_id, name) - .map_err(|e| err("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; post_apply::delete(database_id, tenant_id, name, state); Ok(()) }) diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs index 8eb0dd41a..d0dbaa05f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs @@ -125,12 +125,12 @@ pub fn create_api_key( }); let entry = crate::control::catalog_entry::CatalogEntry::PutApiKey(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); catalog .put_api_key(&stored) - .map_err(|e| err("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.api_keys.install_replicated_key(&stored); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs index 3069870d2..886e55c1d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs @@ -54,13 +54,13 @@ pub fn revoke_api_key( key_id: key_id.to_string(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; let revoked = if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); state .api_keys .revoke_key(key_id, Some(catalog)) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? } else { // Cluster mode: trust the committed log index — the // in-memory cache update runs in a spawned tokio task and diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs index a807a353d..0d8174447 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs @@ -108,7 +108,7 @@ pub(super) fn parse_with_databases( for name in raw_names { let resolved: Option = catalog .get_database_id_by_name(name) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match resolved { Some(id) => ids.push(id), None => { @@ -141,6 +141,6 @@ pub(super) fn build_owner_database_set_for_user( .credentials .catalog() .list_user_grant_databases(user.user_id) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; Ok(DatabaseSet::Some(SmallVec::from_iter(db_ids))) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs b/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs index 66e84db8f..525f2f81b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs @@ -75,7 +75,7 @@ fn deactivate_auth_user( let found = state .auth_users .deactivate(user_id) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("auth user '{user_id}' not found"))); @@ -112,7 +112,7 @@ fn alter_auth_user_status( let found = state .auth_users .set_status(user_id, status_val) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("auth user '{user_id}' not found"))); @@ -165,7 +165,7 @@ pub fn purge_auth_users( let purged = state .auth_users .purge_inactive(cutoff) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs b/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs index 5ce7301da..770ea9f62 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs @@ -93,7 +93,7 @@ fn handle_blacklist_user( state .blacklist .blacklist_user(user_id, &reason, &identity.username, expires_at) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // WITH KILL SESSIONS — terminate active sessions immediately. let kill_sessions = parts.iter().any(|p| p.to_uppercase() == "KILL"); @@ -140,7 +140,7 @@ fn handle_blacklist_ip( state .blacklist .blacklist_ip(addr, &reason, &identity.username, expires_at) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, @@ -221,7 +221,7 @@ fn lift( &crate::control::security::blacklist::store::BlacklistStore, ) -> crate::Result, ) -> Result, DdlError> { - let removed = remove(&state.blacklist).map_err(|e| err("XX000", e.to_string()))?; + let removed = remove(&state.blacklist).map_err(|e| DdlError::from_error(&e))?; if !removed { return Err(err( "42704", diff --git a/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs b/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs index 8c9eb134c..a72684551 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs @@ -130,7 +130,7 @@ pub fn create_change_stream( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock before UNIX epoch"))? + .map_err(|_| DdlError::internal("system clock before UNIX epoch"))? .as_secs(); // Capture the creating principal's roles onto the subscription record. diff --git a/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs index 80795fc90..db30f3e3e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs @@ -81,11 +81,11 @@ pub fn drop_change_stream( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let _ = catalog .delete_change_stream(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; state .stream_registry .unregister(database_id, tenant_id, &name); diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs index 19c2977da..08362b1d4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs @@ -260,7 +260,7 @@ pub fn alter_raft_group( }; let data = change .to_entry_data() - .map_err(|e| ddl_err("XX000", format!("conf_change encode: {e}")))?; + .map_err(|e| DdlError::internal(format!("conf_change encode: {e}")))?; // Find a vShard that maps to this group to propose through Raft. let routing = match &state.cluster_routing { @@ -286,6 +286,6 @@ pub fn alter_raft_group( command: "ALTER RAFT GROUP".to_string(), rows_affected: None, }]), - Err(e) => Err(ddl_err("XX000", format!("propose failed: {e}"))), + Err(e) => Err(DdlError::from_error_in_context("propose failed", &e)), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs index 593379f46..a9086ecd4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs @@ -51,7 +51,7 @@ pub fn rebalance( let topo = topo.read().unwrap_or_else(|p| p.into_inner()); let plan = nodedb_cluster::compute_plan(&routing, &topo) - .map_err(|e| ddl_err("XX000", format!("rebalance planning failed: {e}")))?; + .map_err(|e| DdlError::internal(format!("rebalance planning failed: {e}")))?; if plan.is_empty() { let mut row = Map::new(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs index 6976f6749..d2facea4d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs @@ -126,7 +126,7 @@ pub(super) async fn alter_table_add_column( if let Some(ref coll) = updated { super::super::register::dispatch_register_from_stored(state, coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; super::strict_schema::recompile_rls_policies(state, coll)?; } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs index 89fdbc1aa..e9aa64422 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs @@ -90,7 +90,7 @@ pub(super) async fn add_materialized_sum( let existing_bindings: Vec = catalog .load_collections_for_tenant(database_id, tenant_id) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .into_iter() .flat_map(|c| c.materialized_sums) .collect(); @@ -107,7 +107,7 @@ pub(super) async fn add_materialized_sum( // maintenance write is still rejected on an unknown field. super::super::register::dispatch_register_from_stored(state, &coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // The SOURCE has to be re-registered too, and it is the half that decides // whether anything is folded at all: the binding is stored here on the @@ -117,7 +117,7 @@ pub(super) async fn add_materialized_sum( // nothing and the total silently stays where it was. super::super::register::dispatch_register_for_sum_sources(state, &coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -153,13 +153,13 @@ fn declare_target_column( } let config_json = coll.timeseries_config.as_deref().ok_or_else(|| { - err( - "XX000", - format!("strict collection '{}' has no stored schema", coll.name), - ) + DdlError::internal(format!( + "strict collection '{}' has no stored schema", + coll.name + )) })?; let mut schema: StrictSchema = sonic_rs::from_str(config_json) - .map_err(|e| err("XX000", format!("strict schema decode: {e}")))?; + .map_err(|e| DdlError::internal(format!("strict schema decode: {e}")))?; if schema.columns.iter().any(|c| c.name == column) { return Err(err( diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs index 0592adb88..81a61fe70 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs @@ -76,11 +76,11 @@ pub(super) fn alter_collection_owner( stored.owner = new_owner.to_string(); let entry = CatalogEntry::PutCollection(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .put_collection(database_id, &stored) - .map_err(|e| err("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.permissions.install_replicated_owner( &crate::control::security::catalog::StoredOwner { database_id: stored.database_id.as_u64(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs index 2e48e4b01..c4699df0f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs @@ -49,7 +49,7 @@ pub(super) fn load_strict_collection( .timeseries_config .as_deref() .and_then(|s| sonic_rs::from_str(s).ok()) - .ok_or_else(|| err("XX000", "strict schema missing or malformed"))?; + .ok_or_else(|| DdlError::internal("strict schema missing or malformed"))?; Ok((coll, schema)) } @@ -120,7 +120,7 @@ pub(super) async fn persist_schema_change( super::super::register::dispatch_register_from_stored(state, updated) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; recompile_rls_policies(state, updated)?; state.schema_version.bump(); Ok(()) @@ -144,5 +144,5 @@ pub(super) fn recompile_rls_policies( updated.tenant_id, &updated.name, ) - .map_err(|e| err("XX000", format!("rls recompile: {e}"))) + .map_err(|e| DdlError::from_error_in_context("rls recompile", &e)) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs index 7e3254363..81f13cc47 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs @@ -6,7 +6,8 @@ //! and messages the pgwire handlers produced), the single-row `ALTER`-status //! result builder, and the neutral `propose_and_apply` mirror of the pgwire //! `ddl::catalog_propose::propose_and_apply` (same propose + local-apply -//! ordering, same `XX000` / `"metadata propose: {e}"` error). +//! ordering). A propose error keeps its own SQLSTATE under a +//! `"metadata propose"` prefix. use nodedb_types::DatabaseId; @@ -52,7 +53,7 @@ pub(super) fn load_active_collection( .credentials .catalog() .get_collection(database_id, tenant_id, name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .filter(|c| c.is_active) .ok_or_else(|| err("42P01", format!("collection '{name}' does not exist"))) } @@ -67,7 +68,7 @@ pub(super) fn propose_and_apply( entry: &CatalogEntry, ) -> Result { let outcome = propose_catalog_entry(state, entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; apply_locally_if_needed(state, entry, outcome); Ok(outcome) } @@ -91,7 +92,7 @@ pub(super) async fn propose_and_apply_async( entry: CatalogEntry, ) -> Result { let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { // Clone only the cheap `Arc` handle (not `SharedState`) so // the blocking closure owns exactly what the apply needs. @@ -100,8 +101,8 @@ pub(super) async fn propose_and_apply_async( crate::control::catalog_entry::apply::apply_to(&entry, &catalog) }) .await - .map_err(|e| err("XX000", format!("catalog apply join: {e}")))? - .map_err(|e| err("XX000", format!("catalog apply: {e}")))?; + .map_err(|e| DdlError::internal(format!("catalog apply join: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog apply", &e))?; } Ok(outcome) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs index 9bb4b65a3..b3885a5d8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs @@ -18,7 +18,6 @@ use crate::control::server::shared::ddl::result::DdlError; use crate::control::state::SharedState; use super::super::super::vector_replicate::{propose_delete_model, propose_put_model}; -use super::support::err; /// Drop `column`'s embedding-model row on every node. pub(super) fn drop_vector_model_row( @@ -52,7 +51,7 @@ pub(super) fn move_vector_model_row( .credentials .catalog() .get_vector_model(db, tenant_id, collection, old_column) - .map_err(|e| err("XX000", format!("read vector model: {e}")))? + .map_err(|e| DdlError::from_error_in_context("read vector model", &e))? else { return Ok(()); }; @@ -74,7 +73,7 @@ fn model_row_exists( .catalog() .get_vector_model(database_id, tenant_id, collection, column) .map(|row| row.is_some()) - .map_err(|e| err("XX000", format!("read vector model: {e}"))) + .map_err(|e| DdlError::from_error_in_context("read vector model", &e)) } #[cfg(test)] diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs index a351d2ff8..fe76a4f37 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs @@ -162,9 +162,9 @@ fn check_engine_support( Ok(Some(c)) => c, Ok(None) => return Ok(()), // Collection doesn't exist yet — will fail at INSERT. Err(e) => { - return Err(ddl_err( - "XX000", - format!("COPY: catalog lookup failed: {e}"), + return Err(DdlError::from_error_in_context( + "COPY: catalog lookup failed", + &e, )); } }; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs index 6692abece..07b65fdee 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs @@ -127,9 +127,9 @@ fn check_collection_exists( "42P01", format!("COPY TO: collection \"{collection}\" does not exist"), )), - Err(e) => Err(ddl_err( - "XX000", - format!("COPY TO: catalog lookup failed: {e}"), + Err(e) => Err(DdlError::from_error_in_context( + "COPY TO: catalog lookup failed", + &e, )), } } @@ -194,7 +194,7 @@ async fn execute_and_collect( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", format!("COPY TO: dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("COPY TO: dispatch failed", &e))?; if resp.payload.is_empty() { continue; @@ -224,12 +224,8 @@ fn extract_json_rows( if json.is_empty() { return Ok(()); } - let mut parsed: serde_json::Value = sonic_rs::from_str(json).map_err(|e| { - ddl_err( - "XX000", - format!("COPY TO: failed to decode result rows: {e}"), - ) - })?; + let mut parsed: serde_json::Value = sonic_rs::from_str(json) + .map_err(|e| DdlError::internal(format!("COPY TO: failed to decode result rows: {e}")))?; redact_decoded_value(Some(redaction), store, &mut parsed); match parsed { serde_json::Value::Array(items) => { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs index 7d2003dc3..b552612ac 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs @@ -36,7 +36,7 @@ fn serialize_ndjson(rows: &[serde_json::Value]) -> Result, DdlError> { let mut out = Vec::with_capacity(rows.len() * 64); for row in rows { let line = sonic_rs::to_vec(row) - .map_err(|e| ddl_err("XX000", format!("COPY TO: JSON serialization error: {e}")))?; + .map_err(|e| DdlError::internal(format!("COPY TO: JSON serialization error: {e}")))?; out.extend_from_slice(&line); out.push(b'\n'); } @@ -46,12 +46,8 @@ fn serialize_ndjson(rows: &[serde_json::Value]) -> Result, DdlError> { fn serialize_json_array(rows: &[serde_json::Value]) -> Result, DdlError> { // Build a serde_json::Value::Array and serialize once. let arr = serde_json::Value::Array(rows.to_vec()); - let bytes = sonic_rs::to_vec(&arr).map_err(|e| { - ddl_err( - "XX000", - format!("COPY TO: JSON array serialization error: {e}"), - ) - })?; + let bytes = sonic_rs::to_vec(&arr) + .map_err(|e| DdlError::internal(format!("COPY TO: JSON array serialization error: {e}")))?; Ok(bytes) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs index 324cd03e6..836260ad6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs @@ -116,7 +116,7 @@ pub async fn build_and_persist( let catalog = state.credentials.catalog(); if catalog .get_materialized_view(database_id.as_u64(), tenant_id.as_u64(), name) - .map_err(|error| err("XX000", error.to_string()))? + .map_err(|error| DdlError::from_error(&error))? .is_some() { return Err(err( @@ -130,7 +130,7 @@ pub async fn build_and_persist( // over a soft-deleted incarnation's still-present storage. let existing = catalog .get_collection(database_id, tenant_id.as_u64(), name) - .map_err(|error| err("XX000", error.to_string()))?; + .map_err(|error| DdlError::from_error(&error))?; if let Some(existing) = existing { if existing.is_active { return Err(err( @@ -176,7 +176,7 @@ pub async fn build_and_persist( { guard.disarm(); } - return Err(err("XX000", failure.error.to_string())); + return Err(DdlError::from_error(&failure.error)); } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs index d978ef904..f9a4d928c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/indexed_vector_fields.rs @@ -51,9 +51,9 @@ pub(super) fn indexed_vector_fields( .catalog() .list_vector_index_params_in_database(database_id.as_u64()) .map_err(|e| { - DdlError::new( - "XX000", - format!("read vector indexes of \"{collection}\" for INSERT: {e}"), + DdlError::from_error_in_context( + &format!("read vector indexes of \"{collection}\" for INSERT"), + &e, ) })?; let mut named = HashSet::new(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs index a51501b33..0d6573484 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs @@ -101,9 +101,10 @@ pub async fn insert_document( fields.insert(field_def.name.clone(), typed_val); } Err(e) => { - return Some(Err(ddl_err( - "XX000", - format!("sequence '{seq_name}' error: {e}"), + return Some(Err(DdlError::from_error( + &crate::control::sequence::error_map::sequence_error_to_error( + seq_name, e, + ), ))); } } @@ -225,9 +226,9 @@ pub async fn insert_document( &pending.fields, ) { - return Some(Err(ddl_err( - "XX000", - format!("record inferred schema fields: {e}"), + return Some(Err(DdlError::from_error_in_context( + "record inferred schema fields", + &e, ))); } } @@ -298,7 +299,7 @@ pub async fn insert_document( ) { Ok(s) => s, Err(e) => { - return Some(Err(ddl_err("XX000", format!("surrogate assign: {e}")))); + return Some(Err(DdlError::from_error_in_context("surrogate assign", &e))); } }; let vec_plan = crate::bridge::envelope::PhysicalPlan::Vector(VectorOp::Insert { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index eb1f7ef8d..92f0373ff 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -84,11 +84,13 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn dispat } // A refusal arrives as an error status inside an `Ok` response. Ok(response) if response.status == crate::bridge::envelope::Status::Error => { - let (_, sqlstate, message) = match response.error_code.as_deref() { - Some(code) => error_code_to_sqlstate(code), - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), - }; - Some(Err(ddl_err(sqlstate, message))) + Some(Err(match response.error_code.as_deref() { + Some(code) => { + let (_, sqlstate, message) = error_code_to_sqlstate(code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + })) } Ok(_) => None, } @@ -349,8 +351,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a Arc::clone(&plan_lease_scope), ) { - return Err(ddl_err( - "XX000", + return Err(DdlError::internal( "internal error: failed to retain descriptor leases for buffered transaction tasks", )); } @@ -375,11 +376,13 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a return Err(ddl_err(sqlstate, message)); } Err(StagingGateError::Rejected { code }) => { - let (_, sqlstate, message) = match code { - Some(code) => error_code_to_sqlstate(&code), - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), - }; - return Err(ddl_err(sqlstate, message)); + return Err(match code { + Some(code) => { + let (_, sqlstate, message) = error_code_to_sqlstate(&code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + }); } }; @@ -419,15 +422,13 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a }; if response.status == crate::bridge::envelope::Status::Error { - let (_, sqlstate, message) = match response.error_code.as_deref() { - Some(code) => error_code_to_sqlstate(code), - None => ( - "ERROR", - "XX000", - String::from_utf8_lossy(&response.payload).into_owned(), - ), - }; - return Err(ddl_err(sqlstate, message)); + return Err(match response.error_code.as_deref() { + Some(code) => { + let (_, sqlstate, message) = error_code_to_sqlstate(code); + ddl_err(sqlstate, message) + } + None => DdlError::internal(String::from_utf8_lossy(&response.payload)), + }); } // Shape the STORED rows the write returned, redacted for the caller — @@ -458,7 +459,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a redaction: Some(redaction.ctx(&state.redaction)), sequences: Some(&sequences), }) - .map_err(|error| ddl_err("XX000", error.message().to_string()))?; + .map_err(|error| DdlError::from_error(&crate::Error::from(error)))?; // Folded rather than pushed: a statement is ONE result set, however // many tasks it planned to. if let ShapeOutcome::Rows(shaped) = outcome { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs index 16c4bac48..b9292a1e3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs @@ -42,7 +42,7 @@ pub(super) async fn fire_sync_after_triggers( .await .into_result() { - return Some(Err(ddl_err("XX000", &format!("trigger error: {e}")))); + return Some(Err(DdlError::from_error_in_context("trigger error", &e))); } None } @@ -84,7 +84,7 @@ pub(super) async fn fire_sync_after_update_triggers( .await .into_result() { - return Some(Err(ddl_err("XX000", &format!("trigger error: {e}")))); + return Some(Err(DdlError::from_error_in_context("trigger error", &e))); } None } @@ -120,7 +120,7 @@ pub(super) async fn fire_instead_triggers( }])) } Ok(crate::control::trigger::fire_instead::InsteadOfResult::NoTrigger) => None, - Err(e) => Some(Err(ddl_err("XX000", &format!("trigger error: {e}")))), + Err(e) => Some(Err(DdlError::from_error_in_context("trigger error", &e))), } } @@ -146,10 +146,9 @@ pub(super) async fn fire_before_triggers( .await { Ok(f) => Ok(f), - Err(e) => Err(Err(ddl_err("XX000", &format!("BEFORE trigger error: {e}")))), + Err(e) => Err(Err(DdlError::from_error_in_context( + "BEFORE trigger error", + &e, + ))), } } - -fn ddl_err(sqlstate: &str, msg: &str) -> DdlError { - DdlError::new(sqlstate, msg) -} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs index 7fb2afb8b..6a479f455 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs @@ -98,7 +98,7 @@ pub fn drop_collection( name, &mut visited, ) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? }; // Implicit SERIAL/BIGSERIAL sequences (`{collection}_{field}_seq`) @@ -187,7 +187,7 @@ pub fn drop_collection( let catalog = state.credentials.catalog(); if catalog .get_materialized_view(database_id.as_u64(), tenant_id.as_u64(), name) - .map_err(|error| err("XX000", error.to_string()))? + .map_err(|error| DdlError::from_error(&error))? .is_some() { return Err(err( @@ -271,7 +271,7 @@ pub fn drop_collection( None }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|error| err("XX000", error.to_string()))?; + .map_err(|error| DdlError::from_error(&error))?; if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); if purge { @@ -323,7 +323,9 @@ pub fn drop_collection( }, catalog, ) - .map_err(|error| err("XX000", format!("catalog deactivate failed: {error}")))?; + .map_err(|error| { + DdlError::from_error_in_context("catalog deactivate failed", &error) + })?; } } @@ -337,9 +339,9 @@ pub fn drop_collection( catalog .delete_sequence(database_id.as_u64(), tenant_id.as_u64(), &seq.name) .map_err(|e| { - err( - "XX000", - format!("failed to drop sequence '{}': {e}", seq.name), + DdlError::from_error_in_context( + &format!("failed to drop sequence '{}'", seq.name), + &e, ) })?; // Best-effort: registry removal is non-critical since catalog diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs index d60408641..d350dc8f0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs @@ -13,7 +13,7 @@ use crate::control::state::SharedState; use crate::types::TraceId; use super::super::super::super::result::DdlError; -use super::commit::{commit_collection_mutation, err}; +use super::commit::commit_collection_mutation; /// Backfill `build` on every node and flip it to `Ready`. /// @@ -40,7 +40,7 @@ pub(crate) async fn build_secondary_index( let catalog = state.credentials.catalog(); let Some(coll) = catalog .get_collection(database_id, tenant_id.as_u64(), collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .filter(|coll| coll.indexes.iter().any(|i| &i.name == index_name)) else { return Ok(()); @@ -67,7 +67,7 @@ pub(crate) async fn build_secondary_index( // register of a transaction's buffered entry runs asynchronously. super::super::dispatch_register_from_stored(state, &coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // The backfill runs on the local Data Plane (single node) or the leader // (cluster), vShard-local per core. @@ -91,20 +91,20 @@ pub(crate) async fn build_secondary_index( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if backfill_resp.status == crate::bridge::envelope::Status::Error { - let detail = match backfill_resp.error_code.as_deref() { - Some(crate::bridge::envelope::ErrorCode::Internal { detail, .. }) => detail.clone(), - Some(other) => format!("{other:?}"), - None => String::from_utf8_lossy(&backfill_resp.payload).into_owned(), - }; - let code = if detail.to_lowercase().contains("unique") { - "23505" - } else { - "XX000" - }; - return Err(err(code, detail)); + // A coded refusal keeps its SQLSTATE: a duplicate key is `23505`. + return Err(match backfill_resp.error_code.as_deref() { + Some(code) => DdlError::from_error_in_context( + "index backfill", + &crate::Error::DataPlane(code.clone()), + ), + None => DdlError::internal(format!( + "index backfill: {}", + String::from_utf8_lossy(&backfill_resp.payload) + )), + }); } // Every other node backfills the rows it hosts. Single-node and peerless @@ -147,7 +147,7 @@ async fn mark_ready(state: &SharedState, build: &SecondaryIndexBuild) -> Result< build.tenant_id.as_u64(), &build.collection, ) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? { for idx in ready_coll.indexes.iter_mut() { if idx.name == build.index_name { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs index bc21d03ec..ff7ebfa1f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs @@ -24,20 +24,20 @@ pub(super) async fn commit_collection_mutation( ) -> Result<(), DdlError> { let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(coll.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_collection(database_id, coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; } // Single-node path bypasses the applier post-apply hook, so the // Register refresh has to be fired here. In cluster mode the // applier's `put_async` does it on every node. super::super::dispatch_register_from_stored(state, coll) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs index 6fc00a172..9c4f5c330 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs @@ -149,7 +149,7 @@ pub async fn create_index( // loudly — only a genuine name collision is absorbed by `IF NOT EXISTS`. if let Some(existing) = catalog .get_index_record(database_id.as_u64(), tenant_id.as_u64(), &index_name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? { if if_not_exists { return Ok(create_index_ok()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs index 3d23952ea..c348f59a1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs @@ -68,7 +68,7 @@ pub async fn drop_index( .credentials .catalog() .get_index_record(database_id.as_u64(), tenant_id.as_u64(), index_name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? // An index whose collection is soft-deleted is not listed and cannot // be dropped on its own: the collection owns its lifecycle, and // UNDROP must bring it back intact. diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs index 501ab7f24..ce0ba9a2a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs @@ -31,7 +31,7 @@ use super::super::super::super::result::DdlError; use crate::control::server::shared::session::ddl_buffer; use crate::control::server::shared::session::ddl_effect::DeferredDdlEffect; -use super::commit::{commit_collection_mutation, err}; +use super::commit::commit_collection_mutation; /// Remove every piece of engine and catalog state belonging to `record`, /// except the registry and ownership rows the caller removes afterwards. @@ -92,7 +92,7 @@ async fn secondary( let catalog = state.credentials.catalog(); let Some(mut coll) = catalog .get_collection(database_id, tenant_id.as_u64(), &record.collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? else { // The registry outlived its collection — the collection teardown // path already reclaimed every engine surface, so there is nothing @@ -199,13 +199,12 @@ async fn vector( Ok(appended) => appended, Err(e) => { // Any record appended before the error never reaches a core. - minted - .cancel(&state.wal, owner, 0) - .await - .map_err(|c| err("XX000", format!("cancel vector index drop record: {c}")))?; - return Err(err( - "XX000", - format!("persist vector index drop to WAL: {e}"), + minted.cancel(&state.wal, owner, 0).await.map_err(|c| { + DdlError::from_error_in_context("cancel vector index drop record", &c) + })?; + return Err(DdlError::from_error_in_context( + "persist vector index drop to WAL", + &e, )); } }; @@ -215,12 +214,15 @@ async fn vector( // restart while replay still rebuilds the index from those records. let Some(lsn) = appended.lsn else { minted.settle(); - return Err(err("XX000", "vector index drop minted no WAL record")); + return Err(DdlError::internal("vector index drop minted no WAL record")); }; if let Err(e) = state.wal.wait_durable(lsn).await { // The record can still be on disk, so restart replay can reach it. minted.hold(); - return Err(err("XX000", format!("fsync vector index drop: {e}"))); + return Err(DdlError::from_error_in_context( + "fsync vector index drop", + &e, + )); } dispatch( @@ -255,7 +257,7 @@ async fn fulltext( tenant_id.as_u64(), &record.collection, ) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .into_iter() .filter(|r| r.kind == IndexKind::FullText && r.name != record.name) .count(); @@ -324,15 +326,19 @@ pub(crate) async fn dispatch( }, ) .await - .map_err(|e| err("XX000", format!("index teardown dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("index teardown dispatch failed", &e))?; if response.status == crate::bridge::envelope::Status::Error { - let detail = match response.error_code.as_deref() { - Some(crate::bridge::envelope::ErrorCode::Internal { detail, .. }) => detail.clone(), - Some(other) => format!("{other:?}"), - None => String::from_utf8_lossy(&response.payload).into_owned(), - }; - return Err(err("XX000", format!("index teardown failed: {detail}"))); + return Err(match response.error_code.as_deref() { + Some(code) => DdlError::from_error_in_context( + "index teardown failed", + &crate::Error::DataPlane(code.clone()), + ), + None => DdlError::internal(format!( + "index teardown failed: {}", + String::from_utf8_lossy(&response.payload) + )), + }); } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs index 98df897c4..6ae018608 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs @@ -30,10 +30,6 @@ use nodedb_physical::physical_plan::wire as plan_wire; use super::super::super::result::DdlError; -fn err(sqlstate: &str, message: impl Into) -> DdlError { - DdlError::new(sqlstate, message) -} - /// Remaining budget for per-peer RPCs. Chosen to cover backfill on /// collections with up to ~1M rows at the Data Plane's current /// throughput; large production collections will need a streaming @@ -42,9 +38,9 @@ const PEER_BACKFILL_DEADLINE: Duration = Duration::from_secs(120); /// Run `DocumentOp::BackfillIndex` on every cluster node other than /// this coordinator. Returns `Ok(())` only when every peer reports -/// success; any peer failure is returned as a DDL error with -/// SQLSTATE 23505 for duplicates and XX000 otherwise, matching the -/// single-node path. +/// success. A peer's typed refusal keeps its SQLSTATE, so a duplicate +/// key is `23505` as on the single-node path. A transport fault is +/// `XX000`. /// /// Single-node clusters (no peers) return `Ok(())` immediately — the /// coordinator's local dispatch already covered everything. @@ -106,8 +102,8 @@ pub(super) async fn backfill_on_peers( case_insensitive: args.case_insensitive, predicate: args.predicate.map(str::to_string), }); - let plan_bytes = - plan_wire::encode(&plan).map_err(|e| err("XX000", format!("backfill plan encode: {e}")))?; + let plan_bytes = plan_wire::encode(&plan) + .map_err(|e| DdlError::internal(format!("backfill plan encode: {e}")))?; // Fan out in parallel; collect per-peer outcomes. Any failure // aborts the commit — we do NOT compensate by dropping the index @@ -139,27 +135,20 @@ pub(super) async fn backfill_on_peers( for join in joins { let (node_id, outcome) = join .await - .map_err(|e| err("XX000", format!("peer backfill join: {e}")))?; + .map_err(|e| DdlError::internal(format!("peer backfill join: {e}")))?; let resp = outcome.map_err(|e| { - err( - "XX000", - format!("peer backfill transport to node {node_id}: {e}"), - ) + DdlError::internal(format!("peer backfill transport to node {node_id}: {e}")) })?; let RaftRpc::ExecuteResponse(resp) = resp else { - return Err(err( - "XX000", - format!("peer backfill on node {node_id}: unexpected RPC variant {resp:?}"), - )); + return Err(DdlError::internal(format!( + "peer backfill on node {node_id}: unexpected RPC variant {resp:?}" + ))); }; if let Some(e) = resp.error { - let detail = format!("peer backfill on node {node_id}: {e:?}"); - let code = if detail.to_lowercase().contains("unique") { - "23505" - } else { - "XX000" - }; - return Err(err(code, detail)); + return Err(DdlError::from_error_in_context( + &format!("peer backfill on node {node_id}"), + &crate::Error::from(e), + )); } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/show_indexes.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/show_indexes.rs index 199757094..18f7c0904 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/show_indexes.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/show_indexes.rs @@ -55,7 +55,7 @@ pub fn show_indexes( .credentials .catalog() .list_index_records(database_id.as_u64(), tenant_id.as_u64()) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; records.retain(StoredIndexRecord::is_visible); if let Some(collection) = filter_collection.as_deref() { records.retain(|r| r.collection == collection); diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs index 5122fd97a..feeac7c9b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs @@ -76,7 +76,7 @@ pub fn undrop_collection( )); } Err(e) => { - return Err(DdlError::new("XX000", e.to_string())); + return Err(DdlError::from_error(&e)); } }; if stored.is_active { @@ -133,7 +133,7 @@ pub fn undrop_collection( let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if outcome.needs_local_apply() { // Single-node fallback: run the same applier the replicated path runs // on every node, so the restore carries every invariant of a @@ -141,7 +141,7 @@ pub fn undrop_collection( // visibility of the indexes the soft-delete hid. Writing the row // directly here restored a collection whose indexes stayed hidden. crate::control::catalog_entry::apply::collection::put(&stored, catalog) - .map_err(|e| DdlError::new("XX000", format!("catalog restore failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog restore failed", &e))?; } let completion = UndropAuditDetail::new(name, UndropStage::Completed, owner_user_missing) diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs index e77c3a0ba..a2779e14e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs @@ -178,7 +178,7 @@ pub fn handle_show_vector_models( let entries = catalog .list_vector_models(database_id.as_u64(), tenant_id) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let columns = vec![ "collection".to_string(), @@ -235,7 +235,7 @@ pub fn handle_vector_metadata_query( let entry = catalog .get_vector_model(database_id.as_u64(), tenant_id, collection, column) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let json = match entry { Some(e) => { diff --git a/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs b/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs index daaa8571f..20ce9e541 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs @@ -60,14 +60,14 @@ pub async fn alter_set_on_conflict( let catalog = state.credentials.catalog(); let mut coll = catalog .get_collection(database_id, tenant_id, collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' not found")))?; // Step 1: read the durable policy, falling back to the same ephemeral // default the in-memory `PolicyRegistry` uses for an unregistered // collection. let mut policy: CollectionPolicy = match &coll.conflict_policy { - Some(json) => sonic_rs::from_str(json).map_err(|e| err("XX000", e.to_string()))?, + Some(json) => sonic_rs::from_str(json).map_err(|e| DdlError::internal(e.to_string()))?, None => CollectionPolicy::ephemeral(), }; @@ -76,7 +76,8 @@ pub async fn alter_set_on_conflict( apply_conflict_policy(&mut policy, constraint_kind, new_conflict_policy); // Step 3: persist on the catalog record and re-broadcast. - let policy_json = sonic_rs::to_string(&policy).map_err(|e| err("XX000", e.to_string()))?; + let policy_json = + sonic_rs::to_string(&policy).map_err(|e| DdlError::internal(e.to_string()))?; coll.conflict_policy = Some(policy_json); let entry = CatalogEntry::PutCollection(Box::new(coll)); propose_and_apply(state, &entry)?; @@ -105,14 +106,14 @@ pub async fn show_conflict_policy( let catalog = state.credentials.catalog(); let coll = catalog .get_collection(database_id, tenant_id, collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' not found")))?; let policy: CollectionPolicy = match &coll.conflict_policy { - Some(json) => sonic_rs::from_str(json).map_err(|e| err("XX000", e.to_string()))?, + Some(json) => sonic_rs::from_str(json).map_err(|e| DdlError::internal(e.to_string()))?, None => CollectionPolicy::ephemeral(), }; - let text = sonic_rs::to_string(&policy).map_err(|e| err("XX000", e.to_string()))?; + let text = sonic_rs::to_string(&policy).map_err(|e| DdlError::internal(e.to_string()))?; let mut row = Map::new(); row.insert("policy".to_string(), JsonValue::String(text)); diff --git a/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs index 8c4eb3ebe..164697490 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs @@ -63,7 +63,7 @@ pub fn add_state_constraint( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; if coll @@ -79,7 +79,7 @@ pub fn add_state_constraint( coll.state_constraints.push(def); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", &e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -127,7 +127,7 @@ pub fn add_transition_check( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; if coll.transition_checks.iter().any(|c| c.name == check_name) { @@ -139,7 +139,7 @@ pub fn add_transition_check( coll.transition_checks.push(def); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", &e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -203,7 +203,7 @@ pub fn add_check_constraint( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; if coll @@ -227,7 +227,7 @@ pub fn add_check_constraint( coll.check_constraints.push(def); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", &e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -263,7 +263,7 @@ pub fn drop_constraint( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; let before_state = coll.state_constraints.len(); @@ -285,7 +285,7 @@ pub fn drop_constraint( } persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", &e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs b/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs index 995fed066..28e2e1cc7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs @@ -38,7 +38,7 @@ pub fn show_constraints( let tenant_id = identity.tenant_id.as_u64(); let coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &coll_name) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{coll_name}' not found")))?; let columns = vec![ diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs index 824c27de6..33d1d705c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs @@ -246,7 +246,7 @@ pub async fn commit_offset( partition_id, offset, ) - .map_err(|e| DdlError::new("XX000", format!("offset commit: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("offset commit", &e))?; } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs index 267634371..d88fd4a86 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs @@ -99,7 +99,7 @@ pub async fn create_consumer_group( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock error"))? + .map_err(|_| DdlError::internal("system clock error"))? .as_secs(); let def = ConsumerGroupDef { diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs index d98822d3b..921b01586 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs @@ -71,7 +71,7 @@ pub fn migrate_legacy_topic_group( canonical_stream, group, ) - .map_err(|error| DdlError::new("XX000", format!("consumer-group migration: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("consumer-group migration", &error))?; super::replicate::propose_migrate(state, &def, legacy_stream)?; Ok(true) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs index 2827161ab..2e78a053d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs @@ -22,7 +22,7 @@ pub(super) fn propose_create(state: &SharedState, def: &ConsumerGroupDef) -> Res let entry = CatalogEntry::PutConsumerGroupIfAbsent(Box::new(def.clone())); propose_and_apply(state, &entry, || { apply::put_if_absent(def, state.credentials.catalog()) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; post_apply::put_if_absent(def, state); Ok(()) }) @@ -52,7 +52,7 @@ pub(super) fn propose_delete( name, state.credentials.catalog(), ) - .map_err(|e| DdlError::new("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; post_apply::delete(database_id, tenant_id, stream_name, name, state); Ok(()) }) @@ -73,7 +73,7 @@ pub(super) fn propose_migrate( }; propose_and_apply(state, &entry, || { apply::migrate_stream(def, legacy_stream, state.credentials.catalog()) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; post_apply::migrate_stream(def, legacy_stream, state); Ok(()) }) diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs index 8f8540f15..059fd935e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs @@ -144,7 +144,7 @@ pub async fn create_continuous_aggregate( // fields — the def is decoded on register dispatch in // `post_apply::async_dispatch::continuous_aggregate::put_async`. let def_bytes = zerompk::to_msgpack_vec(&def) - .map_err(|e| err("XX000", format!("serialize continuous aggregate def: {e}")))?; + .map_err(|e| DdlError::internal(format!("serialize continuous aggregate def: {e}")))?; let stored = StoredContinuousAggregate { database_id: database_id.as_u64(), @@ -226,7 +226,7 @@ pub async fn create_continuous_aggregate( propose_and_apply(state, &coll_entry)?; collection::dispatch_register_from_stored(state, &target) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; } // Single-node / no-applier path: the async post-apply dispatcher diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs index 324a2eecc..54a6c77a6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs @@ -75,7 +75,7 @@ pub async fn drop_continuous_aggregate( .credentials .catalog() .get_continuous_aggregate(database_id.as_u64(), tenant_id.as_u64(), &name) - .map_err(|e| err("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .ok_or_else(|| { err( "42704", diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs index e65c7c397..b590371cc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs @@ -86,7 +86,7 @@ pub async fn show_continuous_aggregates( { Ok(payload) => { crate::data::executor::response_codec::decode_payload(&payload).map_err(|e| { - DdlError::new("XX000", format!("continuous aggregate runtime stats: {e}")) + DdlError::from_error_in_context("continuous aggregate runtime stats", &e) })? } Err(_) => Vec::new(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs index 649498ac8..a2fb189cb 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs @@ -41,7 +41,7 @@ pub async fn convert_collection( let mut coll = catalog .get_collection(database_id, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; // Build columns before dispatch — needed for both Data Plane and catalog. @@ -63,7 +63,8 @@ pub async fn convert_collection( }; let schema_json_for_dp = if let Some(ref cols) = columns { - sonic_rs::to_string(cols).map_err(|e| err("XX000", format!("schema serialization: {e}")))? + sonic_rs::to_string(cols) + .map_err(|e| DdlError::internal(format!("schema serialization: {e}")))? } else { String::new() }; @@ -181,7 +182,7 @@ pub async fn convert_collection( } persist_collection_replicated(state, database_id, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // Refresh this node's Data Plane `doc_configs` entry to the NEW storage // mode. Without this, every later read of the collection resolves its @@ -190,7 +191,7 @@ pub async fn convert_collection( state, &coll, ) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; tracing::info!( %collection, diff --git a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs index 6e652b1f0..73f460d31 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs @@ -167,7 +167,7 @@ pub async fn crdt_apply( tenant_id, document_id.as_bytes(), ) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new(database_id, collection), @@ -200,7 +200,7 @@ pub async fn crdt_apply( .into_tasks() .into_iter() .next() - .ok_or_else(|| DdlError::new("XX000", "authorization returned no capability"))?; + .ok_or_else(|| DdlError::internal("authorization returned no capability"))?; // Route through the Raft proposer gate so the delta is quorum-durable under // replication. A local-only dispatch would land the delta on the receiving diff --git a/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs b/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs index f23ea6683..837ccf027 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs @@ -185,11 +185,11 @@ pub fn drop_type( name: name.to_string(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", &format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .delete_custom_type(database_id_u64, tenant_id, name) - .map_err(|e| err("XX000", &format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; } state @@ -289,11 +289,11 @@ fn persist_and_register(state: &SharedState, stored: StoredCustomType) -> Result let entry = crate::control::catalog_entry::CatalogEntry::PutCustomType(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| err("XX000", &format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let written = catalog .put_custom_type_assigning_oid(&stored) - .map_err(|e| err("XX000", &format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.custom_type_registry.register(written); } @@ -332,7 +332,7 @@ fn current_epoch_secs() -> Result { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map(|d| d.as_secs()) - .map_err(|_| err("XX000", "system clock error")) + .map_err(|_| DdlError::internal("system clock error")) } fn type_summary(def: &CustomTypeDef) -> (String, String) { diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs index 980dec440..78232cc62 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs @@ -34,13 +34,15 @@ pub fn alter_database( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; + // A name row whose descriptor is gone is a database a concurrent DROP + // removed between the two reads. let mut descriptor = catalog .get_database(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))? - .ok_or_else(|| ddl_err("XX000", format!("database '{name}' descriptor missing")))?; + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))? + .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; match operation { AlterDatabaseOperation::Rename { new_name } => { @@ -61,7 +63,7 @@ pub fn alter_database( } Ok(_) => {} Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } } descriptor.name = new_name.clone(); @@ -71,7 +73,7 @@ pub fn alter_database( || { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}"))) + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) }, )?; @@ -96,7 +98,7 @@ pub fn alter_database( // before/after diff so operators can reconstruct what changed. let before = catalog .get_database_quota(db_id) - .map_err(|e| ddl_err("XX000", format!("quota read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("quota read failed", &e))? .unwrap_or(QuotaRecord::DEFAULT); let mut record = before.clone(); record.merge(spec); @@ -107,7 +109,7 @@ pub fn alter_database( let ceiling = state.quota_ceiling_snapshot(); catalog .check_database_quota(db_id, &record, &ceiling) - .map_err(|e| ddl_err("53400", format!("{e}")))?; + .map_err(|e| DdlError::from_error(&e))?; // Replicated: every node writes the row and installs the quota in // its live enforcement components via post-apply. @@ -120,7 +122,7 @@ pub fn alter_database( || { catalog .write_database_quota(db_id, &record) - .map_err(|e| ddl_err("53400", format!("{e}")))?; + .map_err(|e| DdlError::from_error(&e))?; crate::control::catalog_entry::post_apply::quota::put_database( db_id, &record, state, ); @@ -175,7 +177,7 @@ pub fn alter_database( || { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}"))) + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) }, )?; @@ -208,7 +210,7 @@ pub fn alter_database( || { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}"))) + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) }, )?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs b/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs index 967d89ff4..3a3950f5f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs @@ -35,7 +35,7 @@ pub fn backup_database( )); } Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } }; require_database_owner_or_higher(state, identity, db_id, &format!("BACKUP DATABASE {name}"))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs b/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs index ffd340982..8803683b9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs @@ -52,7 +52,7 @@ pub async fn clone_database( // ── Resolve source database ─────────────────────────────────────────────── let source_db_id = catalog .get_database_id_by_name(params.source_name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| { ddl_err( "42P01", @@ -65,7 +65,7 @@ pub async fn clone_database( let source_descriptor = catalog .get_database(source_db_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))? .ok_or_else(|| { ddl_err( "42P01", @@ -92,7 +92,7 @@ pub async fn clone_database( // ── Enforce MAX_CLONE_DEPTH ──────────────────────────────────────────────── let depth = clone_chain_depth(state, source_db_id) - .map_err(|e| ddl_err("XX000", format!("clone depth check failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("clone depth check failed", &e))?; if depth >= MAX_CLONE_DEPTH { return Err(ddl_err( @@ -115,7 +115,7 @@ pub async fn clone_database( } Ok(None) => {} Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } } @@ -129,7 +129,7 @@ pub async fn clone_database( // empty the WAL frontier is used as the best available approximation, // which is correct for recent timestamps (within the same server session). let now_ms = - current_wall_ms().map_err(|e| ddl_err("XX000", format!("clock read failed: {e}")))?; + current_wall_ms().map_err(|e| DdlError::from_error_in_context("clock read failed", &e))?; let (as_of_lsn, as_of_ms) = match params.as_of { CloneAsOf::Latest => (state.wal.next_lsn(), now_ms), CloneAsOf::SystemTimeMs(ms) => { @@ -175,7 +175,7 @@ pub async fn clone_database( }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; // Single-node fast path (`LocalOnly` means "no Raft, apply directly"). // @@ -187,32 +187,34 @@ pub async fn clone_database( if outcome.needs_local_apply() { catalog .add_clone_child(source_db_id, target_db_id) - .map_err(|e| ddl_err("XX000", format!("lineage write failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("lineage write failed", &e))?; if let Err(put_err) = catalog.put_database(&target_descriptor) { // Compensate: remove the lineage edge we just wrote. A failure here // is fatal — surface both errors so on-call can repair the catalog. if let Err(rb_err) = catalog.remove_clone_child(source_db_id, target_db_id) { - return Err(ddl_err( - "XX000", - format!( - "catalog write failed: {put_err}; \ - lineage rollback ALSO failed: {rb_err} — \ + return Err(DdlError::from_error_in_context( + &format!( + "lineage rollback ALSO failed: {rb_err} — \ catalog left with orphan lineage edge \ - (source={source_db_id}, target={target_db_id})", + (source={source_db_id}, target={target_db_id}); catalog write failed", ), + &put_err, )); } - return Err(ddl_err("XX000", format!("catalog write failed: {put_err}"))); + return Err(DdlError::from_error_in_context( + "catalog write failed", + &put_err, + )); } // Stamp every active source collection into the target database with // `cloned_from` set. This lets the SQL planner resolve collection // names against the clone without knowing about clone indirection; // CoW delegation happens at dispatch time. - let source_colls = catalog - .load_all_collections(source_db_id) - .map_err(|e| ddl_err("XX000", format!("clone: enumerate source collections: {e}")))?; + let source_colls = catalog.load_all_collections(source_db_id).map_err(|e| { + DdlError::from_error_in_context("clone: enumerate source collections", &e) + })?; let kv_surrogate_ceiling = Some(state.surrogate_assigner.current_hwm()); for mut coll in source_colls.into_iter().filter(|c| c.is_active) { coll.database_id = target_db_id; @@ -231,12 +233,12 @@ pub async fn clone_database( // the failure is the only way the caller learns the clone is // incomplete. catalog.put_collection(target_db_id, &coll).map_err(|e| { - ddl_err( - "XX000", - format!( - "clone: stamping shadow descriptor for collection '{}' failed: {e}", + DdlError::from_error_in_context( + &format!( + "clone: stamping shadow descriptor for collection '{}' failed", coll.name ), + &e, ) })?; @@ -251,12 +253,12 @@ pub async fn clone_database( owner_username: coll.owner.clone(), }; catalog.put_owner(&owner).map_err(|e| { - ddl_err( - "XX000", - format!( - "clone: stamping owner for collection '{}' failed: {e}", + DdlError::from_error_in_context( + &format!( + "clone: stamping owner for collection '{}' failed", coll.name ), + &e, ) })?; } @@ -272,7 +274,7 @@ pub async fn clone_database( // answers queries the source answers differently, and nothing later // re-copies the row. copy_database_metadata(catalog, source_db_id, target_db_id) - .map_err(|e| ddl_err("XX000", format!("clone: copying catalog metadata: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("clone: copying catalog metadata", &e))?; } // Synonym groups and custom types travel as proposed entries, not as a @@ -329,24 +331,14 @@ async fn copy_synonym_groups( let groups = catalog .load_synonym_groups_in_database(source.as_u64()) .map_err(|e| { - ddl_err( - "XX000", - format!("clone: enumerate source synonym groups: {e}"), - ) + DdlError::from_error_in_context("clone: enumerate source synonym groups", &e) })?; for mut group in groups { group.database_id = target.as_u64(); let entry = CatalogEntry::PutSynonymGroup(Box::new(group.clone())); - let outcome = propose_and_apply(state, &entry).map_err(|e| { - ddl_err( - "XX000", - format!( - "clone: copying synonym group '{}': {}", - group.name, e.message - ), - ) - })?; + let outcome = propose_and_apply(state, &entry) + .map_err(|e| e.in_context(&format!("clone: copying synonym group '{}'", group.name)))?; if outcome.needs_local_apply() { state.synonym_registry.register(group.clone()); crate::control::catalog_entry::post_apply::install_synonym_group(group, state).await; @@ -372,25 +364,17 @@ fn copy_custom_types( let catalog = state.credentials.catalog(); let types = catalog .load_custom_types_in_database(source.as_u64()) - .map_err(|e| { - ddl_err( - "XX000", - format!("clone: enumerate source custom types: {e}"), - ) - })?; + .map_err(|e| DdlError::from_error_in_context("clone: enumerate source custom types", &e))?; for mut custom_type in types { custom_type.database_id = target.as_u64(); custom_type.oid = UNASSIGNED_OID; let entry = CatalogEntry::PutCustomType(Box::new(custom_type.clone())); let outcome = propose_and_apply(state, &entry).map_err(|e| { - ddl_err( - "XX000", - format!( - "clone: copying custom type '{}': {}", - custom_type.name, e.message - ), - ) + e.in_context(&format!( + "clone: copying custom type '{}'", + custom_type.name + )) })?; if outcome.needs_local_apply() { register_written( @@ -430,12 +414,7 @@ fn clone_chain_depth(state: &SharedState, start_db_id: DatabaseId) -> crate::Res if depth > MAX_CLONE_DEPTH { return Ok(depth); } - let desc = catalog - .get_database(current) - .map_err(|e| crate::Error::Storage { - engine: "catalog".into(), - detail: format!("depth walk get_database failed: {e}"), - })?; + let desc = catalog.get_database(current)?; match desc.and_then(|d| d.parent_clone) { None => return Ok(depth), Some(parent) => { diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/create.rs b/nodedb/src/control/server/shared/ddl/neutral/database/create.rs index a4f761fe3..add19995c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/create.rs @@ -79,7 +79,7 @@ pub fn create_database( } Ok(None) => {} Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } } @@ -112,14 +112,14 @@ pub fn create_database( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; // Direct write for single-node mode (`LocalOnly`) or as a fallback // when the cluster is in mixed-version compat mode. if outcome.needs_local_apply() { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; } // Flush the allocator hwm on the periodic threshold so restarts @@ -151,3 +151,57 @@ pub fn create_database( Ok(status("CREATE DATABASE")) } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_types::error::ErrorCode; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::control::security::identity::{DatabaseSet, Role}; + use crate::types::TenantId; + use crate::wal::WalManager; + + fn test_state() -> (tempfile::TempDir, Arc) { + let dir = tempfile::tempdir().expect("create test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("create-database.wal")) + .expect("open test WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let state = SharedState::new(dispatcher, wal).expect("construct shared state"); + (dir, state) + } + + fn admin() -> AuthenticatedIdentity { + AuthenticatedIdentity::new_internal_service( + 0, + "create_database_test", + TenantId::new(1), + vec![Role::Superuser], + true, + None, + DatabaseSet::All, + ) + } + + /// CREATE DATABASE of a name already taken is `duplicate_database` + /// (`42P04`) with the already-exists code, never an internal error. + #[test] + fn creating_an_existing_database_is_a_duplicate_database() { + let (_dir, state) = test_state(); + let identity = admin(); + create_database(&state, &identity, "orders", false, &[]).expect("first create succeeds"); + + let err = create_database(&state, &identity, "orders", false, &[]) + .expect_err("a second create of the same name is refused"); + assert_eq!(err.sqlstate, "42P04", "{err:?}"); + assert_eq!(err.code, ErrorCode::ALREADY_EXISTS); + + let existing = create_database(&state, &identity, "orders", true, &[]) + .expect("IF NOT EXISTS on an existing name succeeds"); + assert_eq!(existing.len(), 1); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs index f05865413..0931d9ae3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs @@ -47,7 +47,7 @@ pub fn drop_database( let db_id = match catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? { Some(id) => id, None => { @@ -92,7 +92,7 @@ pub fn drop_database( { let descriptor_for_mirror = catalog .get_database(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))?; if let Some(descriptor) = descriptor_for_mirror && let Some(origin) = descriptor.mirror_origin.as_ref() // Promoted mirrors are now standalone writable databases — the @@ -131,7 +131,7 @@ pub fn drop_database( // before proceeding. let dependent_ids = catalog .get_clone_children(db_id) - .map_err(|e| ddl_err("XX000", format!("lineage check failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("lineage check failed", &e))?; if !dependent_ids.is_empty() { if !cascade { @@ -181,7 +181,7 @@ pub fn drop_database( // ── Cascade: drop all collections ──────────────────────────────────────── let collections = catalog .load_all_collections(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog scan failed", &e))?; if !cascade && !collections.is_empty() { return Err(ddl_err( @@ -216,16 +216,16 @@ pub fn drop_database( db_id: db_id.as_u64(), }, ) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; if outcome.needs_local_apply() { catalog .delete_database(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog delete failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete failed", &e))?; // Single-node path: no applier runs, so this branch owns both the // quota row deletion and the live cap release. crate::control::catalog_entry::apply::quota::purge_database_scope(db_id.as_u64(), catalog) - .map_err(|e| ddl_err("XX000", format!("quota purge failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("quota purge failed", &e))?; crate::control::catalog_entry::post_apply::quota::release_database_scope(db_id, state); } @@ -260,13 +260,13 @@ fn drop_all_collections_in_database( catalog .delete_collection(db_id, coll.tenant_id, &coll.name) .map_err(|e| { - ddl_err( - "XX000", - format!( - "CASCADE DROP DATABASE {}: failed to delete collection '{}': {e}", + DdlError::from_error_in_context( + &format!( + "CASCADE DROP DATABASE {}: failed to delete collection '{}'", db_id.as_u64(), coll.name ), + &e, ) })?; } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs b/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs index 2e5a4d862..ba30a415d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs @@ -33,7 +33,7 @@ pub fn alter_database_materialize( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; require_database_owner_or_higher( diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs index d64c066e3..ebcd19e18 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs @@ -47,7 +47,7 @@ pub fn mirror_database( } Ok(None) => {} Err(e) => { - return Err(ddl_err("XX000", format!("catalog lookup failed: {e}"))); + return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } } @@ -110,12 +110,12 @@ pub fn mirror_database( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; if outcome.needs_local_apply() { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; } // Flush allocator hwm on threshold. diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs index 85bad9a70..6e120b02f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs @@ -34,7 +34,7 @@ pub fn promote_database( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; // Gate after db_id resolution so the audit record carries the database id. @@ -45,10 +45,12 @@ pub fn promote_database( &format!("ALTER DATABASE {name} PROMOTE"), )?; + // A name row whose descriptor is gone is a database a concurrent DROP + // removed between the two reads. let mut descriptor = catalog .get_database(db_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))? - .ok_or_else(|| ddl_err("XX000", format!("database '{name}' descriptor missing")))?; + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))? + .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; // Idempotent: if already promoted (or Active without any mirror_origin), // return success immediately. @@ -95,12 +97,12 @@ pub fn promote_database( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) - .map_err(|e| ddl_err("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; if outcome.needs_local_apply() { catalog .put_database(&descriptor) - .map_err(|e| ddl_err("XX000", format!("catalog write failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; } // The database is now writable. Clear the mirror-only catalog state so @@ -113,15 +115,15 @@ pub fn promote_database( // lineage (origin cluster, mode, last applied LSN at promotion). DROP // DATABASE relies on this cleanup having happened — see drop.rs. if let Err(e) = catalog.delete_mirror_collection_map(db_id) { - return Err(ddl_err( - "XX000", - format!("PROMOTE: failed to clear mirror_collection_map: {e}"), + return Err(DdlError::from_error_in_context( + "PROMOTE: failed to clear mirror_collection_map", + &e, )); } if let Err(e) = catalog.delete_mirror_lag(db_id) { - return Err(ddl_err( - "XX000", - format!("PROMOTE: failed to clear mirror_lag: {e}"), + return Err(DdlError::from_error_in_context( + "PROMOTE: failed to clear mirror_lag", + &e, )); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs index 25599022d..8ccc5f73f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs @@ -34,7 +34,7 @@ pub fn show_database_mirror_status( let all_databases = catalog .list_databases() - .map_err(|e| ddl_err("XX000", format!("catalog list failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog list failed", &e))?; let columns = vec![ "name".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show.rs index 7c47eb155..a8d9acb9c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show.rs @@ -17,7 +17,7 @@ use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; use super::gate::require_tenant_admin; -use super::support::{ddl_err, text_rows}; +use super::support::text_rows; /// Handle `SHOW DATABASES`. pub fn show_databases( @@ -30,7 +30,7 @@ pub fn show_databases( let databases = catalog .list_databases() - .map_err(|e| ddl_err("XX000", format!("catalog list failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog list failed", &e))?; let columns = vec![ "name".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs index 21334a02b..d72067aba 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs @@ -42,7 +42,7 @@ pub fn show_database_lineage( let start_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; // Walk the parent_clone chain, bounded by MAX_CLONE_DEPTH to prevent @@ -54,12 +54,12 @@ pub fn show_database_lineage( for _ in 0..max_hops { let desc = catalog .get_database(current_id) - .map_err(|e| ddl_err("XX000", format!("catalog read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))? .ok_or_else(|| { - ddl_err( - "XX000", - format!("database id {} descriptor missing", current_id.as_u64()), - ) + DdlError::internal(format!( + "database id {} descriptor missing", + current_id.as_u64() + )) })?; let (as_of_lsn, clone_created_at_lsn) = match &desc.parent_clone { diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs index 4edb16541..14ae8ca99 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs @@ -32,12 +32,12 @@ pub fn show_database_quota( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; let record = catalog .get_database_quota(db_id) - .map_err(|e| ddl_err("XX000", format!("quota read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("quota read failed", &e))? .unwrap_or(QuotaRecord::DEFAULT); let columns = vec![ diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs index cf1772332..9e8ffbb01 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs @@ -36,12 +36,12 @@ pub fn show_database_usage( let db_id = catalog .get_database_id_by_name(name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{name}' does not exist")))?; let record = catalog .get_database_quota(db_id) - .map_err(|e| ddl_err("XX000", format!("quota read failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("quota read failed", &e))? .unwrap_or(QuotaRecord::DEFAULT); // Pull live gauges from the system metrics registry. Dimensions without a diff --git a/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs b/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs index 7f1a7abda..553c8ce8f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/deferred_effects.rs @@ -137,7 +137,7 @@ mod tests { error, crate::Error::RejectedConstraint { ref constraint, .. } if constraint == "unique" )); - let other = effect_error("users", DdlError::new("XX000", "core gone")); + let other = effect_error("users", DdlError::internal("core gone")); assert!(matches!(other, crate::Error::Internal { .. })); } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs index 8c29757c9..40e7d84d7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs @@ -102,7 +102,7 @@ pub async fn crdt_merge( tenant_id, target_id.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let apply_plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new(database_id, collection), @@ -135,7 +135,7 @@ pub async fn crdt_merge( .into_tasks() .into_iter() .next() - .ok_or_else(|| ddl_err("XX000", "authorization returned no capability"))?; + .ok_or_else(|| DdlError::internal("authorization returned no capability"))?; // Route the merge result through the Raft proposer gate so the applied delta // is quorum-durable under replication, not lost to followers on failover. diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs index c69cff0a6..c998b4c0e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs @@ -64,7 +64,9 @@ pub fn create_sparse_index( .credentials .catalog() .get_index_record(database_id.as_u64(), tenant_id.as_u64(), &index_name) - .map_err(|e| ddl_err("XX000", format!("{CONTEXT}: read index registry: {e}")))? + .map_err(|e| { + DdlError::from_error_in_context(&format!("{CONTEXT}: read index registry"), &e) + })? { if stmt.header.if_not_exists && taken.kind == IndexKind::Sparse { return Ok(vec![DdlResult::Status { diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs index d1d8172cb..84b2622c6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs @@ -132,7 +132,9 @@ async fn create_text_index( .credentials .catalog() .get_index_record(database_id.as_u64(), tenant_id.as_u64(), &index_name) - .map_err(|e| ddl_err("XX000", format!("{command}: read index registry: {e}")))? + .map_err(|e| { + DdlError::from_error_in_context(&format!("{command}: read index registry"), &e) + })? { if stmt.header.if_not_exists && taken.kind == IndexKind::FullText { return Ok(vec![DdlResult::Status { diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs index 1d2b8679e..c68bc8501 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs @@ -120,7 +120,7 @@ pub async fn create_vector_index( collection, &field_name, ) - .map_err(|e| ddl_err("XX000", format!("read vector index params: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("read vector index params", &e))?; if existing.is_some() { if stmt.header.if_not_exists { return Ok(vec![status()]); @@ -141,7 +141,7 @@ pub async fn create_vector_index( .credentials .catalog() .get_index_record(database_id.as_u64(), tenant_id.as_u64(), index_name) - .map_err(|e| ddl_err("XX000", format!("read index registry: {e}")))? + .map_err(|e| DdlError::from_error_in_context("read index registry", &e))? { if stmt.header.if_not_exists && taken.kind == IndexKind::Vector { return Ok(vec![status()]); @@ -237,7 +237,7 @@ pub async fn create_vector_index( if outcome.needs_local_apply() { let shared = state .self_arc() - .map_err(|e| ddl_err("XX000", format!("install vector index params: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("install vector index params", &e))?; crate::control::catalog_entry::post_apply::install_vector_index_params(stored, shared) .await; } diff --git a/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs b/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs index faf303eb3..bf8d60d79 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/estimate_count.rs @@ -83,7 +83,7 @@ pub async fn estimate_count( ))]); } Err(e) => { - return Err(DdlError::new("XX000", e.to_string())); + return Err(DdlError::from_error(&e)); } } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs index 58d646541..ca4f88bbd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs @@ -210,7 +210,7 @@ pub fn assert_visible( collection, scope.auth(), ) - .map_err(|e| DdlError::new("XX000", format!("rls compile: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("rls compile", &e))?; let visible = rls_bytes.is_some_and(|b| b.is_empty()); // No filters = visible. diff --git a/nodedb/src/control/server/shared/ddl/neutral/field_def.rs b/nodedb/src/control/server/shared/ddl/neutral/field_def.rs index 2be5dd838..df47ab0e5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/field_def.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/field_def.rs @@ -126,7 +126,7 @@ pub fn define_field( database_id, &coll, ) { - return Err(err("XX000", &format!("save collection: {e}"))); + return Err(DdlError::from_error_in_context("save collection", &e)); } } _ => { @@ -229,7 +229,7 @@ pub fn define_event( database_id, &coll, ) { - return Err(err("XX000", &format!("save collection: {e}"))); + return Err(DdlError::from_error_in_context("save collection", &e)); } } _ => { @@ -301,7 +301,7 @@ pub fn remove_event( &format!("collection '{collection}' does not exist"), )); } - Err(e) => return Err(err("XX000", &format!("read collection: {e}"))), + Err(e) => return Err(DdlError::from_error_in_context("read collection", &e)), }; let before = coll.event_defs.len(); coll.event_defs.retain(|e| e.name != event_name); @@ -312,7 +312,7 @@ pub fn remove_event( )); } crate::control::catalog_entry::persist_collection_replicated(state, database_id, &coll) - .map_err(|e| err("XX000", &format!("save collection: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("save collection", &e))?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs index 8e060c2cf..2c234659e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs @@ -87,7 +87,7 @@ pub fn alter_function( let mut func = catalog .get_function_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| DdlError::new("42883", format!("function '{name}' does not exist")))?; let old_owner = func.owner.clone(); @@ -129,7 +129,7 @@ fn alter_function_limits( let mut func = catalog .get_function_in_database(database_id, tenant_id, name) - .map_err(|e| DdlError::new("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| DdlError::new("42883", format!("function '{name}' does not exist")))?; // Parse SET (...) from remaining parts. diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs b/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs index bed8f49bb..c44bb069c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs @@ -85,7 +85,7 @@ pub fn create_function( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock before UNIX epoch"))? + .map_err(|_| DdlError::internal("system clock before UNIX epoch"))? .as_secs(); let mut stored = StoredFunction { diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs index f8128f2d8..a5327e759 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs @@ -36,7 +36,7 @@ pub fn drop_function( // Check if function exists. let func_exists = catalog .get_function_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .is_some(); if !func_exists && !if_exists { @@ -54,7 +54,7 @@ pub fn drop_function( // Check dependencies: block DROP if other objects depend on this function. let dependents = catalog .find_dependents(database_id, tenant_id, "function", &name) - .map_err(|e| DdlError::new("XX000", format!("dependency check: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("dependency check", &e))?; if !dependents.is_empty() { let dep_list: Vec = dependents .iter() @@ -79,7 +79,7 @@ pub fn drop_function( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; crate::control::catalog_entry::apply::local::apply_locally_if_needed(state, &entry, outcome); // Broadcast deletion to connected Lite sessions. diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs index 916168b53..a7675db7f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs @@ -57,17 +57,16 @@ pub fn create_wasm_aggregate( .map_err(|e| DdlError::new("42601", e.to_string()))?; // Validate aggregate exports (init, accumulate, merge, finalize). - let runtime = - wasm::runtime::WasmRuntime::new().map_err(|e| DdlError::new("XX000", e.to_string()))?; + let runtime = wasm::runtime::WasmRuntime::new().map_err(|e| DdlError::from_error(&e))?; let module = runtime .get_or_compile(&wasm_bytes) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; wasm::wit::validate_aggregate_exports(&module) .map_err(|e| DdlError::new("42601", e.to_string()))?; let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock"))? + .map_err(|_| DdlError::internal("system clock"))? .as_secs(); // Store as a function with language=WASM. The "aggregate" nature is diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs index ea10769d1..493e14ff1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs @@ -57,7 +57,7 @@ pub fn create_wasm_function( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock before UNIX epoch"))? + .map_err(|_| DdlError::internal("system clock before UNIX epoch"))? .as_secs(); let stored = StoredFunction { diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs index b83c4a99c..df64a28aa 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs @@ -45,7 +45,7 @@ pub fn grant_database( let db_id = catalog .get_database_id_by_name(db_name) - .map_err(|e| DdlError::new("XX000", format!("catalog lookup: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup", &e))? .ok_or_else(|| DdlError::new("42704", format!("database '{db_name}' does not exist")))?; // Resolve the target user_id from the grantee name, as this statement @@ -68,12 +68,12 @@ pub fn grant_database( privilege: priv_name.to_string(), }, ) - .map_err(|e| DdlError::new("XX000", format!("catalog propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose", &e))?; if outcome.needs_local_apply() { catalog .put_database_grant(db_id, user_record.user_id, priv_name) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } } @@ -101,7 +101,7 @@ pub fn revoke_database( let db_id = catalog .get_database_id_by_name(db_name) - .map_err(|e| DdlError::new("XX000", format!("catalog lookup: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup", &e))? .ok_or_else(|| DdlError::new("42704", format!("database '{db_name}' does not exist")))?; let user_record = super::super::role_checks::visible_user(state, grantee) @@ -122,12 +122,12 @@ pub fn revoke_database( privilege: priv_name.to_string(), }, ) - .map_err(|e| DdlError::new("XX000", format!("catalog propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose", &e))?; if outcome.needs_local_apply() { catalog .delete_database_grant(db_id, user_record.user_id, priv_name) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs index 3ddf80bf9..192cc7875 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs @@ -67,13 +67,13 @@ fn propose_grant( .prepare_permission(target, grantee, perm, granted_by); let entry = CatalogEntry::PutPermission(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_permission(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.permissions.install_replicated_permission(&stored); } @@ -93,13 +93,13 @@ fn propose_revoke( permission: perm_str.clone(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .delete_permission(target, grantee, &perm_str) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state .permissions @@ -119,7 +119,7 @@ fn resolve_tenant_id(state: &SharedState, name: &str) -> Result Ok(algo_payload_to_rows(&payload, algorithm)?), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), }; } @@ -147,7 +147,7 @@ pub async fn algo( .await { Ok(resp) => Ok(algo_payload_to_rows(&resp.payload, algorithm)?), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -244,7 +244,7 @@ fn algo_payload_to_rows( let json_text = response_codec::decode_payload_to_json(payload); let rows: Vec = sonic_rs::from_str(&json_text) - .map_err(|e| ddl_err("XX000", format!("invalid algorithm result JSON: {e}")))?; + .map_err(|e| DdlError::internal(format!("invalid algorithm result JSON: {e}")))?; let mut shaped_rows = Vec::with_capacity(rows.len()); for row in &rows { diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs index b146666c8..fcf346e98 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs @@ -23,14 +23,11 @@ use super::super::super::result::{DdlError, DdlResult}; use super::edge_parse::{properties_to_json, validate_edge_label}; use super::support::{data_plane_verdict, ddl_err}; -/// Read the affected count off a Data-Plane response, mapping a missing count -/// to a [`DdlError`] via `ddl_err` — never a default. +/// Read the affected count off a Data-Plane response. A missing count is an +/// error, never a default. fn response_affected(response: &crate::bridge::envelope::Response) -> Result { require_affected_count(response.payload.as_bytes()).map_err(|e| { - ddl_err( - "XX000", - format!("edge write response is missing its affected count: {e}"), - ) + DdlError::from_error_in_context("edge write response is missing its affected count", &e) }) } @@ -73,7 +70,7 @@ pub async fn insert_edge( &collection, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // Dual-home: a cross-shard edge must be written on the home vShard of both src // and dst, or reverse/IN traversal never finds it. @@ -84,11 +81,11 @@ pub async fn insert_edge( let src_surrogate = assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), TraceId::ZERO) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let dst_surrogate = assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), TraceId::ZERO) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // Write policy decides the `PROPERTIES` image before staging: this handler // dispatches as trusted internal work, so nothing downstream resolves a policy. @@ -149,7 +146,7 @@ pub async fn insert_edge( crate::event::EventSource::User, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; data_plane_verdict(&response)?; response_affected(&response)? } else { @@ -163,11 +160,11 @@ pub async fn insert_edge( post_set_op: PostSetOp::None, txn_id: None, }; - let tx_class = build_static_tx_class(&[task], tenant_id, &[]) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + let tx_class = + build_static_tx_class(&[task], tenant_id, &[]).map_err(|e| DdlError::from_error(&e))?; let response = submit_calvin_routed(state, tx_class) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match response { Some(response) => { data_plane_verdict(&response)?; @@ -179,8 +176,7 @@ pub async fn insert_edge( // participant's applied response. A missing deposit here is a // scheduler invariant violation, never a value to guess. None => { - return Err(ddl_err( - "XX000", + return Err(DdlError::internal( "cross-shard edge insert completed with no applied response to read \ its affected count from", )); @@ -247,11 +243,11 @@ pub async fn delete_edge( let src_surrogate = assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), TraceId::ZERO) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let dst_surrogate = assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), TraceId::ZERO) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // A delete carries no image, so the policy compiles into the plan's write-gate // slot and is decided in the Data Plane against the edge's stored properties. @@ -310,7 +306,7 @@ pub async fn delete_edge( }; let response = crate::control::write_resolve::run_write_resolve(state, ctx, &*resolver) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; return Ok(vec![DdlResult::Status { command: "DELETE EDGE".to_string(), rows_affected: Some(response_affected(&response)?), @@ -331,7 +327,7 @@ pub async fn delete_edge( crate::event::EventSource::User, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; data_plane_verdict(&response)?; response_affected(&response)? } else { @@ -345,11 +341,11 @@ pub async fn delete_edge( post_set_op: PostSetOp::None, txn_id: None, }; - let tx_class = build_static_tx_class(&[task], tenant_id, &[]) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + let tx_class = + build_static_tx_class(&[task], tenant_id, &[]).map_err(|e| DdlError::from_error(&e))?; let response = submit_calvin_routed(state, tx_class) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match response { Some(response) => { data_plane_verdict(&response)?; @@ -363,8 +359,7 @@ pub async fn delete_edge( // absent), so a missing deposit here is a scheduler invariant // violation, never a value to guess. None => { - return Err(ddl_err( - "XX000", + return Err(DdlError::internal( "cross-shard edge delete completed with no applied response to read \ its affected count from", )); @@ -427,8 +422,8 @@ pub async fn set_node_labels( minted .cancel(&state.wal, owner, 0) .await - .map_err(|c| ddl_err("XX000", c.to_string()))?; - return Err(ddl_err("XX000", e.to_string())); + .map_err(|c| DdlError::from_error(&c))?; + return Err(DdlError::from_error(&e)); } let response = @@ -440,7 +435,7 @@ pub async fn set_node_labels( minted, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; data_plane_verdict(&response)?; let tag = if remove { "UNLABEL" } else { "LABEL" }; diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_parse.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_parse.rs index 7735b21c5..4795a9372 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_parse.rs @@ -57,7 +57,7 @@ pub(super) fn properties_to_json(properties: GraphProperties) -> Result sonic_rs::to_string(&nodedb_types::Value::Object(fields)) - .map_err(|e| ddl_err("XX000", format!("PROPERTIES serialize error: {e}"))), + .map_err(|e| DdlError::internal(format!("PROPERTIES serialize error: {e}"))), Some(Err(msg)) => Err(ddl_err( "42601", format!("PROPERTIES object literal error: {msg}"), diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_rls.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_rls.rs index 2b08b4220..436cfa378 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_rls.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_rls.rs @@ -22,7 +22,6 @@ use crate::control::state::SharedState; use crate::types::DatabaseId; use super::super::super::result::DdlError; -use super::support::ddl_err; /// Resolve the collection's write policy against a hand-built edge write. /// @@ -39,9 +38,8 @@ pub(super) fn resolve_edge_write_rls( CollectionReadGate::for_request(state, identity, database_id).inject_rls(&mut plan)?; match plan { PhysicalPlan::Graph(op) => Ok(op), - other => Err(ddl_err( - "XX000", - format!("edge write plan changed shape during RLS resolution: {other:?}"), - )), + other => Err(DdlError::internal(format!( + "edge write plan changed shape during RLS resolution: {other:?}" + ))), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs index 1b5bb30d3..5022b7d0c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs @@ -154,22 +154,22 @@ pub(super) async fn stage_edge_write_in_txn( // transaction block the gate returns `Staged`. Any other route means // the caller's `InBlock` check and the gate disagree. A report of zero // rows drops the write silently, so the statement fails instead. - Ok(InTxnRoute::Read(_) | InTxnRoute::Autocommit(_) | InTxnRoute::Buffered) => Err(ddl_err( - "XX000", - "a graph edge write reached the transaction staging gate and was not staged", - )), + Ok(InTxnRoute::Read(_) | InTxnRoute::Autocommit(_) | InTxnRoute::Buffered) => { + Err(DdlError::internal( + "a graph edge write reached the transaction staging gate and was not staged", + )) + } Err(StagingGateError::Dispatch(e)) => { let (_, sqlstate, message) = error_to_sqlstate(&e); Err(ddl_err(sqlstate, message)) } - Err(StagingGateError::Rejected { code }) => { - let (_, sqlstate, message) = match code { - Some(code) => { - crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code) - } - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), - }; - Err(ddl_err(sqlstate, message)) - } + Err(StagingGateError::Rejected { code }) => Err(match code { + Some(code) => { + let (_, sqlstate, message) = + crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + }), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs index 0f0e07a63..51ed521be 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs @@ -48,7 +48,6 @@ use nodedb_physical::physical_plan::GraphOp; use super::super::super::result::{DdlError, DdlResult}; use super::super::refuse_gate::RefusingReadGate; -use super::support::ddl_err; /// Names the collection-scoped stats read in the refusal a read policy raises. const STATS_WHAT: &str = "graph statistics, which are counters over the collection's edges"; @@ -120,10 +119,10 @@ pub async fn show_graph_stats( let resp = broadcast_to_all_cores(state, identity.tenant_id, database_id, plan, TraceId::ZERO) .await - .map_err(|e| ddl_err("58000", format!("graph stats dispatch failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("graph stats dispatch failed", &e))?; let merged: Vec = decode_merged_stats(resp.payload.as_bytes()) - .map_err(|e| ddl_err("XX000", format!("graph stats decode failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("graph stats decode failed", &e))?; let aggregated = aggregate_by_collection(merged); diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs index 4d14a25bc..2cf23600c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs @@ -36,11 +36,14 @@ pub(super) fn data_plane_verdict( if response.status != crate::bridge::envelope::Status::Error { return Ok(()); } - let (_, sqlstate, message) = match response.error_code.as_deref() { - Some(code) => super::super::super::sqlstate::error_code_to_sqlstate(code), - None => ("ERROR", "XX000", "unknown data plane error".to_owned()), - }; - Err(ddl_err(sqlstate, message)) + Err(match response.error_code.as_deref() { + Some(code) => { + let (_, sqlstate, message) = + super::super::super::sqlstate::error_code_to_sqlstate(code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + }) } /// Gate a named collection on catalog `is_active`: a plain `DROP COLLECTION` diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs index 81ec6d094..60a63b52b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs @@ -168,7 +168,7 @@ pub async fn traverse( .await { Ok(resp) => Ok(payload_to_rows(&resp.payload)), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -229,7 +229,7 @@ pub async fn neighbors( .await { Ok(resp) => Ok(payload_to_rows(&resp.payload)), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -287,7 +287,7 @@ pub async fn shortest_path( .await { Ok(resp) => Ok(payload_to_rows(&resp.payload)), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs b/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs index 6c8aba052..7ebd69c64 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs @@ -75,7 +75,7 @@ pub fn show_audit_log( let entries = catalog .load_recent_audit_entries(limit) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let (columns, column_types) = audit_columns(); let mut rows = Vec::with_capacity(entries.len()); @@ -236,7 +236,7 @@ pub fn show_audit_in_database( let db_id = catalog .get_database_id_by_name(db_name) - .map_err(|e| ddl_err("XX000", format!("catalog lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))? .ok_or_else(|| ddl_err("3D000", format!("database '{db_name}' does not exist")))?; let (columns, column_types) = audit_columns(); @@ -289,7 +289,7 @@ pub fn show_audit_in_database( let remaining = limit - rows.len(); let all_entries = catalog .load_recent_audit_entries(remaining * 10) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; for entry in all_entries.iter().rev() { if rows.len() >= limit { break; diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs index 063177913..4863e4a11 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/dispatch.rs @@ -118,21 +118,17 @@ pub(crate) async fn dispatch_and_respond( // durable apply, and a buffered route has no value to answer with, so // either one is a classification break, never an answer. Ok(InTxnRoute::Read(_)) => { - return Err(ddl_err( - "XX000", - format!("{func_name}: the staging gate classified this write as a read"), - )); + return Err(DdlError::internal(format!( + "{func_name}: the staging gate classified this write as a read" + ))); } Ok(InTxnRoute::Buffered) => { - return Err(ddl_err( - "XX000", - format!( - "{func_name}: the staging gate buffered this write, so it has no value to \ - return at the statement" - ), - )); + return Err(DdlError::internal(format!( + "{func_name}: the staging gate buffered this write, so it has no value to \ + return at the statement" + ))); } - Err(StagingGateError::Dispatch(e)) => return Err(ddl_err("XX000", e.to_string())), + Err(StagingGateError::Dispatch(e)) => return Err(DdlError::from_error(&e)), Err(StagingGateError::Rejected { code }) => return Err(data_plane_error(code)), }; @@ -217,7 +213,7 @@ fn error_to_ddl(error: &crate::Error) -> DdlError { /// silently downgraded to success. fn data_plane_error(code: Option) -> DdlError { let Some(code) = code else { - return ddl_err("XX000", "unknown data plane error"); + return DdlError::internal("unknown data plane error"); }; let (_, sqlstate, message) = crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code); diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs index baba35e3a..af622d164 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs @@ -60,7 +60,7 @@ pub async fn kv_incr( identity.tenant_id, key.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let shape = counter_shape(state, identity, &collection, &key, KvCounterKind::Integer)?; let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), @@ -127,7 +127,7 @@ pub async fn kv_incr_float( identity.tenant_id, key.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let shape = counter_shape(state, identity, &collection, &key, KvCounterKind::Float)?; let plan = PhysicalPlan::Kv(KvOp::IncrFloat { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), @@ -219,7 +219,7 @@ pub async fn kv_cas( identity.tenant_id, key.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Kv(KvOp::Cas { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), @@ -272,7 +272,7 @@ pub async fn kv_getset( identity.tenant_id, key.as_bytes(), ) - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Kv(KvOp::GetSet { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs index 3f7da3752..c039056b2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs @@ -87,7 +87,7 @@ pub async fn create_sorted_index( .credentials .catalog() .get_collection(database_id, tenant_id.as_u64(), &collection) - .map_err(|e| ddl_err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .is_none() { return Err(ddl_err( @@ -170,8 +170,7 @@ pub async fn create_sorted_index( plan, }); if !deferred { - return Err(ddl_err( - "XX000", + return Err(DdlError::internal( "CREATE SORTED INDEX: the transaction buffer took no entry to defer the \ index build on", )); @@ -266,8 +265,7 @@ pub async fn drop_sorted_index( index_name: index_name.clone(), }) { - return Err(ddl_err( - "XX000", + return Err(DdlError::internal( "DROP SORTED INDEX: the transaction buffer took no entry to defer the index \ drop on", )); diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs index ca2b0d8af..b33617619 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/dispatch.rs @@ -74,8 +74,8 @@ fn refusal(target: &SortedIndexTarget<'_>, resp: &Response) -> Option target.collection ), ), - Some(other) => ddl_err("XX000", format!("{other:?}")), - None => ddl_err("XX000", String::from_utf8_lossy(&resp.payload).into_owned()), + Some(other) => DdlError::from_error(&crate::Error::DataPlane(other.clone())), + None => DdlError::internal(String::from_utf8_lossy(&resp.payload)), }) } @@ -98,7 +98,7 @@ pub(super) async fn dispatch_read( read.txn_id, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match refusal(target, &resp) { Some(error) => Err(error), @@ -141,7 +141,7 @@ async fn dispatch_durable( match dispatched { Ok(resp) => Ok(resp), Err(crate::Error::DataPlane(code)) => Ok(verdict_response(code)), - Err(e) => Err(ddl_err("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -170,7 +170,7 @@ fn verdict_response(code: ErrorCode) -> Response { /// report an empty leaderboard for every query, whatever the index held. fn decode_rows(payload: &[u8]) -> Result, DdlError> { crate::data::executor::response_codec::decode_payload(payload) - .map_err(|e| ddl_err("XX000", format!("sorted index reply: {e}"))) + .map_err(|e| DdlError::from_error_in_context("sorted index reply", &e)) } /// Build the index's tree on the core that owns its collection's rows, and diff --git a/nodedb/src/control/server/shared/ddl/neutral/last_value.rs b/nodedb/src/control/server/shared/ddl/neutral/last_value.rs index 2eee908ab..4d48c6e8e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/last_value.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/last_value.rs @@ -60,7 +60,7 @@ pub async fn query_last_values( // them. let entries: Vec<(u64, i64, f64)> = crate::data::executor::response_codec::decode_payload(&payload) - .map_err(|e| ddl_err("XX000", format!("LAST_VALUES reply: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("LAST_VALUES reply", &e))?; let mut rows = Vec::with_capacity(entries.len()); for (series_id, ts, value) in &entries { @@ -126,7 +126,7 @@ pub async fn query_last_value( // encoded as a null (decoding to `None`), which is a different fact from a // payload that could not be read at all. let entry: Option<(i64, f64)> = crate::data::executor::response_codec::decode_payload(&payload) - .map_err(|e| ddl_err("XX000", format!("LAST_VALUE reply: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("LAST_VALUE reply", &e))?; let mut rows = Vec::new(); if let Some((ts, value)) = entry { @@ -148,7 +148,3 @@ pub async fn query_last_value( rows, ))]) } - -fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { - DdlError::new(sqlstate, message) -} diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs index 50a3c28d3..a0270d05a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs @@ -48,7 +48,7 @@ pub async fn handle_analyze( let coll = catalog .get_collection(database_id, tenant_id, &collection) - .map_err(|e| ddl_err("XX000", format!("catalog error: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog error", &e))? .ok_or_else(|| { ddl_err( "42P01", @@ -89,7 +89,7 @@ pub async fn handle_analyze( TraceId::ZERO, ) .await - .map_err(|error| ddl_err("XX000", format!("ANALYZE scan failed: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("ANALYZE scan failed", &error))?; if !resp.payload.is_empty() { let json = crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); push_scan_rows(&json, &mut rows); @@ -141,7 +141,7 @@ pub async fn handle_analyze( super::super::replicate::propose_and_apply(state, &entry, || { catalog .put_column_stats_batch(&local_rows) - .map_err(|e| ddl_err("XX000", format!("failed to store column stats: {e}"))) + .map_err(|e| DdlError::from_error_in_context("failed to store column stats", &e)) })?; state diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs index 36eefb70f..f4e40b619 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs @@ -23,7 +23,6 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; use super::super::super::result::DdlError; -use super::support::ddl_err; /// Duration estimate handed to the maintenance budget pre-screen. /// @@ -273,10 +272,7 @@ fn blocking_analyze( collection: &str, ) -> Result<(), DdlError> { let handle = tokio::runtime::Handle::try_current().map_err(|error| { - ddl_err( - "XX000", - format!("auto-ANALYZE needs a Tokio runtime: {error}"), - ) + DdlError::internal(format!("auto-ANALYZE needs a Tokio runtime: {error}")) })?; // `handle_analyze` reads the collection name off the second whitespace // token and lowercases it, so the bare name is what it expects. diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs index b3afd3f67..2a517dd17 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/reindex.rs @@ -78,7 +78,7 @@ pub async fn handle_reindex( trace_id, ) .await - .map_err(|e| ddl_err("XX000", format!("REINDEX failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("REINDEX failed", &e))?; tracing::info!(%collection, concurrent, "REINDEX acknowledged by all cores"); diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs index 3a593f440..e63e2e2f7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs @@ -59,7 +59,7 @@ pub async fn handle_show_vector_index( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if resp.payload.is_empty() { return Err(ddl_err( @@ -69,7 +69,7 @@ pub async fn handle_show_vector_index( } let stats: nodedb_types::VectorIndexStats = zerompk::from_msgpack(&resp.payload) - .map_err(|e| ddl_err("XX000", format!("decode vector stats: {e}")))?; + .map_err(|e| DdlError::internal(format!("decode vector stats: {e}")))?; let columns = vec!["property".to_string(), "value".to_string()]; @@ -156,7 +156,7 @@ pub async fn handle_alter_vector_index_seal( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; Ok(vec![DdlResult::Status { command: "SEAL".to_string(), @@ -189,7 +189,7 @@ pub async fn handle_alter_vector_index_compact( TraceId::ZERO, ) .await - .map_err(|e| ddl_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; Ok(vec![DdlResult::Status { command: "COMPACT".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs index 364575f06..b743cf1d7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs @@ -60,7 +60,7 @@ pub async fn handle_alter_vector_index_set( &collection, &field_name, ) - .map_err(|e| ddl_err("XX000", format!("read vector index params: {e}")))? + .map_err(|e| DdlError::from_error_in_context("read vector index params", &e))? .ok_or_else(|| { ddl_err( "42704", @@ -81,7 +81,7 @@ pub async fn handle_alter_vector_index_set( if outcome.needs_local_apply() { let shared = state .self_arc() - .map_err(|e| ddl_err("XX000", format!("install vector index params: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("install vector index params", &e))?; crate::control::catalog_entry::post_apply::install_vector_index_params(merged, shared) .await; } diff --git a/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs index ab45188ef..e8229b71d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs @@ -135,7 +135,7 @@ pub async fn match_query( // Serialize the MatchQuery for SPSC transport. let query_bytes = zerompk::to_msgpack_vec(&query) - .map_err(|e| DdlError::new("XX000", format!("serialize match query: {e}")))?; + .map_err(|e| DdlError::internal(format!("serialize match query: {e}")))?; let tenant_id = identity.tenant_id; @@ -189,7 +189,7 @@ pub async fn match_query( match_payload_to_rows(&outcome.rows_payload, &column_names) } } - Err(e) => Err(DdlError::new("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), }; } @@ -222,7 +222,7 @@ pub async fn match_query( match_payload_to_rows(&outcome.rows_payload, &column_names) } } - Err(e) => Err(DdlError::new("XX000", e.to_string())), + Err(e) => Err(DdlError::from_error(&e)), } } @@ -242,7 +242,7 @@ fn match_payload_to_rows( let json_text = response_codec::decode_payload_to_json(payload); let rows: Vec = sonic_rs::from_str(&json_text) - .map_err(|e| DdlError::new("XX000", format!("invalid match result JSON: {e}")))?; + .map_err(|e| DdlError::internal(format!("invalid match result JSON: {e}")))?; let mut out_rows = Vec::with_capacity(rows.len()); for row in &rows { diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs index e2ee74129..b72f90b88 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs @@ -81,7 +81,7 @@ pub async fn create_materialized_view( // failed. if catalog .get_materialized_view(database_id.as_u64(), tenant_id.as_u64(), &name) - .map_err(|error| err("XX000", error.to_string()))? + .map_err(|error| DdlError::from_error(&error))? .is_some() { return Err(err( @@ -91,7 +91,7 @@ pub async fn create_materialized_view( } if catalog .get_collection(database_id, tenant_id.as_u64(), &name) - .map_err(|error| err("XX000", error.to_string()))? + .map_err(|error| DdlError::from_error(&error))? .is_some() { return Err(err("42P07", format!("collection '{name}' already exists"))); @@ -178,7 +178,7 @@ pub async fn create_materialized_view( propose_and_apply(state, &coll_entry)?; super::super::collection::dispatch_register_from_stored(state, &target) .await - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; tracing::info!( view = name, @@ -277,7 +277,7 @@ async fn create_streaming_mv( Box::new(def.clone()), ); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|error| err("XX000", format!("metadata propose: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; crate::control::catalog_entry::apply::local::apply_locally_if_needed(state, &entry, outcome); if outcome.needs_local_apply() { state.permissions.install_replicated_owner( diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs index b7b0c57a1..6b3daf6df 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs @@ -70,7 +70,7 @@ pub fn drop_materialized_view( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|error| err("XX000", format!("metadata propose: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; crate::control::catalog_entry::apply::local::apply_locally_if_needed( state, &entry, outcome, ); @@ -136,14 +136,14 @@ pub fn drop_materialized_view( None }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|error| err("XX000", format!("metadata propose: {error}")))?; + .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; if outcome.needs_local_apply() { // No metadata Raft is active, so apply the same compound catalog // deletion locally and synchronously reclaim the implementation-owned // target collection. A reclaim failure after catalog deletion is // fatal: continuing would permit a same-name CREATE over stale rows. crate::control::catalog_entry::apply::apply_to(&entry, state.credentials.catalog()) - .map_err(|e| err("XX000", format!("catalog apply: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog apply", &e))?; let purge_lsn = state.wal.next_lsn().as_u64(); let purge_result = tokio::task::block_in_place(|| { tokio::runtime::Handle::current().block_on(async { diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs index b3f1db060..efbb38f3c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs @@ -19,10 +19,6 @@ use crate::types::DatabaseId; use super::super::super::result::{DdlError, DdlResult}; -fn err(sqlstate: &str, message: String) -> DdlError { - DdlError::new(sqlstate, message) -} - pub fn show_materialized_views( state: &SharedState, identity: &AuthenticatedIdentity, @@ -49,7 +45,7 @@ pub fn show_materialized_views( .credentials .catalog() .list_materialized_views(database_id.as_u64(), tenant_id.as_u64()) - .map_err(|e| err("XX000", format!("catalog read failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))?; let mut rows = Vec::new(); for view in &views { diff --git a/nodedb/src/control/server/shared/ddl/neutral/oidc.rs b/nodedb/src/control/server/shared/ddl/neutral/oidc.rs index fab4e8aed..192188157 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/oidc.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/oidc.rs @@ -141,7 +141,7 @@ pub fn create_oidc_provider( let tenant_exists = catalog .load_all_tenants() - .map_err(|e| DdlError::new("XX000", format!("tenant lookup: {e}")))? + .map_err(|e| DdlError::from_error_in_context("tenant lookup", &e))? .iter() .any(|tenant| tenant.tenant_id == tenant_id); if !tenant_exists { @@ -162,7 +162,7 @@ pub fn create_oidc_provider( } Ok(None) => {} Err(e) => { - return Err(DdlError::new("XX000", format!("catalog read: {e}"))); + return Err(DdlError::from_error_in_context("catalog read", &e)); } } @@ -182,7 +182,7 @@ pub fn create_oidc_provider( } } Err(e) => { - return Err(DdlError::new("XX000", format!("catalog list: {e}"))); + return Err(DdlError::from_error_in_context("catalog list", &e)); } } @@ -209,11 +209,11 @@ pub fn create_oidc_provider( let entry = CatalogEntry::PutOidcProvider(Box::new(provider.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .put_oidc_provider(&provider) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.audit_record( @@ -241,7 +241,7 @@ pub fn alter_oidc_provider_claim_mapping( let mut provider = catalog .get_oidc_provider(name) - .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .ok_or_else(|| DdlError::new("42704", format!("OIDC provider '{name}' does not exist")))?; validate_claim_mapping_roles(state, claim_mappings, provider.tenant_id)?; @@ -260,11 +260,11 @@ pub fn alter_oidc_provider_claim_mapping( let entry = CatalogEntry::PutOidcProvider(Box::new(provider.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .put_oidc_provider(&provider) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.audit_record( @@ -293,7 +293,7 @@ pub fn drop_oidc_provider( if catalog .get_oidc_provider(name) - .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .is_none() { if if_exists { @@ -309,11 +309,11 @@ pub fn drop_oidc_provider( name: name.to_string(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { catalog .delete_oidc_provider(name) - .map_err(|e| DdlError::new("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; } state.audit_record( @@ -337,7 +337,7 @@ pub fn show_oidc_providers( let providers = catalog .list_oidc_providers() - .map_err(|e| DdlError::new("XX000", format!("catalog list: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog list", &e))?; let columns = vec![ "name".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs index 2c4fb528a..85b9b9b7f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs @@ -109,7 +109,7 @@ fn alter_org( let found = state .orgs .set_status(org_id, &status_val) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("org '{org_id}' not found"))); } @@ -141,7 +141,7 @@ fn drop_org( let found = state .orgs .drop_org(org_id) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("org '{org_id}' not found"))); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs b/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs index e61fe8265..eb62a6e27 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs @@ -100,13 +100,13 @@ pub fn add_period_lock( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{name}' not found")))?; coll.period_lock = Some(def); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -141,13 +141,13 @@ pub fn drop_period_lock( let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id, &name) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{name}' not found")))?; coll.period_lock = None; persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs index d7a835cf4..e101f7f0a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs @@ -79,7 +79,7 @@ pub async fn set_permission_tree( let catalog = state.credentials.catalog(); let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; if !coll.is_active { @@ -91,10 +91,10 @@ pub async fn set_permission_tree( // Serialize and persist. let def_json = sonic_rs::to_string(&def) - .map_err(|e| err("XX000", format!("serialize PERMISSION_TREE: {e}")))?; + .map_err(|e| DdlError::internal(format!("serialize PERMISSION_TREE: {e}")))?; coll.permission_tree_def = Some(def_json); persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; let sources = [collection.clone(), def.permission_table.clone()]; @@ -168,12 +168,12 @@ pub async fn drop_permission_tree( let catalog = state.credentials.catalog(); let mut coll = catalog .get_collection(DatabaseId::DEFAULT, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; coll.permission_tree_def = None; persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; // Update in-memory cache. state diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs index 71bf5f7fe..7296fa088 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs @@ -34,7 +34,7 @@ pub async fn call_procedure( let proc = catalog .get_procedure_in_database(database_id, tenant_id.as_u64(), &name) - .map_err(|e| DdlError::new("XX000", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| DdlError::new("42883", format!("procedure '{name}' does not exist")))?; // Validate argument count matches IN parameters. diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs index dd5e07d46..0a3bbff7f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs @@ -50,7 +50,7 @@ pub fn create_procedure( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock before UNIX epoch"))? + .map_err(|_| DdlError::internal("system clock before UNIX epoch"))? .as_secs(); let routability = extract_routability(&parsed.body_sql); diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs index eaad03580..fa75f4579 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs @@ -57,7 +57,7 @@ pub fn drop_procedure( // a clean no-op that never touches raft. let exists_before = catalog .get_procedure_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog read: {e}")))? + .map_err(|e| DdlError::from_error_in_context("catalog read", &e))? .is_some(); if !exists_before && !if_exists { return Err(DdlError::new( @@ -75,11 +75,11 @@ pub fn drop_procedure( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let _ = catalog .delete_procedure_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } // Broadcast deletion to connected Lite sessions. diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs index 042c54865..81e5261c6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs @@ -60,7 +60,7 @@ pub async fn balance_as_of( tenant_id, &pk_bytes, ) - .map_err(|e| err("XX000", &format!("surrogate lookup failed: {e}")))? + .map_err(|e| DdlError::from_error_in_context("surrogate lookup failed", &e))? .unwrap_or(nodedb_types::Surrogate::ZERO); let mut get_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::PointGet { @@ -83,7 +83,7 @@ pub async fn balance_as_of( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("point get failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("point get failed", &e))?; let doc_json = crate::data::executor::response_codec::decode_payload_to_json(&get_resp.payload); let doc: serde_json::Value = sonic_rs::from_str(&doc_json).unwrap_or(serde_json::Value::Null); @@ -97,7 +97,7 @@ pub async fn balance_as_of( let catalog = state.credentials.catalog(); let coll = catalog .get_collection(database_id, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{collection}' not found")))?; let Some(mat_def) = coll @@ -149,7 +149,7 @@ pub async fn balance_as_of( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("source scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("source scan failed", &e))?; let source_json = crate::data::executor::response_codec::decode_payload_to_json(&source_resp.payload); @@ -170,7 +170,7 @@ pub async fn balance_as_of( let src_doc = serde_json::Value::Object(obj.clone()); let created_at = crate::data::executor::enforcement::retention::extract_created_at_secs( &sonic_rs::to_vec(&src_doc) - .map_err(|e| err("XX000", &format!("serialization failed: {e}")))?, + .map_err(|e| DdlError::internal(format!("serialization failed: {e}")))?, ); if let Some(ts) = created_at { if ts <= as_of_secs { diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs index 62a1cff67..653c3c79c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/convert_currency_lookup.rs @@ -98,7 +98,7 @@ pub async fn convert_currency_lookup( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("rate table scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("rate table scan failed", &e))?; let payload_json = crate::data::executor::response_codec::decode_payload_to_json(&scan_resp.payload); diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs index 28cc868e6..dc3141fdd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs @@ -94,7 +94,7 @@ pub fn single_result(value: &str) -> Vec { pub fn unwrap_scan_docs(docs: Vec) -> Result>, DdlError> { let mut rows = Vec::with_capacity(docs.len()); for doc in docs { - push_flat_rows(Value::from(doc), &mut rows).map_err(|e| err("XX000", &e.to_string()))?; + push_flat_rows(Value::from(doc), &mut rows).map_err(|e| DdlError::from_error(&e))?; } Ok(rows.iter().map(row_to_wire_json).collect()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs index 2660b40c2..333b23d33 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/temporal_lookup.rs @@ -72,7 +72,7 @@ pub async fn temporal_lookup( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("scan failed", &e))?; let payload_json = crate::data::executor::response_codec::decode_payload_to_json(&scan_resp.payload); diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_audit_chain.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_audit_chain.rs index 682996ef0..a8aa3617a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_audit_chain.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_audit_chain.rs @@ -71,7 +71,7 @@ pub async fn verify_audit_chain( let audit_entries = state .wal .recover_audit_entries() - .map_err(|e| err("XX000", &format!("audit WAL recovery failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("audit WAL recovery failed", &e))?; let mut valid = true; let mut checked = 0u64; diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs index 70229796d..33c50bb46 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_balance.rs @@ -51,7 +51,7 @@ pub async fn verify_balance( let catalog = state.credentials.catalog(); let coll = catalog .get_collection(database_id, tenant_id.as_u64(), &collection) - .map_err(|e| err("XX000", &e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", &format!("collection '{collection}' not found")))?; let Some(mat_def) = coll @@ -96,7 +96,7 @@ pub async fn verify_balance( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("target scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("target scan failed", &e))?; let target_json = crate::data::executor::response_codec::decode_payload_to_json(&target_resp.payload); let target_docs: Vec = sonic_rs::from_str(&target_json) @@ -136,7 +136,7 @@ pub async fn verify_balance( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("source scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("source scan failed", &e))?; let source_json = crate::data::executor::response_codec::decode_payload_to_json(&source_resp.payload); let source_docs: Vec = sonic_rs::from_str(&source_json) diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs index e97086ffe..e1bfea4a1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs @@ -70,7 +70,7 @@ pub async fn verify_hash_chain( TraceId::ZERO, ) .await - .map_err(|e| err("XX000", &format!("scan failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("scan failed", &e))?; let payload_json = crate::data::executor::response_codec::decode_payload_to_json(&scan_resp.payload); @@ -118,7 +118,7 @@ pub async fn verify_hash_chain( obj.remove("_chain_hash"); } let doc_bytes = sonic_rs::to_vec(&doc_for_hash) - .map_err(|e| err("XX000", &format!("failed to serialize document: {e}")))?; + .map_err(|e| DdlError::internal(format!("failed to serialize document: {e}")))?; let expected = crate::data::executor::enforcement::hash_chain::compute_chain_hash( &prev_hash, &doc_id, &doc_bytes, diff --git a/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs index 566c09f7b..cb5cd6e95 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs @@ -129,7 +129,7 @@ pub fn define_quota( .credentials .catalog() .put_scope_quota(&stored) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; scope_quota_post_apply::put(&stored, state); Ok(()) }, @@ -185,7 +185,7 @@ pub fn drop_quota( .credentials .catalog() .delete_scope_quota(&scope_name) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; scope_quota_post_apply::delete(&scope_name, state); Ok(()) }, diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index 0b9de8806..5cb11a926 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -117,10 +117,9 @@ pub async fn rate_check( let current: i64 = sonic_rs::from_str::(&payload_text) .ok() .and_then(|v| v.get("value")?.as_i64()) - .ok_or(ddl_err( - "XX000", - format!("RATE_CHECK: counter '{rate_key}' increment answered no integer value"), - ))?; + .ok_or(DdlError::internal(format!( + "RATE_CHECK: counter '{rate_key}' increment answered no integer value" + )))?; if current > max_count { // Read TTL to compute retry_after_ms. @@ -316,10 +315,9 @@ fn ttl_from( let ttl_ms = sonic_rs::from_str::(&text) .ok() .and_then(|v| v.get("ttl_ms")?.as_i64()) - .ok_or(ddl_err( - "XX000", - format!("{context}: TTL read answered no integer ttl_ms: {text}"), - ))?; + .ok_or(DdlError::internal(format!( + "{context}: TTL read answered no integer ttl_ms: {text}" + )))?; Ok((ttl_ms != TTL_ABSENT).then_some(ttl_ms)) } @@ -337,7 +335,7 @@ fn counter_from( .ok() .and_then(|text| text.parse::().ok()) .ok_or(ddl_err( - "XX000", + "22P02", format!( "RATE_REMAINING: counter '{rate_key}' does not hold decimal text; \ reset the gate with RATE_RESET" @@ -574,6 +572,20 @@ mod tests { ); } + /// A counter that holds text other than a decimal is a stored value the + /// read cannot parse: `22P02`, never an internal error. + #[test] + fn a_non_decimal_counter_is_invalid_text() { + let err = counter_from("_rate:g:k", Ok(answer(b"abc".to_vec()))) + .expect_err("a non-decimal counter fails the read"); + assert_eq!( + err.sqlstate, + sqlstate::INVALID_TEXT_REPRESENTATION, + "{err:?}" + ); + assert_eq!(err.code, nodedb_types::error::ErrorCode::BAD_REQUEST); + } + /// The TTL read behind `retry after` propagates a dispatch error instead /// of reading it as no time left. #[test] diff --git a/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs index 8cdcb2d25..8e8b9a2e6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/read_gate.rs @@ -50,8 +50,6 @@ const INSUFFICIENT_PRIVILEGE: &str = "42501"; const FEATURE_NOT_SUPPORTED: &str = "0A000"; /// SQLSTATE for a collection the catalog does not hold. const UNDEFINED_TABLE: &str = "42P01"; -/// SQLSTATE for a policy set that could not be compiled. -const INTERNAL_ERROR: &str = "XX000"; fn gate_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) @@ -192,7 +190,7 @@ impl<'a> CollectionReadGate<'a> { self.tenant_id().as_u64(), collection, ) - .map_err(|e| gate_err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match stored { None => Err(gate_err( UNDEFINED_TABLE, @@ -222,7 +220,7 @@ impl<'a> CollectionReadGate<'a> { collection, self.scope.auth(), ) - .map_err(|e| gate_err(INTERNAL_ERROR, format!("rls compile: {e}")))? + .map_err(|e| DdlError::from_error_in_context("rls compile", &e))? .is_some_and(|filters| filters.is_empty()); if unrestricted { return Ok(()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs b/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs index cad899278..791b56250 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs @@ -130,17 +130,17 @@ pub fn create_redaction_policy( }; let stored = StoredRedactionPolicy::from_runtime(&policy) - .map_err(|e| DdlError::new("XX000", format!("redaction serialize: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("redaction serialize", &e))?; let entry = CatalogEntry::PutRedactionPolicy(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_redaction_policy(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.redaction.install_replicated_policy(policy); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs b/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs index c5323cc87..0baa11a0b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs @@ -54,13 +54,13 @@ pub fn drop_redaction_policy( for_role: for_role.to_string(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .delete_redaction_policy(tenant_id, &qualified_collection, for_role) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state .redaction diff --git a/nodedb/src/control/server/shared/ddl/neutral/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/replicate.rs index d92262523..ddbefd261 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/replicate.rs @@ -16,7 +16,7 @@ use super::super::result::DdlError; /// Propose `entry`, then run `local` when this node owns the catalog write. /// -/// A propose failure maps to SQLSTATE `XX000`. +/// A propose failure keeps the SQLSTATE class of its typed error. pub(crate) fn propose_and_apply( state: &SharedState, entry: &CatalogEntry, @@ -35,7 +35,7 @@ pub(crate) fn propose_and_apply_outcome( local: impl FnOnce() -> Result<(), DdlError>, ) -> Result { let outcome = propose_catalog_entry(state, entry) - .map_err(|e| DdlError::new("XX000", format!("catalog propose failed: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; if outcome.needs_local_apply() { local()?; } diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs index 0f87dbf35..5cb97b7c1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs @@ -125,7 +125,7 @@ pub async fn create_retention_policy( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| err("XX000", "system clock error".to_string()))? + .map_err(|_| DdlError::internal("system clock error"))? .as_secs(); let def = RetentionPolicyDef { @@ -170,15 +170,18 @@ pub async fn create_retention_policy( // Roll back through the same replicated path that created it, so the // policy disappears on every node, not only on this one. if let Err(rollback) = propose_delete(state, &def) { - return Err(err( - "XX000", - format!( - "failed to auto-wire aggregates: {e}; rollback left the policy in place: {}", + return Err(DdlError::from_error_in_context( + &format!( + "rollback left the policy in place: {}; failed to auto-wire aggregates", rollback.message ), + &e, )); } - return Err(err("XX000", format!("failed to auto-wire aggregates: {e}"))); + return Err(DdlError::from_error_in_context( + "failed to auto-wire aggregates", + &e, + )); } state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs index 47b8f00e6..4b5d42aba 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs @@ -14,10 +14,6 @@ use crate::engine::timeseries::retention_policy::RetentionPolicyDef; use super::super::super::result::DdlError; use super::super::replicate::propose_and_apply; -fn err(sqlstate: &str, message: String) -> DdlError { - DdlError::new(sqlstate, message) -} - /// Propose the full policy record. CREATE and ALTER both re-put the row. /// /// The leader validates before proposing, so apply never rejects. @@ -28,7 +24,7 @@ pub(super) fn propose_put(state: &SharedState, def: &RetentionPolicyDef) -> Resu .credentials .catalog() .put_retention_policy(def) - .map_err(|e| err("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; post_apply::put(def, state); Ok(()) }) @@ -50,7 +46,7 @@ pub(super) fn propose_delete( .credentials .catalog() .delete_retention_policy(def.database_id, def.tenant_id, &def.name) - .map_err(|e| err("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; post_apply::delete(def.database_id, def.tenant_id, &def.name, state); Ok(()) }) diff --git a/nodedb/src/control/server/shared/ddl/neutral/rls.rs b/nodedb/src/control/server/shared/ddl/neutral/rls.rs index 68a31ccad..eace83d3d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rls.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rls.rs @@ -229,17 +229,17 @@ pub fn create_rls_policy( }; let stored = StoredRlsPolicy::from_runtime(&policy, database_id, predicate_raw) - .map_err(|e| DdlError::new("XX000", format!("rls serialize: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("rls serialize", &e))?; let entry = CatalogEntry::PutRlsPolicy(Box::new(stored.clone())); let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .put_rls_policy(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state.rls.install_replicated_policy(policy); } @@ -290,13 +290,13 @@ pub fn drop_rls_policy( name: name.to_string(), }; let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { { let catalog = state.credentials.catalog(); catalog .delete_rls_policy(tenant_id, &qualified_collection, name) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; } state .rls diff --git a/nodedb/src/control/server/shared/ddl/neutral/role.rs b/nodedb/src/control/server/shared/ddl/neutral/role.rs index c8bde6f4a..2e7a0daaf 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/role.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/role.rs @@ -68,12 +68,12 @@ pub fn create_role( let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); catalog .put_role(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.roles.install_replicated_role(&stored); } else if outcome.is_replicated() { super::role_checks::confirm_role(state, name, parent)?; @@ -131,13 +131,13 @@ pub fn drop_role( name: name.to_string(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; let dropped = if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); state .roles .drop_role(name, Some(catalog)) - .map_err(|e| DdlError::new("2BP01", e.to_string()))? + .map_err(|e| DdlError::from_error(&e))? } else if outcome.is_replicated() { // The synchronous post-apply removed the role from this node's // cache before the applied index advanced. A role still present @@ -287,12 +287,12 @@ pub fn set_role_parent( let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored.clone())); let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let catalog = state.credentials.catalog(); catalog .put_role(&stored) - .map_err(|e| DdlError::new("XX000", format!("catalog write: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; state.roles.install_replicated_role(&stored); } else if outcome.is_replicated() { super::role_checks::confirm_role(state, role_name, parent)?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs index 9b08071a7..ec80f408f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs @@ -364,10 +364,7 @@ fn crdt_apply_forbidden_in_transaction(txn_ctx: &DmlTxnCtx<'_>) -> bool { } fn crdt_transaction_error() -> DdlError { - DdlError::new( - "25001", - crate::Error::CrdtApplyForbiddenInTransaction.to_string(), - ) + DdlError::from_error(&crate::Error::CrdtApplyForbiddenInTransaction) } #[cfg(test)] @@ -395,6 +392,10 @@ mod tests { let error = crdt_transaction_error(); assert_eq!(error.sqlstate, "25001"); + assert_eq!( + error.code, + nodedb_types::error::ErrorCode::ACTIVE_SQL_TRANSACTION + ); assert_eq!( error.message, crate::Error::CrdtApplyForbiddenInTransaction.to_string() diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs index c507646f3..bd566e950 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs @@ -63,9 +63,8 @@ pub(super) async fn try_string( } if upper.starts_with("RESTORE ") && upper.contains("SET VERSION") { if restore_forbidden_in_transaction(txn_ctx) { - return Some(Err(DdlError::new( - "25001", - crate::Error::CrdtApplyForbiddenInTransaction.to_string(), + return Some(Err(DdlError::from_error( + &crate::Error::CrdtApplyForbiddenInTransaction, ))); } return Some( diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs index 9fab7bf78..6cc918345 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs @@ -65,7 +65,7 @@ pub(super) async fn try_typed( collection::dispatch_register_by_name(state, identity, name, database_id) .await .map(|()| resp) - .map_err(|e| DdlError::new("XX000", e.to_string())) + .map_err(|e| DdlError::from_error(&e)) } Err(e) => Err(e), }; @@ -106,7 +106,7 @@ pub(super) async fn try_typed( collection::dispatch_register_by_name(state, identity, name, database_id) .await .map(|()| resp) - .map_err(|e| DdlError::new("XX000", e.to_string())) + .map_err(|e| DdlError::from_error(&e)) } Err(e) => Err(e), }; diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs index b1d2fe469..72eb6ec46 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs @@ -179,10 +179,11 @@ pub(super) async fn try_typed( // USE DATABASE is intercepted in `execute_single_sql` before the DDL // router runs; reaching this arm means the intercept did not fire. - NodedbStatement::Database(DatabaseStmt::UseDatabase { name }) => Some(Err(DdlError::new( - "XX000", - format!("USE DATABASE {name}: reached router after expected intercept"), - ))), + NodedbStatement::Database(DatabaseStmt::UseDatabase { name }) => { + Some(Err(DdlError::internal(format!( + "USE DATABASE {name}: reached router after expected intercept" + )))) + } _ => None, } diff --git a/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs b/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs index 33be9087d..59be7aa7b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs @@ -78,7 +78,7 @@ pub fn create_schedule( let now = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| DdlError::new("XX000", "system clock error"))? + .map_err(|_| DdlError::internal("system clock error"))? .as_secs(); let target_collection = extract_target_collection(body_sql); diff --git a/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs index d0c66b8b1..6848c0c3c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs @@ -81,14 +81,14 @@ pub fn drop_schedule( name: name.clone(), }; let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::new("XX000", format!("metadata propose: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; if outcome.needs_local_apply() { let _ = catalog .delete_schedule_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; catalog .delete_owner("schedule", database_id.as_u64(), tenant_id, &name) - .map_err(|e| DdlError::new("XX000", format!("catalog owner delete: {e}")))?; + .map_err(|e| DdlError::from_error_in_context("catalog owner delete", &e))?; state .schedule_registry .unregister(database_id, tenant_id, &name); diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs index 7661e91fe..1897ea1a0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs @@ -99,7 +99,7 @@ pub fn drop_scope( let found = state .scope_defs .drop_scope(name) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; if !found { return Err(err("42704", format!("scope '{name}' not found"))); } diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs index a746f6795..d4fb7d62a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs @@ -23,7 +23,7 @@ use super::support::{err, status}; /// the expiry sweep and the DDL handlers cannot drift apart; this wrapper only /// translates the error into what pgwire reports. fn propose_scope_grant(state: &SharedState, stored: &StoredScopeGrant) -> Result<(), DdlError> { - propose_grant(state, stored).map_err(|e| err("XX000", e.to_string())) + propose_grant(state, stored).map_err(|e| DdlError::from_error(&e)) } /// Replicate a scope-grant removal. Same dual path as [`propose_scope_grant`]. @@ -34,7 +34,7 @@ fn propose_scope_revoke( grantee_id: &str, ) -> Result<(), DdlError> { propose_revoke(state, scope_name, grantee_type, grantee_id) - .map_err(|e| err("XX000", e.to_string())) + .map_err(|e| DdlError::from_error(&e)) } /// GRANT SCOPE '' TO '' @@ -92,7 +92,7 @@ pub fn grant_scope( on_expire_action: &on_expire_action, conditions, }) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; propose_scope_grant(state, &stored)?; state.audit_record( @@ -168,7 +168,7 @@ pub fn renew_scope( let outcome = state .scope_grants .prepare_renew(scope_name, &grantee_type, grantee_id, extend_secs) - .map_err(|e| err("XX000", e.to_string()))?; + .map_err(|e| DdlError::from_error(&e))?; match outcome { RenewOutcome::NotFound => return Err(err("42704", "scope grant not found")), // Nothing to move: a permanent grant has no deadline to extend. diff --git a/nodedb/src/control/server/shared/ddl/neutral/sequence.rs b/nodedb/src/control/server/shared/ddl/neutral/sequence.rs index 6622fec5d..fdde1d9a9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/sequence.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/sequence.rs @@ -12,6 +12,8 @@ use serde_json::{Map, Value as JsonValue}; use crate::control::security::catalog::sequence_types::StoredSequence; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::sequence::SequenceError; +use crate::control::sequence::error_map::sequence_error_to_error; use crate::control::server::response_shape::types::ShapedRows; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; use crate::control::state::SharedState; @@ -132,10 +134,11 @@ pub fn create_sequence( let entry = crate::control::catalog_entry::CatalogEntry::PutSequence(Box::new(def.clone())); let outcome = propose_and_apply(state, &entry)?; if outcome.needs_local_apply() { + let name = def.name.clone(); state .sequence_registry .create(def) - .map_err(|e| DdlError::new("XX000", e.to_string()))?; + .map_err(|e| create_refusal(&name, e))?; } state.schema_version.bump(); @@ -143,6 +146,24 @@ pub fn create_sequence( Ok(status("CREATE SEQUENCE")) } +/// The DDL error for a registry refusal of a new sequence. A name taken since +/// the existence check is a duplicate object. Every other refusal keeps the +/// SQLSTATE the sequence error map gives it. +fn create_refusal(name: &str, error: SequenceError) -> DdlError { + match error { + SequenceError::AlreadyExists { .. } => DdlError::new("42P07", error.to_string()), + other @ (SequenceError::Exhausted { .. } + | SequenceError::NotYetCalled { .. } + | SequenceError::OutOfRange { .. } + | SequenceError::NotFound { .. } + | SequenceError::InvalidDefinition { .. } + | SequenceError::FormatParse { .. } + | SequenceError::InvalidResetScope { .. }) => { + DdlError::from_error(&sequence_error_to_error(name, other)) + } + } +} + /// Handle `ALTER SEQUENCE RESTART [WITH ] | FORMAT '