diff --git a/Cargo.lock b/Cargo.lock index 319fda30..7f0a3755 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -309,7 +309,7 @@ version = "0.1.0" dependencies = [ "asap-frontend-promql", "asap-types", - "asap_sketchlib", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", "serde", "serde_json", "thiserror 2.0.18", @@ -368,12 +368,29 @@ dependencies = [ "asap-aware-mapping", "asap-frontend-promql", "asap-frontend-sql", + "asap-physical-operators", "asap-types", - "asap_sketchlib", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib)", + "futures", "serde_json", "tokio", ] +[[package]] +name = "asap-physical-operators" +version = "0.1.0" +dependencies = [ + "asap-aware-mapping", + "asap-frontend-promql", + "asap-types", + "asap_sketchlib 0.3.0 (git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369)", + "futures", + "serde", + "serde_json", + "thiserror 2.0.18", + "tracing", +] + [[package]] name = "asap-planner" version = "0.1.0" @@ -400,6 +417,24 @@ dependencies = [ "thiserror 2.0.18", ] +[[package]] +name = "asap_sketchlib" +version = "0.3.0" +source = "git+https://github.com/ProjectASAP/asap_sketchlib?rev=5f03ccbd798ed5fec62bdd839bcb331123cab369#5f03ccbd798ed5fec62bdd839bcb331123cab369" +dependencies = [ + "bincode", + "bytes", + "prost", + "rand 0.9.5", + "rmp-serde", + "serde", + "serde-big-array", + "serde_bytes", + "smallvec", + "twox-hash 2.1.2", + "xxhash-rust", +] + [[package]] name = "asap_sketchlib" version = "0.3.0" diff --git a/Cargo.toml b/Cargo.toml index a1daf49d..a2b019af 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,6 @@ [workspace] members = [ + "crates/asap-physical-operators", "crates/types", "crates/sql-function-catalog", "crates/asap-aware-mapping", diff --git a/crates/asap-physical-operators/Cargo.toml b/crates/asap-physical-operators/Cargo.toml new file mode 100644 index 00000000..74c172e3 --- /dev/null +++ b/crates/asap-physical-operators/Cargo.toml @@ -0,0 +1,18 @@ +[package] +name = "asap-physical-operators" +version = "0.1.0" +edition = "2021" + +[dependencies] +futures = "0.3" +planner-types = { package = "asap-types", path = "../types" } +asap_sketchlib = { git = "https://github.com/ProjectASAP/asap_sketchlib", rev = "5f03ccbd798ed5fec62bdd839bcb331123cab369" } +serde = { version = "1", features = ["derive", "rc"] } +serde_json = "1" +tracing = "0.1" +thiserror = "2" + + +[dev-dependencies] +asap-aware-mapping = { path = "../asap-aware-mapping" } +asap-frontend-promql = { path = "../frontend-promql" } diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md new file mode 100644 index 00000000..5f638465 --- /dev/null +++ b/crates/asap-physical-operators/README.md @@ -0,0 +1,121 @@ +# ASAP physical operators + +An independent Rust physical operator DAG runtime shared by ingestion time and +query time execution. The library requires neither backend engine, a server, +a storage implementation, Arrow nor DataFusion. DataFusion informed the design; +it is not the execution framework. + +`plan::PhysicalDag` binds typed operator inputs to node IDs. Each execution starts +one producer per reachable node, shares output batches among its consumers, and +bounds buffering. Dropping one consumer does not cancel other consumers. A +`RunContext` carries query or ingestion scope, cancellation and byte accounting. +Executions use the caller's worker and worker-local streams, with no internal +thread pool. Poll multiple root streams concurrently when they share inputs. + +`operators::Operator` implements native batch sources, scalar values, +projection, filtering, grouped exact aggregation, semi-join, grouped Sort and +Limit, vector-to-scalar conversion, Union, and summary construction/merge/readout. +Sort followed by Limit implements grouped ranking; no dedicated TopK physical +operator is needed. Summary construction updates state batch by batch. End of +input means the supplied query range or ingestion window is complete. + +```rust +use asap_physical_operators::{ + expressions::Expression, + operators::Operator, + values::Value, + plan::PhysicalDag, + runtime::{Limits, RunContext, Scope}, +}; +use asap_physical_operators::planner::pre_asap::DataType; +use futures::{executor::block_on, StreamExt}; + +let source = Operator::scalar(Value::Int64(7), DataType::Int64)?; +let negate = Operator::project(source.schema(), vec![ + ("value".into(), Expression::Negate(Box::new(Expression::Column(0)))), +])?; +let mut plan = PhysicalDag::default(); +plan.add(0, vec![], source)?; +plan.add(1, vec![0], negate)?; +let run = RunContext::new( + Scope::Query { evaluation_time_ms: 1000, revision: 1 }, + Limits::default(), +)?; +let mut output = plan.execute(&[1], run)?.remove(0); +let batch = block_on(output.next()).unwrap()?; +assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); +# Ok::<(), asap_physical_operators::dag::Error>(()) +``` + +`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDag`) and typed input contracts. +The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and +schema mismatches before starting a source. Implement `PhysicalOperator` for a +deployment source, including asynchronous I/O; computation operators remain in +the library. The public `planner` export identifies the exact Planner types used +by the crate. The physical compiler currently supports a subset of those types and +operations; it does not interpret an unknown node as external fallback. + +Plain values preserve Planner scalar/collection types and nullability. Numeric +arithmetic uses matching Int64 or Float64 inputs; integer overflow is an error. +Boolean predicates use three-valued logic. Native summary states currently cover +exact Sum/Count/Min/Max/Rate/Increase, KLL, DDSketch, HLL and Float64 weighted CMS and CountSketch with candidate heaps. Binding checks family, +parameters and readout compatibility; source batches also validate state payloads. +Existing accumulator algorithms are reused as kernels behind these operators. + +This crate is owned by ASAPPlanner. Its `planner-types` dependency is the local +IR crate, so a contract change and its execution tests belong in the same PR. +Deployments supply storage/ingestion sources and adapt output protocols. The +library has no ASAPQuery-backend dependency. Backend raw Scan remains a separate +deployment capability. + +See [the design](../../docs/design_docs/physical-planning-and-deployment.md). + +## Module boundaries + +- `plan`: immutable graph, operator interface, schemas and execution properties. +- `runtime`: per-run streams, shared producers, memory reservations and cancellation. +- `expressions`: scalar evaluation; typed builders and the Planner expression adapter. +- `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. +- `sources`: raw-source interface, Scan and the memory connector. +- `physical_planner`: native operator lowering, typed input contracts and checked instantiation. +- `summary_kernels`: in-memory summary state over `asap_sketchlib` and exact Planner state: merge, typed readout and update adapters. +- `readout`: readouts over merged exact summary states. +- `capability`: explicit kernel, native-batch and typed readout validation. + +The `dag`, `factory`, `traits` and `arithmetic` paths are re-exports. They contain no alternative +execution implementations. + +Sketch algorithms and their state encodings belong to `asap_sketchlib`. Wire decoding, delta +frames, edge sampling and storage statistics belong to deployments. Kernels hold one population's +state; operators own grouping. + +A source must declare `Boundedness::Bounded` to feed a blocking operator. +The default for a custom raw source is `Unknown`; query or ingestion scope alone +does not promise that its cursor ends. `PhysicalDag::properties` validates these +requirements before any source starts and returns boundedness and emission mode +for every reachable node. The memory connector declares finite input. Custom +physical sources expose the same facts through `PhysicalOperator::properties`. + +Blocking operators reserve estimated workspace and yield cooperatively during +row processing and sort merges. Cancellation releases reservations when the +stream is polled or dropped. Individual scalar evaluations, bounded sort chunks +and sketch kernel calls are synchronous; this is not preemptive execution. +There is no spill or partitioned parallel execution in this implementation. + +## Physical compilation and deployment inputs + +`physical_planner::compile` accepts a Planner `PostAsapDag`, typed +`InputContract`s and output roots. It returns a reusable `CompiledPhysicalDag` +containing selected native operators and no live readers. Compilation validates +schemas, input ordering, sharing and boundedness before deployment source access. + +A deployment calls `CompiledPhysicalDag::instantiate` with exactly the declared +inputs. This checks source schemas and execution properties and constructs the +runnable graph without repeating logical lowering. The graph executes through +the shared runtime with independent per-run state. Window coverage, revision and +maintenance-policy admission remain deployment/planning contracts; this compiler +does not discover storage or silently change a selected maintenance strategy. + +Pane construction and geometry belong to deployment. A deployment runs the +precompute DAG once per pane it constructs and binds the selected pane states to +query input slots; the query DAG merges and reads them out as computation. diff --git a/crates/asap-physical-operators/src/capability.rs b/crates/asap-physical-operators/src/capability.rs new file mode 100644 index 00000000..2a1cc095 --- /dev/null +++ b/crates/asap-physical-operators/src/capability.rs @@ -0,0 +1,241 @@ +//! Capability boundaries, checked without constructing accumulator state. +//! +//! `validate_summary_kernel` checks update kernels, including families without a +//! native batch representation. `validate_native_family` and +//! `validate_sketch_readout` / `validate_exact_readout` check native state and readout support. +//! Keyed weighted-frequency readouts are checked by `Operator::keyed_readout`. +//! A successful kernel check alone does not mean a physical DAG will bind. +//! +//! Stored-state encodings belong to deployments. Full plan acceptance is +//! owned by `binding`, which also validates schemas, expressions and inputs. +use crate::Error; +use planner_types::post_asap::{ + ExactKind, ExactParams, GroupingStrategy, SketchAlgorithm, SketchParams, SketchQuery, + SummaryFamilyType, SummaryUpdate, +}; + +/// Check the same contract used by `create_planner_accumulator` before a plan +/// is accepted. Execution timing is deliberately not a kernel property. +pub fn validate_summary_kernel( + family: &SummaryFamilyType, + input: &SummaryUpdate, + grouping: &GroupingStrategy, +) -> Result<(), String> { + if grouping != &GroupingStrategy::PerSubpopulationInstance { + return Err("shared summary grouping has no registered kernel".into()); + } + let keyed = match family { + SummaryFamilyType::ExactAggregate(kind, params) => { + use ExactKind as K; + use ExactParams as P; + if !matches!( + (kind, params), + (K::Sum, P::Sum) + | (K::Count, P::Count) + | (K::Min, P::Min) + | (K::Max, P::Max) + | (K::Rate, P::Rate) + | (K::Increase, P::Increase) + ) { + return Err(format!("unsupported exact kernel {family:?}")); + } + input.item.is_some() + } + SummaryFamilyType::Sketch(kind, layout) => { + if layout != grouping { + return Err("Planner family and operator grouping disagree".into()); + } + use SketchAlgorithm as A; + use SketchParams as P; + match (kind.algorithm(), kind.params()) { + (A::Kll, P::Kll { k }) if (8..=u16::MAX as u32).contains(k) => false, + (A::DDSketch, P::DDSketch { alpha }) + if alpha.is_finite() && *alpha > 0.0 && *alpha < 1.0 => + { + false + } + (A::Hll, P::Hll { precision }) if (4..=18).contains(precision) => false, + (A::Cms, P::Cms { width, depth }) + | (A::CountSketch, P::CountSketch { width, depth }) + if valid_matrix(*width, *depth) => + { + true + } + ( + A::CmsWithHeap, + P::CmsWithHeap { + width, + depth, + heap_size, + }, + ) + | ( + A::CountSketchWithHeap, + P::CountSketchWithHeap { + width, + depth, + heap_size, + }, + ) if valid_matrix(*width, *depth) && *heap_size > 0 => true, + ( + A::UnivMon, + P::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + ) if *heap_size > 0 + && *sketch_cols > 0 + && (1..=20).contains(sketch_rows) + && (1..=64).contains(layers) + && (*sketch_rows as usize) + .checked_mul(*sketch_cols as usize) + .and_then(|n| n.checked_mul(*layers as usize)) + .is_some() => + { + false + } + _ => { + return Err(format!( + "unsupported kernel or invalid parameters: {kind:?}" + )) + } + } + } + _ => return Err(format!("unsupported summary kernel {family:?}")), + }; + if keyed != input.item.is_some() && !is_unit_sample_frequency(input) { + return Err("Planner item expression does not match kernel layout".into()); + } + Ok(()) +} + +fn valid_matrix(width: u32, depth: u32) -> bool { + // Construction uses the kernel's native row hashing, so no encoded-size + // limit applies here. + width > 0 + && depth > 0 + && (width as usize) + .checked_mul(depth as usize) + .and_then(|n| n.checked_mul(std::mem::size_of::())) + .is_some() +} + +pub(crate) fn is_unit_sample_frequency(update: &planner_types::post_asap::SummaryUpdate) -> bool { + use planner_types::post_asap::{NonNegativeWeightProof, SummaryInputExpr, WeightDomain}; + matches!( + update.item, + Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::SampleValue + )) + ) && matches!(update.weight, SummaryInputExpr::Constant(1.0)) + && matches!( + update.weight_domain, + WeightDomain::NonNegative { + proof: NonNegativeWeightProof::UnitCount + } + ) +} + +pub fn validate_native_family(family: &SummaryFamilyType) -> Result<(), Error> { + use planner_types::post_asap::SketchAlgorithm as A; + if let SummaryFamilyType::Sketch(kind, grouping) = family { + if matches!(kind.algorithm(), A::CmsWithHeap | A::CountSketchWithHeap) { + let (_, width, depth, _) = + crate::summary_kernels::weighted_frequency::WeightedFrequency::configuration(kind)?; + return if valid_matrix(width as u32, depth as u32) && grouping == &Default::default() { + Ok(()) + } else { + Err(Error::Invalid( + "invalid weighted frequency dimensions or grouping strategy".into(), + )) + }; + } + } + match family { + SummaryFamilyType::ExactAggregate(..) => {} + SummaryFamilyType::Sketch(kind, _) + if matches!(kind.algorithm(), A::Kll | A::DDSketch | A::Hll) => {} + _ => { + return Err(Error::Invalid( + "summary family has no native DAG state implementation".into(), + )) + } + } + crate::capability::validate_summary_kernel( + family, + &planner_types::post_asap::SummaryUpdate::column( + planner_types::pre_asap::ColumnRef::SampleValue, + ), + &Default::default(), + ) + .map_err(Error::Invalid) +} + +/// A sketch readout is native only for the families Planner can read directly. +pub fn validate_sketch_readout( + family: &SummaryFamilyType, + query: &SketchQuery, +) -> Result<(), Error> { + validate_native_family(family)?; + use planner_types::post_asap::SketchAlgorithm as A; + // A point count without an item value reads the total count. + let bare_count = matches!(query, SketchQuery::PointCount { value: None, .. }); + let supported = match family { + SummaryFamilyType::Sketch(kind, _) => match (kind.algorithm(), query) { + (A::Kll, SketchQuery::Quantile { q }) | (A::DDSketch, SketchQuery::Quantile { q }) => { + if !(0.0..=1.0).contains(q) { + return Err(Error::Invalid( + "quantile readout requires quantile in [0,1]".into(), + )); + } + true + } + (A::DDSketch, _) => bare_count, + (A::Hll, SketchQuery::Cardinality) => true, + (A::Hll, _) => bare_count, + _ => false, + }, + _ => false, + }; + if !supported { + return Err(Error::Invalid( + "readout is not implemented for this summary family".into(), + )); + } + Ok(()) +} + +/// An exact readout must match the exact family it reads. +pub fn validate_exact_readout( + family: &SummaryFamilyType, + readout: &crate::summary_kernels::exact::ExactReadout, +) -> Result<(), Error> { + validate_native_family(family)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let supported = matches!( + (family, readout.statistic), + (SummaryFamilyType::ExactAggregate(E::Sum, _), S::Sum) + | (SummaryFamilyType::ExactAggregate(E::Count, _), S::Count) + | (SummaryFamilyType::ExactAggregate(E::Min, _), S::Min) + | (SummaryFamilyType::ExactAggregate(E::Max, _), S::Max) + | (SummaryFamilyType::ExactAggregate(E::Rate, _), S::Rate) + | ( + SummaryFamilyType::ExactAggregate(E::Increase, _), + S::Increase + ) + ); + if !supported { + return Err(Error::Invalid( + "readout is not implemented for this summary family".into(), + )); + } + if readout.lookback_ms.is_some_and(|lookback| { + lookback <= 0 || !matches!(readout.statistic, S::Rate | S::Increase) + }) { + return Err(Error::Invalid("invalid exact counter lookback".into())); + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs new file mode 100644 index 00000000..c7837672 --- /dev/null +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -0,0 +1,8 @@ +//! Compatibility imports. New code should use plan, runtime, operators, physical_planner and sources directly. +pub use crate::plan::{NodeId, PhysicalDag, PhysicalOperator}; +pub use crate::runtime::batch_execution; +pub use crate::runtime::{ + Input, Limits, OutputStream, Reservation, RunContext, Scope, SharedValue, +}; +pub use crate::Error; +pub use crate::{expressions, operators, physical_planner as planner, sources as scan, values}; diff --git a/crates/asap-physical-operators/src/error.rs b/crates/asap-physical-operators/src/error.rs new file mode 100644 index 00000000..afee33d8 --- /dev/null +++ b/crates/asap-physical-operators/src/error.rs @@ -0,0 +1,18 @@ +use crate::plan::NodeId; +#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] +pub enum Error { + #[error("invalid DAG: {0}")] + Invalid(String), + #[error("operator failed: {0}")] + Operator(String), + #[error("node {node} ({operation}) failed: {source}")] + AtNode { + node: NodeId, + operation: String, + source: Box, + }, + #[error("execution memory limit exceeded")] + MemoryLimit, + #[error("execution cancelled")] + Cancelled, +} diff --git a/crates/asap-physical-operators/src/expressions/arithmetic.rs b/crates/asap-physical-operators/src/expressions/arithmetic.rs new file mode 100644 index 00000000..30e277d4 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/arithmetic.rs @@ -0,0 +1,63 @@ +//! Float64 arithmetic shared by ASAP execution engines. +//! Preserve IEEE non-finite results; callers own their output policies. + +pub fn evaluate_float64_arithmetic( + operator: &planner_types::pre_asap::ArithmeticOpKind, + left: f64, + right: f64, +) -> f64 { + use planner_types::pre_asap::ArithmeticOpKind::*; + match operator { + Add => left + right, + Sub => left - right, + Mul => left * right, + Div => left / right, + Mod => left % right, + Pow => left.powf(right), + Atan2 => left.atan2(right), + } +} + +/// Execute the Planner binary contract after a deployment has resolved matching rows. +pub fn evaluate_binary( + operator: &planner_types::post_asap::BinaryOperator, + left: f64, + right: f64, +) -> Result { + use crate::{values::Value, Error}; + use planner_types::pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}; + let invalid = + || Error::Invalid("unsupported binary operation or invalid checked-division domain".into()); + if operator.vector_match.is_some() { + return Err(invalid()); + } + if operator.checked_relative_division || operator.checked_finite_division { + if operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + || !left.is_finite() + || !right.is_finite() + || right == 0. + { + return Err(invalid()); + } + let value = left / right; + if !value.is_finite() || (operator.checked_relative_division && !value.is_normal()) { + return Err(invalid()); + } + return Ok(Value::Float64(value)); + } + Ok(match operator.kind { + BinaryOpKind::Arithmetic(ref op) => { + Value::Float64(evaluate_float64_arithmetic(op, left, right)) + } + BinaryOpKind::Compare(ref op) => Value::Bool(match op { + CompareOpKind::Eq => left == right, + CompareOpKind::Ne => left != right, + CompareOpKind::Lt => left < right, + CompareOpKind::Le => left <= right, + CompareOpKind::Gt => left > right, + CompareOpKind::Ge => left >= right, + _ => return Err(invalid()), + }), + _ => return Err(invalid()), + }) +} diff --git a/crates/asap-physical-operators/src/expressions/mod.rs b/crates/asap-physical-operators/src/expressions/mod.rs new file mode 100644 index 00000000..2916cca7 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/mod.rs @@ -0,0 +1,348 @@ +//! Scalar semantics and typed expression binding. Planner expressions enter through CompiledExpression. +use crate::{ + values::{plain, Schema, Value}, + Error, +}; +use planner_types::pre_asap::{ArithmeticOpKind, DataType}; +pub mod arithmetic; +mod planner; +pub use planner::CompiledExpression; +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub enum Expression { + Binary { + operator: planner_types::post_asap::BinaryOperator, + left: Box, + right: Box, + }, + Planner(Box), + Column(usize), + ExactFloat64(usize), + FiniteFloat64(Box), + LabelSet { + column: usize, + labels: Vec, + without: bool, + }, + Literal { + value: Value, + dtype: DataType, + }, + Negate(Box), + Arithmetic { + op: ArithmeticOpKind, + left: Box, + right: Box, + }, + Equal(Box, Box), + Less(Box, Box), + And(Box, Box), + Or(Box, Box), + Not(Box), + IsNull(Box), +} +impl Expression { + pub fn planner(expression: crate::expressions::CompiledExpression) -> Self { + Self::Planner(Box::new(expression)) + } + pub(crate) fn dtype(&self, input: &Schema) -> Result<(DataType, bool), Error> { + use Expression::*; + match self { + Binary { + operator, + left, + right, + } => { + use planner_types::pre_asap::{BinaryOpKind, CompareOpKind}; + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a != DataType::Float64 || b != a || operator.vector_match.is_some() { + return Err(invalid( + "binary expression requires resolved Float64 operands", + )); + } + if (operator.checked_relative_division || operator.checked_finite_division) + && operator.kind != BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + { + return Err(invalid("checked division contract on non-division")); + } + let dtype = match operator.kind { + BinaryOpKind::Arithmetic(_) => DataType::Float64, + BinaryOpKind::Compare( + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge, + ) => DataType::Bool, + _ => return Err(invalid("unsupported binary operation")), + }; + Ok((dtype, n || m)) + } + Planner(expression) => { + expression.validate_input(input)?; + Ok(expression.dtype()) + } + FiniteFloat64(expression) => { + if expression.dtype(input)? != (DataType::Float64, false) { + return Err(invalid("finite update requires non-null Float64")); + } + Ok((DataType::Float64, false)) + } + ExactFloat64(column) => { + let (dtype, nullable) = plain(input, *column)?; + if nullable || !matches!(dtype, DataType::Int64 | DataType::Float64) { + return Err(invalid( + "exact Float64 conversion requires non-null numeric input", + )); + } + Ok((DataType::Float64, false)) + } + LabelSet { column, labels, .. } => { + let (dtype, nullable) = plain(input, *column)?; + let expected = DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }; + if dtype != &expected + || nullable + || labels + .iter() + .collect::>() + .len() + != labels.len() + { + return Err(invalid( + "label projection requires a non-null Utf8 map and unique label names", + )); + } + Ok((expected, false)) + } + Column(i) => { + let (t, n) = plain(input, *i)?; + Ok((t.clone(), n)) + } + Literal { value, dtype } => { + if value.matches(dtype, true) { + Ok((dtype.clone(), matches!(value, Value::Null))) + } else { + Err(invalid("literal type mismatch")) + } + } + Negate(v) => { + let (t, n) = v.dtype(input)?; + if matches!(t, DataType::Int64 | DataType::Float64) { + Ok((t, n)) + } else { + Err(invalid("numeric negation required")) + } + } + Arithmetic { op, left, right } => { + let (a, n) = left.dtype(input)?; + let (b, m) = right.dtype(input)?; + if a == b + && matches!(a, DataType::Int64 | DataType::Float64) + && !(a == DataType::Int64 && *op == ArithmeticOpKind::Atan2) + { + Ok((a, n || m)) + } else { + Err(invalid("arithmetic requires matching numeric types")) + } + } + Equal(a, b) | Less(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == b && ordered(&a) { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("comparison requires matching ordered types")) + } + } + And(a, b) | Or(a, b) => { + let (a, n) = a.dtype(input)?; + let (b, m) = b.dtype(input)?; + if a == DataType::Bool && b == DataType::Bool { + Ok((DataType::Bool, n || m)) + } else { + Err(invalid("boolean operands required")) + } + } + Not(v) => { + let (t, n) = v.dtype(input)?; + if t == DataType::Bool { + Ok((t, n)) + } else { + Err(invalid("boolean operand required")) + } + } + IsNull(v) => { + v.dtype(input)?; + Ok((DataType::Bool, false)) + } + } + } + pub(crate) fn evaluate(&self, row: &[Value]) -> Result { + use Expression::*; + Ok(match self { + FiniteFloat64(expression) => match expression.evaluate(row)? { + Value::Float64(value) if value.is_finite() => Value::Float64(value), + _ => return Err(invalid("summary update must be finite")), + }, + ExactFloat64(column) => match row[*column] { + Value::Float64(value) => Value::Float64(value), + Value::Int64(value) if value.unsigned_abs() <= (1u64 << 53) => { + Value::Float64(value as f64) + } + _ => { + return Err(invalid( + "numeric result cannot be represented exactly as Float64", + )) + } + }, + LabelSet { + column, + labels, + without, + } => { + let Value::Map(entries) = &row[*column] else { + return Err(invalid("label projection requires a map")); + }; + let mut selected = std::collections::BTreeMap::new(); + let mut seen = std::collections::BTreeSet::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("label projection requires Utf8 entries")); + }; + if !seen.insert(key.clone()) { + return Err(invalid("duplicate label name")); + } + let keep = if *without { + key.as_ref() != "__name__" + && !labels.iter().any(|label| label.as_str() == key.as_ref()) + } else { + labels.iter().any(|label| label.as_str() == key.as_ref()) + }; + if keep && !value.is_empty() { + selected.insert(key.clone(), value.clone()); + } + } + Value::Map( + selected + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + ) + } + Binary { + operator, + left, + right, + } => { + let (a, b) = (left.evaluate(row)?, right.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else { + let (Value::Float64(a), Value::Float64(b)) = (a, b) else { + return Err(invalid("binary value schema mismatch")); + }; + arithmetic::evaluate_binary(operator, a, b)? + } + } + Planner(expression) => expression.evaluate(row)?, + Column(i) => row[*i].clone(), + Literal { value, .. } => value.clone(), + Negate(v) => match v.evaluate(row)? { + Value::Int64(v) => Value::Int64( + v.checked_neg() + .ok_or_else(|| invalid("integer negation overflow"))?, + ), + Value::Float64(v) => Value::Float64(-v), + Value::Null => Value::Null, + _ => return Err(invalid("numeric negation required")), + }, + Arithmetic { op, left, right } => { + numeric(op, left.evaluate(row)?, right.evaluate(row)?)? + } + Equal(a, b) | Less(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + if matches!(a, Value::Null) || matches!(b, Value::Null) { + Value::Null + } else if matches!((&a,&b),(Value::Float64(a),Value::Float64(b)) if a.is_nan() || b.is_nan()) + { + Value::Bool(false) + } else { + let c = a.compare(&b)?; + Value::Bool(if matches!(self, Equal(..)) { + c.is_eq() + } else { + c.is_lt() + }) + } + } + And(a, b) | Or(a, b) => { + let (a, b) = (a.evaluate(row)?, b.evaluate(row)?); + match (a, b, matches!(self, And(..))) { + (Value::Bool(false), _, true) | (_, Value::Bool(false), true) => { + Value::Bool(false) + } + (Value::Bool(true), _, false) | (_, Value::Bool(true), false) => { + Value::Bool(true) + } + (Value::Null, _, _) | (_, Value::Null, _) => Value::Null, + (Value::Bool(a), Value::Bool(b), true) => Value::Bool(a && b), + (Value::Bool(a), Value::Bool(b), false) => Value::Bool(a || b), + _ => return Err(invalid("boolean operands required")), + } + } + Not(v) => match v.evaluate(row)? { + Value::Bool(v) => Value::Bool(!v), + Value::Null => Value::Null, + _ => return Err(invalid("boolean operand required")), + }, + IsNull(v) => Value::Bool(matches!(v.evaluate(row)?, Value::Null)), + }) + } +} +pub(crate) fn ordered(dtype: &DataType) -> bool { + if let DataType::Map { key, value, .. } = dtype { + return ordered(key) && ordered(value); + } + matches!( + dtype, + DataType::Null + | DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp + | DataType::Date + ) +} +pub(crate) fn numeric(op: &ArithmeticOpKind, a: Value, b: Value) -> Result { + use ArithmeticOpKind::*; + Ok(match (a, b) { + (Value::Null, _) | (_, Value::Null) => Value::Null, + (Value::Float64(a), Value::Float64(b)) => { + Value::Float64(arithmetic::evaluate_float64_arithmetic(op, a, b)) + } + (Value::Int64(a), Value::Int64(b)) => Value::Int64( + match op { + Add => a.checked_add(b), + Sub => a.checked_sub(b), + Mul => a.checked_mul(b), + Div => a.checked_div(b), + Mod => a.checked_rem(b), + Pow => u32::try_from(b).ok().and_then(|b| a.checked_pow(b)), + Atan2 => None, + } + .ok_or_else(|| invalid("invalid integer arithmetic or overflow"))?, + ), + _ => return Err(invalid("arithmetic type mismatch")), + }) +} + +fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} diff --git a/crates/asap-physical-operators/src/expressions/planner.rs b/crates/asap-physical-operators/src/expressions/planner.rs new file mode 100644 index 00000000..2130a2f7 --- /dev/null +++ b/crates/asap-physical-operators/src/expressions/planner.rs @@ -0,0 +1,549 @@ +//! Planner scalar expressions evaluated over native typed rows. +use crate::{ + values::{Schema, Value}, + Error, +}; +use planner_types::pre_asap::{ArithmeticOpKind, CompareOpKind, DataType, QueryExpr, ScalarValue}; +use std::{cmp::Ordering, sync::Arc}; + +pub(super) fn evaluate( + expr: &QueryExpr, + row: &[Value], + schema: &planner_types::pre_asap::Schema, +) -> Result { + match expr { + QueryExpr::Column(index) => row.get(*index).cloned().ok_or(Error::Invalid(format!( + "column {index} outside row width {}", + row.len() + ))), + QueryExpr::Literal(value) => Ok(match value { + ScalarValue::Interval { + months, + days, + nanos, + } => Value::Interval { + months: *months, + days: *days, + nanos: *nanos, + }, + ScalarValue::Int64(value) => Value::Int64(*value), + ScalarValue::Float64(value) => Value::Float64(*value), + ScalarValue::Utf8(value) => Value::Utf8(value.clone().into()), + ScalarValue::Boolean(value) => Value::Bool(*value), + ScalarValue::Null => Value::Null, + }), + QueryExpr::Compare { left, op, right } => { + let left = evaluate(left, row, schema)?; + let right = evaluate(right, row, schema)?; + compare(op, left, right) + } + QueryExpr::Arithmetic { op, left, right } => arithmetic( + op, + evaluate(left, row, schema)?, + evaluate(right, row, schema)?, + ), + QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + let and = matches!(expr, QueryExpr::BoolAnd(_)); + let mut null = false; + for part in parts { + match evaluate(part, row, schema)? { + Value::Bool(value) if value != and => return Ok(Value::Bool(value)), + Value::Bool(_) => {} + Value::Null => null = true, + _ => return Err(Error::Invalid("boolean predicate required".into())), + } + } + Ok(if null { Value::Null } else { Value::Bool(and) }) + } + QueryExpr::Not(value) => match evaluate(value, row, schema)? { + Value::Bool(value) => Ok(Value::Bool(!value)), + Value::Null => Ok(Value::Null), + _ => Err(Error::Invalid("boolean predicate required".into())), + }, + QueryExpr::IsNull(value) => Ok(Value::Bool(matches!( + evaluate(value, row, schema)?, + Value::Null + ))), + QueryExpr::IsNotNull(value) => Ok(Value::Bool(!matches!( + evaluate(value, row, schema)?, + Value::Null + ))), + QueryExpr::FunctionCall { name, args } => { + use planner_types::pre_asap::scalar_signature::MapScalarFunction; + if name.eq_ignore_ascii_case("asap_struct_field") { + expr.scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + let DataType::Struct { fields } = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + .0 + else { + unreachable!() + }; + let offset = match &args[1] { + QueryExpr::Literal(ScalarValue::Int64(index)) => { + usize::try_from(index - 1).ok() + } + QueryExpr::Literal(ScalarValue::Utf8(name)) => { + fields.iter().position(|field| &field.name == name) + } + _ => None, + } + .ok_or_else(|| Error::Invalid("struct field selector".into()))?; + let Value::Struct(values) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("struct field input".into())); + }; + return values + .get(offset) + .cloned() + .ok_or_else(|| Error::Invalid("struct field value".into())); + } + if name.eq_ignore_ascii_case("asap_element_access") { + let (output_type, _) = expr + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + if let DataType::List { element } = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + .0 + { + let Value::List(values) = evaluate(&args[0], row, schema)? else { + return Err(Error::Invalid("array access input".into())); + }; + let index = match evaluate(&args[1], row, schema)? { + Value::Null => return Ok(Value::Null), + Value::Int64(index) => index, + _ => return Err(Error::Invalid("array access index".into())), + }; + let offset = if index > 0 { + usize::try_from(index - 1).ok() + } else if index < 0 { + usize::try_from(index.unsigned_abs()) + .ok() + .and_then(|distance| values.len().checked_sub(distance)) + } else { + None + }; + return match offset.and_then(|offset| values.get(offset)) { + Some(value) => Ok(value.clone()), + None => default_collection_element(&output_type, element.nullable), + }; + } + } + let function = (if name.eq_ignore_ascii_case("asap_element_access") { + Some(MapScalarFunction::Access) + } else { + MapScalarFunction::from_name(name) + }) + .ok_or_else(|| Error::Invalid(format!("scalar function {name}")))?; + expr.scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))?; + let values = args + .iter() + .map(|arg| evaluate(arg, row, schema)) + .collect::, _>>()?; + match function { + MapScalarFunction::Construct => { + let mut values = values.into_iter(); + let mut entries = Vec::new(); + while let Some(key) = values.next() { + if !matches!(key, Value::Int64(_) | Value::Utf8(_) | Value::Bool(_)) { + return Err(Error::Invalid("map key value type".into())); + } + entries.push(( + key, + values + .next() + .ok_or_else(|| Error::Invalid("odd map argument count".into()))?, + )); + } + Ok(Value::Map(entries.into())) + } + MapScalarFunction::Concat => { + let mut entries = Vec::new(); + for value in values { + let Value::Map(next) = value else { + return Err(Error::Invalid("map concat argument".into())); + }; + entries.extend(next.iter().cloned()); + } + Ok(Value::Map(entries.into())) + } + MapScalarFunction::Access => { + let [Value::Map(entries), key] = values.as_slice() else { + return Err(Error::Invalid("map access arguments".into())); + }; + if matches!(key, Value::Null) { + return Ok(Value::Null); + } + if !matches!(key, Value::Int64(_) | Value::Utf8(_) | Value::Bool(_)) { + return Err(Error::Invalid("map lookup key type".into())); + } + if let Some((_, value)) = entries + .iter() + .find(|(candidate, _)| cell_cmp(candidate, key) == Some(Ordering::Equal)) + { + return Ok(value.clone()); + } + let ( + DataType::Map { + value, + value_nullable, + .. + }, + _, + ) = args[0] + .scalar_type(schema) + .map_err(|error| Error::Invalid(error.to_string()))? + else { + unreachable!() + }; + default_collection_element(&value, value_nullable) + } + } + } + other => Err(Error::Invalid(format!("scalar expression {other:?}"))), + } +} + +fn default_collection_element(dtype: &DataType, nullable: bool) -> Result { + if nullable { + return Ok(Value::Null); + } + Ok(match dtype { + DataType::Interval | DataType::Date => { + return Err(Error::Invalid("temporal value transport".into())) + } + DataType::Null => Value::Null, + DataType::Int64 => Value::Int64(0), + DataType::Float64 => Value::Float64(0.0), + DataType::Utf8 => Value::Utf8("".into()), + DataType::Bool => Value::Bool(false), + DataType::Map { .. } => Value::Map(Arc::from([])), + DataType::List { .. } => Value::List(Arc::from([])), + DataType::Struct { fields } => Value::Struct( + fields + .iter() + .map(|field| default_collection_element(&field.dtype, field.nullable)) + .collect::, _>>()? + .into(), + ), + _ => { + return Err(Error::Invalid( + "collection missing-element default type".into(), + )) + } + }) +} + +fn compare(op: &CompareOpKind, left: Value, right: Value) -> Result { + if matches!(left, Value::Null) || matches!(right, Value::Null) { + return Ok(Value::Null); + } + // NaN is unordered, not a type mismatch. Match the native scalar path. + if matches!(&left, Value::Float64(v) if v.is_nan()) + || matches!(&right, Value::Float64(v) if v.is_nan()) + { + return match op { + CompareOpKind::Ne => Ok(Value::Bool(true)), + CompareOpKind::Eq + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge => Ok(Value::Bool(false)), + _ => Err(Error::Invalid(format!("comparison {op:?}"))), + }; + } + let ordering = cell_cmp(&left, &right) + .ok_or_else(|| Error::Invalid("comparison of incompatible values".into()))?; + let value = match op { + CompareOpKind::Eq => ordering == Ordering::Equal, + CompareOpKind::Ne => ordering != Ordering::Equal, + CompareOpKind::Lt => ordering == Ordering::Less, + CompareOpKind::Le => ordering != Ordering::Greater, + CompareOpKind::Gt => ordering == Ordering::Greater, + CompareOpKind::Ge => ordering != Ordering::Less, + _ => return Err(Error::Invalid(format!("comparison {op:?}"))), + }; + Ok(Value::Bool(value)) +} + +fn arithmetic(op: &ArithmeticOpKind, left: Value, right: Value) -> Result { + let (left, right) = match (left, right) { + (Value::Int64(a), Value::Float64(b)) => (Value::Float64(a as f64), Value::Float64(b)), + (Value::Float64(a), Value::Int64(b)) => (Value::Float64(a), Value::Float64(b as f64)), + pair => pair, + }; + super::numeric(op, left, right) +} + +fn integer_float_cmp(integer: i64, float: f64) -> Option { + if float.is_nan() { + return None; + } + // These bounds are powers of two, exactly representable as Float64. + if float >= 9_223_372_036_854_775_808.0 { + return Some(Ordering::Less); + } + if float < -9_223_372_036_854_775_808.0 { + return Some(Ordering::Greater); + } + let integral = float as i64; + match integer.cmp(&integral) { + Ordering::Equal => 0.0_f64.partial_cmp(&float.fract()), + other => Some(other), + } +} + +fn cell_cmp(left: &Value, right: &Value) -> Option { + match (left, right) { + (Value::Int64(left), Value::Int64(right)) => Some(left.cmp(right)), + (Value::Float64(left), Value::Float64(right)) => left.partial_cmp(right), + (Value::Int64(left), Value::Float64(right)) => integer_float_cmp(*left, *right), + (Value::Float64(left), Value::Int64(right)) => { + integer_float_cmp(*right, *left).map(Ordering::reverse) + } + (Value::Utf8(left), Value::Utf8(right)) => Some(left.cmp(right)), + (Value::Bool(left), Value::Bool(right)) => Some(left.cmp(right)), + (Value::Timestamp(left), Value::Timestamp(right)) => Some(left.cmp(right)), + (Value::Map(left), Value::Map(right)) => { + for ((left_key, left_value), (right_key, right_value)) in left.iter().zip(right.iter()) + { + let order = cell_cmp(left_key, right_key)?; + if order != Ordering::Equal { + return Some(order); + } + let order = match (left_value, right_value) { + (Value::Null, Value::Null) => Ordering::Equal, + (Value::Null, _) => Ordering::Greater, + (_, Value::Null) => Ordering::Less, + _ => cell_cmp(left_value, right_value)?, + }; + if order != Ordering::Equal { + return Some(order); + } + } + Some(left.len().cmp(&right.len())) + } + _ => None, + } +} + +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub struct CompiledExpression { + expression: QueryExpr, + schema: planner_types::pre_asap::Schema, + output: (DataType, bool), +} +impl CompiledExpression { + pub(crate) fn expression(&self) -> &QueryExpr { + &self.expression + } + + pub fn compile(expression: &QueryExpr, input: &Schema) -> Result { + let schema = input + .fields + .iter() + .map(|field| { + let planner_types::post_asap::SummaryFamilyType::Plain(dtype) = &field.dtype else { + return Err(Error::Invalid( + "scalar expression cannot consume opaque summary state".into(), + )); + }; + Ok(planner_types::pre_asap::Column::new( + field.name.clone(), + dtype.clone(), + field.nullable, + )) + }) + .collect::, Error>>()?; + let schema = planner_types::pre_asap::Schema::new(schema); + validate(expression, &schema)?; + let output = expression + .scalar_type(&schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + Ok(Self { + expression: expression.clone(), + schema, + output, + }) + } + pub(crate) fn dtype(&self) -> (DataType, bool) { + self.output.clone() + } + pub(crate) fn validate_input(&self, input: &Schema) -> Result<(), Error> { + let checked = Self::compile(&self.expression, input)?; + if checked.output != self.output { + return Err(Error::Invalid( + "persisted expression type differs from its semantics".into(), + )); + } + if input.fields.len() != self.schema.columns.len() + || input + .fields + .iter() + .zip(&self.schema.columns) + .any(|(field, column)| { + field.dtype + != planner_types::post_asap::SummaryFamilyType::Plain(column.dtype.clone()) + || field.nullable != column.nullable + }) + { + return Err(Error::Invalid( + "expression input differs from its bound schema".into(), + )); + } + Ok(()) + } + /// Evaluate a row under the same typed schema used when binding the expression. + pub fn evaluate(&self, row: &[Value]) -> Result { + if row.len() != self.schema.columns.len() + || row + .iter() + .zip(&self.schema.columns) + .any(|(value, column)| !value.matches(&column.dtype, column.nullable)) + { + return Err(Error::Invalid( + "expression input differs from its bound schema".into(), + )); + } + evaluate(&self.expression, row, &self.schema) + } +} +fn validate(expr: &QueryExpr, schema: &planner_types::pre_asap::Schema) -> Result<(), Error> { + let invalid = || Error::Invalid(format!("unsupported scalar expression: {expr:?}")); + expr.scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + match expr { + QueryExpr::Column(_) | QueryExpr::Literal(_) => Ok(()), + QueryExpr::Arithmetic { left, right, .. } => { + for value in [left, right] { + validate(value, schema)?; + if !matches!( + value + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Int64 | DataType::Float64 | DataType::Null + ) { + return Err(invalid()); + } + } + Ok(()) + } + QueryExpr::Compare { left, right, op } => { + if !matches!( + op, + CompareOpKind::Eq + | CompareOpKind::Ne + | CompareOpKind::Lt + | CompareOpKind::Le + | CompareOpKind::Gt + | CompareOpKind::Ge + ) { + return Err(invalid()); + } + validate(left, schema)?; + validate(right, schema)?; + let (a, _) = left + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + let (b, _) = right + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))?; + fn comparable(dtype: &DataType) -> bool { + match dtype { + DataType::Null + | DataType::Int64 + | DataType::Float64 + | DataType::Utf8 + | DataType::Bool + | DataType::Timestamp => true, + DataType::Map { key, value, .. } => comparable(key) && comparable(value), + _ => false, + } + } + let numeric = |dtype: &DataType| matches!(dtype, DataType::Int64 | DataType::Float64); + if !comparable(&a) + || !comparable(&b) + || (a != b + && !matches!(a, DataType::Null) + && !matches!(b, DataType::Null) + && !(numeric(&a) && numeric(&b))) + { + return Err(invalid()); + } + Ok(()) + } + QueryExpr::FunctionCall { name, args } => { + if name != "asap_struct_field" + && name != "asap_element_access" + && planner_types::pre_asap::scalar_signature::MapScalarFunction::from_name(name) + .is_none() + { + return Err(invalid()); + } + for arg in args { + validate(arg, schema)?; + } + Ok(()) + } + QueryExpr::BoolAnd(parts) | QueryExpr::BoolOr(parts) => { + for part in parts { + validate(part, schema)?; + if !matches!( + part.scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Bool | DataType::Null + ) { + return Err(invalid()); + } + } + Ok(()) + } + QueryExpr::Not(value) => { + validate(value, schema)?; + if !matches!( + value + .scalar_type(schema) + .map_err(|e| Error::Invalid(e.to_string()))? + .0, + DataType::Bool | DataType::Null + ) { + return Err(invalid()); + } + Ok(()) + } + QueryExpr::IsNull(value) | QueryExpr::IsNotNull(value) => validate(value, schema), + _ => Err(invalid()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + #[test] + fn mixed_comparison_preserves_integer_precision_and_boundaries() { + assert_eq!( + integer_float_cmp(9_007_199_254_740_993, 9_007_199_254_740_992.0), + Some(Ordering::Greater) + ); + assert_eq!( + integer_float_cmp(i64::MAX, 9_223_372_036_854_775_808.0), + Some(Ordering::Less) + ); + assert_eq!( + integer_float_cmp(i64::MIN, -9_223_372_036_854_775_808.0), + Some(Ordering::Equal) + ); + assert_eq!(integer_float_cmp(-1, -1.5), Some(Ordering::Greater)); + assert_eq!(integer_float_cmp(1, 1.5), Some(Ordering::Less)); + assert_eq!(integer_float_cmp(0, f64::INFINITY), Some(Ordering::Less)); + assert_eq!( + integer_float_cmp(0, f64::NEG_INFINITY), + Some(Ordering::Greater) + ); + assert_eq!(integer_float_cmp(0, f64::NAN), None); + } +} diff --git a/crates/asap-physical-operators/src/key_by_label_values.rs b/crates/asap-physical-operators/src/key_by_label_values.rs new file mode 100644 index 00000000..e574da23 --- /dev/null +++ b/crates/asap-physical-operators/src/key_by_label_values.rs @@ -0,0 +1,126 @@ +use serde::{Deserialize, Serialize}; +// use std::collections::HashMap; +use std::hash::{Hash, Hasher}; + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct KeyByLabelValues { + // pub labels: HashMap, + pub labels: Vec, +} + +impl KeyByLabelValues { + pub fn new() -> Self { + Self { labels: Vec::new() } + } + + pub fn new_with_labels(labels: Vec) -> Self { + Self { labels } + } + + pub fn insert(&mut self, value: String) { + self.labels.push(value); + } + + pub fn get(&self, index: usize) -> Option<&String> { + self.labels.get(index) + } + + /// Encode labels as a semicolon-joined string — the canonical key format used + /// for sketch item hashing (CountMinSketch, CountSketch, HydraKLL). + pub fn to_semicolon_str(&self) -> String { + self.labels.join(";") + } + + #[cfg(test)] + /// Decode a semicolon-joined string back into a KeyByLabelValues. + pub fn from_semicolon_str(s: &str) -> Self { + Self { + labels: s.split(';').map(|s| s.to_string()).collect(), + } + } + + pub fn is_empty(&self) -> bool { + self.labels.is_empty() + } + + pub fn len(&self) -> usize { + self.labels.len() + } +} + +impl Hash for KeyByLabelValues { + fn hash(&self, state: &mut H) { + // Create a sorted vector of key-value pairs for consistent hashing + let mut sorted_pairs: Vec<_> = self.labels.iter().collect(); + sorted_pairs.sort(); + + for value in sorted_pairs { + value.hash(state); + } + } +} + +impl Default for KeyByLabelValues { + fn default() -> Self { + Self::new() + } +} + +impl std::fmt::Display for KeyByLabelValues { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{{")?; + let mut first = true; + for value in &self.labels { + if !first { + write!(f, ", ")?; + } + write!(f, "{value}")?; + first = false; + } + write!(f, "}}") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_key_by_label_values() { + let mut key = KeyByLabelValues::new(); + key.insert("localhost:8080".to_string()); + key.insert("prometheus".to_string()); + + assert_eq!(key.len(), 2); + assert_eq!(key.get(0), Some(&"localhost:8080".to_string())); + assert_eq!(key.get(1), Some(&"prometheus".to_string())); + } + + #[test] + fn test_semicolon_roundtrip() { + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string(), "prod".to_string()]); + assert_eq!(key.to_semicolon_str(), "web;prod"); + let roundtripped = KeyByLabelValues::from_semicolon_str("web;prod"); + assert_eq!(roundtripped, key); + } + + #[test] + fn test_hash_consistency() { + let mut key1 = KeyByLabelValues::new(); + key1.insert("a".to_string()); + key1.insert("b".to_string()); + + let mut key2 = KeyByLabelValues::new(); + key2.insert("b".to_string()); + key2.insert("a".to_string()); + + // Should hash to the same value regardless of insertion order + let mut hasher1 = std::collections::hash_map::DefaultHasher::new(); + let mut hasher2 = std::collections::hash_map::DefaultHasher::new(); + + key1.hash(&mut hasher1); + key2.hash(&mut hasher2); + + assert_eq!(hasher1.finish(), hasher2.finish()); + } +} diff --git a/crates/asap-physical-operators/src/lib.rs b/crates/asap-physical-operators/src/lib.rs new file mode 100644 index 00000000..345ee762 --- /dev/null +++ b/crates/asap-physical-operators/src/lib.rs @@ -0,0 +1,33 @@ +#![doc = include_str!("../README.md")] + +pub mod key_by_label_values; +pub mod measurement; +pub mod summary_kernels; +pub use summary_kernels::traits; + +mod statistic; +pub use key_by_label_values::KeyByLabelValues; +pub use measurement::Measurement; +pub use statistic::Statistic; +pub use traits::*; + +pub use expressions::arithmetic; +pub mod capability; +pub use summary_kernels::factory; + +/// The exact Planner contract used by these kernels. +pub use planner_types as planner; + +pub mod dag; + +pub mod readout; + +mod error; +pub use error::Error; +pub mod expressions; +pub mod operators; +pub mod physical_planner; +pub mod plan; +pub mod runtime; +pub mod sources; +pub mod values; diff --git a/crates/asap-physical-operators/src/measurement.rs b/crates/asap-physical-operators/src/measurement.rs new file mode 100644 index 00000000..57234f01 --- /dev/null +++ b/crates/asap-physical-operators/src/measurement.rs @@ -0,0 +1,48 @@ +use serde::{Deserialize, Serialize}; +use std::ops::Add; + +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct Measurement { + pub value: f64, +} + +impl Measurement { + pub fn new(value: f64) -> Self { + Self { value } + } +} + +impl Add for Measurement { + type Output = Measurement; + + fn add(self, other: Measurement) -> Measurement { + Measurement::new(self.value + other.value) + } +} + +impl Add for &Measurement { + type Output = Measurement; + + fn add(self, other: &Measurement) -> Measurement { + Measurement::new(self.value + other.value) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_measurement_creation() { + let measurement = Measurement::new(42.5); + assert_eq!(measurement.value, 42.5); + } + + #[test] + fn test_measurement_addition() { + let m1 = Measurement::new(10.0); + let m2 = Measurement::new(20.0); + let result = m1 + m2; + assert_eq!(result.value, 30.0); + } +} diff --git a/crates/asap-physical-operators/src/operators/aggregate/mod.rs b/crates/asap-physical-operators/src/operators/aggregate/mod.rs new file mode 100644 index 00000000..71e7134c --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aggregate/mod.rs @@ -0,0 +1,293 @@ +use super::*; +impl Operator { + pub fn aggregate( + input: Schema, + groups: Vec, + measures: Vec<(String, Reduction)>, + ) -> Result { + validate_groups(&input, &groups)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + for (name, reduction) in &measures { + let (t, n) = match reduction { + Reduction::Count => (DataType::Int64, false), + Reduction::Sum(i) | Reduction::Avg(i) => { + let (t, _) = plain(&input, *i)?; + if !matches!(t, DataType::Int64 | DataType::Float64) { + return Err(invalid("numeric aggregate input required")); + } + ( + if matches!(reduction, Reduction::Avg(_)) { + DataType::Float64 + } else { + t.clone() + }, + false, + ) + } + Reduction::Min(i) | Reduction::Max(i) => { + let (t, nullable) = plain(&input, *i)?; + if !ordered(t) { + return Err(invalid("ordered aggregate input required")); + } + (t.clone(), nullable || groups.is_empty()) + } + }; + fields.push(result_field(name, t, n)); + } + Ok(Self { + kind: Kind::Aggregate { + groups, + measures: measures.into_iter().map(|(_, r)| r).collect(), + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn window( + input: Schema, + intent: planner_types::pre_asap::AggIntent, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + ) -> Result { + use planner_types::pre_asap::AggIntent; + validate_groups(&input, &groups)?; + let histogram = matches!(intent, AggIntent::HistogramQuantile { .. }); + if !matches!( + intent, + AggIntent::Rate + | AggIntent::Increase + | AggIntent::Count { .. } + | AggIntent::Sum { col: None } + | AggIntent::Avg { col: None } + | AggIntent::Min { col: None } + | AggIntent::Max { col: None } + | AggIntent::HistogramQuantile { .. } + ) { + return Err(invalid( + "unsupported temporal intent or unresolved value column", + )); + } + let coordinate_type = if histogram { + DataType::Float64 + } else { + DataType::Timestamp + }; + if plain(&input, coordinate)? != (&coordinate_type, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("window coordinate/value schema mismatch")); + } + if (!histogram && !matches!(window, Some((start, end)) if start < end)) + || (histogram && window.is_some()) + { + return Err(invalid("invalid temporal window")); + } + let mut fields = groups + .iter() + .map(|i| input.fields[*i].clone()) + .collect::>(); + fields.push(result_field( + "value", + if matches!(intent, AggIntent::Count { .. }) { + DataType::Int64 + } else { + DataType::Float64 + }, + false, + )); + Ok(Self { + kind: Kind::Window { + intent: Box::new(intent), + coordinate, + value, + groups, + window, + }, + inputs: vec![input], + output: schema(fields), + }) + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub enum Reduction { + Count, + Sum(usize), + Avg(usize), + Min(usize), + Max(usize), +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => { + crate::operators::aggregate::temporal::reduce( + rows, + intent, + groups, + *coordinate, + *value, + *window, + &context, + ) + .await? + } + Kind::Aggregate { groups, measures } => { + reduce(rows, groups, measures, &operator.inputs[0], &context).await? + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +pub(super) mod temporal; +async fn reduce( + rows: Vec>, + groups: &[usize], + measures: &[Reduction], + input: &Schema, + context: &RunContext, +) -> Result>, Error> { + let mut work = Cooperative::new(context); + let mut workspace = Workspace::new(context)?; + let mut grouped = BTreeMap::>, Vec>>::new(); + if rows.is_empty() && groups.is_empty() { + grouped.insert(vec![], vec![]); + } + for row in rows { + work.checkpoint().await?; + let key = group_key(&row, groups)?; + workspace.grow(std::mem::size_of::>())?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key))?; + } + grouped.entry(key).or_default().push(row); + } + let mut output = Vec::new(); + for rows in grouped.into_values() { + work.checkpoint().await?; + let mut result = groups + .iter() + .map(|&i| rows[0][i].clone()) + .collect::>(); + for measure in measures { + result.push(reduce_one(&rows, measure, input, &mut work).await?); + } + workspace.grow(row_bytes(&result))?; + output.push(result); + } + Ok(output) +} + +async fn reduce_one( + rows: &[Vec], + measure: &Reduction, + input: &Schema, + work: &mut Cooperative, +) -> Result { + let column = match measure { + Reduction::Count => { + return Ok(Value::Int64( + i64::try_from(rows.len()).map_err(|_| invalid("count overflow"))?, + )) + } + Reduction::Sum(i) | Reduction::Avg(i) | Reduction::Min(i) | Reduction::Max(i) => *i, + }; + let values = rows + .iter() + .map(|r| &r[column]) + .filter(|v| !matches!(v, Value::Null)); + if matches!(measure, Reduction::Min(_) | Reduction::Max(_)) { + if plain(input, column)?.0 == &DataType::Float64 { + // Match exact-state kernels: ignore NaN when a numeric value exists. + let mut best: Option = None; + for value in values { + work.checkpoint().await?; + let Value::Float64(value) = value else { + return Err(invalid("floating aggregate value required")); + }; + best = Some(best.map_or(*value, |old| { + if matches!(measure, Reduction::Min(_)) { + old.min(*value) + } else { + old.max(*value) + } + })); + } + return Ok(best.map(Value::Float64).unwrap_or(Value::Null)); + } + let mut best: Option<&Value> = None; + for value in values { + work.checkpoint().await?; + if best + .map(|b| value.compare(b)) + .transpose()? + .is_none_or(|order| { + if matches!(measure, Reduction::Min(_)) { + order.is_lt() + } else { + order.is_gt() + } + }) + { + best = Some(value); + } + } + return Ok(best.cloned().unwrap_or(Value::Null)); + } + let mut count = 0usize; + let dtype = plain(input, column)?.0; + if dtype == &DataType::Int64 { + let mut sum = 0i128; + for v in values { + work.checkpoint().await?; + let Value::Int64(v) = v else { + return Err(invalid("integer aggregate value required")); + }; + sum = sum + .checked_add(i128::from(*v)) + .ok_or_else(|| invalid("integer aggregate overflow"))?; + count += 1; + } + return if matches!(measure, Reduction::Avg(_)) { + Ok(Value::Float64(sum as f64 / count as f64)) + } else { + Ok(Value::Int64( + i64::try_from(sum).map_err(|_| invalid("integer sum overflow"))?, + )) + }; + } + let mut sum = -0.0; + for v in values { + work.checkpoint().await?; + let Value::Float64(v) = v else { + return Err(invalid("floating aggregate value required")); + }; + sum += v; + count += 1; + } + Ok(Value::Float64(if matches!(measure, Reduction::Avg(_)) { + sum / count as f64 + } else { + sum + })) +} diff --git a/crates/asap-physical-operators/src/operators/aggregate/temporal.rs b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs new file mode 100644 index 00000000..dbf66108 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aggregate/temporal.rs @@ -0,0 +1,300 @@ +//! Windowed computations use Planner intents; deployments supply the input window. +use crate::{ + operators::{ + common::{key_bytes, row_bytes, Workspace}, + sort::cooperative_sort, + }, + runtime::{Cooperative, RunContext}, +}; +use crate::{ + values::{group_key, Value}, + Error, +}; +use planner_types::pre_asap::{AggIntent, ColumnRef}; +use std::collections::BTreeMap; + +pub(in crate::operators) async fn reduce( + rows: Vec>, + intent: &AggIntent, + groups: &[usize], + coordinate: usize, + value: usize, + window: Option<(i64, i64)>, + context: &RunContext, +) -> Result>, Error> { + let mut work = Cooperative::new(context); + let mut workspace = Workspace::new(context)?; + let mut grouped = + BTreeMap::>, (Vec, Vec<(f64, f64)>, Vec<(i64, f64)>)>::new(); + for row in rows { + work.checkpoint().await?; + let key = group_key(&row, groups)?; + workspace.grow(32)?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key) + row_bytes(&row))?; + } + let entry = grouped.entry(key).or_insert_with(|| { + ( + groups.iter().map(|i| row[*i].clone()).collect(), + vec![], + vec![], + ) + }); + let Value::Float64(v) = row[value] else { + return Err(Error::Invalid("window value must be Float64".into())); + }; + match row[coordinate] { + Value::Timestamp(t) => entry.2.push((t, v)), + Value::Float64(bound) => entry.1.push((bound, v)), + _ => return Err(Error::Invalid("invalid window coordinate".into())), + } + } + let mut output = Vec::new(); + for (_, (mut keys, buckets, points)) in grouped { + work.checkpoint().await?; + let result = if let AggIntent::HistogramQuantile { q } = intent { + Some(Value::Float64(bucket_quantile(*q, buckets, context).await?)) + } else { + let points = cooperative_sort(points, |a, b| a.0.cmp(&b.0), context).await?; + let (start, end) = + window.ok_or_else(|| Error::Invalid("missing temporal window".into()))?; + if points.iter().any(|p| p.0 < start || p.0 > end) + || points.windows(2).any(|p| p[0].0 == p[1].0) + { + return Err(Error::Invalid( + "duplicate or out-of-window timestamp".into(), + )); + } + match intent { + AggIntent::Rate => rate(&points, start, end).map(Value::Float64), + AggIntent::Increase => rate(&points, start, end) + .map(|v| Value::Float64(v * (end as f64 - start as f64) / 1000.)), + AggIntent::Count { .. } => Some(Value::Int64( + i64::try_from(points.len()) + .map_err(|_| Error::Invalid("count overflow".into()))?, + )), + AggIntent::Sum { .. } => Some(Value::Float64(points.iter().map(|p| p.1).sum())), + AggIntent::Avg { .. } => Some(Value::Float64( + points.iter().map(|p| p.1).sum::() / points.len() as f64, + )), + AggIntent::Min { .. } => { + Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 < a { + p.1 + } else { + a + } + }))) + } + AggIntent::Max { .. } => { + Some(Value::Float64(points.iter().fold(f64::NAN, |a, p| { + if a.is_nan() || p.1 > a { + p.1 + } else { + a + } + }))) + } + _ => return Err(Error::Invalid("unsupported temporal intent".into())), + } + }; + if let Some(result) = result { + keys.push(result); + output.push(keys); + } + } + Ok(output) +} + +fn rate(points: &[(i64, f64)], start: i64, end: i64) -> Option { + if points.len() < 2 { + return None; + } + let (first_t, first) = points[0]; + let (last_t, last) = *points.last()?; + let span = (last_t as f64 - first_t as f64) / 1000.; + if span <= 0. { + return None; + } + let mut delta = last - first; + for pair in points.windows(2) { + if pair[1].1 < pair[0].1 { + delta += pair[0].1; + } + } + let average = span / (points.len() - 1) as f64; + let mut to_start = (first_t as f64 - start as f64) / 1000.; + let mut to_end = (end as f64 - last_t as f64) / 1000.; + if to_start >= average * 1.1 { + to_start = average / 2.; + } + // Apply the zero bound after the sparse-window half-interval cap. + if delta > 0. && first >= 0. { + to_start = to_start.min(span * first / delta); + } + if to_end >= average * 1.1 { + to_end = average / 2.; + } + Some(delta * (span + to_start + to_end) / span / ((end as f64 - start as f64) / 1000.)) +} + +async fn bucket_quantile( + q: f64, + mut b: Vec<(f64, f64)>, + context: &RunContext, +) -> Result { + let mut work = Cooperative::new(context); + let _scratch = context.reserve(b.len().checked_mul(16).ok_or(Error::MemoryLimit)?)?; + if q.is_nan() { + return Ok(f64::NAN); + } + if q < 0. { + return Ok(f64::NEG_INFINITY); + } + if q > 1. { + return Ok(f64::INFINITY); + } + b.retain(|p| !p.0.is_nan()); + b = cooperative_sort(b, |a, b| a.0.total_cmp(&b.0), context).await?; + let mut buckets: Vec<(f64, f64)> = Vec::new(); + for p in b { + work.checkpoint().await?; + if let Some(last) = buckets.last_mut() { + if last.0 == p.0 { + last.1 += p.1; + continue; + } + } + buckets.push(p); + } + if buckets.len() < 2 || buckets.last().unwrap().0 != f64::INFINITY { + return Ok(f64::NAN); + } + let mut prev = buckets[0].1; + for p in buckets.iter_mut().skip(1) { + work.checkpoint().await?; + if p.1 < prev || (p.1 - prev).abs() <= 1e-12 * (p.1.abs() + prev.abs()) { + p.1 = prev; + } + prev = p.1; + } + let count = buckets.last().unwrap().1; + if count == 0. { + return Ok(f64::NAN); + } + let rank = q * count; + let idx = buckets[..buckets.len() - 1].partition_point(|p| p.1 < rank); + if idx == buckets.len() - 1 { + return Ok(buckets[idx - 1].0); + } + if idx == 0 && buckets[0].0 <= 0. { + return Ok(buckets[0].0); + } + let (start, base) = if idx == 0 { (0., 0.) } else { buckets[idx - 1] }; + let (end, upper) = buckets[idx]; + Ok(start + (end - start) * (rank - base) / (upper - base)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + operators::Operator, + runtime::{batch_execution::evaluate_batch, Limits, RunContext, Scope}, + values::Batch, + }; + use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, + types::AccuracyTarget, + }; + use std::sync::Arc; + + // The same window operator must give the same answer in either engine phase. + #[test] + fn temporal_windows_execute_in_both_phases_and_count_is_integer() { + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "time".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }, + SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: Some(0), + }); + for scope in [ + Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + ] { + for (intent, expected) in [ + (AggIntent::Rate, Value::Float64(2.)), + (AggIntent::Increase, Value::Float64(4.)), + ( + AggIntent::Count { + accuracy: AccuracyTarget::Exact, + }, + Value::Int64(3), + ), + ] { + let batch = Batch::try_new( + schema.clone(), + vec![ + vec![Value::Timestamp(0), Value::Float64(2.)], + vec![Value::Timestamp(1000), Value::Float64(4.)], + vec![Value::Timestamp(2000), Value::Float64(2.)], + ], + ) + .unwrap(); + let operator = + Operator::window(schema.clone(), intent, 0, 1, vec![], Some((0, 2000))) + .unwrap(); + let result = evaluate_batch( + batch, + vec![operator], + RunContext::new(scope.clone(), Limits::default()).unwrap(), + ) + .unwrap(); + assert_eq!( + format!("{:?}", result[0].rows()[0][0]), + format!("{expected:?}") + ); + } + } + assert!(Operator::window(schema, AggIntent::Rate, 0, 1, vec![], Some((1, 1))).is_err()); + } + + // Histogram interpolation requires an infinite terminal bucket and coalesces duplicates. + #[test] + fn histogram_boundaries_and_duplicate_buckets() { + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let bucket_quantile = |q, buckets| { + futures::executor::block_on(super::bucket_quantile(q, buckets, &context)).unwrap() + }; + assert_eq!( + bucket_quantile(0.5, vec![(1., 1.), (1., 1.), (2., 4.), (f64::INFINITY, 4.)]), + 1. + ); + assert!(bucket_quantile(0.5, vec![(1., 2.), (2., 4.)]).is_nan()); + assert_eq!(bucket_quantile(-0.1, vec![]), f64::NEG_INFINITY); + } +} diff --git a/crates/asap-physical-operators/src/operators/aligned_binary.rs b/crates/asap-physical-operators/src/operators/aligned_binary.rs new file mode 100644 index 00000000..941a9f61 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/aligned_binary.rs @@ -0,0 +1,144 @@ +//! Arithmetic on complete, aligned population/window rows used by precomputation. +use super::*; +use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; +use std::collections::BTreeSet; + +impl Operator { + /// Match every row by the declared identity columns. Unlike an inner join, + /// incomplete or duplicate keys are errors: dropping an update changes state. + pub fn aligned_binary( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + values: (usize, usize), + operator: BinaryOperator, + ) -> Result { + if keys.is_empty() + || !matches!(operator.kind, BinaryOpKind::Arithmetic(_)) + || operator.vector_match.is_some() + { + return Err(invalid( + "aligned arithmetic requires explicit keys and arithmetic semantics", + )); + } + for (input, value) in [(&left, values.0), (&right, values.1)] { + if input.fields.get(value).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Float64) + }) { + return Err(invalid( + "aligned arithmetic requires non-null Float64 values", + )); + } + } + let mut left_keys = BTreeSet::new(); + let mut right_keys = BTreeSet::new(); + for &(l, r) in &keys { + if l == values.0 + || r == values.1 + || !left_keys.insert(l) + || !right_keys.insert(r) + || left + .fields + .get(l) + .zip(right.fields.get(r)) + .is_none_or(|(l, r)| l.nullable || r.nullable || l.dtype != r.dtype) + { + return Err(invalid("invalid aligned arithmetic keys")); + } + } + if left_keys.len() + 1 != left.fields.len() || right_keys.len() + 1 != right.fields.len() { + return Err(invalid( + "aligned arithmetic must account for every input column", + )); + } + Ok(Self { + output: left.clone(), + inputs: vec![left, right], + kind: Kind::AlignedBinary { + keys, + values, + operator, + }, + }) + } +} + +pub(super) fn execute<'a>( + op: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::AlignedBinary { + keys, + values, + operator, + } = &op.kind + else { + unreachable!() + }; + let right = inputs + .pop() + .ok_or_else(|| invalid("missing aligned right input"))?; + let left = inputs + .pop() + .ok_or_else(|| invalid("missing aligned left input"))?; + Ok(futures::stream::once(async move { + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + if left.is_empty() || left.len() != right.len() { + return Err(invalid( + "aligned arithmetic requires matching nonempty key sets", + )); + } + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + let columns = |side: bool| { + keys.iter() + .map(|&(l, r)| if side { r } else { l }) + .collect::>() + }; + let left_columns = columns(false); + let right_columns = columns(true); + let mut indexed = BTreeMap::new(); + for row in right { + work.checkpoint().await?; + let key = group_key(&row, &right_columns)?; + let Value::Float64(value) = row[values.1] else { + return Err(invalid("invalid aligned value")); + }; + if !value.is_finite() { + return Err(invalid("aligned arithmetic input is non-finite")); + } + workspace.grow(key_bytes(&key) + 64)?; + if indexed.insert(key, value).is_some() { + return Err(invalid("aligned arithmetic input has duplicate keys")); + } + } + let mut rows = Vec::new(); + for mut row in left { + work.checkpoint().await?; + let key = group_key(&row, &left_columns)?; + let right = indexed + .remove(&key) + .ok_or_else(|| invalid("aligned arithmetic input has missing or duplicate keys"))?; + let Value::Float64(left) = row[values.0] else { + return Err(invalid("invalid aligned value")); + }; + if !left.is_finite() { + return Err(invalid("aligned arithmetic input is non-finite")); + } + let result = crate::expressions::arithmetic::evaluate_binary(operator, left, right)?; + if !matches!(result, Value::Float64(value) if value.is_finite()) { + return Err(invalid("aligned arithmetic produced a non-finite update")); + } + row[values.0] = result; + workspace.grow(std::mem::size_of::>())?; + rows.push(row); + } + if !indexed.is_empty() { + return Err(invalid("aligned arithmetic has unmatched input keys")); + } + Batch::try_new(op.output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/common.rs b/crates/asap-physical-operators/src/operators/common.rs new file mode 100644 index 00000000..9f1709da --- /dev/null +++ b/crates/asap-physical-operators/src/operators/common.rs @@ -0,0 +1,75 @@ +use super::*; +pub(super) fn invalid(message: &str) -> Error { + Error::Invalid(message.into()) +} +pub(super) fn schema(fields: Vec) -> Schema { + Arc::new(SummarySchema { + fields, + time_index: None, + }) +} +pub(super) fn result_field(name: &str, dtype: DataType, nullable: bool) -> SummaryField { + SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable, + } +} + +pub(super) fn validate_groups(input: &Schema, groups: &[usize]) -> Result<(), Error> { + for &i in groups { + plain(input, i)?; + } + if groups + .iter() + .collect::>() + .len() + != groups.len() + { + return Err(invalid("duplicate group columns")); + } + Ok(()) +} +pub(super) async fn collect_rows( + mut input: Input<'_, Batch>, + context: &RunContext, +) -> Result<(Vec>, Vec), Error> { + let mut rows = Vec::new(); + let mut work = Cooperative::new(context); + let mut reservations = Vec::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + reservations.push(context.reserve(batch.bytes())?); + for row in batch.rows() { + work.checkpoint().await?; + rows.push(row.clone()); + } + } + Ok((rows, reservations)) +} +/// Estimates retained workspace before growing collections. It is not an RSS limit. +pub(super) struct Workspace { + reservation: Reservation, + bytes: usize, +} +impl Workspace { + pub(super) fn new(context: &RunContext) -> Result { + Ok(Self { + reservation: context.reserve(0)?, + bytes: 0, + }) + } + pub(super) fn grow(&mut self, bytes: usize) -> Result<(), Error> { + self.bytes = self.bytes.checked_add(bytes).ok_or(Error::MemoryLimit)?; + self.reservation.resize(self.bytes) + } +} +pub(super) fn row_bytes(row: &[Value]) -> usize { + std::mem::size_of::>() + row.iter().map(Value::bytes).sum::() +} +pub(super) fn key_bytes(key: &[Vec]) -> usize { + 64 + key + .iter() + .map(|part| std::mem::size_of::>() + part.len()) + .sum::() +} diff --git a/crates/asap-physical-operators/src/operators/current_series.rs b/crates/asap-physical-operators/src/operators/current_series.rs new file mode 100644 index 00000000..937036c5 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/current_series.rs @@ -0,0 +1,135 @@ +//! A bounded instant-vector snapshot: select latest before removing stale markers. +use super::*; + +impl Operator { + pub fn current_series( + input: Schema, + identity: usize, + coordinate: usize, + value: usize, + lookback_ms: i64, + ) -> Result { + if lookback_ms <= 0 + || plain(&input, identity)? != (&DataType::Utf8, false) + || plain(&input, coordinate)? != (&DataType::Timestamp, false) + || plain(&input, value)? != (&DataType::Float64, false) + { + return Err(invalid("invalid current-series input contract")); + } + Ok(Self { + kind: Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + }, + inputs: vec![input.clone()], + output: input, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + } = operator.kind + else { + unreachable!() + }; + let input = inputs + .pop() + .ok_or_else(|| invalid("current-series input missing"))?; + let output = operator.output.clone(); + let (start, end) = window(lookback_ms, &context)?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut latest = BTreeMap::, usize>::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for (index, row) in rows.iter().enumerate() { + work.checkpoint().await?; + let Value::Timestamp(timestamp) = row[coordinate] else { + unreachable!() + }; + if timestamp <= start || timestamp > end { + continue; + } + let key = row[identity].key()?; + if let Some(&previous) = latest.get(&key) { + let Value::Timestamp(previous_time) = rows[previous][coordinate] else { + unreachable!() + }; + if timestamp < previous_time { + continue; + } + if timestamp == previous_time { + let (Value::Float64(a), Value::Float64(b)) = + (&row[value], &rows[previous][value]) + else { + unreachable!() + }; + if a.to_bits() != b.to_bits() { + return Err(invalid("conflicting samples for one series timestamp")); + } + continue; + } + } else { + workspace.grow(64 + key.len())?; + } + latest.insert(key, index); + } + let mut result = Vec::new(); + for index in latest.into_values() { + work.checkpoint().await?; + let Value::Float64(sample) = rows[index][value] else { + unreachable!() + }; + if sample.to_bits() == 0x7ff0_0000_0000_0002 { + continue; + } + workspace.grow(row_bytes(&rows[index]))?; + let mut row = rows[index].clone(); + row[coordinate] = Value::Timestamp(end); + result.push(row); + } + Batch::try_new(output, result) + }) + .boxed_local()) +} + +fn window(lookback_ms: i64, context: &RunContext) -> Result<(i64, i64), Error> { + let end = match context.scope { + crate::runtime::Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + crate::runtime::Scope::Ingestion { window_end_ms, .. } => window_end_ms, + }; + let start = end + .checked_sub(lookback_ms) + .ok_or_else(|| invalid("current-series window overflows"))?; + if let crate::runtime::Scope::Ingestion { + window_start_ms, .. + } = context.scope + { + if window_start_ms != start { + return Err(invalid( + "current-series maintenance window differs from lookback", + )); + } + } + Ok((start, end)) +} + +pub(super) fn validate_context(operator: &Operator, context: &RunContext) -> Result<(), Error> { + if let Kind::CurrentSeries { lookback_ms, .. } = operator.kind { + window(lookback_ms, context)?; + } + Ok(()) +} diff --git a/crates/asap-physical-operators/src/operators/filter.rs b/crates/asap-physical-operators/src/operators/filter.rs new file mode 100644 index 00000000..8862aacc --- /dev/null +++ b/crates/asap-physical-operators/src/operators/filter.rs @@ -0,0 +1,39 @@ +use super::*; +impl Operator { + pub fn filter(input: Schema, predicate: Expression) -> Result { + if predicate.dtype(&input)?.0 != DataType::Bool { + return Err(invalid("filter predicate must be boolean")); + } + Ok(Self { + kind: Kind::Filter(predicate), + inputs: vec![input.clone()], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Filter(predicate) => Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + if matches!(predicate.evaluate(row)?, Value::Bool(true)) { + rows.push(row.clone()); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/joins/mod.rs b/crates/asap-physical-operators/src/operators/joins/mod.rs new file mode 100644 index 00000000..a4397033 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/joins/mod.rs @@ -0,0 +1,228 @@ +use super::*; +impl Operator { + pub fn semi_join( + left: Schema, + right: Schema, + keys: Vec<(usize, usize)>, + ) -> Result { + if keys.is_empty() { + return Err(invalid("semi-join needs matching keys")); + } + for &(l, r) in &keys { + if plain(&left, l)?.0 != plain(&right, r)?.0 { + return Err(invalid("join key types differ")); + } + } + Ok(Self { + kind: Kind::SemiJoin { + keys, + require_complete_right: false, + }, + inputs: vec![left.clone(), right], + output: left, + }) + } + pub(crate) fn require_complete_right(mut self) -> Self { + if let Kind::SemiJoin { + require_complete_right, + .. + } = &mut self.kind + { + *require_complete_right = true; + } + self + } + pub(crate) fn certified_pruning_keys(&self) -> Option<&[(usize, usize)]> { + match &self.kind { + Kind::SemiJoin { + keys, + require_complete_right: true, + } => Some(keys), + _ => None, + } + } + pub fn relational_join( + left: Schema, + right: Schema, + kind: planner_types::pre_asap::JoinKind, + predicate: &planner_types::pre_asap::Predicate, + output: Schema, + ) -> Result { + use planner_types::pre_asap::JoinKind; + let mut joined = left.fields.clone(); + joined.extend(right.fields.clone()); + let predicate = + crate::expressions::CompiledExpression::compile(&predicate.0, &schema(joined.clone()))?; + if predicate.dtype().0 != DataType::Bool { + return Err(invalid("join predicate must be boolean")); + } + let fields = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + left.fields.clone() + } else { + for field in &mut joined[..left.fields.len()] { + if matches!(kind, JoinKind::Right | JoinKind::Full) { + field.nullable = true; + } + } + for field in &mut joined[left.fields.len()..] { + if matches!(kind, JoinKind::Left | JoinKind::Full) { + field.nullable = true; + } + } + joined + }; + Self { + kind: Kind::Join { + kind, + predicate: Box::new(predicate), + }, + inputs: vec![left, right], + output: schema(fields), + } + .with_output_schema(output) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + if let Kind::Join { kind, predicate } = &operator.kind { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + use planner_types::pre_asap::JoinKind; + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + workspace.grow(right.len())?; + let mut result = Vec::new(); + let mut right_matched = vec![false; right.len()]; + for left_row in &left { + work.checkpoint().await?; + let mut matched = false; + for (i, right_row) in right.iter().enumerate() { + work.checkpoint().await?; + let mut joined = left_row.clone(); + joined.extend(right_row.iter().cloned()); + if *kind == JoinKind::Cross + || matches!(predicate.evaluate(&joined)?, Value::Bool(true)) + { + matched = true; + right_matched[i] = true; + match kind { + JoinKind::Semi => { + workspace.grow(row_bytes(left_row))?; + result.push(left_row.clone()); + break; + } + JoinKind::Anti => break, + _ => { + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + } + } + } + if !matched { + match kind { + JoinKind::Left | JoinKind::Full => { + let mut joined = left_row.clone(); + joined.resize( + joined.len() + operator.inputs[1].fields.len(), + Value::Null, + ); + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + JoinKind::Anti => { + workspace.grow(row_bytes(left_row))?; + result.push(left_row.clone()); + } + _ => {} + } + } + } + if matches!(kind, JoinKind::Right | JoinKind::Full) { + for (matched, row) in right_matched.into_iter().zip(right) { + work.checkpoint().await?; + if !matched { + let mut joined = vec![Value::Null; operator.inputs[0].fields.len()]; + joined.extend(row); + workspace.grow(row_bytes(&joined))?; + result.push(joined); + } + } + } + Batch::try_new(output, result) + }) + .boxed_local()); + } + if let Kind::SemiJoin { + keys, + require_complete_right, + } = &operator.kind + { + let right = inputs.pop().ok_or_else(|| invalid("right input missing"))?; + let left = inputs.pop().ok_or_else(|| invalid("left input missing"))?; + return Ok(futures::stream::once(async move { + // Poll both branches together: either may depend on a common producer. + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + let right_cols = keys.iter().map(|(_, r)| *r).collect::>(); + let left_cols = keys.iter().map(|(l, _)| *l).collect::>(); + let mut members = std::collections::BTreeSet::new(); + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + for row in &right { + work.checkpoint().await?; + if right_cols.iter().all(|&i| matchable_key(&row[i])) { + let key = group_key(row, &right_cols)?; + if !members.contains(&key) { + workspace.grow(key_bytes(&key))?; + members.insert(key); + } + } else if *require_complete_right { + return Err(invalid( + "certified pruning candidate has an unmatchable key", + )); + } + } + let mut rows = Vec::new(); + let mut covered = std::collections::BTreeSet::new(); + for row in left { + work.checkpoint().await?; + if left_cols.iter().all(|&i| matchable_key(&row[i])) + && members.contains(&group_key(&row, &left_cols)?) + { + if *require_complete_right { + let key = group_key(&row, &left_cols)?; + if !covered.contains(&key) { + workspace.grow(key_bytes(&key))?; + covered.insert(key); + } + } + workspace.grow(std::mem::size_of::>())?; + rows.push(row); + } + } + if *require_complete_right && members != covered { + return Err(invalid("certified pruning key has no authoritative value")); + } + Batch::try_new(output, rows) + }) + .boxed_local()); + } + unreachable!() +} + +// Group keys canonicalize NaNs, but equality joins must not match them. +fn matchable_key(value: &Value) -> bool { + match value { + Value::Null => false, + Value::Float64(v) => !v.is_nan(), + _ => true, + } +} diff --git a/crates/asap-physical-operators/src/operators/limit.rs b/crates/asap-physical-operators/src/operators/limit.rs new file mode 100644 index 00000000..5f5e601f --- /dev/null +++ b/crates/asap-physical-operators/src/operators/limit.rs @@ -0,0 +1,69 @@ +use super::*; +impl Operator { + pub fn limit(input: Schema, n: u64, offset: u64, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + Ok(Self { + kind: Kind::Limit { n, offset, groups }, + inputs: vec![input.clone()], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Limit { n, offset, groups } => { + let counts = BTreeMap::>, u64>::new(); + Ok(futures::stream::try_unfold( + (input, counts, Vec::::new(), false), + move |(mut input, mut counts, mut memory, done)| { + let output = output.clone(); + let context = context.clone(); + async move { + if done || *n == 0 { + return Ok(None); + } + let Some(batch) = input.next().await else { + return Ok(None); + }; + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let key = group_key(row, groups)?; + if !counts.contains_key(&key) { + memory.push( + context.reserve( + key.iter() + .map(|part| part.len() + std::mem::size_of::>()) + .sum::() + + 64, + )?, + ); + } + let count = counts.entry(key).or_default(); + if *count >= *offset && count.saturating_sub(*offset) < *n { + rows.push(row.clone()); + } + *count = count.saturating_add(1); + } + let done = groups.is_empty() + && counts + .get(&vec![]) + .is_some_and(|count| count.saturating_sub(*offset) >= *n); + Ok(Some(( + Batch::try_new(output, rows)?, + (input, counts, memory, done), + ))) + } + }, + ) + .boxed_local()) + } + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/mod.rs b/crates/asap-physical-operators/src/operators/mod.rs new file mode 100644 index 00000000..9ae0772c --- /dev/null +++ b/crates/asap-physical-operators/src/operators/mod.rs @@ -0,0 +1,340 @@ +//! Native physical operators. Each module owns its constructors and execution. +mod aligned_binary; +use crate::plan::{Boundedness, Emission, PhysicalOperator, PlanProperties}; +use crate::{ + runtime::{Cooperative, Input, OutputStream, Reservation, RunContext}, + values::{field, group_key, plain, Batch, Schema, Value}, + Error, +}; +use futures::StreamExt; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema, SummaryUpdate}, + pre_asap::{ColumnRef, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; +pub(crate) mod common; +use crate::expressions::ordered; +pub use crate::expressions::Expression; +use common::*; +mod aggregate; +mod current_series; +mod filter; +mod joins; +mod limit; +mod projection; +mod scope_timestamp; +mod sort; +mod source; +mod summary; +mod unchecked; +pub(crate) mod vector_binary; +pub(crate) mod vector_window; +pub use aggregate::Reduction; +pub use sort::SortKey; +pub use summary::ReadoutQuery; +#[derive(Clone, serde::Serialize, serde::Deserialize)] +enum Kind { + #[serde(skip)] + Source(Vec), + Constant { + value: Value, + dtype: DataType, + }, + ScopeTimestamp { + columns: Vec>, + }, + CurrentSeries { + identity: usize, + coordinate: usize, + value: usize, + lookback_ms: i64, + }, + Union, + VectorToScalar { + column: usize, + }, + VectorBinary { + operator: planner_types::post_asap::BinaryOperator, + return_bool: bool, + }, + AlignedBinary { + keys: Vec<(usize, usize)>, + values: (usize, usize), + operator: planner_types::post_asap::BinaryOperator, + }, + RangeWindow { + intent: Box>, + }, + HistogramQuantile, + Project(Vec), + Filter(Expression), + Limit { + n: u64, + offset: u64, + groups: Vec, + }, + Sort { + keys: Vec, + groups: Vec, + }, + Window { + intent: Box>, + coordinate: usize, + value: usize, + groups: Vec, + window: Option<(i64, i64)>, + }, + Aggregate { + groups: Vec, + measures: Vec, + }, + SemiJoin { + keys: Vec<(usize, usize)>, + require_complete_right: bool, + }, + Join { + kind: planner_types::pre_asap::JoinKind, + predicate: Box, + }, + SummaryBuild { + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + }, + KeyedSummaryBuild { + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + }, + KeyedReadout { + state: usize, + k: usize, + }, + SummaryMerge { + state: usize, + groups: Vec, + }, + Readout { + state: usize, + query: ReadoutQuery, + }, +} +/// A bound operation has a fully checked input/output contract before execution. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "unchecked::UncheckedOperator")] +pub struct Operator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl Operator { + pub(crate) fn row_preserving_input(&self) -> Option { + match self.kind { + Kind::Filter(_) | Kind::Sort { .. } | Kind::Limit { .. } | Kind::SemiJoin { .. } => { + Some(0) + } + _ => None, + } + } + + pub(crate) fn is_counter_readout(&self) -> bool { + matches!( + self.kind, + Kind::Readout { + query: ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + statistic: crate::Statistic::Rate | crate::Statistic::Increase, + .. + }), + .. + } + ) + } + + pub(crate) fn with_counter_lookback(mut self, lookback: i64) -> Result { + if lookback <= 0 { + return Err(invalid("counter lookback must be positive")); + } + if let Kind::Readout { + query: ReadoutQuery::Exact(readout), + .. + } = &mut self.kind + { + readout.lookback_ms = Some(lookback); + } + Ok(self) + } + + /// Resolve a counter readout's logical lookback to this run's evaluation range. + pub(super) fn readout_range(&self, context: &RunContext) -> Result, Error> { + let Kind::Readout { + query: + ReadoutQuery::Exact(crate::summary_kernels::exact::ExactReadout { + lookback_ms: Some(lookback), + .. + }), + .. + } = &self.kind + else { + return Ok(None); + }; + let end = match context.scope { + crate::runtime::Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + crate::runtime::Scope::Ingestion { window_end_ms, .. } => window_end_ms, + }; + let start = end + .checked_sub(*lookback) + .ok_or_else(|| invalid("counter window overflows Int64"))?; + if let crate::runtime::Scope::Ingestion { + window_start_ms, .. + } = context.scope + { + if window_start_ms != start { + return Err(invalid( + "maintenance window differs from logical counter window", + )); + } + } + Ok(Some((start, end))) + } + + pub(crate) fn with_output_schema(mut self, output: Schema) -> Result { + if self.output.fields.len() != output.fields.len() + || self + .output + .fields + .iter() + .zip(&output.fields) + .any(|(actual, declared)| { + actual.dtype != declared.dtype || (actual.nullable && !declared.nullable) + }) + { + return Err(invalid("native output type differs from Planner output")); + } + if output.time_index.is_some_and(|i| { + i >= output.fields.len() + || output.fields[i].dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) { + return Err(invalid("invalid output time column")); + } + self.output = output; + Ok(self) + } + pub fn schema(&self) -> Schema { + self.output.clone() + } +} +impl PhysicalOperator for Operator { + fn requires_bounded_input(&self) -> bool { + matches!( + self.kind, + Kind::Sort { .. } + | Kind::AlignedBinary { .. } + | Kind::VectorBinary { .. } + | Kind::RangeWindow { .. } + | Kind::HistogramQuantile + | Kind::CurrentSeries { .. } + | Kind::Aggregate { .. } + | Kind::Window { .. } + | Kind::Join { .. } + | Kind::SemiJoin { .. } + | Kind::SummaryBuild { .. } + | Kind::KeyedSummaryBuild { .. } + | Kind::SummaryMerge { .. } + | Kind::VectorToScalar { .. } + ) + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + let boundedness = match &self.kind { + Kind::Source(_) | Kind::Constant { .. } => Boundedness::Bounded, + Kind::Limit { groups, .. } if groups.is_empty() => Boundedness::Bounded, + _ => Boundedness::from_inputs(inputs), + }; + PlanProperties { + boundedness, + emission: if matches!(self.kind, Kind::ScopeTimestamp { .. }) { + inputs + .first() + .map_or(Emission::Unknown, |input| input.emission) + } else if self.requires_bounded_input() { + Emission::AfterInput + } else { + Emission::Incremental + }, + } + } + + fn name(&self) -> &str { + match self.kind { + Kind::Source(_) => "Source", + Kind::Constant { .. } => "Constant", + Kind::ScopeTimestamp { .. } => "ScopeTimestamp", + Kind::Union => "Union", + Kind::CurrentSeries { .. } => "CurrentSeries", + Kind::VectorToScalar { .. } => "VectorToScalar", + Kind::VectorBinary { .. } => "VectorBinary", + Kind::AlignedBinary { .. } => "AlignedBinary", + Kind::RangeWindow { .. } => "RangeWindow", + Kind::HistogramQuantile => "HistogramQuantile", + Kind::Project(_) => "Project", + Kind::Filter(_) => "Filter", + Kind::Limit { .. } => "Limit", + Kind::Sort { .. } => "Sort", + Kind::Aggregate { .. } => "Aggregate", + Kind::Window { .. } => "WindowAggregate", + Kind::SemiJoin { .. } => "SemiJoin", + Kind::Join { .. } => "RelationalJoin", + Kind::SummaryBuild { .. } | Kind::KeyedSummaryBuild { .. } => "SummaryAgg", + Kind::KeyedReadout { .. } => "SummaryEstimate", + Kind::SummaryMerge { .. } => "SummaryMerge", + Kind::Readout { .. } => "SummaryReadout", + } + } + fn validate_context(&self, context: &RunContext) -> Result<(), Error> { + current_series::validate_context(self, context)?; + self.readout_range(context).map(|_| ()) + } + fn input_schemas(&self) -> Vec { + self.inputs.clone() + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + match self.kind { + Kind::Source(_) | Kind::Constant { .. } | Kind::Union | Kind::VectorToScalar { .. } => { + source::execute(self, inputs, context) + } + Kind::VectorBinary { .. } => vector_binary::execute(self, inputs, context), + Kind::AlignedBinary { .. } => aligned_binary::execute(self, inputs, context), + Kind::RangeWindow { .. } | Kind::HistogramQuantile => { + vector_window::execute(self, inputs, context) + } + Kind::Project(_) => projection::execute(self, inputs, context), + Kind::CurrentSeries { .. } => current_series::execute(self, inputs, context), + Kind::ScopeTimestamp { .. } => scope_timestamp::execute(self, inputs, context), + Kind::Filter(_) => filter::execute(self, inputs, context), + Kind::Limit { .. } => limit::execute(self, inputs, context), + Kind::Sort { .. } => sort::execute(self, inputs, context), + Kind::Window { .. } | Kind::Aggregate { .. } => { + aggregate::execute(self, inputs, context) + } + Kind::Join { .. } | Kind::SemiJoin { .. } => joins::execute(self, inputs, context), + Kind::SummaryMerge { .. } => summary::execute_merge(self, inputs, context), + Kind::SummaryBuild { .. } + | Kind::Readout { .. } + | Kind::KeyedSummaryBuild { .. } + | Kind::KeyedReadout { .. } => summary::execute(self, inputs, context), + } + } +} diff --git a/crates/asap-physical-operators/src/operators/projection.rs b/crates/asap-physical-operators/src/operators/projection.rs new file mode 100644 index 00000000..4a174ed9 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/projection.rs @@ -0,0 +1,56 @@ +use super::*; +impl Operator { + pub fn project(input: Schema, columns: Vec<(String, Expression)>) -> Result { + let fields = columns + .iter() + .map(|(name, e)| { + if let Expression::Column(index) = e { + let mut field = input + .fields + .get(*index) + .ok_or_else(|| invalid("projection column out of range"))? + .clone(); + field.name = name.clone(); + return Ok(field); + } + let (t, n) = e.dtype(&input)?; + Ok(result_field(name, t, n)) + }) + .collect::>()?; + Ok(Self { + kind: Kind::Project(columns.into_iter().map(|(_, e)| e).collect()), + inputs: vec![input], + output: schema(fields), + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::Project(expressions) => Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let rows = batch + .rows() + .iter() + .map(|r| { + expressions + .iter() + .map(|e| e.evaluate(r)) + .collect::, _>>() + }) + .collect::, _>>()?; + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/scope_timestamp.rs b/crates/asap-physical-operators/src/operators/scope_timestamp.rs new file mode 100644 index 00000000..3561d6a6 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/scope_timestamp.rs @@ -0,0 +1,91 @@ +//! Run-scoped timestamp restoration after reduction. +use super::*; +use crate::runtime::Scope; + +impl Operator { + pub(crate) fn scope_timestamp(input: Schema, output: Schema) -> Result { + crate::values::validate_schema(&output)?; + let coordinate = output + .time_index + .ok_or_else(|| invalid("temporal output requires a time index"))?; + if plain(&output, coordinate)? != (&DataType::Timestamp, false) { + return Err(invalid("temporal output requires a non-null timestamp")); + } + let mut columns = Vec::new(); + let mut used = std::collections::BTreeSet::new(); + for (index, field) in output.fields.iter().enumerate() { + if index == coordinate { + columns.push(None); + continue; + } + let matches: Vec<_> = input + .fields + .iter() + .enumerate() + .filter(|(_, candidate)| { + candidate.dtype == field.dtype + && candidate.nullable == field.nullable + && (candidate.name == field.name + || !matches!(field.dtype, SummaryFamilyType::Plain(_))) + }) + .map(|(index, _)| index) + .collect(); + let [column] = matches.as_slice() else { + return Err(invalid("temporal output column missing or ambiguous")); + }; + if !used.insert(*column) { + return Err(invalid("temporal output repeats an input column")); + } + columns.push(Some(*column)); + } + if used.len() != input.fields.len() { + return Err(invalid("temporal output drops an input column")); + } + Ok(Self { + kind: Kind::ScopeTimestamp { columns }, + inputs: vec![input], + output, + }) + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::ScopeTimestamp { columns } = &operator.kind else { + return Err(invalid("scope timestamp operator required")); + }; + let input = inputs + .pop() + .ok_or_else(|| invalid("scope timestamp input missing"))?; + let output = operator.output.clone(); + let timestamp = match context.scope { + Scope::Ingestion { window_end_ms, .. } => window_end_ms, + Scope::Query { + evaluation_time_ms, .. + } => evaluation_time_ms, + }; + Ok(input + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + let rows = batch + .rows() + .iter() + .map(|row| { + columns + .iter() + .map(|column| { + column.map_or(Value::Timestamp(timestamp), |column| row[column].clone()) + }) + .collect() + }) + .collect(); + Batch::try_new(output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/sort.rs b/crates/asap-physical-operators/src/operators/sort.rs new file mode 100644 index 00000000..71f81de0 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/sort.rs @@ -0,0 +1,172 @@ +use super::*; +impl Operator { + pub fn sort(input: Schema, keys: Vec, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + for key in &keys { + if !ordered(plain(&input, key.column)?.0) { + return Err(invalid("unsupported sort type")); + } + } + Ok(Self { + kind: Kind::Sort { keys, groups }, + inputs: vec![input.clone()], + output: input, + }) + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Debug)] +pub struct SortKey { + pub column: usize, + pub descending: bool, + pub nulls_first: bool, +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::Sort { keys, groups } => { + let mut grouped = BTreeMap::>, Vec>>::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for row in rows { + work.checkpoint().await?; + for key in keys { + if matches!(row[key.column], Value::Map(_)) && nested_nan(&row[key.column]) + { + return Err(invalid("NaN in collection sort key")); + } + } + let key = group_key(&row, groups)?; + workspace.grow(std::mem::size_of::>())?; + if !grouped.contains_key(&key) { + workspace.grow(key_bytes(&key))?; + } + grouped.entry(key).or_default().push(row); + } + let mut result = Vec::new(); + for rows in grouped.into_values() { + let rows = + cooperative_sort(rows, |a, b| compare_rows(a, b, keys), &context).await?; + result.extend(rows); + } + result + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> std::cmp::Ordering { + use std::cmp::Ordering::*; + for key in keys { + let (a, b) = (&a[key.column], &b[key.column]); + let order = match (a, b) { + (Value::Null, Value::Null) => Equal, + (Value::Null, _) => { + if key.nulls_first { + Less + } else { + Greater + } + } + (_, Value::Null) => { + if key.nulls_first { + Greater + } else { + Less + } + } + (Value::Float64(a), Value::Float64(b)) if a.is_nan() || b.is_nan() => { + match (a.is_nan(), b.is_nan()) { + (true, true) => Equal, + (true, false) => Greater, + _ => Less, + } + } + _ => { + let order = a.compare(b).expect("bound ordered types"); + if key.descending { + order.reverse() + } else { + order + } + } + }; + if order != Equal { + return order; + } + } + Equal +} +fn nested_nan(value: &Value) -> bool { + match value { + Value::Float64(value) => value.is_nan(), + Value::Map(values) => values + .iter() + .any(|(key, value)| nested_nan(key) || nested_nan(value)), + Value::List(values) | Value::Struct(values) => values.iter().any(nested_nan), + _ => false, + } +} +/// Stable in-memory merge sort with bounded synchronous chunks. Scratch storage +/// is reserved before allocation; comparisons yield between merge steps. +pub(super) async fn cooperative_sort( + rows: Vec, + compare: impl Fn(&T, &T) -> std::cmp::Ordering, + context: &RunContext, +) -> Result, Error> { + use std::collections::VecDeque; + let bytes = rows + .len() + .checked_mul(std::mem::size_of::() + std::mem::size_of::>()) + .and_then(|n| n.checked_mul(3)) + .ok_or(Error::MemoryLimit)?; + let _scratch = context.reserve(bytes)?; + let mut work = Cooperative::new(context); + let mut rows = rows.into_iter(); + let mut runs = VecDeque::new(); + loop { + work.checkpoint().await?; + let mut chunk = rows.by_ref().take(256).collect::>(); + if chunk.is_empty() { + break; + } + chunk.sort_by(&compare); + runs.push_back(VecDeque::from(chunk)); + } + // Merge adjacent runs in rounds to preserve ties in original input order. + while runs.len() > 1 { + let mut next = VecDeque::new(); + while let Some(mut left) = runs.pop_front() { + let Some(mut right) = runs.pop_front() else { + next.push_back(left); + break; + }; + let mut merged = VecDeque::with_capacity(left.len() + right.len()); + while !left.is_empty() || !right.is_empty() { + work.checkpoint().await?; + let take_left = match (left.front(), right.front()) { + (Some(a), Some(b)) => !compare(a, b).is_gt(), + (Some(_), None) => true, + _ => false, + }; + merged.push_back(if take_left { + left.pop_front().unwrap() + } else { + right.pop_front().unwrap() + }); + } + next.push_back(merged); + } + runs = next; + } + Ok(runs.pop_front().unwrap_or_default().into()) +} diff --git a/crates/asap-physical-operators/src/operators/source.rs b/crates/asap-physical-operators/src/operators/source.rs new file mode 100644 index 00000000..42701e17 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/source.rs @@ -0,0 +1,96 @@ +use super::*; +impl Operator { + pub fn source(output: Schema, batches: Vec) -> Result { + crate::values::validate_schema(&output)?; + if batches.iter().any(|b| b.schema() != &output) { + return Err(invalid("source schema mismatch")); + } + Ok(Self { + kind: Kind::Source(batches), + inputs: vec![], + output, + }) + } + pub fn scalar(value: Value, dtype: DataType) -> Result { + let output = schema(vec![result_field( + "value", + dtype.clone(), + matches!(value, Value::Null), + )]); + Batch::try_new(output.clone(), vec![vec![value.clone()]])?; + Ok(Self { + kind: Kind::Constant { value, dtype }, + inputs: vec![], + output, + }) + } + pub fn vector_to_scalar(input: Schema, column: usize) -> Result { + if plain(&input, column)? != (&DataType::Float64, false) { + return Err(invalid("scalar conversion requires non-null Float64")); + } + Ok(Self { + kind: Kind::VectorToScalar { column }, + inputs: vec![input], + output: schema(vec![result_field("value", DataType::Float64, false)]), + }) + } + pub fn union(input: Schema, arity: usize) -> Result { + if arity == 0 { + return Err(invalid("union needs at least one input")); + } + Ok(Self { + kind: Kind::Union, + inputs: vec![input.clone(); arity], + output: input, + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + if let Kind::Source(batches) = &operator.kind { + return Ok(futures::stream::iter(batches.iter().cloned().map(Ok)).boxed_local()); + } + if let Kind::Constant { value, .. } = &operator.kind { + return Ok(futures::stream::once(async move { + Batch::try_new(output, vec![vec![value.clone()]]) + }) + .boxed_local()); + } + if matches!(operator.kind, Kind::Union) { + return Ok(futures::stream::select_all(inputs) + .map(|batch| batch.map(|batch| batch.value().clone())) + .boxed_local()); + } + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::VectorToScalar { column } => Ok(futures::stream::once(async move { + let mut input = input; + let mut work = Cooperative::new(&context); + let mut value = f64::NAN; + let mut count = 0usize; + while let Some(batch) = input.next().await { + for row in batch?.rows() { + work.checkpoint().await?; + count = count.saturating_add(1); + if let Value::Float64(v) = row[*column] { + value = v; + } + } + } + Batch::try_new( + output, + vec![vec![Value::Float64(if count == 1 { + value + } else { + f64::NAN + })]], + ) + }) + .boxed_local()), + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/operators/summary/mod.rs b/crates/asap-physical-operators/src/operators/summary/mod.rs new file mode 100644 index 00000000..cc87a75a --- /dev/null +++ b/crates/asap-physical-operators/src/operators/summary/mod.rs @@ -0,0 +1,567 @@ +use super::*; +/// A summary readout: a sketch query, or an exact readout with typed parameters. +#[derive(Clone, Debug, PartialEq, serde::Serialize, serde::Deserialize)] +pub enum ReadoutQuery { + Sketch(planner_types::post_asap::SketchQuery), + Exact(crate::summary_kernels::exact::ExactReadout), +} + +impl Operator { + pub fn keyed_summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + items: Vec, + groups: Vec, + ) -> Result { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + crate::values::validate_family(&family)?; + let SummaryFamilyType::Sketch(kind, _) = &family else { + return Err(invalid("keyed sketch required")); + }; + WeightedFrequency::configuration(kind)?; + validate_groups(&input, &groups)?; + if items.is_empty() || plain(&input, value)? != (&DataType::Float64, false) { + return Err(invalid( + "keyed summary requires identities and non-null Float64 weights", + )); + } + for &item in &items { + if !matches!( + plain(&input, item)?.0, + DataType::Utf8 + | DataType::Timestamp + | DataType::Int64 + | DataType::Float64 + | DataType::Bool + | DataType::Null + ) { + return Err(invalid("unsupported keyed summary identity type")); + } + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn keyed_readout( + input: Schema, + state: usize, + k: usize, + output: Schema, + ) -> Result { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + crate::values::validate_family(&field(&input, state)?.dtype)?; + let SummaryFamilyType::Sketch(kind, _) = &field(&input, state)?.dtype else { + return Err(invalid("keyed readout requires summary state")); + }; + let (_, _, _, capacity) = WeightedFrequency::configuration(kind)?; + if k > capacity || output.fields.len() <= input.fields.len() { + return Err(invalid("invalid keyed readout shape or capacity")); + } + if state + 1 != input.fields.len() + || output.fields[..state] != input.fields[..state] + || output.fields.last().unwrap().dtype != SummaryFamilyType::Plain(DataType::Float64) + { + return Err(invalid( + "keyed readout must preserve partitions and return a Float64 score", + )); + } + crate::values::validate_schema(&output)?; + Ok(Self { + kind: Kind::KeyedReadout { state, k }, + inputs: vec![input], + output, + }) + } + pub fn summary_build( + input: Schema, + family: SummaryFamilyType, + value: usize, + time: Option, + groups: Vec, + ) -> Result { + crate::values::validate_family(&family)?; + validate_groups(&input, &groups)?; + if plain(&input, value)?.0 != &DataType::Float64 { + return Err(invalid("summary numeric update requires Float64")); + } + if let Some(time) = time { + if plain(&input, time)? != (&DataType::Timestamp, false) { + return Err(invalid("summary time column must be a timestamp")); + } + } + if time.is_none() + && matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ) + { + return Err(invalid("counter summary requires a timestamp column")); + } + crate::capability::validate_summary_kernel( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Invalid)?; + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }); + Ok(Self { + kind: Kind::SummaryBuild { + family, + value, + time, + groups, + }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn summary_merge(input: Schema, state: usize, groups: Vec) -> Result { + validate_groups(&input, &groups)?; + crate::values::validate_family(&field(&input, state)?.dtype)?; + if matches!(field(&input, state)?.dtype, SummaryFamilyType::Plain(_)) { + return Err(invalid("summary state required")); + } + let mut fields = groups + .iter() + .map(|&i| input.fields[i].clone()) + .collect::>(); + fields.push(input.fields[state].clone()); + Ok(Self { + kind: Kind::SummaryMerge { state, groups }, + inputs: vec![input], + output: schema(fields), + }) + } + pub fn readout(input: Schema, state: usize, query: ReadoutQuery) -> Result { + let family = &field(&input, state)?.dtype; + crate::values::validate_family(family)?; + match &query { + ReadoutQuery::Sketch(query) => { + crate::capability::validate_sketch_readout(family, query)? + } + ReadoutQuery::Exact(readout) => { + crate::capability::validate_exact_readout(family, readout)? + } + } + let mut fields = input.fields.clone(); + let result_type = if matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Count, _) + ) { + DataType::Int64 + } else { + DataType::Float64 + }; + // A state-only row represents the global population. Its extrema may + // be empty, just like an ordinary ungrouped MIN/MAX aggregate. + let nullable = fields.len() == 1 + && matches!( + fields[state].dtype, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Min + | planner_types::post_asap::ExactKind::Max, + _ + ) + ); + fields[state] = result_field("value", result_type, nullable); + Ok(Self { + kind: Kind::Readout { state, query }, + inputs: vec![input], + output: schema(fields), + }) + } +} +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let range_ms = operator.readout_range(&context)?; + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + match &operator.kind { + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_summary(input, family, *value, *time, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Ok(futures::stream::once(async move { + Batch::try_new( + output, + build_keyed_summary(input, family, *value, items, groups, &context).await?, + ) + }) + .boxed_local()), + Kind::KeyedReadout { state, k } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = Vec::new(); + for row in batch.rows() { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + let summary = summary + .as_any() + .downcast_ref::() + .ok_or_else(|| invalid("weighted frequency typed state required"))?; + for items in summary.rows(*k) { + let mut values = row[..*state].to_vec(); + values.extend(items); + // The typed output schema restores epoch-millisecond + // timestamp keys from the kernel's Int64 representation. + for (value, field) in values.iter_mut().zip(&output.fields) { + if field.dtype == SummaryFamilyType::Plain(DataType::Timestamp) { + if let Value::Int64(time) = value { + *value = Value::Timestamp(*time); + } + } + } + rows.push(values); + } + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + Kind::Readout { state, query } => Ok(input + .map(move |batch| { + let batch = batch?; + let mut rows = batch.rows().to_vec(); + if let ReadoutQuery::Exact(readout) = query { + rows.retain(|row| !matches!(&row[*state], Value::Summary { state: summary, .. } + if crate::readout::insufficient_counter_samples(summary.as_ref(), readout.statistic))); + } + for row in &mut rows { + let Value::Summary { state: summary, .. } = &row[*state] else { + return Err(invalid("summary value required")); + }; + row[*state] = match query { + ReadoutQuery::Sketch(query) => Value::Float64( + summary + .estimate(query) + .map_err(|e| Error::Operator(e.to_string()))?, + ), + ReadoutQuery::Exact(readout) => { + let exact = summary + .as_any() + .downcast_ref::() + .ok_or_else(|| invalid("exact readout requires exact state"))?; + if output.fields[*state].dtype == SummaryFamilyType::Plain(DataType::Int64) { + let count = exact.count().ok_or_else(|| { + Error::Operator("exact count state lacks an integer count".into()) + })?; + Value::Int64(i64::try_from(count).map_err(|_| { + Error::Operator("exact count exceeds Int64".into()) + })?) + } else { + match exact + .readout(readout.statistic, range_ms, None) + .map_err(|e| Error::Operator(e.to_string()))? + { + Some(value) => Value::Float64(value), + None if output.fields[*state].nullable => Value::Null, + None => { + return Err(Error::Operator( + "empty exact population".into(), + )) + } + } + } + } + }; + } + Batch::try_new(output.clone(), rows) + }) + .boxed_local()), + _ => unreachable!(), + } +} +pub(super) fn execute_merge<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let output = operator.output.clone(); + let input = inputs.pop().ok_or_else(|| invalid("input missing"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let result = match &operator.kind { + Kind::SummaryMerge { state, groups } => { + merge_summary(rows, *state, groups, &context).await? + } + _ => unreachable!(), + }; + Batch::try_new(output, result) + }) + .boxed_local()) +} + +async fn build_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + time: Option, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type State = ( + Vec, + Box, + Reservation, + usize, + Option, + ); + let create = |labels: Vec, key_bytes: usize| -> Result { + let updater = crate::factory::create_planner_accumulator( + family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .map_err(Error::Operator)?; + let overhead = labels.iter().map(Value::bytes).sum::() + key_bytes + 64; + let memory = context.reserve(updater.memory_usage_bytes() + overhead)?; + Ok((labels, updater, memory, overhead, None)) + }; + let mut work = Cooperative::new(context); + let mut states = BTreeMap::>, State>::new(); + if groups.is_empty() { + states.insert(vec![], create(vec![], 0)?); + } + let ordered_time = matches!( + family, + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Rate + | planner_types::post_asap::ExactKind::Increase, + _ + ) + ); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + work.checkpoint().await?; + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect(); + let state = create( + labels, + key.iter() + .map(|v| v.len() + std::mem::size_of::>()) + .sum(), + )?; + states.insert(key.clone(), state); + } + let (_, updater, memory, overhead, previous) = + states.get_mut(&key).expect("inserted group"); + // SQL aggregates ignore NULL samples while retaining the group. + // A missing counter sample also contributes no observation. + let value = match row[value] { + Value::Float64(value) => value, + Value::Null => continue, + _ => return Err(invalid("summary update type")), + }; + let timestamp = if let Some(time) = time { + let Value::Timestamp(time) = row[time] else { + return Err(invalid("summary time type")); + }; + time + } else { + 0 + }; + if ordered_time && previous.is_some_and(|prior| timestamp <= prior) { + return Err(Error::Operator( + "counter samples must have strictly increasing timestamps within each group" + .into(), + )); + } + updater + .validate_single_input(value) + .map_err(Error::Operator)?; + updater.update_single(value, timestamp); + *previous = Some(timestamp); + memory.resize(updater.memory_usage_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, updater, _memory, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::from(updater.into_accumulator()), + }); + labels + }) + .collect()) +} +async fn merge_summary( + rows: Vec>, + state_column: usize, + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + type GroupState = (Vec, SummaryFamilyType, Arc); + let mut states: BTreeMap>, GroupState> = BTreeMap::new(); + let mut work = Cooperative::new(context); + let mut memory = context.reserve(0)?; + let mut retained = 0usize; + for row in rows { + work.checkpoint().await?; + let Value::Summary { family, state } = &row[state_column] else { + return Err(invalid("summary state required")); + }; + let key = group_key(&row, groups)?; + if let Some((_, expected, existing)) = states.get_mut(&key) { + if expected != family { + return Err(invalid("incompatible summary family")); + } + let old_bytes = existing.approx_memory_bytes(); + // Reserve an estimate for the replacement while both input states remain live. + memory.resize( + retained + .checked_add(old_bytes) + .and_then(|n| n.checked_add(state.approx_memory_bytes())) + .ok_or(Error::MemoryLimit)?, + )?; + *existing = Arc::from( + existing + .merge_with(state.as_ref()) + .map_err(|e| Error::Operator(e.to_string()))?, + ); + retained = retained + .checked_sub(old_bytes) + .and_then(|n| n.checked_add(existing.approx_memory_bytes())) + .ok_or(Error::MemoryLimit)?; + memory.resize(retained)?; + } else { + retained = retained + .checked_add(key_bytes(&key) + row_bytes(&row)) + .ok_or(Error::MemoryLimit)?; + memory.resize(retained)?; + states.insert( + key, + ( + groups.iter().map(|&i| row[i].clone()).collect(), + family.clone(), + state.clone(), + ), + ); + } + } + Ok(states + .into_values() + .map(|(mut keys, family, state)| { + keys.push(Value::Summary { family, state }); + keys + }) + .collect()) +} + +async fn build_keyed_summary( + mut input: Input<'_, Batch>, + family: &SummaryFamilyType, + value: usize, + items: &[usize], + groups: &[usize], + context: &RunContext, +) -> Result>, Error> { + use crate::{summary_kernels::weighted_frequency::WeightedFrequency, AggregateCore}; + let SummaryFamilyType::Sketch(kind, _) = family else { + unreachable!() + }; + let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; + let mut work = Cooperative::new(context); + let mut states = + BTreeMap::>, (Vec, WeightedFrequency, Reservation, usize)>::new(); + while let Some(batch) = input.next().await { + let batch = batch?; + for row in batch.rows() { + work.checkpoint().await?; + let key = group_key(row, groups)?; + if !states.contains_key(&key) { + let labels = groups.iter().map(|&i| row[i].clone()).collect::>(); + let overhead = labels.iter().map(Value::bytes).sum::() + + key.iter().map(|v| v.len() + 24).sum::() + + 128; + let bytes = width + .checked_mul(depth) + .and_then(|n| n.checked_mul(8)) + .and_then(|n| n.checked_add(overhead)) + .ok_or_else(|| invalid("weighted frequency memory size overflow"))?; + let reservation = context.reserve(bytes)?; + states.insert( + key.clone(), + ( + labels, + WeightedFrequency::new(algorithm, width, depth, capacity)?, + reservation, + overhead, + ), + ); + } + let (_, summary, reservation, overhead) = states.get_mut(&key).unwrap(); + let Value::Float64(weight) = row[value] else { + return Err(invalid("weighted frequency weight type")); + }; + summary.update( + &items + .iter() + .map(|&i| match &row[i] { + Value::Timestamp(time) => Value::Int64(*time), + value => value.clone(), + }) + .collect::>(), + weight, + )?; + reservation.resize(summary.approx_memory_bytes() + *overhead)?; + } + } + Ok(states + .into_values() + .map(|(mut labels, summary, _, _)| { + labels.push(Value::Summary { + family: family.clone(), + state: Arc::new(summary), + }); + labels + }) + .collect()) +} diff --git a/crates/asap-physical-operators/src/operators/unchecked.rs b/crates/asap-physical-operators/src/operators/unchecked.rs new file mode 100644 index 00000000..f927a4e4 --- /dev/null +++ b/crates/asap-physical-operators/src/operators/unchecked.rs @@ -0,0 +1,135 @@ +//! Deserialized operators are validated before use, whatever the encoding. +use super::*; + +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct UncheckedOperator { + kind: Kind, + inputs: Vec, + output: Schema, +} +impl TryFrom for Operator { + type Error = Error; + fn try_from(unchecked: UncheckedOperator) -> Result { + let expected_kind = + serde_json::to_value(&unchecked.kind).map_err(|error| invalid(&error.to_string()))?; + let UncheckedOperator { + kind, + inputs, + output, + } = unchecked; + for schema in inputs.iter().chain(std::iter::once(&output)) { + crate::values::validate_schema(schema)?; + } + let input = |index| { + inputs + .get(index) + .cloned() + .ok_or_else(|| invalid("missing operator input")) + }; + let op = match kind { + Kind::Source(_) => return Err(invalid("physical plans cannot serialize live sources")), + Kind::Constant { value, dtype } => Operator::scalar(value, dtype)?, + Kind::ScopeTimestamp { .. } => Operator::scope_timestamp(input(0)?, output.clone())?, + Kind::Union => Operator::union(input(0)?, inputs.len())?, + Kind::CurrentSeries { + identity, + coordinate, + value, + lookback_ms, + } => Operator::current_series(input(0)?, identity, coordinate, value, lookback_ms)?, + Kind::VectorToScalar { column } => Operator::vector_to_scalar(input(0)?, column)?, + Kind::VectorBinary { + operator, + return_bool, + } => Operator::vector_binary(input(0)?, input(1)?, operator, return_bool)?, + Kind::AlignedBinary { + keys, + values, + operator, + } => Operator::aligned_binary(input(0)?, input(1)?, keys, values, operator)?, + Kind::RangeWindow { intent } => Operator::range_window(*intent)?, + Kind::HistogramQuantile => Operator::histogram_quantile(), + Kind::Project(expressions) => { + if expressions.len() != output.fields.len() { + return Err(invalid("projection width mismatch")); + } + Operator::project( + input(0)?, + output + .fields + .iter() + .zip(expressions) + .map(|(f, e)| (f.name.clone(), e)) + .collect(), + )? + } + Kind::Filter(expression) => Operator::filter(input(0)?, expression)?, + Kind::Limit { n, offset, groups } => Operator::limit(input(0)?, n, offset, groups)?, + Kind::Sort { keys, groups } => Operator::sort(input(0)?, keys, groups)?, + Kind::Window { + intent, + coordinate, + value, + groups, + window, + } => Operator::window(input(0)?, *intent, coordinate, value, groups, window)?, + Kind::Aggregate { groups, measures } => { + if groups.len() + measures.len() != output.fields.len() { + return Err(invalid("aggregate width mismatch")); + } + let names = output.fields[groups.len()..].iter().map(|f| f.name.clone()); + Operator::aggregate(input(0)?, groups, names.zip(measures).collect())? + } + Kind::SemiJoin { + keys, + require_complete_right, + } => { + let operator = Operator::semi_join(input(0)?, input(1)?, keys)?; + if require_complete_right { + operator.require_complete_right() + } else { + operator + } + } + Kind::Join { kind, predicate } => Operator::relational_join( + input(0)?, + input(1)?, + kind, + &planner_types::pre_asap::Predicate(std::rc::Rc::new( + predicate.expression().clone(), + )), + output.clone(), + )?, + Kind::SummaryBuild { + family, + value, + time, + groups, + } => Operator::summary_build(input(0)?, family, value, time, groups)?, + Kind::KeyedSummaryBuild { + family, + value, + items, + groups, + } => Operator::keyed_summary_build(input(0)?, family, value, items, groups)?, + Kind::KeyedReadout { state, k } => { + Operator::keyed_readout(input(0)?, state, k, output.clone())? + } + Kind::SummaryMerge { state, groups } => { + Operator::summary_merge(input(0)?, state, groups)? + } + Kind::Readout { state, query } => Operator::readout(input(0)?, state, query)?, + } + .with_output_schema(output)?; + if serde_json::to_value(&op.kind).map_err(|error| invalid(&error.to_string()))? + != expected_kind + { + return Err(invalid("operator contains inconsistent compiled fields")); + } + if op.inputs != inputs { + return Err(invalid("operator input contracts differ")); + } + Ok(op) + } +} diff --git a/crates/asap-physical-operators/src/operators/vector_binary.rs b/crates/asap-physical-operators/src/operators/vector_binary.rs new file mode 100644 index 00000000..cc60ff4c --- /dev/null +++ b/crates/asap-physical-operators/src/operators/vector_binary.rs @@ -0,0 +1,237 @@ +//! Label matching and scalar broadcasting are physical computation, not source binding. +use super::*; +use planner_types::{post_asap::BinaryOperator, pre_asap::BinaryOpKind}; + +pub(crate) fn value_schema(scalar: bool) -> Schema { + let mut fields = Vec::new(); + if !scalar { + fields.push(result_field( + "labels", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }, + false, + )); + } + fields.push(result_field( + if scalar { "$promql_scalar" } else { "value" }, + DataType::Float64, + false, + )); + schema(fields) +} + +fn is_scalar(input: &Schema) -> Result { + for scalar in [true, false] { + let expected = value_schema(scalar); + if input.fields.len() == expected.fields.len() + && input + .fields + .iter() + .zip(&expected.fields) + .all(|(a, b)| a.dtype == b.dtype && !a.nullable) + { + return Ok(scalar); + } + } + Err(invalid( + "vector binary requires Float64 scalars or complete label-map vectors", + )) +} + +impl Operator { + pub fn vector_binary( + left: Schema, + right: Schema, + operator: BinaryOperator, + return_bool: bool, + ) -> Result { + let scalar = is_scalar(&left)? && is_scalar(&right)?; + is_scalar(&right)?; + let expression = Expression::Binary { + operator: operator.clone(), + left: Box::new(Expression::Column(0)), + right: Box::new(Expression::Column(1)), + }; + expression.dtype(&schema(vec![ + result_field("left", DataType::Float64, false), + result_field("right", DataType::Float64, false), + ]))?; + let comparison = matches!(operator.kind, BinaryOpKind::Compare(_)); + if (return_bool && !comparison) || (scalar && comparison && !return_bool) { + return Err(invalid("invalid scalar/vector comparison bool mode")); + } + Ok(Self { + inputs: vec![left, right], + output: value_schema(scalar), + kind: Kind::VectorBinary { + operator, + return_bool, + }, + }) + } +} + +type Labels = BTreeMap, Arc>; +fn labels(row: &[Value]) -> Result { + let Some(Value::Map(entries)) = row.first() else { + return Err(invalid("vector requires label map")); + }; + let mut result = BTreeMap::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("labels must be Utf8")); + }; + if result.insert(key.clone(), value.clone()).is_some() { + return Err(invalid("duplicate label name")); + } + } + Ok(result) +} +fn identity(mut labels: Labels) -> Labels { + labels.remove("__name__"); + labels.retain(|_, value| !value.is_empty()); + labels +} +fn value(row: &[Value]) -> Result { + match row.last() { + Some(Value::Float64(value)) => Ok(*value), + _ => Err(invalid("binary value must be Float64")), + } +} +fn label_bytes(labels: &Labels) -> usize { + labels.iter().map(|(k, v)| 64 + k.len() + v.len()).sum() +} + +pub(super) fn execute<'a>( + op: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + let Kind::VectorBinary { + operator, + return_bool, + } = &op.kind + else { + unreachable!() + }; + let left_scalar = is_scalar(&op.inputs[0])?; + let right_scalar = is_scalar(&op.inputs[1])?; + let right = inputs.pop().ok_or_else(|| invalid("missing right input"))?; + let left = inputs.pop().ok_or_else(|| invalid("missing left input"))?; + Ok(futures::stream::once(async move { + let ((left, _left_memory), (right, _right_memory)) = + futures::try_join!(collect_rows(left, &context), collect_rows(right, &context))?; + if (left_scalar && left.len() != 1) || (right_scalar && right.len() != 1) { + return Err(invalid("scalar input must contain exactly one value")); + } + let mut workspace = Workspace::new(&context)?; + let mut work = Cooperative::new(&context); + let mut rows = Vec::new(); + let mut result_identities = std::collections::BTreeSet::new(); + let mut emit = |labels: Labels, a: f64, b: f64| -> Result<(), Error> { + let arithmetic = matches!(operator.kind, BinaryOpKind::Arithmetic(_)); + let result = match crate::expressions::arithmetic::evaluate_binary(operator, a, b)? { + Value::Float64(value) => value, + Value::Bool(value) if *return_bool => { + if value { + 1. + } else { + 0. + } + } + Value::Bool(true) => { + if left_scalar { + b + } else { + a + } + } + Value::Bool(false) => return Ok(()), + _ => return Err(invalid("invalid binary result")), + }; + let mut row = Vec::new(); + if !left_scalar || !right_scalar { + let labels = if arithmetic || *return_bool { + identity(labels) + } else { + labels + }; + workspace.grow(label_bytes(&labels) + 64)?; + if !result_identities.insert(labels.clone()) { + return Err(invalid("duplicate vector result labels")); + } + workspace.grow( + label_bytes(&labels) + + std::mem::size_of::>() + + 2 * std::mem::size_of::(), + )?; + row.push(Value::Map( + labels + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + )); + } else { + workspace.grow(std::mem::size_of::>() + std::mem::size_of::())?; + } + row.push(Value::Float64(result)); + rows.push(row); + Ok(()) + }; + if left_scalar || right_scalar { + let vectors = if left_scalar { &right } else { &left }; + for row in vectors { + work.checkpoint().await?; + let labels = if left_scalar && right_scalar { + Labels::new() + } else { + labels(row)? + }; + emit( + labels, + if left_scalar { + value(&left[0])? + } else { + value(row)? + }, + if right_scalar { + value(&right[0])? + } else { + value(row)? + }, + )?; + } + } else { + let mut rhs = BTreeMap::new(); + // Keep matching workspace separate from the output reservation captured by emit. + let mut matching = Workspace::new(&context)?; + for row in &right { + work.checkpoint().await?; + let key = identity(labels(row)?); + matching.grow(label_bytes(&key) + 64)?; + if rhs.insert(key, value(row)?).is_some() { + return Err(invalid("duplicate vector matching labels")); + } + } + let mut seen = std::collections::BTreeSet::new(); + for row in &left { + work.checkpoint().await?; + let labels = labels(row)?; + let key = identity(labels.clone()); + matching.grow(label_bytes(&key) + 64)?; + if !seen.insert(key.clone()) { + return Err(invalid("duplicate vector matching labels")); + } + if let Some(b) = rhs.get(&key) { + emit(labels, value(row)?, *b)?; + } + } + } + Batch::try_new(op.output.clone(), rows) + }) + .boxed_local()) +} diff --git a/crates/asap-physical-operators/src/operators/vector_window.rs b/crates/asap-physical-operators/src/operators/vector_window.rs new file mode 100644 index 00000000..407f04ea --- /dev/null +++ b/crates/asap-physical-operators/src/operators/vector_window.rs @@ -0,0 +1,147 @@ +//! Window bounds are typed input data; aggregation and histogram semantics stay native. +use super::*; +use planner_types::pre_asap::AggIntent; + +pub(crate) fn matrix_schema() -> Schema { + let mut fields = vector_binary::value_schema(false).fields.clone(); + fields.insert(1, result_field("timestamp", DataType::Timestamp, false)); + fields.push(result_field("window_start", DataType::Timestamp, false)); + fields.push(result_field("window_end", DataType::Timestamp, false)); + Arc::new(SummarySchema { + fields, + time_index: Some(1), + }) +} + +impl Operator { + pub fn range_window(intent: AggIntent) -> Result { + // Reuse the window constructor's semantic admission, without fixing request time. + Self::window(matrix_schema(), intent.clone(), 1, 2, vec![0], Some((0, 1)))?; + Ok(Self { + kind: Kind::RangeWindow { + intent: Box::new(intent), + }, + inputs: vec![matrix_schema()], + output: vector_binary::value_schema(false), + }) + } + pub fn histogram_quantile() -> Self { + Self { + kind: Kind::HistogramQuantile, + inputs: vec![ + vector_binary::value_schema(true), + vector_binary::value_schema(false), + ], + output: vector_binary::value_schema(false), + } + } +} + +pub(super) fn execute<'a>( + operator: &'a Operator, + mut inputs: Vec>, + context: RunContext, +) -> Result, Error> { + match &operator.kind { + Kind::RangeWindow { intent } => { + let input = inputs + .pop() + .ok_or_else(|| invalid("missing matrix input"))?; + Ok(futures::stream::once(async move { + let (rows, _memory) = collect_rows(input, &context).await?; + let mut window = None; + let mut work = Cooperative::new(&context); + for row in &rows { + work.checkpoint().await?; + let (Value::Timestamp(start), Value::Timestamp(end)) = (&row[3], &row[4]) + else { + return Err(invalid("missing matrix window bounds")); + }; + if start >= end || window.is_some_and(|bounds| bounds != (*start, *end)) { + return Err(invalid("matrix rows must share one nonempty window")); + } + window = Some((*start, *end)); + } + let mut rows = + aggregate::temporal::reduce(rows, intent, &[0], 1, 2, window, &context).await?; + for row in &mut rows { + work.checkpoint().await?; + row[1] = Expression::ExactFloat64(1).evaluate(row)?; + } + Batch::try_new(operator.output.clone(), rows) + }) + .boxed_local()) + } + Kind::HistogramQuantile => { + let buckets = inputs + .pop() + .ok_or_else(|| invalid("missing histogram buckets"))?; + let quantile = inputs.pop().ok_or_else(|| invalid("missing quantile"))?; + Ok(futures::stream::once(async move { + let ((quantile, _q_memory), (buckets, _bucket_memory)) = futures::try_join!( + collect_rows(quantile, &context), + collect_rows(buckets, &context) + )?; + let [row] = quantile.as_slice() else { + return Err(invalid("histogram quantile requires one scalar")); + }; + let [Value::Float64(q)] = row.as_slice() else { + return Err(invalid("invalid quantile scalar")); + }; + let mut rows = Vec::new(); + let mut work = Cooperative::new(&context); + let mut workspace = Workspace::new(&context)?; + for row in buckets { + work.checkpoint().await?; + let Value::Map(entries) = &row[0] else { + return Err(invalid("histogram buckets require labels")); + }; + let mut bound = None; + let mut labels = BTreeMap::new(); + let mut seen = std::collections::BTreeSet::new(); + for (key, value) in entries.iter() { + let (Value::Utf8(key), Value::Utf8(value)) = (key, value) else { + return Err(invalid("histogram labels must be Utf8")); + }; + if !seen.insert(key) { + return Err(invalid("duplicate histogram label")); + } + if key.as_ref() == "le" { + bound = value.parse::().ok(); + } else if key.as_ref() != "__name__" && !value.is_empty() { + labels.insert(key.clone(), value.clone()); + } + } + if let Some(bound) = bound { + let projected = vec![ + Value::Map( + labels + .into_iter() + .map(|(k, v)| (Value::Utf8(k), Value::Utf8(v))) + .collect::>() + .into(), + ), + Value::Float64(bound), + row[1].clone(), + ]; + workspace.grow(row_bytes(&projected))?; + rows.push(projected); + } + } + let result = aggregate::temporal::reduce( + rows, + &AggIntent::HistogramQuantile { q: *q }, + &[0], + 1, + 2, + None, + &context, + ) + .await?; + Batch::try_new(operator.output.clone(), result) + }) + .boxed_local()) + } + _ => unreachable!(), + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs new file mode 100644 index 00000000..a8f476f3 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -0,0 +1,275 @@ +//! Compile maintenance-selected frontiers without deployment-specific graph rewrites. +use super::*; + +/// One computation realization; lifecycle/window/revision requirements accompany +/// it during optimization and deployment. Stored outputs have no storage identity. +/// Deserialization validates the producer/reader boundary. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "UncheckedCandidate")] +pub struct PhysicalCandidate { + pub precompute: Option, + pub query: CompiledPhysicalDag, + pub materialized_outputs: BTreeMap, +} + +/// Compile an explicit materialization frontier selected by Planner maintenance +/// search. Operators upstream of that frontier run in precompute, including +/// readouts/reductions; query execution receives their typed output values. +/// Empty frontiers retain the full computation in the query DAG. +/// +/// Repeated windows must be instantiated with the same evaluation/population +/// contract used to build each output. This API never treats a result from a +/// different window or revision as interchangeable merely because types match. +pub fn compile_candidate( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], + frontier: &[NodeId], +) -> Result { + if frontier.is_empty() { + return Ok(PhysicalCandidate { + precompute: None, + query: compile(dag, inputs, roots)?, + materialized_outputs: BTreeMap::new(), + }); + } + let frontier_set: BTreeSet<_> = frontier.iter().copied().collect(); + if frontier_set.len() != frontier.len() || frontier.iter().any(|id| inputs.contains_key(id)) { + return Err(invalid("frontier must contain distinct computed outputs")); + } + let full = compile(dag, inputs.clone(), roots)?; + let precompute = compile(dag, inputs.clone(), frontier)?; + let mut materialized_outputs = BTreeMap::new(); + for &id in frontier { + // Also proves that the frontier is reachable from the requested roots. + full.output_contract(id)?; + let mut output = precompute.output_contract(id)?; + if output.properties.boundedness != Boundedness::Bounded { + return Err(invalid("materialized output requires bounded execution")); + } + // A stored reader may stream batches even when the producer blocked. + // Its timing is independent; the retained result still must be finite. + output.properties.emission = Emission::Unknown; + materialized_outputs.insert(id, output); + } + let mut query_inputs = inputs; + query_inputs.extend(materialized_outputs.clone()); + let query = compile(dag, query_inputs, roots)?; + let used: BTreeSet<_> = query.input_contracts().map(|(id, _)| id).collect(); + if !frontier.iter().all(|id| used.contains(id)) { + return Err(invalid( + "frontier contains an output shadowed by another boundary", + )); + } + Ok(PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs, + }) +} + +/// Enumerate bounded, reachable materialization frontiers above explicit inputs. +/// Each frontier is an antichain: storing an output and its ancestor together +/// would leave the ancestor unused by query execution. Lifecycle eligibility +/// and deployment feasibility are evaluated separately before cost selection. +/// Exceeding the search budget returns an error, never a partial inventory. +pub fn enumerate_frontiers( + dag: &PostAsapDag, + inputs: &BTreeMap, + roots: &[NodeId], + max_candidates: usize, +) -> Result>, Error> { + if max_candidates == 0 { + return Err(invalid( + "frontier search requires a positive candidate budget", + )); + } + let compiled = compile(dag, inputs.clone(), roots)?; + let mut ancestors = BTreeMap::>::new(); + let mut eligible = Vec::new(); + for node in &dag.nodes { + let id = u64::from(node.id.0); + if inputs.contains_key(&id) { + continue; + } + let Ok(contract) = compiled.output_contract(id) else { + continue; + }; + if contract.properties.boundedness != Boundedness::Bounded { + continue; + } + let mut seen = BTreeSet::new(); + let mut pending = vec![id]; + while let Some(current) = pending.pop() { + if !seen.insert(current) || inputs.contains_key(¤t) { + continue; + } + pending.extend( + dag.edges + .iter() + .filter(|edge| u64::from(edge.consumer.0) == current) + .map(|edge| u64::from(edge.producer.0)), + ); + } + ancestors.insert(id, seen); + eligible.push(id); + } + eligible.sort_unstable(); + let mut frontiers = vec![vec![]]; + for id in eligible { + let additions = frontiers + .iter() + .filter(|frontier| { + frontier.iter().all(|previous| { + !ancestors[&id].contains(previous) && !ancestors[previous].contains(&id) + }) + }) + .map(|frontier| { + let mut next = frontier.clone(); + next.push(id); + next + }) + .collect::>(); + if additions.len() > max_candidates.saturating_sub(frontiers.len()) { + return Err(invalid( + "materialization frontier search exceeds candidate budget", + )); + } + frontiers.extend(additions); + } + Ok(frontiers) +} + +/// Lower every maintenance candidate before feasibility/cost evaluation. Keep +/// individual failures visible; do not substitute another computation on error. +pub fn compile_candidates( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], + frontiers: &[Vec], +) -> Vec> { + frontiers + .iter() + .map(|frontier| compile_candidate(dag, inputs.clone(), roots, frontier)) + .collect() +} + +/// Complete workload cost supplied by scoped optimizer/deployment evidence. +/// The evaluator includes build/update work, retained state, shared producers +/// and recurrent reads over the same horizon; these are not per-query timings. +#[derive(Clone, Debug)] +pub struct CandidateCost { + pub workload_scope: String, + pub horizon_seconds: f64, + pub total_cost: f64, +} + +pub struct CandidateSelection { + pub candidate: T, + pub candidate_index: usize, + pub cost: CandidateCost, +} + +/// Select only compiled and deployment-feasible physical candidates. `None` +/// rejects an unbindable candidate before pricing. Comparable scoped costs are +/// required; deployment never rewrites the selected frontier after this step. +/// The payload is generic so deployments can retain binding/diagnostic metadata +/// alongside each compiled computation without duplicating winner selection. +pub fn select_candidate( + candidates: Vec>, + mut evaluate: impl FnMut(&T) -> Result, Error>, +) -> Result, Error> { + let mut scope: Option<(String, f64)> = None; + let mut selected: Option> = None; + for (candidate_index, candidate) in candidates.into_iter().enumerate() { + let Ok(candidate) = candidate else { continue }; + let Some(cost) = evaluate(&candidate)? else { + continue; + }; + if cost.workload_scope.is_empty() + || !cost.horizon_seconds.is_finite() + || cost.horizon_seconds <= 0. + || !cost.total_cost.is_finite() + || cost.total_cost < 0. + { + return Err(invalid( + "candidate cost lacks a valid workload scope/horizon", + )); + } + let current_scope = (cost.workload_scope.clone(), cost.horizon_seconds); + if scope.as_ref().is_some_and(|scope| scope != ¤t_scope) { + return Err(invalid( + "candidate costs describe different workloads or horizons", + )); + } + scope = Some(current_scope); + if selected + .as_ref() + .is_none_or(|selected| cost.total_cost < selected.cost.total_cost) + { + selected = Some(CandidateSelection { + candidate, + candidate_index, + cost, + }); + } + } + selected.ok_or_else(|| invalid("no feasible priced physical candidate")) +} + +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct UncheckedCandidate { + precompute: Option, + query: CompiledPhysicalDag, + materialized_outputs: BTreeMap, +} +impl TryFrom for PhysicalCandidate { + type Error = Error; + fn try_from(candidate: UncheckedCandidate) -> Result { + let result = Self { + precompute: candidate.precompute, + query: candidate.query, + materialized_outputs: candidate.materialized_outputs, + }; + result.validate()?; + Ok(result) + } +} + +impl PhysicalCandidate { + /// Validate the physical handoff, including the producer/reader boundary. + pub fn validate(&self) -> Result<(), Error> { + self.query.validate()?; + let Some(precompute) = &self.precompute else { + return if self.materialized_outputs.is_empty() { + Ok(()) + } else { + Err(invalid("materialized outputs have no producer DAG")) + }; + }; + precompute.validate()?; + let outputs: BTreeSet<_> = self.materialized_outputs.keys().copied().collect(); + if outputs.is_empty() || outputs != precompute.roots().iter().copied().collect() { + return Err(invalid("physical frontier differs from precompute outputs")); + } + let readers: BTreeMap<_, _> = self.query.input_contracts().collect(); + for (&id, contract) in &self.materialized_outputs { + let produced = precompute.output_contract(id)?; + // Direct frontiers retain their node IDs. Temporal candidates can + // read several window instances through distinct input slots; + // their deployment bindings must validate those slots separately. + let reader = readers.get(&id); + if contract.schema != produced.schema + || reader.is_some_and(|reader| contract.schema != reader.schema) + || produced.properties.boundedness != Boundedness::Bounded + || contract.properties.boundedness != Boundedness::Bounded + || reader + .is_some_and(|reader| reader.properties.boundedness != Boundedness::Bounded) + { + return Err(invalid("physical frontier schema or boundedness mismatch")); + } + } + Ok(()) + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs new file mode 100644 index 00000000..70d6a9ff --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -0,0 +1,310 @@ +//! Reader-independent physical computation and checked deployment instantiation. +use super::*; + +/// A typed execution boundary, without storage identity or a live reader. +#[derive(Clone, Debug, serde::Serialize, serde::Deserialize)] +pub struct InputContract { + pub schema: Schema, + pub properties: PlanProperties, +} +impl InputContract { + pub fn bounded(schema: Schema) -> Self { + Self { + schema, + properties: PlanProperties { + boundedness: Boundedness::Bounded, + emission: Emission::Unknown, + }, + } + } + pub fn from_source(source: &dyn PhysicalOperator) -> Self { + Self { + schema: source.output_schema(), + properties: source.properties(&[]), + } + } +} +#[derive(Clone, serde::Serialize, serde::Deserialize)] +enum Node { + Input(InputContract), + Operator { + inputs: Vec, + operator: Operator, + }, +} + +/// Selected native operators and input slots. Rebinding never repeats lowering. +/// Serde is format-agnostic; deployments choose the encoding and its versioning. +/// Deserialization validates the graph before it is usable. +#[derive(Clone, serde::Serialize, serde::Deserialize)] +#[serde(try_from = "UncheckedDag")] +pub struct CompiledPhysicalDag { + nodes: BTreeMap, + roots: Vec, +} +#[derive(serde::Deserialize)] +#[serde(deny_unknown_fields)] +struct UncheckedDag { + nodes: BTreeMap, + roots: Vec, +} +impl TryFrom for CompiledPhysicalDag { + type Error = Error; + fn try_from(dag: UncheckedDag) -> Result { + let result = Self { + nodes: dag.nodes, + roots: dag.roots, + }; + result.validate()?; + Ok(result) + } +} + +impl CompiledPhysicalDag { + /// Link already-selected physical fragments without lowering operators again. + /// Fragment keys and source keys share a namespace; repeated dependency IDs + /// therefore remain one producer in the composed graph. + pub fn compose( + sources: BTreeMap, + fragments: BTreeMap, Self)>, + roots: Vec, + ) -> Result { + if sources.keys().any(|id| fragments.contains_key(id)) { + return Err(invalid("physical source and fragment IDs overlap")); + } + let mut contracts = sources.clone(); + for (&id, (_, fragment)) in &fragments { + fragment.validate()?; + let [root] = fragment.roots() else { + return Err(invalid("composed fragment requires one root")); + }; + if fragment.input_contracts().any(|(id, _)| id == *root) { + return Err(invalid("fragment root must be a computed output")); + } + contracts.insert(id, fragment.output_contract(*root)?); + } + let mut next = contracts + .keys() + .next_back() + .copied() + .unwrap_or(0) + .checked_add(1) + .ok_or_else(|| invalid("physical node ID overflow"))?; + let mut result = Self::new(roots); + for (id, contract) in sources { + result.add_input(id, contract)?; + } + for (id, (inputs, fragment)) in fragments { + if inputs.len() != fragment.input_contracts().count() { + return Err(invalid("physical fragment input arity mismatch")); + } + let mut mapping = BTreeMap::new(); + for ((local, expected), global) in fragment.input_contracts().zip(inputs) { + let actual = contracts + .get(&global) + .ok_or_else(|| invalid("missing physical fragment dependency"))?; + if expected.schema != actual.schema + || (expected.properties.boundedness == Boundedness::Bounded + && actual.properties.boundedness != Boundedness::Bounded) + { + return Err(invalid("physical fragment dependency contract mismatch")); + } + mapping.insert(local, global); + } + mapping.insert(fragment.roots[0], id); + for local in fragment.nodes.keys() { + if !mapping.contains_key(local) { + mapping.insert(*local, next); + next = next + .checked_add(1) + .ok_or_else(|| invalid("physical node ID overflow"))?; + } + } + for (local, node) in fragment.nodes { + if let Node::Operator { inputs, operator } = node { + result.add( + mapping[&local], + inputs.into_iter().map(|input| mapping[&input]).collect(), + operator, + )?; + } + } + } + result.validate()?; + Ok(result) + } + + /// Assemble already-lowered operators and typed external inputs. This is + /// useful for engines that compose multiple compiled computation fragments. + pub fn from_operators( + inputs: BTreeMap, + operators: BTreeMap, Operator)>, + roots: Vec, + ) -> Result { + let mut result = Self::new(roots); + for (id, contract) in inputs { + result.add_input(id, contract)?; + } + for (id, (inputs, operator)) in operators { + result.add(id, inputs, operator)?; + } + result.validate()?; + Ok(result) + } + pub(super) fn new(roots: Vec) -> Self { + Self { + nodes: BTreeMap::new(), + roots, + } + } + pub(super) fn add_input(&mut self, id: NodeId, contract: InputContract) -> Result<(), Error> { + self.insert(id, Node::Input(contract)) + } + pub(super) fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: Operator, + ) -> Result<(), Error> { + self.insert(id, Node::Operator { inputs, operator }) + } + fn insert(&mut self, id: NodeId, node: Node) -> Result<(), Error> { + if self.nodes.insert(id, node).is_some() { + return Err(invalid(format!("duplicate physical node {id}"))); + } + Ok(()) + } + /// Identify the external input whose rows survive unchanged at this output. + /// Protocol adapters can retain labels that are outside a closed physical schema. + pub fn row_source(&self, id: NodeId) -> Option { + match self.nodes.get(&id)? { + Node::Input(_) => Some(id), + Node::Operator { inputs, operator } => { + let index = operator.row_preserving_input()?; + self.row_source(*inputs.get(index)?) + } + } + } + + /// Selected operator name, for plan inspection without decoding its wire format. + /// Certified candidate pruning checks authoritative-key coverage inside this operator. + pub fn certified_pruning_keys(&self, id: NodeId) -> Option<&[(usize, usize)]> { + match self.nodes.get(&id)? { + Node::Operator { operator, .. } => operator.certified_pruning_keys(), + Node::Input(_) => None, + } + } + pub fn operator_name(&self, id: NodeId) -> Option<&str> { + match self.nodes.get(&id)? { + Node::Input(_) => Some("Input"), + Node::Operator { operator, .. } => Some(operator.name()), + } + } + + pub fn roots(&self) -> &[NodeId] { + &self.roots + } + pub fn input_contracts(&self) -> impl Iterator { + self.nodes.iter().filter_map(|(&id, node)| match node { + Node::Input(contract) => Some((id, contract)), + Node::Operator { .. } => None, + }) + } + /// Derive a reachable output contract without opening deployment readers. + pub fn output_contract(&self, id: NodeId) -> Result { + let sources = self + .input_contracts() + .map(|(id, contract)| (id, Box::new(contract.clone()) as Source<'_>)) + .collect(); + let graph = self.instantiate(sources)?; + let properties = graph.properties(&self.roots)?; + let properties = *properties + .get(&id) + .ok_or_else(|| invalid("output is not reachable"))?; + let schema = match self + .nodes + .get(&id) + .ok_or_else(|| invalid("missing output"))? + { + Node::Input(contract) => contract.schema.clone(), + Node::Operator { operator, .. } => operator.output_schema(), + }; + Ok(InputContract { schema, properties }) + } + /// Validate using contract-only sources. No deployment reader is available. + pub fn validate(&self) -> Result<(), Error> { + let sources = self + .input_contracts() + .map(|(id, c)| (id, Box::new(c.clone()) as Source<'_>)) + .collect(); + self.instantiate(sources).map(|_| ()) + } + /// Resolve exactly the declared inputs and validate before any source starts. + pub fn instantiate<'a>( + &self, + mut sources: BTreeMap>, + ) -> Result, Error> { + let mut graph = PhysicalDag::default(); + for (&id, node) in &self.nodes { + match node { + Node::Input(contract) => { + let source = sources + .remove(&id) + .ok_or_else(|| invalid(format!("missing physical input {id}")))?; + let actual = source.properties(&[]); + if !source.input_schemas().is_empty() + || source.output_schema() != contract.schema + || (contract.properties.boundedness != Boundedness::Unknown + && actual.boundedness != contract.properties.boundedness) + || (contract.properties.emission != Emission::Unknown + && actual.emission != contract.properties.emission) + { + return Err(invalid(format!( + "physical input {id} violates its compiled contract" + ))); + } + graph.add_boxed( + id, + vec![], + Box::new(CheckedSource { + source, + output: contract.schema.clone(), + }), + )?; + } + Node::Operator { inputs, operator } => { + graph.add(id, inputs.clone(), operator.clone())?; + } + } + } + if !sources.is_empty() { + return Err(invalid("unexpected physical input binding")); + } + graph.validate(&self.roots)?; + Ok(graph) + } +} +impl PhysicalOperator for InputContract { + fn name(&self) -> &str { + "UnresolvedInput" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.schema.clone() + } + fn properties(&self, _: &[PlanProperties]) -> PlanProperties { + self.properties + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + _: Vec>, + _: crate::runtime::RunContext, + ) -> Result, Error> { + Err(invalid("physical input must be resolved before execution")) + } +} diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs new file mode 100644 index 00000000..112e1715 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -0,0 +1,894 @@ +//! Compile logical computation to native operators with typed external inputs. +//! Compilation needs no readers; deployment resolves inputs after selection. +use crate::operators::ReadoutQuery; +use crate::summary_kernels::exact::ExactReadout; +use crate::{ + operators::{Expression, Operator, Reduction, SortKey}, + plan::{Boundedness, Emission, NodeId, PhysicalDag, PhysicalOperator, PlanProperties}, + values::{Batch, Schema}, + Error, +}; +use planner_types::{ + post_asap::{ + ExactOperation, PostAsapDag, PostAsapDagNode, PostAsapOperatorPayload as Payload, + SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, + }, + pre_asap::{ + AggIntent, ColumnRef, CompareOpKind, GroupKeys, QueryExpr, Reduction as PlannerReduction, + }, +}; +use std::{ + collections::{BTreeMap, BTreeSet}, + sync::Arc, +}; +fn invalid(message: impl Into) -> Error { + Error::Invalid(message.into()) +} + +/// Source nodes cut the DAG at an installed storage/ingestion frontier. The +/// binding must have exactly the declared schema and no upstream dependencies. +/// A deployment must authorize these frontiers before calling this function. +pub type Source<'a> = Box + 'a>; + +pub mod precompute; +pub mod promql_rows; +pub mod promql_values; + +mod candidates; +pub use candidates::{ + compile_candidate, compile_candidates, enumerate_frontiers, select_candidate, CandidateCost, + CandidateSelection, PhysicalCandidate, +}; + +mod compiled; +pub use compiled::{CompiledPhysicalDag, InputContract}; + +/// Compile computation without opening or retaining deployment readers. +/// Input contracts identify explicit boundaries selected by maintenance planning. +pub fn compile( + dag: &PostAsapDag, + inputs: BTreeMap, + roots: &[NodeId], +) -> Result { + compile_internal(dag, inputs, roots) +} + +/// Convenience for callers that already resolved inputs. Lowering still uses +/// only their contracts, and instantiation checks those contracts again. +pub fn bind<'a>( + dag: &PostAsapDag, + sources: BTreeMap>, + roots: &[NodeId], +) -> Result, Error> { + let inputs = sources + .iter() + .map(|(&id, source)| (id, InputContract::from_source(source.as_ref()))) + .collect(); + compile(dag, inputs, roots)?.instantiate(sources) +} + +/// Resolve raw scan connectors before invoking the reader-independent compiler. +pub fn bind_with_data_sources<'a>( + dag: &PostAsapDag, + mut sources: BTreeMap>, + roots: &[NodeId], + data_sources: &crate::sources::DataSources, +) -> Result, Error> { + // Only resolve scans reachable below the selected input boundaries. + let mut pending = roots.to_vec(); + let mut seen = BTreeSet::new(); + while let Some(id) = pending.pop() { + if !seen.insert(id) || sources.contains_key(&id) { + continue; + } + let node = dag + .nodes + .iter() + .find(|n| u64::from(n.id.0) == id) + .ok_or_else(|| invalid(format!("missing node {id}")))?; + if let Payload::Fallback { + expression: expression @ QueryExpr::Scan { .. }, + } = &node.payload + { + sources.insert(id, Box::new(data_sources.bind(expression)?)); + } else { + pending.extend( + dag.edges + .iter() + .filter(|e| u64::from(e.consumer.0) == id) + .map(|e| u64::from(e.producer.0)), + ); + } + } + bind(dag, sources, roots) +} + +fn compile_internal( + dag: &PostAsapDag, + mut sources: BTreeMap, + roots: &[NodeId], +) -> Result { + preflight_depth(dag)?; + dag.validate().map_err(|e| invalid(e.to_string()))?; + let nodes = dag + .nodes + .iter() + .map(|node| (u64::from(node.id.0), node)) + .collect::>(); + let mut dependencies = BTreeMap::>::new(); + // Binary input order is semantic; serialized edge order is not. + let mut edges = dag.edges.iter().collect::>(); + edges.sort_by_key(|edge| { + ( + edge.consumer.0, + match edge.role { + planner_types::post_asap::EdgeRole::Left => 0, + planner_types::post_asap::EdgeRole::Input => 1, + planner_types::post_asap::EdgeRole::Right => 2, + }, + ) + }); + for edge in edges { + dependencies + .entry(u64::from(edge.consumer.0)) + .or_default() + .push(u64::from(edge.producer.0)); + } + if sources.keys().any(|id| !nodes.contains_key(id)) { + return Err(invalid("source binding names an unknown node")); + } + let mut ordered = Vec::new(); + let mut seen = BTreeSet::new(); + let mut pending = roots.iter().map(|&id| (id, false)).collect::>(); + while let Some((id, expanded)) = pending.pop() { + if expanded { + ordered.push(id); + continue; + } + if !seen.insert(id) { + continue; + } + if !nodes.contains_key(&id) { + return Err(invalid(format!("missing root {id}"))); + } + pending.push((id, true)); + if !sources.contains_key(&id) { + for &input in dependencies.get(&id).into_iter().flatten() { + pending.push((input, false)); + } + } + } + let mut graph = CompiledPhysicalDag::new(roots.to_vec()); + let mut auxiliary = u64::MAX; + for id in ordered { + let node = nodes[&id]; + let output = Arc::new(node.output_schema.clone()); + crate::values::validate_schema(&output)?; + if let Some(source) = sources.remove(&id) { + if source.schema != output { + return Err(invalid("frontier does not have the declared schema")); + } + graph.add_input(id, source)?; + } else { + let mut inputs = dependencies.get(&id).cloned().unwrap_or_default(); + let mut schemas = inputs + .iter() + .map(|id| Arc::new(nodes[id].output_schema.clone())) + .collect::>(); + if matches!(node.payload, Payload::SummaryMerge) && inputs.len() > 1 { + if schemas.iter().any(|s| s != &schemas[0]) { + return Err(invalid("summary merge inputs have different schemas")); + } + graph.add( + auxiliary, + inputs, + Operator::union(schemas[0].clone(), schemas.len())?, + )?; + inputs = vec![auxiliary]; + auxiliary -= 1; + schemas.truncate(1); + } + if let Payload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &node.payload + { + use planner_types::post_asap::maintained_population::PopulationInput; + let PopulationInput::CurrentSeries(spec) = &population.input else { + return Err(invalid( + "native maintained population requires a current-series input", + )); + }; + let [input] = schemas.as_slice() else { + return Err(invalid("current-series population requires one input")); + }; + if spec.without { + return Err(invalid( + "dynamic without grouping requires label-set projection", + )); + } + let identity = named_column( + input, + &ColumnRef::Named(promql_rows::SERIES_IDENTITY_COLUMN.into()), + )?; + let coordinate = input + .time_index + .ok_or_else(|| invalid("current-series input lacks timestamp"))?; + let value = named_column(input, &ColumnRef::SampleValue)?; + let lookback = i64::try_from(spec.lookback_ms) + .map_err(|_| invalid("current-series lookback overflows"))?; + graph.add( + id, + inputs, + Operator::current_series(input.clone(), identity, coordinate, value, lookback)? + .with_output_schema(output)?, + )?; + continue; + } + if let Payload::Value { + operation: ValueOperation::ReadPopulation { readout }, + } = &node.payload + { + use planner_types::post_asap::maintained_population::{ + PopulationInput, PopulationReadout, + }; + let PopulationReadout::TopK { k } = readout else { + return Err(invalid( + "native population readout does not support this operation", + )); + }; + let [producer] = inputs.as_slice() else { + return Err(invalid("population readout requires one input")); + }; + let Payload::Value { + operation: ValueOperation::MaintainPopulation { population }, + } = &nodes[producer].payload + else { + return Err(invalid( + "population readout requires its declared population", + )); + }; + let PopulationInput::CurrentSeries(spec) = &population.input else { + return Err(invalid("current-series population required")); + }; + if spec.without { + return Err(invalid( + "dynamic without ranking requires label-set projection", + )); + } + let input = schemas[0].clone(); + let groups = spec + .grouping + .iter() + .map(|name| named_column(&input, &ColumnRef::Named(name.clone()))) + .collect::, _>>()?; + let value = named_column(&input, &ColumnRef::SampleValue)?; + graph.add( + auxiliary, + inputs, + Operator::sort( + input.clone(), + vec![SortKey { + column: value, + descending: true, + nulls_first: false, + }], + groups.clone(), + )?, + )?; + graph.add( + id, + vec![auxiliary], + Operator::limit(input, *k as u64, 0, groups)?.with_output_schema(output)?, + )?; + auxiliary -= 1; + continue; + } + // A closed row must include either all source labels or the explicit + // complete-label identity. Projected labels alone are insufficient. + if let Payload::SummaryAgg { + family, + input: update, + reduction: PlannerReduction::PerEntity, + grouping, + } = &node.payload + { + let [input_id] = inputs.as_slice() else { + return Err(invalid("per-entity summary requires one input")); + }; + let Payload::Fallback { + expression: QueryExpr::TimeRange { child, .. }, + } = &nodes[input_id].payload + else { + return Err(invalid( + "per-entity summary requires a resolved raw time range", + )); + }; + let QueryExpr::Scan { schema, .. } = child.as_ref() else { + return Err(invalid("per-entity summary requires a resolved source")); + }; + if !schema.closed || update.item.is_some() { + return Err(invalid( + "per-entity summary requires complete source identity", + )); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let SummaryInputExpr::Column(value) = &update.weight else { + return Err(invalid( + "per-entity update requires a projected value column", + )); + }; + let input = schemas[0].clone(); + let value = named_column(&input, value)?; + let coordinate = input + .time_index + .ok_or_else(|| invalid("temporal input lacks time"))?; + let groups = (0..input.fields.len()) + .filter(|&column| column != value && column != coordinate) + .collect(); + let build = Operator::summary_build( + input, + family.clone(), + value, + Some(coordinate), + groups, + )?; + let compact = build.schema(); + graph.add(auxiliary, inputs, build)?; + graph.add( + id, + vec![auxiliary], + Operator::scope_timestamp(compact, output)?, + )?; + auxiliary -= 1; + continue; + } + let mut operator = compile_node(node, &schemas) + .map_err(|error| invalid(format!("node {id}: {error}")))?; + if operator.is_counter_readout() { + let mut pending = vec![id]; + let mut visited = BTreeSet::new(); + let mut ranges = BTreeSet::new(); + while let Some(ancestor) = pending.pop() { + if !visited.insert(ancestor) { + continue; + } + if let Payload::Fallback { + expression: QueryExpr::TimeRange { range, .. }, + } = &nodes[&ancestor].payload + { + ranges.insert( + i64::try_from(range.as_millis()) + .map_err(|_| invalid("counter lookback exceeds Int64"))?, + ); + continue; + } + pending.extend(dependencies.get(&ancestor).into_iter().flatten().copied()); + } + if ranges.len() > 1 { + return Err(invalid("counter readout has ambiguous logical windows")); + } + if let Some(lookback) = ranges.into_iter().next() { + operator = operator.with_counter_lookback(lookback)?; + } + } + graph.add(id, inputs, operator)?; + } + } + graph.validate()?; + Ok(graph) +} + +/// Bind a Planner node against the schemas supplied by its deployment edges. +/// This is the same checked path used by complete DAG binding. +pub fn compile_node(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { + for schema in inputs { + crate::values::validate_schema(schema)?; + } + bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) +} + +fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { + if let Payload::Binary { operator } = &node.payload { + let [left, right] = inputs else { + return Err(invalid("binary requires two inputs")); + }; + if node.output_state.timing == planner_types::post_asap::ExecutionTiming::IngestionTime { + let value = |schema: &Schema| -> Result { + let columns = schema + .fields + .iter() + .enumerate() + .filter(|(_, field)| { + field.dtype + == SummaryFamilyType::Plain(planner_types::pre_asap::DataType::Float64) + }) + .map(|(i, _)| i) + .collect::>(); + match columns.as_slice() { + [value] => Ok(*value), + _ => Err(invalid("aligned binary requires one value column")), + } + }; + let (l, r) = (value(left)?, value(right)?); + let keys = left + .fields + .iter() + .enumerate() + .filter(|(i, _)| *i != l) + .map(|(i, field)| { + right + .fields + .iter() + .position(|other| other.name == field.name && other.dtype == field.dtype) + .map(|j| (i, j)) + .ok_or_else(|| invalid("aligned input identities differ")) + }) + .collect::, _>>()?; + return Operator::aligned_binary( + left.clone(), + right.clone(), + keys, + (l, r), + operator.clone(), + ); + } + return Operator::vector_binary(left.clone(), right.clone(), operator.clone(), false); + } + if let Payload::RelationalJoin { + join_kind, + pred, + pruning, + } = &node.payload + { + use planner_types::{post_asap::CandidateCompleteness, pre_asap::JoinKind}; + if pruning.is_some() && *join_kind != JoinKind::Semi { + return Err(invalid("pruning certificate requires a semi-join")); + } + if matches!(pruning,Some(CandidateCompleteness::Certified { guarantee }) if guarantee.has_unknown() || guarantee.metric != planner_types::post_asap::ErrorMetric::TopKMembership) + { + return Err(invalid("invalid pruning certificate")); + } + let [left, right] = inputs else { + return Err(invalid("join requires two inputs")); + }; + if *join_kind == JoinKind::Semi { + if let Ok(keys) = equijoin_keys(pred, left, right) { + let operator = Operator::semi_join(left.clone(), right.clone(), keys)?; + return Ok( + if matches!(pruning, Some(CandidateCompleteness::Certified { .. })) { + operator.require_complete_right() + } else { + operator + }, + ); + } + } + if matches!(pruning, Some(CandidateCompleteness::Certified { .. })) { + return Err(invalid("certified pruning requires explicit equijoin keys")); + } + return Operator::relational_join( + left.clone(), + right.clone(), + join_kind.clone(), + pred, + Arc::new(node.output_schema.clone()), + ); + } + let [input] = inputs else { + return Err(invalid( + "native Planner binding currently requires a unary operation or an explicit source", + )); + }; + match &node.payload { + Payload::Value { operation, .. } => match operation { + ValueOperation::Project { cols, .. } => Operator::project( + input.clone(), + cols.iter() + .enumerate() + .map(|(i, col)| { + Ok(( + node.output_schema + .fields + .get(i) + .ok_or_else(|| invalid("projection width mismatch"))? + .name + .clone(), + match &col.expr { + QueryExpr::Column(index) => Expression::Column(*index), + expr => expression(expr, input)?, + }, + )) + }) + .collect::>()?, + ), + ValueOperation::Filter { pred } => { + Operator::filter(input.clone(), expression(&pred.0, input)?) + } + ValueOperation::Sort { keys, partition_by } => Operator::sort( + input.clone(), + keys.iter() + .map(|key| { + let QueryExpr::Column(column) = key.expr else { + return Err(invalid( + "sort expression must be projected before sorting", + )); + }; + Ok(SortKey { + column, + descending: !key.ascending, + nulls_first: key.nulls_first, + }) + }) + .collect::>()?, + groups(input, partition_by)?, + ), + ValueOperation::Limit { + n, + offset, + partition_by, + } => Operator::limit( + input.clone(), + *n as u64, + *offset as u64, + groups(input, partition_by)?, + ), + ValueOperation::Exact(ExactOperation::Aggregate { + reduction, + measures, + output_names, + having: None, + }) => { + if measures.len() != output_names.len() { + return Err(invalid("aggregate output names differ from measures")); + } + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid( + "per-entity aggregate requires an explicit entity binding", + )); + }; + let measures = measures + .iter() + .zip(output_names) + .map(|(m, name)| { + let column = |col: Option| { + col.map(Ok) + .unwrap_or_else(|| named_column(input, &ColumnRef::SampleValue)) + }; + let m = match m { + AggIntent::Count { .. } => Reduction::Count, + AggIntent::Sum { col } => Reduction::Sum(column(*col)?), + AggIntent::Avg { col } => Reduction::Avg(column(*col)?), + AggIntent::Min { col } => Reduction::Min(column(*col)?), + AggIntent::Max { col } => Reduction::Max(column(*col)?), + _ => { + return Err(invalid( + "aggregate intent has no native implementation", + )) + } + }; + Ok((name.clone(), m)) + }) + .collect::>()?; + Operator::aggregate(input.clone(), groups(input, keys)?, measures) + } + ValueOperation::FinalizeExactAccumulator => { + let state = summary_column(input)?; + use crate::Statistic as S; + use planner_types::post_asap::ExactKind as E; + let statistic = match &input.fields[state].dtype { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { + E::Sum => S::Sum, + E::Count => S::Count, + E::Min => S::Min, + E::Max => S::Max, + E::Rate => S::Rate, + E::Increase => S::Increase, + _ => return Err(invalid("exact family readout is unsupported")), + }, + _ => return Err(invalid("exact finalization requires exact state")), + }; + Operator::readout( + input.clone(), + state, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + ) + } + _ => Err(invalid("value operation has no native implementation")), + }, + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + } => { + if let Some(item) = &update.item { + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid("keyed summary requires explicit partitions")); + }; + let SummaryInputExpr::Column(weight) = &update.weight else { + return Err(invalid( + "keyed summary weight must be a finalized value column", + )); + }; + if matches!(family, SummaryFamilyType::Sketch(kind, _) if kind.algorithm() == &planner_types::post_asap::SketchAlgorithm::CmsWithHeap) + && !matches!( + update.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { .. } + ) + { + return Err(invalid("CMS requires a nonnegative weight contract")); + } + fn columns( + expr: &SummaryInputExpr, + input: &Schema, + result: &mut Vec, + ) -> Result<(), Error> { + match expr { + SummaryInputExpr::Column(column) => { + result.push(named_column(input, column)?) + } + SummaryInputExpr::Tuple(items) => { + for item in items { + columns(item, input, result)?; + } + } + _ => return Err(invalid("keyed summary needs explicit item columns")), + } + Ok(()) + } + let mut items = Vec::new(); + columns(item, input, &mut items)?; + return Operator::keyed_summary_build( + input.clone(), + family.clone(), + named_column(input, weight)?, + items, + groups(input, keys)?, + ); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let SummaryInputExpr::Column(column) = &update.weight else { + return Err(invalid( + "summary update expression must be projected to a column", + )); + }; + let PlannerReduction::Reduce(keys) = reduction else { + return Err(invalid( + "summary construction requires explicit grouping columns", + )); + }; + Operator::summary_build( + input.clone(), + family.clone(), + named_column(input, column)?, + input.time_index, + groups(input, keys)?, + ) + } + Payload::SummaryMerge => { + let state = summary_column(input)?; + Operator::summary_merge( + input.clone(), + state, + (0..input.fields.len()) + .filter(|&i| i != state && Some(i) != input.time_index) + .collect(), + ) + } + Payload::SummaryEstimate { query } => { + if let SketchQuery::TopK { k } = query { + return Operator::keyed_readout( + input.clone(), + summary_column(input)?, + *k, + Arc::new(node.output_schema.clone()), + ); + } + Operator::readout( + input.clone(), + summary_column(input)?, + ReadoutQuery::Sketch(query.clone()), + ) + } + _ => Err(invalid( + "physical operation has no native binding; no fallback is installed", + )), + } +} +fn summary_column(input: &Schema) -> Result { + let columns = input + .fields + .iter() + .enumerate() + .filter(|(_, f)| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .map(|(i, _)| i) + .collect::>(); + match columns.as_slice() { + [column] => Ok(*column), + _ => Err(invalid("one summary state column required")), + } +} +fn named_column(input: &Schema, column: &ColumnRef) -> Result { + let name = match column { + // Executable SummarySchema retains column names, not table qualifiers. + // Frontend binding has resolved the qualifier; still reject ambiguous + // names here rather than guessing a join side. + ColumnRef::Named(name) | ColumnRef::Qualified { name, .. } => name.as_str(), + ColumnRef::SampleValue => "value", + _ => { + return Err(invalid( + "summary update requires an unambiguous bound column", + )) + } + }; + let matches = input + .fields + .iter() + .enumerate() + .filter(|(_, field)| field.name == name) + .map(|(i, _)| i) + .collect::>(); + match matches.as_slice() { + [column] => Ok(*column), + _ => Err(invalid("summary update column missing or ambiguous")), + } +} +fn groups(input: &Schema, groups: &GroupKeys) -> Result, Error> { + if groups.is_without() { + return Err(invalid("grouping without requires resolved label columns")); + } + if groups.keys().iter().any(|&i| i >= input.fields.len()) { + return Err(invalid("grouping column out of range")); + } + Ok(groups.keys().to_vec()) +} +fn expression(expr: &QueryExpr, input: &Schema) -> Result { + Ok(Expression::planner( + crate::expressions::CompiledExpression::compile(expr, input)?, + )) +} + +struct CheckedSource<'a> { + source: Source<'a>, + output: Schema, +} +impl PhysicalOperator for CheckedSource<'_> { + fn properties(&self, inputs: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { + self.source.properties(inputs) + } + + fn name(&self) -> &str { + self.source.name() + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, batch: &Batch) -> usize { + self.source.output_bytes(batch) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: crate::runtime::RunContext, + ) -> Result, Error> { + use futures::StreamExt; + Ok(self + .source + .start(inputs, context)? + .map(|batch| { + let batch = batch?; + if batch.schema() != &self.output { + return Err(invalid("source batch differs from its bound schema")); + } + Ok(batch) + }) + .boxed_local()) + } +} + +// Bound recursion before invoking the upstream recursive provenance validator. +fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { + let mut remaining = dag + .nodes + .iter() + .map(|node| (node.id, 0usize)) + .collect::>(); + if remaining.len() != dag.nodes.len() { + return Err(invalid("duplicate Planner node")); + } + let mut consumers = BTreeMap::<_, Vec<_>>::new(); + for edge in &dag.edges { + if !remaining.contains_key(&edge.producer) { + return Err(invalid("missing Planner edge producer")); + } + *remaining + .get_mut(&edge.consumer) + .ok_or_else(|| invalid("missing Planner edge consumer"))? += 1; + consumers + .entry(edge.producer) + .or_default() + .push(edge.consumer); + } + let mut ready = remaining + .iter() + .filter(|(_, n)| **n == 0) + .map(|(id, _)| *id) + .collect::>(); + let mut depths = BTreeMap::new(); + let mut visited = 0; + while let Some(id) = ready.pop_front() { + visited += 1; + let depth = *depths.get(&id).unwrap_or(&1usize); + if depth > 128 { + return Err(invalid("DAG exceeds the supported execution depth of 128")); + } + for &consumer in consumers.get(&id).into_iter().flatten() { + let next = depths.entry(consumer).or_insert(1); + *next = (*next).max(depth + 1); + let count = remaining.get_mut(&consumer).expect("validated endpoint"); + *count -= 1; + if *count == 0 { + ready.push_back(consumer); + } + } + } + if visited != dag.nodes.len() { + return Err(invalid("Planner DAG contains a cycle")); + } + Ok(()) +} + +/// Join predicates address the concatenated left/right schema. +fn semi_join_keys( + expr: &QueryExpr, + left: usize, + right: usize, + keys: &mut Vec<(usize, usize)>, +) -> Result<(), Error> { + match expr { + QueryExpr::BoolAnd(parts) => { + for part in parts { + semi_join_keys(part, left, right, keys)?; + } + } + QueryExpr::Compare { + left: a, + op: CompareOpKind::Eq, + right: b, + } => { + let (QueryExpr::Column(a), QueryExpr::Column(b)) = (a.as_ref(), b.as_ref()) else { + return Err(invalid("semi-join requires column equality keys")); + }; + let (a, b) = if a < b { (*a, *b) } else { (*b, *a) }; + if a >= left || b < left || b >= left + right { + return Err(invalid("semi-join key must match left to right")); + } + keys.push((a, b - left)); + } + _ => return Err(invalid("unsupported semi-join predicate")), + } + Ok(()) +} + +/// Resolve equality keys against the Planner join's concatenated input schema. +/// Deployments may use these positions to bind their source columns. +pub fn equijoin_keys( + pred: &planner_types::pre_asap::Predicate, + left: &planner_types::post_asap::SummarySchema, + right: &planner_types::post_asap::SummarySchema, +) -> Result, Error> { + let mut keys = Vec::new(); + semi_join_keys(&pred.0, left.fields.len(), right.fields.len(), &mut keys)?; + if keys.is_empty() { + return Err(invalid("semi-join requires explicit matching keys")); + } + Ok(keys) +} diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs new file mode 100644 index 00000000..063bd0b8 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -0,0 +1,384 @@ +//! Compile immutable summary-input computation with explicit population and pane identity. +use super::*; +use planner_types::{ + post_asap::{ExecutionTiming, GroupingStrategy, SummarySchema}, + pre_asap::DataType, +}; + +/// Physical rows carry the population and pane coordinate alongside the logical value. +/// These fields preserve identities which are implicit in a stored summary instance. +pub fn population_schema(family: SummaryFamilyType) -> Schema { + Arc::new(SummarySchema { + fields: vec![ + planner_types::post_asap::SummaryField { + name: "$population".into(), + dtype: SummaryFamilyType::Plain(DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }), + nullable: false, + }, + planner_types::post_asap::SummaryField { + name: "$window_end".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }, + planner_types::post_asap::SummaryField { + name: "value".into(), + dtype: family, + nullable: false, + }, + ], + time_index: Some(1), + }) +} + +/// Validate the adapter layout during installed-plan recovery without lowering operators. +pub fn source_schema(logical: &SummarySchema) -> Result { + let states = logical + .fields + .iter() + .filter(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + .collect::>(); + let [state] = states.as_slice() else { + return Err(invalid( + "stored population requires one typed summary state", + )); + }; + if logical.fields.iter().enumerate().any(|(i, field)| matches!(&field.dtype, SummaryFamilyType::Plain(dtype) + if field.nullable || !matches!(dtype, DataType::Utf8) && !(Some(i) == logical.time_index && *dtype == DataType::Timestamp))) { + return Err(invalid("stored population metadata cannot reconstruct extra value columns")); + } + if state.nullable { + return Err(invalid("stored population state cannot be null")); + } + Ok(population_schema(state.dtype.clone())) +} + +pub fn is_population_schema(schema: &Schema) -> bool { + schema + .fields + .get(2) + .is_some_and(|field| *schema == population_schema(field.dtype.clone())) +} + +/// Compile a complete selected precompute sub-DAG. Inputs are already-computed +/// state boundaries; the deployment supplies groups, panes and states, never operations. +pub fn compile( + dag: &PostAsapDag, + frontiers: &[NodeId], + roots: &[NodeId], +) -> Result { + preflight_depth(dag)?; + dag.validate().map_err(|e| invalid(e.to_string()))?; + let nodes = dag + .nodes + .iter() + .map(|n| (u64::from(n.id.0), n)) + .collect::>(); + let frontier = frontiers.iter().copied().collect::>(); + if frontier.len() != frontiers.len() || roots.iter().any(|r| frontier.contains(r)) { + return Err(invalid( + "precompute boundaries must be distinct from outputs", + )); + } + let mut dependencies = BTreeMap::>::new(); + let mut edges = dag.edges.iter().collect::>(); + edges.sort_by_key(|edge| { + ( + edge.consumer.0, + match edge.role { + planner_types::post_asap::EdgeRole::Left => 0, + planner_types::post_asap::EdgeRole::Input => 1, + planner_types::post_asap::EdgeRole::Right => 2, + }, + ) + }); + for edge in edges { + dependencies + .entry(u64::from(edge.consumer.0)) + .or_default() + .push(u64::from(edge.producer.0)); + } + let mut ordered = Vec::new(); + let mut seen = BTreeSet::new(); + let mut pending = roots.iter().map(|&id| (id, false)).collect::>(); + while let Some((id, expanded)) = pending.pop() { + if expanded { + ordered.push(id); + continue; + } + if !seen.insert(id) { + continue; + } + if !nodes.contains_key(&id) { + return Err(invalid("missing precompute node")); + } + pending.push((id, true)); + if !frontier.contains(&id) { + pending.extend( + dependencies + .get(&id) + .into_iter() + .flatten() + .map(|id| (*id, false)), + ); + } + } + let mut sources = BTreeMap::new(); + let mut fragments = BTreeMap::new(); + let mut outputs = BTreeMap::::new(); + for id in ordered { + let node = nodes[&id]; + if frontier.contains(&id) { + let schema = source_schema(&node.output_schema)?; + sources.insert(id, InputContract::bounded(schema.clone())); + outputs.insert(id, schema); + continue; + } + if node.output_state.timing != ExecutionTiming::IngestionTime { + return Err(invalid("precompute graph contains a query-time operation")); + } + let inputs = dependencies.get(&id).cloned().unwrap_or_default(); + let schemas = inputs + .iter() + .map(|id| { + outputs + .get(id) + .cloned() + .ok_or_else(|| invalid("missing precompute input")) + }) + .collect::, _>>()?; + let graph = fragment( + node, + &schemas, + &inputs.iter().map(|id| nodes[id]).collect::>(), + )?; + outputs.insert(id, graph.output_contract(graph.roots()[0])?.schema); + fragments.insert(id, (inputs, graph)); + } + CompiledPhysicalDag::compose(sources, fragments, roots.to_vec()) +} + +fn validate_value_output(node: &PostAsapDagNode) -> Result<(), Error> { + let schema = &node.output_schema; + let values = schema + .fields + .iter() + .enumerate() + .filter(|(i, _)| Some(*i) != schema.time_index) + .collect::>(); + if !matches!(values.as_slice(), [(_, field)] if !field.nullable && field.dtype == SummaryFamilyType::Plain(DataType::Float64)) + || schema.time_index.is_some_and(|i| { + schema.fields.get(i).is_none_or(|f| { + f.nullable || f.dtype != SummaryFamilyType::Plain(DataType::Timestamp) + }) + }) + { + return Err(invalid( + "precompute value schema requires Float64 and an optional declared timestamp", + )); + } + Ok(()) +} + +fn fragment( + node: &PostAsapDagNode, + schemas: &[Schema], + parents: &[&PostAsapDagNode], +) -> Result { + let sources = schemas + .iter() + .enumerate() + .map(|(id, schema)| (id as u64, InputContract::bounded(schema.clone()))) + .collect(); + let mut operators = BTreeMap::new(); + let mut next = schemas.len() as u64; + let mut add = |inputs: Vec, op: Operator| -> Result { + let id = next; + next += 1; + operators.insert(id, (inputs, op)); + Ok(id) + }; + let root = match &node.payload { + Payload::Binary { operator } => { + validate_value_output(node)?; + if node.output_schema.time_index.is_none() + || parents.iter().any(|p| p.output_schema.time_index.is_none()) + { + return Err(invalid( + "precompute binary requires declared window timestamps", + )); + } + let [left, right] = schemas else { + return Err(invalid("precompute binary requires two inputs")); + }; + add( + vec![0, 1], + Operator::aligned_binary( + left.clone(), + right.clone(), + vec![(0, 0), (1, 1)], + (2, 2), + operator.clone(), + )?, + )? + } + Payload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + } => { + let [input] = schemas else { + return Err(invalid("finalize requires one state input")); + }; + validate_value_output(node)?; + let statistic = match &input.fields[2].dtype { + SummaryFamilyType::ExactAggregate(planner_types::post_asap::ExactKind::Sum, _) => { + crate::Statistic::Sum + } + SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Count, + _, + ) => crate::Statistic::Count, + _ => { + return Err(invalid( + "precompute finalization requires explicit Sum or Count semantics", + )) + } + }; + let read = Operator::readout( + input.clone(), + 2, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + )?; + let output = read.schema(); + let read = add(vec![0], read)?; + let project = Operator::project( + output, + vec![ + ("$population".into(), Expression::Column(0)), + ("$window_end".into(), Expression::Column(1)), + ( + "value".into(), + Expression::FiniteFloat64(Box::new(Expression::ExactFloat64(2))), + ), + ], + )? + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; + add(vec![read], project)? + } + Payload::SummaryAgg { + family, + input: update, + reduction, + grouping, + } => { + let [input] = schemas else { + return Err(invalid("summary update requires one input")); + }; + if update.item.is_some() + || !matches!(grouping, GroupingStrategy::PerSubpopulationInstance) + { + return Err(invalid( + "precompute keyed/shared update needs its dedicated physical candidate", + )); + } + crate::capability::validate_summary_kernel(family, update, grouping) + .map_err(Error::Invalid)?; + let labels = match reduction { + PlannerReduction::PerEntity => Expression::Column(0), + PlannerReduction::Reduce(keys) => Expression::LabelSet { + column: 0, + labels: keys + .keys() + .iter() + .map(|key| { + parents[0] + .output_schema + .fields + .get(*key) + .filter(|field| { + !field.nullable + && field.dtype == SummaryFamilyType::Plain(DataType::Utf8) + }) + .map(|f| f.name.clone()) + .ok_or_else(|| { + invalid("summary grouping must identify population labels") + }) + }) + .collect::, _>>()?, + without: keys.is_without(), + }, + }; + let weight = match &update.weight { + SummaryInputExpr::Constant(value) => Expression::Literal { + value: crate::values::Value::Float64(*value), + dtype: DataType::Float64, + }, + SummaryInputExpr::Column(ColumnRef::SampleValue) => Expression::Column(2), + SummaryInputExpr::Column(ColumnRef::Named(name)) + if parents[0].output_schema.fields.iter().any(|f| { + f.name == *name && f.dtype == SummaryFamilyType::Plain(DataType::Float64) + }) => + { + Expression::Column(2) + } + _ => { + return Err(invalid( + "summary weight does not resolve to the input value", + )) + } + }; + let project = Operator::project( + input.clone(), + vec![ + ("$population".into(), labels), + ("$window_end".into(), Expression::Column(1)), + ("value".into(), Expression::FiniteFloat64(Box::new(weight))), + ], + )? + .with_output_schema(population_schema(SummaryFamilyType::Plain( + DataType::Float64, + )))?; + let projected = project.schema(); + let project = add(vec![0], project)?; + let build = Operator::summary_build(projected, family.clone(), 2, Some(1), vec![0])?; + let built = build.schema(); + let build = add(vec![project], build)?; + add( + vec![build], + Operator::scope_timestamp(built, population_schema(family.clone()))?, + )? + } + Payload::SummaryMerge => { + let Some(input) = schemas.first() else { + return Err(invalid("summary merge requires inputs")); + }; + if schemas.iter().any(|s| s != input) { + return Err(invalid("summary merge inputs differ")); + } + let union = add( + (0..schemas.len() as u64).collect(), + Operator::union(input.clone(), schemas.len())?, + )?; + let merge = Operator::summary_merge(input.clone(), 2, vec![0])?; + let merged = merge.schema(); + let merge = add(vec![union], merge)?; + add( + vec![merge], + Operator::scope_timestamp(merged, input.clone())?, + )? + } + _ => { + return Err(invalid( + "precompute operation has no native population implementation", + )) + } + }; + CompiledPhysicalDag::from_operators(sources, operators, vec![root]) +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs new file mode 100644 index 00000000..cb680096 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -0,0 +1,389 @@ +//! A bounded PromQL source row carries the entire label set, not just labels +//! mentioned by the query. The source adapter owns this lossless encoding. +use super::*; +use planner_types::pre_asap::{Column, DataType, Source as LogicalSource}; +use std::rc::Rc; + +/// Not a legal PromQL label name, so it cannot shadow a user label. +pub use planner_types::pre_asap::schema::PROMQL_SERIES_IDENTITY as SERIES_IDENTITY_COLUMN; + +/// Canonical, reversible identity. JSON object encoding preserves label names, +/// empty values and escaping; sorting makes ingestion order irrelevant. +pub fn encode_series_identity(labels: &BTreeMap) -> Result { + serde_json::to_string(labels).map_err(|error| invalid(error.to_string())) +} + +pub fn decode_series_identity(encoded: &str) -> Result, Error> { + let labels: BTreeMap = + serde_json::from_str(encoded).map_err(|error| invalid(error.to_string()))?; + if encode_series_identity(&labels)? != encoded { + return Err(invalid("series identity is not canonically encoded")); + } + Ok(labels) +} + +/// Resolve the row representation before candidate search. `closed` describes +/// physical columns here: the final column contains every dynamic source label. +/// It does not assert that the query's projected labels are the full label set. +/// +/// This realization supports explicit `by` grouping and per-series computation. +/// Operators that rewrite or implicitly match dynamic label sets require their +/// own realization; they must not accidentally treat the opaque identity as a +/// user label or silently discard it. +pub fn with_series_identity(root: &QueryExpr) -> Result { + let mut root = root.clone(); + fn visit(node: &mut QueryExpr) -> Result<(), Error> { + use planner_types::pre_asap::Reduction; + match node { + QueryExpr::Scan { + source: LogicalSource::TimeSeries { .. }, + schema, + .. + } => { + if schema + .columns + .iter() + .any(|column| column.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "source already contains a physical series identity", + )); + } + if schema.closed { + return Err(invalid( + "dynamic series identity requires an open PromQL source", + )); + } + schema + .columns + .push(Column::new(SERIES_IDENTITY_COLUMN, DataType::Utf8, false)); + schema.closed = true; + Ok(()) + } + QueryExpr::TimeRange { child, .. } | QueryExpr::Limit { child, .. } => { + visit(Rc::make_mut(child)) + } + QueryExpr::Aggregate { + child, reduction, .. + } => { + if matches!(reduction, Reduction::Reduce(keys) if keys.is_without()) { + return Err(invalid( + "dynamic without grouping requires label-set projection", + )); + } + visit(Rc::make_mut(child)) + } + QueryExpr::Sort { + child, + partition_by, + .. + } => { + if partition_by.is_without() { + return Err(invalid( + "dynamic without ranking requires label-set projection", + )); + } + visit(Rc::make_mut(child)) + } + _ => Err(invalid( + "operator has no dynamic series-identity realization", + )), + } + } + visit(&mut root)?; + root.output_schema() + .map_err(|error| invalid(error.to_string()))?; + Ok(root) +} + +/// Construct source rows only from full identities. The named label columns +/// are projections of that same identity and cannot independently redefine it. +pub fn series_row( + schema: &Schema, + labels: &BTreeMap, + timestamp: i64, + value: f64, +) -> Result, Error> { + use crate::values::Value; + let identity = encode_series_identity(labels)?; + let mut found = false; + let row = schema + .fields + .iter() + .enumerate() + .map(|(index, field)| { + if field.name == SERIES_IDENTITY_COLUMN { + if field.dtype != SummaryFamilyType::Plain(DataType::Utf8) + || field.nullable + || found + { + return Err(invalid("invalid series identity column")); + } + found = true; + Ok(Value::Utf8(identity.clone().into())) + } else if Some(index) == schema.time_index { + Ok(Value::Timestamp(timestamp)) + } else if field.name == "value" + && field.dtype == SummaryFamilyType::Plain(DataType::Float64) + { + Ok(Value::Float64(value)) + } else if field.dtype == SummaryFamilyType::Plain(DataType::Utf8) { + Ok(labels.get(&field.name).map_or_else( + || Value::Utf8("".into()), + |value| Value::Utf8(value.clone().into()), + )) + } else { + Err(invalid("unsupported PromQL source column")) + } + }) + .collect::, _>>()?; + if !found { + return Err(invalid("source lacks its full series identity")); + } + Ok(row) +} + +/// Compile the selected TopK computation above an existing maintained-population +/// source. The boundary supplies the complete eligible vector, not a truncated +/// TopK result; ranking remains a native physical operator. +pub fn compile_current_series_readout( + selected: &Rc, +) -> Result { + use planner_types::post_asap::{ + compile_post_asap_dag, maintained_population::PopulationReadout, SummaryField, + }; + let mut dag = compile_post_asap_dag(selected).map_err(|error| invalid(error.to_string()))?; + // Typed snapshot candidates already carry full identity throughout the DAG. + // Cut at the population output, preserving all selected heap/readout nodes. + let populations = dag.nodes.iter().filter(|node| matches!(&node.payload, + Payload::Value { operation: ValueOperation::MaintainPopulation { population } } + if matches!(population.input, planner_types::post_asap::maintained_population::PopulationInput::CurrentSeries(_)) + )).collect::>(); + if let [population] = populations.as_slice() { + if population + .output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return compile( + &dag, + BTreeMap::from([( + u64::from(population.id.0), + InputContract::bounded(Arc::new(population.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + ); + } + } + if dag.nodes.len() != 3 + || !dag.nodes.iter().any(|node| { + node.id == dag.root + && matches!( + node.payload, + Payload::Value { + operation: ValueOperation::ReadPopulation { + readout: PopulationReadout::TopK { .. } + } + } + ) + }) + { + return Err(invalid( + "expected one selected current-series TopK computation", + )); + } + let mut frontier = None; + for node in &mut dag.nodes { + match &mut node.payload { + Payload::Fallback { expression } => { + *expression = with_series_identity(expression)?; + } + Payload::Value { + operation: ValueOperation::MaintainPopulation { .. }, + } => { + frontier = Some(u64::from(node.id.0)); + } + Payload::Value { + operation: + ValueOperation::ReadPopulation { + readout: PopulationReadout::TopK { .. }, + }, + } => {} + _ => return Err(invalid("unsupported current-series readout dependency")), + } + if node + .output_schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "current-series input already has a physical identity column", + )); + } + node.output_schema.fields.push(SummaryField { + name: SERIES_IDENTITY_COLUMN.into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }); + } + for edge in &mut dag.edges { + edge.intermediate_schema = dag + .nodes + .iter() + .find(|node| node.id == edge.producer) + .unwrap() + .output_schema + .clone(); + } + let frontier = frontier.ok_or_else(|| invalid("missing current-series population"))?; + let schema = Arc::new( + dag.nodes + .iter() + .find(|node| u64::from(node.id.0) == frontier) + .unwrap() + .output_schema + .clone(), + ); + compile( + &dag, + BTreeMap::from([(frontier, InputContract::bounded(schema))]), + &[u64::from(dag.root.0)], + ) +} + +/// Compile selected ranking or aggregation above an exact per-series Rate +/// readout. Deployments bind complete window readouts at this boundary; +/// the heap is rebuilt independently for each evaluation. This does not move +/// that frontier to ingestion time or authorize combining finalized rates. +pub fn compile_rate_ranking( + selected: &Rc, +) -> Result< + ( + Rc, + CompiledPhysicalDag, + ), + Error, +> { + use planner_types::post_asap::{ + compile_post_asap_dag_with_node_ids, ExactKind, SummaryExpr, SummaryNode, + }; + fn frontier(node: &Rc) -> Option> { + match &node.expr { + SummaryExpr::ValueOperation { + child, + operation: ValueOperation::FinalizeExactAccumulator, + timing: planner_types::post_asap::ExecutionTiming::QueryTime, + } if matches!(&child.expr, SummaryExpr::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, + child: raw, .. + } if matches!(&raw.expr, SummaryExpr::KeepPreAsap(expr) if matches!(expr.as_ref(), QueryExpr::TimeRange { .. }))) => + { + Some(Rc::clone(node)) + } + SummaryExpr::ValueOperation { child, .. } | SummaryExpr::SummaryAgg { child, .. } => { + frontier(child) + } + SummaryExpr::SummaryEstimate { summary_input, .. } => frontier(summary_input), + _ => None, + } + } + let source = frontier(selected) + .ok_or_else(|| invalid("ranking requires one exact per-series Rate frontier"))?; + if !source + .schema + .fields + .iter() + .any(|field| field.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid("Rate ranking requires complete series identity")); + } + let compiled = compile_post_asap_dag_with_node_ids(selected) + .map_err(|error| invalid(error.to_string()))?; + let id = u64::from( + compiled + .node_ids + .node_id(&source) + .ok_or_else(|| invalid("missing Rate frontier"))? + .0, + ); + let program = compile( + &compiled.dag, + BTreeMap::from([(id, InputContract::bounded(Arc::new(source.schema.clone())))]), + &[u64::from(compiled.dag.root.0)], + )?; + Ok((source, program)) +} + +/// The selected logical placement requires fresh aggregate state per closed window. +/// Compile both physical graphs before deployment chooses storage or scheduling. +/// The input is the complete collection of per-series exact counter states. +pub fn compile_fixed_window_rate_aggregation( + selected: &Rc, +) -> Result { + use planner_types::post_asap::{ + compile_post_asap_dag, ExactKind, ExecutionTiming, SketchAlgorithm, + }; + let dag = compile_post_asap_dag(selected).map_err(|e| invalid(e.to_string()))?; + let sources = dag + .nodes + .iter() + .filter(|n| { + matches!( + &n.payload, + Payload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + reduction: planner_types::pre_asap::Reduction::PerEntity, + .. + } + ) + }) + .collect::>(); + let heaps = dag + .nodes + .iter() + .filter(|n| { + n.output_state.timing == ExecutionTiming::IngestionTime + && match &n.payload { + Payload::SummaryAgg { + family: SummaryFamilyType::Sketch(kind, _), + .. + } => matches!( + kind.algorithm(), + SketchAlgorithm::CmsWithHeap | SketchAlgorithm::CountSketchWithHeap + ), + Payload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } => true, + _ => false, + } + }) + .collect::>(); + let ([source], [heap]) = (sources.as_slice(), heaps.as_slice()) else { + return Err(invalid( + "expected one selected fixed-window Rate aggregation", + )); + }; + if !source + .output_schema + .fields + .iter() + .any(|f| f.name == SERIES_IDENTITY_COLUMN) + { + return Err(invalid( + "fixed-window Rate aggregation requires complete series identity", + )); + } + compile_candidate( + &dag, + BTreeMap::from([( + u64::from(source.id.0), + InputContract::bounded(Arc::new(source.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[u64::from(heap.id.0)], + ) +} diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs new file mode 100644 index 00000000..98032505 --- /dev/null +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -0,0 +1,278 @@ +//! Physical scalar/vector contracts preserve complete label sets across native computation. +use super::*; + +pub fn scalar_schema() -> Schema { + crate::operators::vector_binary::value_schema(true) +} +pub fn vector_schema() -> Schema { + crate::operators::vector_binary::value_schema(false) +} + +pub fn matrix_schema() -> Schema { + crate::operators::vector_window::matrix_schema() +} + +pub fn compile_scalar(value: f64) -> Result { + let operator = Operator::scalar( + crate::values::Value::Float64(value), + planner_types::pre_asap::DataType::Float64, + )? + .with_output_schema(scalar_schema())?; + CompiledPhysicalDag::from_operators( + BTreeMap::new(), + BTreeMap::from([(0, (vec![], operator))]), + vec![0], + ) +} + +pub fn compile_temporal( + intent: &AggIntent, + preserve_metric_name: bool, +) -> Result { + let operator = Operator::range_window(intent.clone())?; + let mut operators = vec![operator]; + if !preserve_metric_name { + operators.push(Operator::project( + vector_schema(), + vec![ + ( + "labels".into(), + Expression::LabelSet { + column: 0, + labels: vec![], + without: true, + }, + ), + ("value".into(), Expression::Column(1)), + ], + )?); + } + unary(operators, matrix_schema()) +} + +pub fn compile_histogram_quantile() -> Result { + CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(scalar_schema())), + (1, InputContract::bounded(vector_schema())), + ]), + BTreeMap::from([(2, (vec![0, 1], Operator::histogram_quantile()))]), + vec![2], + ) +} + +/// Compile before deployment chooses readers. Input slots 0 and 1 retain operand order. +pub fn compile_binary( + operator: &planner_types::post_asap::BinaryOperator, + return_bool: bool, + left_scalar: bool, + right_scalar: bool, +) -> Result { + let left = crate::operators::vector_binary::value_schema(left_scalar); + let right = crate::operators::vector_binary::value_schema(right_scalar); + let op = Operator::vector_binary(left.clone(), right.clone(), operator.clone(), return_bool)?; + CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(left)), + (1, InputContract::bounded(right)), + ]), + BTreeMap::from([(2, (vec![0, 1], op))]), + vec![2], + ) +} + +fn unary(operators: Vec, input: Schema) -> Result { + let root = operators.len() as u64; + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(input))]), + operators + .into_iter() + .enumerate() + .map(|(i, op)| ((i + 1) as u64, (vec![i as u64], op))) + .collect(), + vec![root], + ) +} + +fn grouped(grouping: &GroupKeys) -> Result { + let labels = grouping + .keys() + .iter() + .map(|key| match key { + ColumnRef::Named(label) => Ok(label.clone()), + _ => Err(invalid("vector grouping requires label names")), + }) + .collect::, _>>()?; + Operator::project( + vector_schema(), + vec![ + ("labels".into(), Expression::Column(0)), + ("value".into(), Expression::Column(1)), + ( + "group".into(), + Expression::LabelSet { + column: 0, + labels, + without: grouping.is_without(), + }, + ), + ], + ) +} + +fn vector_output(input: Schema, labels: usize, value: usize) -> Result { + let value = Expression::ExactFloat64(value); + Operator::project( + input, + vec![ + ("labels".into(), Expression::Column(labels)), + ("value".into(), value), + ], + ) +} + +pub fn compile_aggregate( + intent: &AggIntent, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let reduction = match intent { + AggIntent::Sum { .. } => Reduction::Sum(1), + AggIntent::Avg { .. } => Reduction::Avg(1), + AggIntent::Count { .. } => Reduction::Count, + AggIntent::Min { .. } => Reduction::Min(1), + AggIntent::Max { .. } => Reduction::Max(1), + _ => return Err(invalid("unsupported vector aggregate")), + }; + let aggregate = + Operator::aggregate(project.schema(), vec![2], vec![("value".into(), reduction)])?; + let output = vector_output(aggregate.schema(), 0, 1)?; + unary(vec![project, aggregate, output], vector_schema()) +} + +pub fn compile_sort( + descending: bool, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let sort = Operator::sort( + project.schema(), + vec![SortKey { + column: 1, + descending, + nulls_first: false, + }], + vec![2], + )?; + let output = vector_output(sort.schema(), 0, 1)?; + unary(vec![project, sort, output], vector_schema()) +} + +pub fn compile_limit( + n: u64, + offset: u64, + grouping: &GroupKeys, +) -> Result { + let project = grouped(grouping)?; + let limit = Operator::limit(project.schema(), n, offset, vec![2])?; + let output = vector_output(limit.schema(), 0, 1)?; + unary(vec![project, limit, output], vector_schema()) +} + +pub fn compile_negate(scalar: bool) -> Result { + let input = if scalar { + scalar_schema() + } else { + vector_schema() + }; + let mut columns = Vec::new(); + if !scalar { + columns.push(("labels".into(), Expression::Column(0))); + } + columns.push(( + if scalar { + "$promql_scalar".into() + } else { + "value".into() + }, + Expression::Negate(Box::new(Expression::Column(if scalar { 0 } else { 1 }))), + )); + unary(vec![Operator::project(input.clone(), columns)?], input) +} + +pub fn compile_vector_to_scalar() -> Result { + unary( + vec![Operator::vector_to_scalar(vector_schema(), 1)?.with_output_schema(scalar_schema())?], + vector_schema(), + ) +} + +/// A stored exact-state input retains the complete population identity. The +/// deployment supplies eligible panes; merging and finalization are computation. +pub fn exact_state_schema(family: SummaryFamilyType) -> Result { + if !matches!(family, SummaryFamilyType::ExactAggregate(..)) { + return Err(invalid("exact-state input requires an exact family")); + } + crate::values::validate_family(&family)?; + let mut schema = (*vector_schema()).clone(); + schema.fields[1].dtype = family; + Ok(Arc::new(schema)) +} + +/// Retain exact readout semantics before any deployment state is opened. +pub fn compile_exact_readout( + family: SummaryFamilyType, + lookback_ms: u64, + preserve_metric_name: bool, +) -> Result { + use planner_types::post_asap::ExactKind; + let statistic = match &family { + SummaryFamilyType::ExactAggregate(kind, _) => match kind { + ExactKind::Sum => crate::Statistic::Sum, + ExactKind::Count => crate::Statistic::Count, + ExactKind::Min => crate::Statistic::Min, + ExactKind::Max => crate::Statistic::Max, + ExactKind::Rate => crate::Statistic::Rate, + ExactKind::Increase => crate::Statistic::Increase, + ExactKind::IRate => return Err(invalid("instant-rate state readout is not supported")), + }, + _ => return Err(invalid("exact readout requires an exact family")), + }; + let input = exact_state_schema(family)?; + let merge = Operator::summary_merge(input.clone(), 1, vec![0])?; + let mut readout = Operator::readout( + merge.schema(), + 1, + ReadoutQuery::Exact(ExactReadout { + statistic, + lookback_ms: None, + }), + )?; + if matches!( + statistic, + crate::Statistic::Rate | crate::Statistic::Increase + ) { + readout = readout.with_counter_lookback( + i64::try_from(lookback_ms).map_err(|_| invalid("counter lookback exceeds Int64"))?, + )?; + } + let project = Operator::project( + readout.schema(), + vec![ + ( + "labels".into(), + if preserve_metric_name { + Expression::Column(0) + } else { + Expression::LabelSet { + column: 0, + labels: vec![], + without: true, + } + }, + ), + ("value".into(), Expression::ExactFloat64(1)), + ], + )?; + unary(vec![merge, readout, project], input) +} diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/asap-physical-operators/src/plan/mod.rs new file mode 100644 index 00000000..0dce4dd8 --- /dev/null +++ b/crates/asap-physical-operators/src/plan/mod.rs @@ -0,0 +1,160 @@ +//! Immutable physical graph, operator contracts and pre-execution validation. +use crate::{ + runtime::{Input, OutputStream, RunContext}, + Error, +}; +use std::{ + collections::{BTreeMap, BTreeSet}, + fmt::Debug, +}; +pub type NodeId = u64; +mod properties; +pub use properties::{Boundedness, Emission, PlanProperties}; +/// Operators own computation. The runtime provides already-connected inputs; +/// an operator must not recursively execute another plan node itself. +pub trait PhysicalOperator { + fn name(&self) -> &str; + /// Source implementations must explicitly declare finite input before feeding blocking operators. + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + PlanProperties { + boundedness: Boundedness::from_inputs(inputs), + emission: Emission::Unknown, + } + } + fn requires_bounded_input(&self) -> bool { + false + } + + /// Validate run-specific contracts before any source is opened. + fn validate_context(&self, _context: &RunContext) -> Result<(), Error> { + Ok(()) + } + fn input_schemas(&self) -> Vec; + fn output_schema(&self) -> S; + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error>; + fn output_bytes(&self, value: &V) -> usize; +} +pub(crate) struct Node<'a, V, S> { + pub(crate) inputs: Vec, + pub(crate) operator: Box + 'a>, +} +pub struct PhysicalDag<'a, V, S> { + pub(crate) nodes: BTreeMap>, +} +impl Default for PhysicalDag<'_, V, S> { + fn default() -> Self { + Self { + nodes: BTreeMap::new(), + } + } +} +impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { + pub fn add( + &mut self, + id: NodeId, + inputs: Vec, + operator: impl PhysicalOperator + 'a, + ) -> Result<(), Error> { + self.add_boxed(id, inputs, Box::new(operator)) + } + pub fn add_boxed( + &mut self, + id: NodeId, + inputs: Vec, + operator: Box + 'a>, + ) -> Result<(), Error> { + if self.nodes.contains_key(&id) { + return Err(Error::Invalid(format!("duplicate node {id}"))); + } + self.nodes.insert(id, Node { inputs, operator }); + Ok(()) + } + pub fn validate(&self, roots: &[NodeId]) -> Result<(), Error> { + self.properties(roots).map(|_| ()) + } + /// Derive properties while checking topology and schemas, before starting sources. + pub fn properties(&self, roots: &[NodeId]) -> Result, Error> { + fn visit( + dag: &PhysicalDag<'_, V, S>, + id: NodeId, + active: &mut BTreeSet, + done: &mut BTreeMap, + ) -> Result { + if let Some((depth, _)) = done.get(&id) { + return Ok(*depth); + } + if active.len() >= 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if !active.insert(id) { + return Err(Error::Invalid(format!("cycle at node {id}"))); + } + let node = dag + .nodes + .get(&id) + .ok_or_else(|| Error::Invalid(format!("missing node {id}")))?; + let expected = node.operator.input_schemas(); + if expected.len() != node.inputs.len() { + return Err(Error::Invalid(format!("node {id} input arity mismatch"))); + } + let mut depth = 1; + let mut input_properties = Vec::new(); + for (input, schema) in node.inputs.iter().zip(expected) { + depth = depth.max(1 + visit(dag, *input, active, done)?); + input_properties.push(done[input].1); + let actual = dag.nodes[input].operator.output_schema(); + if actual != schema { + return Err(Error::Invalid(format!( + "node {id} input {input} schema mismatch: {actual:?} vs {schema:?}" + ))); + } + } + if depth > 128 { + return Err(Error::Invalid( + "DAG exceeds the supported execution depth of 128".into(), + )); + } + if node.operator.requires_bounded_input() + && input_properties + .iter() + .any(|p| p.boundedness != Boundedness::Bounded) + { + return Err(Error::Invalid(format!( + "node {id} ({}) requires bounded inputs", + node.operator.name() + ))); + } + let properties = node.operator.properties(&input_properties); + active.remove(&id); + done.insert(id, (depth, properties)); + Ok(depth) + } + if roots.is_empty() { + return Err(Error::Invalid("execution needs a root".into())); + } + let mut done = BTreeMap::new(); + for &root in roots { + visit(self, root, &mut BTreeSet::new(), &mut done)?; + } + Ok(done + .into_iter() + .map(|(id, (_, properties))| (id, properties)) + .collect()) + } + pub fn execute<'r>( + &'r self, + roots: &[NodeId], + context: RunContext, + ) -> Result>, Error> + where + 'a: 'r, + { + crate::runtime::execute(self, roots, context) + } +} diff --git a/crates/asap-physical-operators/src/plan/properties.rs b/crates/asap-physical-operators/src/plan/properties.rs new file mode 100644 index 00000000..7ec770ea --- /dev/null +++ b/crates/asap-physical-operators/src/plan/properties.rs @@ -0,0 +1,32 @@ +//! Execution facts used to reject operators that cannot finish on their inputs. +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub enum Boundedness { + /// The source or operator promises a finite result for this run. + Bounded, + Unbounded, + /// No finite-input guarantee has been supplied. + Unknown, +} +impl Boundedness { + pub fn from_inputs(inputs: &[PlanProperties]) -> Self { + if inputs.iter().any(|p| p.boundedness == Self::Unbounded) { + Self::Unbounded + } else if inputs.is_empty() || inputs.iter().any(|p| p.boundedness == Self::Unknown) { + Self::Unknown + } else { + Self::Bounded + } + } +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub enum Emission { + Incremental, + /// Produces its result only after all inputs end, even if accumulation is incremental. + AfterInput, + Unknown, +} +#[derive(serde::Serialize, serde::Deserialize, Clone, Copy, Debug, PartialEq, Eq)] +pub struct PlanProperties { + pub boundedness: Boundedness, + pub emission: Emission, +} diff --git a/crates/asap-physical-operators/src/readout.rs b/crates/asap-physical-operators/src/readout.rs new file mode 100644 index 00000000..e884ecc3 --- /dev/null +++ b/crates/asap-physical-operators/src/readout.rs @@ -0,0 +1,111 @@ +//! Readouts over merged exact summary states. +use crate::summary_kernels::exact::ExactAccumulator; +use crate::{AggregateCore, KeyByLabelValues, Statistic}; +use std::sync::Arc; + +fn merge_exact_states( + states: impl IntoIterator>, +) -> Result { + let mut states = states.into_iter(); + let exact = |state: &Arc| { + state + .as_any() + .downcast_ref::() + .cloned() + .ok_or_else(|| "readout requires Planner exact state".to_string()) + }; + let mut merged = exact(&states.next().ok_or("empty exact state input")?)?; + for state in states { + merged + .merge_from(&exact(&state)?) + .map_err(|error| error.to_string())?; + } + Ok(merged) +} + +/// PromQL counter readouts omit a series with fewer than two samples. Other +/// state/type/range failures remain errors rather than empty results. +pub fn insufficient_counter_samples(state: &dyn AggregateCore, statistic: Statistic) -> bool { + matches!(statistic, Statistic::Rate | Statistic::Increase) + && state + .as_any() + .downcast_ref::() + .is_some_and(|state| state.insufficient_counter_samples(statistic, &None)) +} + +/// Merge already selected exact panes and read one population. `None` means +/// the population is absent from the result: a counter with too few samples, +/// or an empty MIN/MAX. +pub fn exact_readout( + states: impl IntoIterator>, + statistic: Statistic, + range_ms: Option<(i64, i64)>, + key: Option<&KeyByLabelValues>, +) -> Result, String> { + let merged = merge_exact_states(states)?; + if merged.insufficient_counter_samples(statistic, &key.cloned()) { + return Ok(None); + } + merged + .readout(statistic, range_ms, key) + .map_err(|error| error.to_string()) +} + +#[cfg(test)] +mod counter_tests { + use super::*; + use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; + + fn counter(kind: ExactKind, params: ExactParams, keyed: bool) -> ExactAccumulator { + ExactAccumulator::new(SummaryFamilyType::ExactAggregate(kind, params), keyed).unwrap() + } + + // A counter population with a single sample is absent, keyed or not. + #[test] + fn planner_counter_population_omits_insufficient_samples() { + for (kind, params, statistic) in [ + (ExactKind::Rate, ExactParams::Rate, Statistic::Rate), + ( + ExactKind::Increase, + ExactParams::Increase, + Statistic::Increase, + ), + ] { + for keyed in [false, true] { + let mut state = counter(kind.clone(), params.clone(), keyed); + let key = keyed.then(|| KeyByLabelValues::new_with_labels(vec!["checkout".into()])); + state.update(key.as_ref(), 10., 10_000); + assert_eq!( + exact_readout( + [Arc::new(state) as Arc], + statistic, + None, + key.as_ref() + ) + .unwrap(), + None + ); + } + } + } + + // Two ordered samples read a rate; an inverted range and empty input fail. + #[test] + fn sparse_counter_is_absent_but_invalid_ranges_still_fail() { + let mut state = counter(ExactKind::Rate, ExactParams::Rate, false); + state.update(None, 10., 10_000); + let rate = Statistic::Rate; + let one = [Arc::new(state.clone()) as Arc]; + assert_eq!( + exact_readout(one, rate, Some((0, 60_000)), None).unwrap(), + None + ); + state.update(None, 20., 20_000); + let two = || [Arc::new(state.clone()) as Arc]; + assert!(exact_readout(two(), rate, Some((0, 60_000)), None) + .unwrap() + .is_some()); + assert!(exact_readout(two(), rate, Some((60_000, 0)), None).is_err()); + assert!(exact_readout([], rate, Some((0, 60_000)), None).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs new file mode 100644 index 00000000..f6300440 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -0,0 +1,210 @@ +//! Execute a bounded in-memory batch through native operators. This is also the +//! bridge for deployments whose boundary values are not yet streaming batches. +use crate::{ + operators::Operator, + plan::PhysicalDag, + runtime::{RunContext, SharedValue}, + values::Batch, + Error, +}; +use futures::{FutureExt, StreamExt}; + +/// Every input is already in memory; the chain contains native operators only. +/// This deliberately does not enter a nested executor when called from a DAG +/// adapter. I/O belongs to source operators in the surrounding execution. +pub fn evaluate_batch( + input: Batch, + operators: Vec, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + graph.add( + 0, + vec![], + Operator::source(input.schema().clone(), vec![input])?, + )?; + let mut root = 0; + for operator in operators { + graph.add(root + 1, vec![root], operator)?; + root += 1; + } + evaluate_graph(graph, root, context) +} + +/// Bind the ordered in-memory inputs of a native multi-input operator. +pub fn evaluate_inputs( + inputs: Vec, + operator: Operator, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + let root = inputs.len() as u64; + for (id, input) in inputs.into_iter().enumerate() { + graph.add( + id as u64, + vec![], + Operator::source(input.schema().clone(), vec![input])?, + )?; + } + graph.add(root, (0..root).collect(), operator)?; + evaluate_graph(graph, root, context) +} + +/// Evaluate a native in-memory source, including scalar sources, in the caller's scope. +pub fn evaluate_source( + source: Operator, + context: RunContext, +) -> Result>, Error> { + let mut graph = PhysicalDag::default(); + graph.add(0, vec![], source)?; + evaluate_graph(graph, 0, context) +} + +fn evaluate_graph( + graph: PhysicalDag<'_, Batch, crate::values::Schema>, + root: crate::plan::NodeId, + context: RunContext, +) -> Result>, Error> { + let mut output = graph.execute(&[root], context)?.remove(0); + let mut batches = Vec::new(); + loop { + match output.next().now_or_never() { + Some(Some(Ok(batch))) => batches.push(batch), + Some(Some(Err(error))) => return Err(error), + Some(None) => return Ok(batches), + // Native operators have no I/O sources here. Pending is the + // shared runtime's cooperative yield after a batch quantum. + None => continue, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{ + operators::Expression, + runtime::{Limits, Scope}, + values::Value, + }; + use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, + }; + use std::sync::Arc; + + // Engine adapters can run the identical native chain from an outer executor. + #[test] + fn same_native_chain_inside_query_and_ingestion_execution() { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + for scope in [ + Scope::Query { + evaluation_time_ms: 20, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: 10, + window_end_ms: 20, + revision: 1, + }, + ] { + let batch = Batch::try_new(schema.clone(), vec![vec![Value::Float64(7.)]]).unwrap(); + let negate = Operator::project( + schema.clone(), + vec![( + "value".into(), + Expression::Negate(Box::new(Expression::Column(0))), + )], + ) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let result = futures::executor::block_on(async { + evaluate_batch(batch, vec![negate], context.clone()) + }) + .unwrap(); + assert!(matches!(result[0].rows()[0][0], Value::Float64(-7.))); + let source = Operator::scalar(Value::Float64(9.), DataType::Float64).unwrap(); + let scalar = evaluate_source(source, context).unwrap(); + assert!(matches!(scalar[0].rows()[0][0], Value::Float64(9.))); + } + } + + // Native sources may cross the runtime's cooperative batch quantum. + #[test] + fn in_memory_source_drives_cooperative_yields() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); + let source = Operator::source(schema, vec![batch; 65]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + assert_eq!(evaluate_source(source, context).unwrap().len(), 65); + } + + // An adapter-held output must retain its parent's reservation after execution. + #[test] + fn returned_batches_keep_their_resource_reservation() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema.clone(), vec![vec![]]).unwrap(); + let bytes = batch.bytes(); + let source = Operator::source(schema, vec![batch]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes: bytes, + max_buffered_batches: 1, + }, + ) + .unwrap(); + let held = evaluate_source(source.clone(), context.clone()).unwrap(); + assert_eq!(context.retained_bytes(), bytes); + assert!(evaluate_source(source.clone(), context.clone()).is_err()); + drop(held); + assert_eq!(context.retained_bytes(), 0); + assert!(evaluate_source(source, context).is_ok()); + } + + // A cancelled surrounding execution also prevents its native computation. + #[test] + fn cancellation_is_not_bypassed_by_in_memory_execution() { + let schema = Arc::new(SummarySchema { + fields: vec![], + time_index: None, + }); + let batch = Batch::try_new(schema, vec![vec![]]).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + context.cancel(); + assert!(matches!( + evaluate_batch(batch, vec![], context), + Err(Error::Cancelled) + )); + } +} diff --git a/crates/asap-physical-operators/src/runtime/context.rs b/crates/asap-physical-operators/src/runtime/context.rs new file mode 100644 index 00000000..c145a457 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/context.rs @@ -0,0 +1,133 @@ +use crate::Error; +use std::{ + cell::{Cell, RefCell}, + rc::Rc, + task::Waker, +}; +/// Scope is part of an execution instance, never mutable state in a reusable plan. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Scope { + Ingestion { + window_start_ms: i64, + window_end_ms: i64, + revision: u64, + }, + Query { + evaluation_time_ms: i64, + revision: u64, + }, +} +#[derive(Clone, Debug)] +pub struct Limits { + pub max_buffered_batches: usize, + pub max_bytes: usize, +} +impl Default for Limits { + fn default() -> Self { + Self { + max_buffered_batches: 8, + max_bytes: 64 * 1024 * 1024, + } + } +} +pub(super) struct Control { + cancelled: Cell, + bytes: Cell, + peak: Cell, + pub(super) limits: Limits, + waiters: RefCell>, +} +#[derive(Clone)] +pub struct RunContext { + pub scope: Scope, + pub(super) control: Rc, +} +impl RunContext { + pub fn new(scope: Scope, limits: Limits) -> Result { + if limits.max_buffered_batches == 0 || limits.max_bytes == 0 { + return Err(Error::Invalid("execution limits must be positive".into())); + } + if matches!(&scope, Scope::Ingestion { window_start_ms, window_end_ms, .. } if window_start_ms > window_end_ms) + { + return Err(Error::Invalid("inverted ingestion window".into())); + } + Ok(Self { + scope, + control: Rc::new(Control { + cancelled: Cell::new(false), + bytes: Cell::new(0), + peak: Cell::new(0), + limits, + waiters: RefCell::new(Vec::new()), + }), + }) + } + pub fn cancel(&self) { + self.control.cancelled.set(true); + for waiter in self.control.waiters.borrow_mut().drain(..) { + waiter.wake(); + } + } + pub fn is_cancelled(&self) -> bool { + self.control.cancelled.get() + } + pub fn retained_bytes(&self) -> usize { + self.control.bytes.get() + } + pub fn peak_bytes(&self) -> usize { + self.control.peak.get() + } + pub fn reserve(&self, bytes: usize) -> Result { + let total = self + .control + .bytes + .get() + .checked_add(bytes) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + Ok(Reservation { + bytes, + control: Rc::clone(&self.control), + }) + } + pub(super) fn register(&self, waker: &Waker) { + let mut waiters = self.control.waiters.borrow_mut(); + if !waiters.iter().any(|old| old.will_wake(waker)) { + waiters.push(waker.clone()); + } + } +} +pub struct Reservation { + bytes: usize, + pub(super) control: Rc, +} +impl Reservation { + /// Adjust an operator-owned allocation without accumulating bookkeeping entries. + pub fn resize(&mut self, bytes: usize) -> Result<(), Error> { + let total = self + .control + .bytes + .get() + .checked_sub(self.bytes) + .and_then(|total| total.checked_add(bytes)) + .ok_or(Error::MemoryLimit)?; + if total > self.control.limits.max_bytes { + return Err(Error::MemoryLimit); + } + self.control.bytes.set(total); + self.control.peak.set(self.control.peak.get().max(total)); + self.bytes = bytes; + Ok(()) + } +} +impl Drop for Reservation { + fn drop(&mut self) { + self.control + .bytes + .set(self.control.bytes.get().saturating_sub(self.bytes)); + } +} diff --git a/crates/asap-physical-operators/src/runtime/cooperative.rs b/crates/asap-physical-operators/src/runtime/cooperative.rs new file mode 100644 index 00000000..fae869d6 --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/cooperative.rs @@ -0,0 +1,40 @@ +//! Worker-local CPU loops yield so other consumers and cancellation can progress. +use super::RunContext; +use crate::Error; +use std::task::Poll; + +pub(crate) struct Cooperative { + context: RunContext, + remaining: usize, +} +impl Cooperative { + pub(crate) fn new(context: &RunContext) -> Self { + Self { + context: context.clone(), + remaining: 1024, + } + } + pub(crate) async fn checkpoint(&mut self) -> Result<(), Error> { + if self.context.is_cancelled() { + return Err(Error::Cancelled); + } + self.remaining -= 1; + if self.remaining == 0 { + self.remaining = 1024; + let mut yielded = false; + futures::future::poll_fn(|cx| { + if self.context.is_cancelled() { + return Poll::Ready(Err(Error::Cancelled)); + } + if yielded { + return Poll::Ready(Ok(())); + } + yielded = true; + cx.waker().wake_by_ref(); + Poll::Pending + }) + .await?; + } + Ok(()) + } +} diff --git a/crates/asap-physical-operators/src/runtime/mod.rs b/crates/asap-physical-operators/src/runtime/mod.rs new file mode 100644 index 00000000..f72a57ef --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/mod.rs @@ -0,0 +1,272 @@ +//! Per-run producer sharing, streams, backpressure and resource ownership. +use crate::{ + plan::{NodeId, PhysicalDag}, + Error, +}; +use futures::{stream::LocalBoxStream, Stream}; +use std::{ + cell::RefCell, + collections::{BTreeMap, VecDeque}, + fmt::Debug, + pin::Pin, + rc::Rc, + sync::Arc, + task::{Context, Poll, Waker}, +}; +mod context; +pub use context::{Limits, Reservation, RunContext, Scope}; +pub type OutputStream<'a, V> = LocalBoxStream<'a, Result>; +/// An output owns its memory reservation even after it leaves the DAG's queue. +pub struct SharedValue { + value: Arc, + _reservation: Rc, +} +impl Clone for SharedValue { + fn clone(&self) -> Self { + Self { + value: Arc::clone(&self.value), + _reservation: Rc::clone(&self._reservation), + } + } +} +impl std::ops::Deref for SharedValue { + type Target = V; + fn deref(&self) -> &V { + &self.value + } +} +impl SharedValue { + pub fn value(&self) -> &V { + &self.value + } +} + +pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( + dag: &'r PhysicalDag<'_, V, S>, + roots: &[NodeId], + context: RunContext, +) -> Result>, Error> { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + dag.validate(roots)?; + let mut pending = roots.to_vec(); + let mut visited = std::collections::BTreeSet::new(); + while let Some(id) = pending.pop() { + if visited.insert(id) { + let node = &dag.nodes[&id]; + node.operator.validate_context(&context)?; + pending.extend(node.inputs.iter().copied()); + } + } + fn build<'r, V: 'r, S: 'r>( + dag: &'r PhysicalDag<'_, V, S>, + id: NodeId, + context: &RunContext, + states: &mut BTreeMap>>>, + ) -> Result>>, Error> { + if let Some(state) = states.get(&id) { + return Ok(Rc::clone(state)); + } + let node = &dag.nodes[&id]; + let mut inputs = Vec::new(); + for &child in &node.inputs { + inputs.push(Input::subscribe(build(dag, child, context, states)?)); + } + let stream = node + .operator + .start(inputs, context.clone()) + .map_err(|source| Error::AtNode { + node: id, + operation: node.operator.name().into(), + source: Box::new(source), + })?; + let op = node.operator.as_ref(); + let state = Rc::new(RefCell::new(Producer { + stream: Some(stream), + node: id, + operation: node.operator.name().into(), + size: Box::new(move |value| op.output_bytes(value)), + context: context.clone(), + queue: VecDeque::new(), + base: 0, + next_reader: 0, + batches_polled: 0, + readers: BTreeMap::new(), + waiters: BTreeMap::new(), + finished: false, + failure: None, + })); + states.insert(id, Rc::clone(&state)); + Ok(state) + } + let mut states = BTreeMap::new(); + roots + .iter() + .map(|&id| build(dag, id, &context, &mut states).map(Input::subscribe)) + .collect() +} +struct Producer<'a, V> { + node: NodeId, + operation: String, + stream: Option>, + size: Box usize + 'a>, + context: RunContext, + queue: VecDeque>, + base: u64, + next_reader: u64, + batches_polled: usize, + readers: BTreeMap, + waiters: BTreeMap, + finished: bool, + failure: Option, +} +impl Producer<'_, V> { + fn trim(&mut self) { + let minimum = self + .readers + .values() + .copied() + .min() + .unwrap_or(self.base + self.queue.len() as u64); + while self.base < minimum { + self.queue.pop_front(); + self.base += 1; + } + for (_, waker) in std::mem::take(&mut self.waiters) { + waker.wake(); + } + if self.readers.is_empty() { + self.stream = None; + self.queue.clear(); + } + } +} +pub struct Input<'a, V> { + producer: Rc>>, + reader: u64, + done: bool, +} +impl<'a, V> Input<'a, V> { + fn subscribe(producer: Rc>>) -> Self { + let reader = { + let mut state = producer.borrow_mut(); + let id = state.next_reader; + state.next_reader += 1; + let base = state.base; + state.readers.insert(id, base); + id + }; + Self { + producer, + reader, + done: false, + } + } +} +impl Drop for Input<'_, V> { + fn drop(&mut self) { + let mut state = self.producer.borrow_mut(); + state.readers.remove(&self.reader); + state.waiters.remove(&self.reader); + state.trim(); + } +} +impl Stream for Input<'_, V> { + type Item = Result, Error>; + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let this = self.get_mut(); + if this.done { + return Poll::Ready(None); + } + let mut state = this.producer.borrow_mut(); + state.context.register(cx.waker()); + if state.context.is_cancelled() { + state.failure = Some(Error::Cancelled); + state.finished = true; + state.stream = None; + state.queue.clear(); + } + let position = state.readers[&this.reader]; + let index = (position - state.base) as usize; + if let Some(value) = state.queue.get(index).cloned() { + state.readers.insert(this.reader, position + 1); + state.trim(); + return Poll::Ready(Some(Ok(value))); + } + if state.finished { + this.done = true; + state.readers.remove(&this.reader); + let failure = state.failure.clone(); + state.trim(); + return Poll::Ready(failure.map(Err)); + } + state.waiters.insert(this.reader, cx.waker().clone()); + if state.queue.len() >= state.context.control.limits.max_buffered_batches { + return Poll::Pending; + } + // Always-ready sources must still give cancellation and other roots a turn. + if state.batches_polled >= 32 { + state.batches_polled = 0; + cx.waker().wake_by_ref(); + return Poll::Pending; + } + let polled = state + .stream + .as_mut() + .expect("unfinished producer") + .as_mut() + .poll_next(cx); + if matches!(&polled, Poll::Ready(Some(Ok(_)))) { + state.batches_polled += 1; + } + match polled { + Poll::Pending => Poll::Pending, + Poll::Ready(Some(Ok(value))) => match state.context.reserve((state.size)(&value)) { + Ok(reservation) => { + let value = SharedValue { + value: Arc::new(value), + _reservation: Rc::new(reservation), + }; + state.queue.push_back(value.clone()); + state.readers.insert(this.reader, position + 1); + state.trim(); + Poll::Ready(Some(Ok(value))) + } + Err(error) => { + state.failure = Some(error.clone()); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(Some(Err(error))) + } + }, + Poll::Ready(result) => { + let error = result.and_then(Result::err).map(|source| match source { + Error::AtNode { .. } | Error::Cancelled | Error::MemoryLimit => source, + source => Error::AtNode { + node: state.node, + operation: state.operation.clone(), + source: Box::new(source), + }, + }); + state.failure = error.clone(); + state.finished = true; + state.stream = None; + this.done = true; + state.readers.remove(&this.reader); + state.trim(); + Poll::Ready(error.map(Err)) + } + } + } +} + +pub mod batch_execution; +#[cfg(test)] +mod tests; + +mod cooperative; +pub(crate) use cooperative::Cooperative; diff --git a/crates/asap-physical-operators/src/runtime/tests.rs b/crates/asap-physical-operators/src/runtime/tests.rs new file mode 100644 index 00000000..f683041c --- /dev/null +++ b/crates/asap-physical-operators/src/runtime/tests.rs @@ -0,0 +1,262 @@ +use super::*; +use crate::plan::PhysicalOperator; +use futures::{executor::block_on, stream, StreamExt}; +use std::cell::Cell; + +struct Source { + starts: Rc>, + polls: Rc>, + fail: bool, + end: u64, +} +impl PhysicalOperator for Source { + fn name(&self) -> &str { + "CountingSource" + } + fn input_schemas(&self) -> Vec<()> { + vec![] + } + fn output_schema(&self) {} + fn output_bytes(&self, _: &u64) -> usize { + 8 + } + fn start<'a>( + &'a self, + _: Vec>, + _: RunContext, + ) -> Result, Error> { + self.starts.set(self.starts.get() + 1); + Ok(stream::iter(0..self.end) + .map(move |n| { + self.polls.set(self.polls.get() + 1); + if self.fail && n == 1 { + Err(Error::Operator("source failure".into())) + } else { + Ok(n) + } + }) + .boxed_local()) + } +} +struct Identity; +impl PhysicalOperator for Identity { + fn name(&self) -> &str { + "Identity" + } + fn input_schemas(&self) -> Vec<()> { + vec![()] + } + fn output_schema(&self) {} + fn output_bytes(&self, _: &u64) -> usize { + 8 + } + fn start<'a>( + &'a self, + mut inputs: Vec>, + _: RunContext, + ) -> Result, Error> { + Ok(inputs + .remove(0) + .map(|value| value.map(|v| *v)) + .boxed_local()) + } +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 100, + revision: 1, + }, + Limits { + max_buffered_batches: 1, + max_bytes: 1024, + }, + ) + .unwrap() +} +fn source(fail: bool) -> (Source, Rc>, Rc>) { + let starts = Rc::new(Cell::new(0)); + let polls = Rc::new(Cell::new(0)); + ( + Source { + starts: starts.clone(), + polls: polls.clone(), + fail, + end: 4, + }, + starts, + polls, + ) +} + +// A shared producer runs once, and the slow reader bounds producer progress. +#[test] +fn shared_source_backpressure_and_reader_drop() { + let (source, starts, polls) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let context = context(); + let mut readers = dag.execute(&[0, 0], context.clone()).unwrap(); + let mut slow = readers.pop().unwrap(); + let mut fast = readers.pop().unwrap(); + assert_eq!(starts.get(), 1); + let first = block_on(fast.next()).unwrap().unwrap(); + assert_eq!(*first, 0); + let mut cx = Context::from_waker(futures::task::noop_waker_ref()); + assert!(Pin::new(&mut fast).poll_next(&mut cx).is_pending()); + assert_eq!(polls.get(), 1); + let same = block_on(slow.next()).unwrap().unwrap(); + assert!(Arc::ptr_eq(&first.value, &same.value)); + drop(same); + drop(first); + assert_eq!(context.retained_bytes(), 0); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 1); + drop(slow); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 2); + assert_eq!(*block_on(fast.next()).unwrap().unwrap(), 3); + assert!(block_on(fast.next()).is_none()); + assert_eq!(polls.get(), 4); + drop(fast); + assert_eq!(context.retained_bytes(), 0); +} + +// Independent branches consume a common node concurrently without duplicate work. +#[test] +fn diamond_and_run_isolation() { + let (source, starts, polls) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + dag.add(1, vec![0], Identity).unwrap(); + dag.add(2, vec![0], Identity).unwrap(); + for _ in 0..2 { + let mut outputs = dag.execute(&[1, 2], context()).unwrap(); + let a = outputs.pop().unwrap(); + let b = outputs.pop().unwrap(); + let (a, b) = + block_on(async { futures::join!(a.collect::>(), b.collect::>()) }); + assert_eq!( + a.iter().map(|v| **v.as_ref().unwrap()).collect::>(), + vec![0, 1, 2, 3] + ); + assert_eq!( + b.iter().map(|v| **v.as_ref().unwrap()).collect::>(), + vec![0, 1, 2, 3] + ); + } + assert_eq!(starts.get(), 2); + assert_eq!(polls.get(), 8); +} + +// Failure reaches every subscriber; cancellation stops further producer work. +#[test] +fn broadcast_error_and_cancel() { + let (source, _, polls) = source(true); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let mut outputs = dag.execute(&[0, 0], context()).unwrap(); + let a = outputs.pop().unwrap(); + let b = outputs.pop().unwrap(); + let (a, b) = block_on(async { futures::join!(a.collect::>(), b.collect::>()) }); + for values in [a, b] { + assert_eq!(values.len(), 2); + assert!(matches!(values[1], Err(Error::AtNode { node: 0, .. }))); + } + assert_eq!(polls.get(), 2); + let run = context(); + let mut output = dag.execute(&[0], run.clone()).unwrap().remove(0); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + assert!(block_on(output.next()).is_none()); + assert_eq!(polls.get(), 2); +} + +// Retaining a consumer output retains its budget lease after queue eviction. +#[test] +fn retained_outputs_count_against_budget() { + let (source, _, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_buffered_batches: 1, + max_bytes: 8, + }, + ) + .unwrap(); + let mut input = dag.execute(&[0], run.clone()).unwrap().remove(0); + let held = block_on(input.next()).unwrap().unwrap(); + assert_eq!(run.retained_bytes(), 8); + assert!(matches!( + block_on(input.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(input); + assert_eq!(run.retained_bytes(), 8); + drop(held); + assert_eq!(run.retained_bytes(), 0); +} + +// Invalid graphs fail before even starting a source. +#[test] +fn invalid_graphs_do_not_start_sources() { + let (source, starts, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + dag.add(1, vec![2], Identity).unwrap(); + dag.add(2, vec![1], Identity).unwrap(); + assert!(dag.execute(&[0, 1], context()).is_err()); + assert_eq!(starts.get(), 0); + let mut missing = PhysicalDag::default(); + missing.add(1, vec![9], Identity).unwrap(); + assert!(missing.validate(&[1]).is_err()); + let mut arity = PhysicalDag::default(); + arity.add(1, vec![], Identity).unwrap(); + assert!(arity.validate(&[1]).is_err()); +} + +// An always-ready source must yield so cancellation can be polled on this worker. +#[test] +fn ready_sources_cooperate_with_cancellation() { + let (mut source, _, polls) = source(false); + source.end = 10_000; + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + let context = context(); + let mut input = dag.execute(&[0], context.clone()).unwrap().remove(0); + block_on(async { + let drain = async { + while let Some(result) = input.next().await { + if let Err(error) = result { + assert_eq!(error, Error::Cancelled); + return; + } + } + panic!("source completed without yielding"); + }; + let cancel = async { + context.cancel(); + }; + futures::join!(drain, cancel); + }); + assert_eq!(polls.get(), 32); + assert_eq!(context.retained_bytes(), 0); +} + +// Cached shorter paths must not hide an over-deep path through shared nodes. +#[test] +fn depth_limit_covers_shared_paths() { + let (source, _, _) = source(false); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], source).unwrap(); + for id in 1..129 { + dag.add(id, vec![id - 1], Identity).unwrap(); + } + assert!(dag.validate(&(0..129).collect::>()).is_err()); +} diff --git a/crates/asap-physical-operators/src/sources/memory.rs b/crates/asap-physical-operators/src/sources/memory.rs new file mode 100644 index 00000000..7f77c54f --- /dev/null +++ b/crates/asap-physical-operators/src/sources/memory.rs @@ -0,0 +1,44 @@ +use super::*; +/// Immutable in-memory raw data. The connector owns the resident input; each +/// cursor clones only the next requested batch, not the entire data set. +pub struct MemorySource { + schema: Schema, + batches: Vec, +} +impl MemorySource { + pub fn new(schema: Schema, batches: Vec) -> Result { + crate::values::validate_schema(&schema)?; + if schema + .fields + .iter() + .any(|f| !matches!(f.dtype, SummaryFamilyType::Plain(_))) + { + return Err(Error::Invalid( + "raw source cannot contain summary states".into(), + )); + } + if batches.iter().any(|batch| batch.schema() != &schema) { + return Err(Error::Invalid("memory source batch schema mismatch".into())); + } + Ok(Self { schema, batches }) + } +} +impl RawSource for MemorySource { + fn boundedness(&self) -> crate::plan::Boundedness { + crate::plan::Boundedness::Bounded + } + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, context: RunContext) -> Result, Error> { + Ok(stream::iter(self.batches.iter()) + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let _allocation = context.reserve(batch.bytes())?; + Ok(batch.clone()) + }) + .boxed_local()) + } +} diff --git a/crates/asap-physical-operators/src/sources/mod.rs b/crates/asap-physical-operators/src/sources/mod.rs new file mode 100644 index 00000000..4da778b0 --- /dev/null +++ b/crates/asap-physical-operators/src/sources/mod.rs @@ -0,0 +1,187 @@ +//! Raw data access. Connectors provide rows; Scan owns Planner predicate semantics. +use crate::{ + expressions::CompiledExpression, + plan::PhysicalOperator, + runtime::{Input, OutputStream, RunContext}, + values::{Batch, Schema, Value}, + Error, +}; +use futures::{stream, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{DataType, QueryExpr, Source}, +}; +use std::sync::Arc; + +/// A bound data source. Metadata must be stable for the lifetime of the binding. +/// Each scan opens an independent cursor. Connectors return raw, unfiltered rows +/// and must honor cancellation and bound their own I/O buffers. Dropping a cursor +/// must release its resources. A connector error is never an empty successful scan. +pub trait RawSource { + fn schema(&self) -> Schema; + /// Declare a finite snapshot/window explicitly; execution scope alone does not bound a cursor. + fn boundedness(&self) -> crate::plan::Boundedness { + crate::plan::Boundedness::Unknown + } + fn scan(&self, context: RunContext) -> Result, Error>; +} + +/// Explicit source identities; no implicit network discovery or fallback. +#[derive(Default)] +pub struct DataSources { + sources: Vec<(Source, Arc)>, +} +impl DataSources { + pub fn register(&mut self, identity: Source, source: Arc) -> Result<(), Error> { + if self.sources.iter().any(|(key, _)| key == &identity) { + return Err(Error::Invalid("duplicate data source".into())); + } + crate::values::validate_schema(&source.schema())?; + self.sources.push((identity, source)); + Ok(()) + } + pub fn bind(&self, expression: &QueryExpr) -> Result { + let QueryExpr::Scan { + source, + predicates, + schema, + } = expression + else { + return Err(Error::Invalid( + "raw Scan requires a Planner Scan leaf".into(), + )); + }; + let output = Arc::new(SummarySchema { + fields: schema + .columns + .iter() + .map(|column| SummaryField { + name: column.name.clone(), + dtype: SummaryFamilyType::Plain(column.dtype.clone()), + nullable: column.nullable, + }) + .collect(), + time_index: schema.time_index, + }); + crate::values::validate_schema(&output)?; + let reader = self + .sources + .iter() + .find(|(key, _)| key == source) + .map(|(_, reader)| reader.clone()) + .ok_or_else(|| Error::Invalid(format!("unbound raw source: {source:?}")))?; + if reader.schema() != output { + return Err(Error::Invalid( + "raw source differs from Planner Scan schema".into(), + )); + } + let predicates = predicates + .iter() + .map(|predicate| { + let predicate = CompiledExpression::compile(&predicate.0, &output)?; + if predicate.dtype().0 != DataType::Bool { + return Err(Error::Invalid("Scan predicate must be boolean".into())); + } + Ok(predicate) + }) + .collect::, Error>>()?; + Ok(Scan { + reader, + output, + predicates, + }) + } +} + +pub struct Scan { + reader: Arc, + output: Schema, + predicates: Vec, +} +impl PhysicalOperator for Scan { + fn properties(&self, _: &[crate::plan::PlanProperties]) -> crate::plan::PlanProperties { + crate::plan::PlanProperties { + boundedness: self.reader.boundedness(), + emission: crate::plan::Emission::Incremental, + } + } + + fn name(&self) -> &str { + "Scan" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.output.clone() + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + if !inputs.is_empty() { + return Err(Error::Invalid("Scan cannot have inputs".into())); + } + if context.is_cancelled() { + return Err(Error::Cancelled); + } + // Opening is lazy: validation and construction of a run perform no I/O. + let opening = context.clone(); + let stream = stream::once(async move { + if opening.is_cancelled() { + return Err(Error::Cancelled); + } + self.reader.scan(opening) + }); + use futures::TryStreamExt; + Ok(stream + .try_flatten() + .map(move |batch| { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let batch = batch?; + if batch.schema() != &self.output { + return Err(Error::Invalid( + "connector returned a different Scan schema".into(), + )); + } + if self.predicates.is_empty() { + return Ok(batch); + } + let _workspace = + context.reserve(batch.bytes().checked_mul(2).ok_or(Error::MemoryLimit)?)?; + let mut rows = Vec::new(); + for row in batch.rows() { + if context.is_cancelled() { + return Err(Error::Cancelled); + } + let mut keep = true; + for predicate in &self.predicates { + match predicate.evaluate(row)? { + Value::Bool(true) => {} + Value::Bool(false) | Value::Null => { + keep = false; + break; + } + _ => { + return Err(Error::Invalid("Scan predicate is not boolean".into())) + } + } + } + if keep { + rows.push(row.clone()); + } + } + Batch::try_new(self.output.clone(), rows) + }) + .boxed_local()) + } +} + +mod memory; +pub use memory::MemorySource; diff --git a/crates/asap-physical-operators/src/statistic.rs b/crates/asap-physical-operators/src/statistic.rs new file mode 100644 index 00000000..7053308d --- /dev/null +++ b/crates/asap-physical-operators/src/statistic.rs @@ -0,0 +1,67 @@ +use std::{fmt, str::FromStr}; +use tracing::debug; +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, serde::Serialize, serde::Deserialize)] +pub enum Statistic { + Count, + Sum, + Cardinality, + FrequencyL2, + FrequencyEntropy, + Increase, + Rate, + Min, + Max, + Quantile, + Topk, +} + +impl fmt::Display for Statistic { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + debug!("Formatting Statistic: {:?}", self); + match self { + Statistic::Count => write!(f, "count"), + Statistic::Sum => write!(f, "sum"), + Statistic::Cardinality => write!(f, "cardinality"), + Statistic::FrequencyL2 => write!(f, "frequency_l2"), + Statistic::FrequencyEntropy => write!(f, "frequency_entropy"), + Statistic::Increase => write!(f, "increase"), + Statistic::Rate => write!(f, "rate"), + Statistic::Min => write!(f, "min"), + Statistic::Max => write!(f, "max"), + Statistic::Quantile => write!(f, "quantile"), + Statistic::Topk => write!(f, "topk"), + } + } +} + +#[allow(clippy::should_implement_trait)] +impl Statistic { + pub fn from_str(s: &str) -> Option { + debug!("Parsing Statistic from string: {}", s); + match s.to_lowercase().as_str() { + "count" => Some(Statistic::Count), + "sum" => Some(Statistic::Sum), + "cardinality" => Some(Statistic::Cardinality), + "frequency_l2" => Some(Statistic::FrequencyL2), + "frequency_entropy" => Some(Statistic::FrequencyEntropy), + "increase" => Some(Statistic::Increase), + "rate" => Some(Statistic::Rate), + "min" => Some(Statistic::Min), + "max" => Some(Statistic::Max), + "quantile" => Some(Statistic::Quantile), + "topk" => Some(Statistic::Topk), + _ => None, + } + } +} + +impl FromStr for Statistic { + type Err = (); + + /// Parse a statistic from a string (case-insensitive). + /// Use `s.parse::()` or `Statistic::from_str(s)`. + fn from_str(s: &str) -> Result { + debug!("FromStr trait parsing Statistic: {}", s); + Statistic::from_str(s).ok_or(()) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs new file mode 100644 index 00000000..5a78fc48 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch.rs @@ -0,0 +1,76 @@ +//! Count-Min Sketch frequency summary over `asap_sketchlib::CountMinSketch`. +use crate::{AggregateCore, KernelError, KeyByLabelValues}; +use asap_sketchlib::CountMinSketch; + +#[derive(Debug, Clone)] +pub struct CountMinSketchAccumulator { + pub inner: CountMinSketch, +} + +impl CountMinSketchAccumulator { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + inner: CountMinSketch::new(row_num, col_num), + } + } + + /// Estimated frequency of one item. + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + self.inner.estimate(&key.to_semicolon_str()) + } +} + +impl AggregateCore for CountMinSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("Count-Min Sketch merges only with Count-Min Sketch")?; + Ok(Box::new(Self { + inner: CountMinSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + fn approx_memory_bytes(&self) -> usize { + 16 * 1024 + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merged point counts add item frequencies and never underestimate. + #[test] + fn merged_point_counts_add() { + let (mut a, mut b) = ( + CountMinSketchAccumulator::new(3, 128), + CountMinSketchAccumulator::new(3, 128), + ); + let key = KeyByLabelValues::new_with_labels(vec!["checkout".into()]); + a.inner.update(&key.to_semicolon_str(), 2.0); + b.inner.update(&key.to_semicolon_str(), 3.0); + let merged = a.merge_with(&b).unwrap(); + let merged = merged + .as_any() + .downcast_ref::() + .unwrap(); + assert!(merged.query_key(&key) >= 5.0); + } + + // Merge rejects a different summary family. + #[test] + fn rejects_foreign_merge() { + let cms = CountMinSketchAccumulator::new(3, 128); + let kll = crate::summary_kernels::DatasketchesKLLAccumulator::new(200); + assert!(cms.merge_with(&kll).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs new file mode 100644 index 00000000..da07aeee --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_min_sketch_with_heap.rs @@ -0,0 +1,234 @@ +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountMinSketchWithHeap; + +/// Count-Min Sketch with a top-k heap over `asap_sketchlib::CountMinSketchWithHeap`. +#[derive(Debug, Clone)] +pub struct CountMinSketchWithHeapAccumulator { + pub inner: CountMinSketchWithHeap, +} + +impl CountMinSketchWithHeapAccumulator { + pub fn new(row_num: usize, col_num: usize, heap_size: usize) -> Self { + Self { + inner: CountMinSketchWithHeap::new(row_num, col_num, heap_size), + } + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + let key_string = key.labels.join(";"); + self.inner.estimate(&key_string) + } + + /// VALUE-WEIGHTED heavy-hitter update (FIX: CountSketch/CMS topk + /// recall-0). The default ingest path inserts `+1` per occurrence keyed + /// by the raw `item`, so the heap ranks groups by OCCURRENCE COUNT — the + /// wrong answer for `topk(k, sum by (label) (metric))`, which asks for + /// the top groups by SUM OF VALUE. This update adds the sample `value` + /// (not `+1`) into both the CMS matrix and the top-k heap, keyed by the + /// GROUP LABEL (e.g. the `host` / `zone` value), so the heap's ranking is + /// by summed value. Repeated calls for the same `group_label` accumulate, + /// so after folding a window the heap holds Σvalue per group. + /// + /// Delegates to the library's value-weighted `CountMinSketchWithHeap:: + /// update(key, value)` (`sketchlib_cms_heap_update` → `insert_many(key, + /// round(value))`), which is the "separate update path" the evaluation + /// plan (Fig 3c) called for. + pub fn insert_value(&mut self, group_label: &str, value: f64) { + self.inner.update(group_label, value); + } + + /// Read the top-`k` GROUPS ranked by summed VALUE (descending), keyed by + /// the group label. Pairs with [`Self::insert_value`]: the heap built by + /// value-weighted updates ranks by Σvalue, so this returns the + /// value-weighted top-k (not the occurrence-count top-k the raw `item` + /// heap would give). Sorted descending by value; ties broken by key for + /// determinism; truncated to `k`. + pub fn topk_by_value(&self, k: usize) -> Vec<(String, f64)> { + let mut items: Vec<(String, f64)> = self + .inner + .topk_heap_items() + .into_iter() + .map(|it| (it.key, it.value)) + .collect(); + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) + }); + items.truncate(k); + items + } + + /// Get all keys from the top-k heap. + pub fn get_topk_keys(&self) -> Vec { + self.inner + .topk_heap_items() + .iter() + .map(|item| { + let labels: Vec = item.key.split(';').map(|s| s.to_string()).collect(); + KeyByLabelValues { labels } + }) + .collect() + } +} + +impl AggregateCore for CountMinSketchWithHeapAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cms = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountMinSketchWithHeapAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&other_cms.inner)?; + Ok(Box::new(merged)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_min_sketch_with_heap_creation() { + let cms = CountMinSketchWithHeapAccumulator::new(4, 1000, 20); + assert_eq!(cms.inner.rows(), 4); + assert_eq!(cms.inner.cols(), 1000); + assert_eq!(cms.inner.heap_size, 20); + assert_eq!(cms.inner.topk_heap_items().len(), 0); + } + + #[test] + fn test_get_topk_keys() { + let mut cms = CountMinSketchWithHeapAccumulator::new(2, 3, 5); + cms.inner.update("label1;label2", 100.0); + cms.inner.update("label3;label4", 50.0); + + let keys = cms.get_topk_keys(); + assert_eq!(keys.len(), 2); + // Heap order is not part of the contract; compare as a set. + let label_sets: std::collections::HashSet<_> = + keys.iter().map(|k| k.labels.clone()).collect(); + assert!(label_sets.contains(&vec!["label1".to_string(), "label2".to_string()])); + assert!(label_sets.contains(&vec!["label3".to_string(), "label4".to_string()])); + } + + // ---------------------------------------------------------------- + // FIX 1 — VALUE-WEIGHTED top-k (recall 0 → correct). + // + // `topk(k, sum by (host) (cpu_load))` asks for the top-k hosts by + // SUM OF VALUE. The heavy-hitter heap built by the default `+1`-per- + // occurrence update ranks by COUNT keyed by `item`, so its recall + // against the value-weighted ground truth is 0 when the busiest host + // (most samples) is NOT the heaviest host (largest Σvalue). + // `insert_value(group_label, value)` adds the sample VALUE keyed by the + // GROUP LABEL, so `topk_by_value` ranks by Σvalue — correct recall. + // ---------------------------------------------------------------- + + /// Crafted adversarial dataset: the host with the MOST samples + /// (`h_chatty`, 100 tiny samples) is NOT the host with the largest + /// value-sum (`h_heavy`, a handful of huge samples). A COUNT-ranked + /// heap would surface `h_chatty`; the value-weighted top-k must surface + /// the true heavy hitters by Σvalue, giving recall 1.0 against the + /// ground-truth top-k-by-value-sum. + #[test] + fn value_weighted_topk_has_full_recall_vs_count_topk() { + // (host, per-sample value, sample count) → true Σvalue: + // h_heavy : 1000 × 3 = 3000 (few samples, huge value) + // h_mid : 200 × 5 = 1000 + // h_small : 50 × 6 = 300 + // h_chatty: 1 × 100 = 100 (MOST samples, tiny value) + let data: &[(&str, f64, usize)] = &[ + ("h_heavy", 1000.0, 3), + ("h_mid", 200.0, 5), + ("h_small", 50.0, 6), + ("h_chatty", 1.0, 100), + ]; + + // Wide CMS + heap large enough to hold every group exactly (4 groups) + // so the estimate equals the true Σvalue with no hash collisions. + let mut acc = CountMinSketchWithHeapAccumulator::new(5, 4096, 16); + let mut truth: std::collections::HashMap<&str, f64> = std::collections::HashMap::new(); + for (host, value, count) in data { + for _ in 0..*count { + acc.insert_value(host, *value); + } + *truth.entry(*host).or_insert(0.0) += value * (*count as f64); + } + + // Ground-truth top-2 by value-sum: h_heavy (3000), h_mid (1000). + let mut truth_ranked: Vec<(&str, f64)> = truth.into_iter().collect(); + truth_ranked.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap()); + let truth_top2: std::collections::HashSet<&str> = + truth_ranked.iter().take(2).map(|(k, _)| *k).collect(); + assert!( + truth_top2.contains("h_heavy") && truth_top2.contains("h_mid"), + "ground-truth top-2 by value-sum should be h_heavy + h_mid" + ); + + // Value-weighted top-2 from the heap. + let got = acc.topk_by_value(2); + assert_eq!(got.len(), 2, "k=2 → two groups: {got:?}"); + let got_keys: std::collections::HashSet<&str> = + got.iter().map(|(k, _)| k.as_str()).collect(); + + // RECALL = |got ∩ truth| / |truth| must be 1.0. + let hits = got_keys.intersection(&truth_top2).count(); + let recall = hits as f64 / truth_top2.len() as f64; + assert_eq!( + recall, 1.0, + "value-weighted top-k recall must be 1.0 (count-ranked heap would \ + surface h_chatty and miss h_heavy → recall < 1): got={got:?}" + ); + + // The busiest-by-count host (h_chatty) must NOT be in the top-2, + // proving we rank by value-sum, not occurrence count. + assert!( + !got_keys.contains("h_chatty"), + "h_chatty (most samples, smallest value-sum) must be excluded: {got:?}" + ); + + // Estimates are exact here (no collisions, heap holds all groups): + // top-1 must be h_heavy with Σvalue 3000. + assert_eq!(got[0].0, "h_heavy"); + assert!( + (got[0].1 - 3000.0).abs() < 1e-6, + "h_heavy value-sum estimate ≈ 3000, got {}", + got[0].1 + ); + assert_eq!(got[1].0, "h_mid"); + assert!( + (got[1].1 - 1000.0).abs() < 1e-6, + "h_mid value-sum estimate ≈ 1000, got {}", + got[1].1 + ); + } + + /// A single value-weighted insert must put the full value (not +1) into + /// the heap, and repeated inserts for the same group must accumulate. + #[test] + fn insert_value_accumulates_summed_value_in_heap() { + let mut acc = CountMinSketchWithHeapAccumulator::new(4, 1024, 8); + acc.insert_value("g", 10.0); + acc.insert_value("g", 25.0); + let top = acc.topk_by_value(1); + assert_eq!(top.len(), 1); + assert_eq!(top[0].0, "g"); + assert!( + (top[0].1 - 35.0).abs() < 1e-6, + "summed value should be 35 (10+25), got {}", + top[0].1 + ); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs new file mode 100644 index 00000000..2728b5f2 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch.rs @@ -0,0 +1,96 @@ +//! CountSketch accumulator backed by `asap_sketchlib::CountSketch`. +//! +//! Per-key queries delegate to sketchlib's median-of-signed-rows estimator. +//! Top-k requires the separate heap-bearing accumulator. + +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountSketch; + +/// Count Sketch accumulator — inner matrix of signed counts. +#[derive(Debug, Clone)] +pub struct CountSketchAccumulator { + pub inner: CountSketch, +} + +impl CountSketchAccumulator { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + inner: CountSketch::new(row_num, col_num), + } + } + + /// Median-of-signed-rows point estimate for `key`, via + /// `asap_sketchlib::CountSketch::estimate`. + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + self.inner.estimate(&key.to_semicolon_str()) + } +} + +impl AggregateCore for CountSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cs = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountSketchAccumulator")?; + + let merged_inner = CountSketch::merge_refs(&[&self.inner, &other_cs.inner])?; + Ok(Box::new(Self { + inner: merged_inner, + })) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_query_key_uses_real_sketchlib_estimator() { + // `query_key` must match sketchlib's estimator and hash specification. + let mut cs = CountSketchAccumulator::new(4, 1000); + let key = KeyByLabelValues::new_with_labels(vec!["web".to_string()]); + cs.inner.update(&key.to_semicolon_str(), 10.0); + assert_eq!( + cs.query_key(&key), + cs.inner.estimate(&key.to_semicolon_str()) + ); + } + + #[test] + fn test_aggregate_core_merge_matches_matrix_add() { + let a = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![1.0, -2.0], vec![3.0, -4.0]], 2, 2), + }; + let b = CountSketchAccumulator { + inner: CountSketch::from_legacy_matrix(vec![vec![-1.0, 2.0], vec![-3.0, 4.0]], 2, 2), + }; + let merged_box = a.merge_with(&b).expect("merge ok"); + let merged = merged_box + .as_any() + .downcast_ref::() + .expect("downcast ok"); + let m = merged.inner.sketch(); + assert_eq!(m[0], vec![0.0, 0.0]); + assert_eq!(m[1], vec![0.0, 0.0]); + } + + #[test] + fn test_aggregate_core_merge_wrong_type_rejects() { + use crate::summary_kernels::count_min_sketch::CountMinSketchAccumulator; + let cs = CountSketchAccumulator::new(2, 3); + let cms = CountMinSketchAccumulator::new(2, 3); + let result = cs.merge_with(&cms); + assert!(result.is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs new file mode 100644 index 00000000..a735e4bd --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/count_sketch_with_heap.rs @@ -0,0 +1,143 @@ +//! CountSketch with a top-k heap over `asap_sketchlib::CountSketchWithHeap` +//! (median-of-signed-rows), distinct from the Count-Min heap variant. + +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::CountSketchWithHeap; + +#[derive(Debug, Clone)] +pub struct CountSketchWithHeapAccumulator { + pub inner: CountSketchWithHeap, +} + +impl CountSketchWithHeapAccumulator { + pub fn new(row_num: usize, col_num: usize, heap_size: usize) -> Self { + Self { + inner: CountSketchWithHeap::new(row_num, col_num, heap_size), + } + } + + pub fn query_key(&self, key: &KeyByLabelValues) -> f64 { + let key_string = key.labels.join(";"); + self.inner.estimate(&key_string) + } + + /// Value-weighted heavy-hitter update -- see + /// `CountMinSketchWithHeapAccumulator::insert_value`'s doc for why + /// this (not a `+1`-per-occurrence update) is the correct semantics + /// for `topk(k, sum by (label) (metric))`-shaped queries. + pub fn insert_value(&mut self, group_label: &str, value: f64) { + self.inner.update(group_label, value); + } + + /// Read the top-`k` groups ranked by summed value (descending, tie-broken + /// by key for determinism). Mirrors `CountMinSketchWithHeapAccumulator::topk_by_value`. + pub fn topk_by_value(&self, k: usize) -> Vec<(String, f64)> { + let mut items: Vec<(String, f64)> = self + .inner + .topk_heap_items() + .into_iter() + .map(|it| (it.key, it.value)) + .collect(); + items.sort_by(|a, b| { + b.1.partial_cmp(&a.1) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.0.cmp(&b.0)) + }); + items.truncate(k); + items + } + + /// Get all keys from the top-k heap. + pub fn get_topk_keys(&self) -> Vec { + self.inner + .topk_heap_items() + .iter() + .map(|item| { + let labels: Vec = item.key.split(';').map(|s| s.to_string()).collect(); + KeyByLabelValues { labels } + }) + .collect() + } +} + +impl AggregateCore for CountSketchWithHeapAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other_cs = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to CountSketchWithHeapAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&other_cs.inner)?; + Ok(Box::new(merged)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_count_sketch_with_heap_creation() { + let cs = CountSketchWithHeapAccumulator::new(4, 1000, 20); + assert_eq!(cs.inner.rows(), 4); + assert_eq!(cs.inner.cols(), 1000); + assert_eq!(cs.inner.heap_size, 20); + assert_eq!(cs.inner.topk_heap_items().len(), 0); + } + + #[test] + fn test_get_topk_keys() { + let mut cs = CountSketchWithHeapAccumulator::new(2, 3, 5); + cs.inner.update("label1;label2", 100.0); + cs.inner.update("label3;label4", 50.0); + + let keys = cs.get_topk_keys(); + assert_eq!(keys.len(), 2); + let label_sets: std::collections::HashSet<_> = + keys.iter().map(|k| k.labels.clone()).collect(); + assert!(label_sets.contains(&vec!["label1".to_string(), "label2".to_string()])); + assert!(label_sets.contains(&vec!["label3".to_string(), "label4".to_string()])); + } + + #[test] + fn insert_value_accumulates_summed_value_in_heap() { + let mut acc = CountSketchWithHeapAccumulator::new(4, 1024, 8); + acc.insert_value("g", 10.0); + acc.insert_value("g", 25.0); + let top = acc.topk_by_value(1); + assert_eq!(top.len(), 1); + assert_eq!(top[0].0, "g"); + assert!( + (top[0].1 - 35.0).abs() < 1e-6, + "summed value should be 35 (10+25), got {}", + top[0].1 + ); + } + + /// CountSketch and Count-Min heap states are distinct families and never merge. + #[test] + fn test_rejects_merge_with_cms_family_accumulator() { + use crate::summary_kernels::count_min_sketch_with_heap::CountMinSketchWithHeapAccumulator; + + let cs = CountSketchWithHeapAccumulator::new(4, 64, 10); + let cms = CountMinSketchWithHeapAccumulator::new(4, 64, 10); + let result = cs.merge_with(&cms); + assert!( + result.is_err(), + "CountSketchWithHeapAccumulator must not merge with CountMinSketchWithHeapAccumulator \ + -- different algorithms sharing only a storage shape" + ); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs new file mode 100644 index 00000000..0f740a9d --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/datasketches_kll.rs @@ -0,0 +1,106 @@ +//! KLL quantile summary over `asap_sketchlib::KllSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::KllSketch; +use planner_types::post_asap::SketchQuery; + +#[derive(Clone)] +pub struct DatasketchesKLLAccumulator { + pub inner: KllSketch, +} + +impl DatasketchesKLLAccumulator { + pub fn new(k: u16) -> Self { + Self { + inner: KllSketch::new(k), + } + } + + pub fn update(&mut self, value: f64) { + self.inner.update(value); + } + + pub fn get_quantile(&self, quantile: f64) -> f64 { + self.inner.quantile(quantile) + } +} + +impl std::fmt::Debug for DatasketchesKLLAccumulator { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DatasketchesKLLAccumulator") + .field("k", &self.inner.k) + .field("sketch_n", &self.inner.count()) + .finish() + } +} + +// SAFETY: `KllSketch` owns its buffers and has no interior mutability; the +// accumulator is only mutated through `&mut self`. +unsafe impl Send for DatasketchesKLLAccumulator {} +unsafe impl Sync for DatasketchesKLLAccumulator {} + +impl AggregateCore for DatasketchesKLLAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("KLL merges only with KLL")?; + Ok(Box::new(Self { + inner: KllSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Quantile { q } if (0.0..=1.0).contains(q) => Ok(self.get_quantile(*q)), + SketchQuery::Quantile { .. } => Err("quantile must be in [0, 1]".into()), + other => Err(format!("KLL does not answer {other:?}").into()), + } + } + + fn approx_memory_bytes(&self) -> usize { + // KLL with default k=200 holds ~2*k items (~3 KiB); round up for overhead. + 4 * 1024 + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merging two KLL states reads like one state built over both inputs. + #[test] + fn merged_quantile_matches_single_build() { + let (mut a, mut b, mut all) = ( + DatasketchesKLLAccumulator::new(200), + DatasketchesKLLAccumulator::new(200), + DatasketchesKLLAccumulator::new(200), + ); + for v in 0..100 { + a.update(f64::from(v)); + all.update(f64::from(v)); + } + for v in 100..200 { + b.update(f64::from(v)); + all.update(f64::from(v)); + } + let merged = a.merge_with(&b).unwrap(); + let q = SketchQuery::Quantile { q: 0.5 }; + assert_eq!(merged.estimate(&q).unwrap(), all.estimate(&q).unwrap()); + } + + // KLL answers only quantiles in [0, 1]. + #[test] + fn rejects_unsupported_or_out_of_range_queries() { + let kll = DatasketchesKLLAccumulator::new(200); + assert!(kll.estimate(&SketchQuery::Quantile { q: 1.5 }).is_err()); + assert!(kll.estimate(&SketchQuery::Cardinality).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs new file mode 100644 index 00000000..1d6be98c --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/dd_sketch.rs @@ -0,0 +1,88 @@ +//! DDSketch quantile summary over `asap_sketchlib::DdSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::DdSketch; +use planner_types::post_asap::SketchQuery; + +#[derive(Debug, Clone)] +pub struct DDSketchAccumulator { + pub inner: DdSketch, +} + +impl DDSketchAccumulator { + pub fn new(alpha: f64) -> Self { + Self { + inner: DdSketch::new(alpha), + } + } +} + +impl AggregateCore for DDSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("DDSketch merges only with DDSketch")?; + Ok(Box::new(Self { + inner: DdSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + /// Quantiles, and the total sample count as a bare `PointCount`. + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Quantile { q } if (0.0..=1.0).contains(q) => self + .inner + .quantile(*q) + .ok_or_else(|| "DDSketch quantile of an empty population".into()), + SketchQuery::Quantile { .. } => Err("quantile must be in [0, 1]".into()), + SketchQuery::PointCount { value: None, .. } => Ok(self.inner.total_count() as f64), + other => Err(format!("DDSketch does not answer {other:?}").into()), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use planner_types::pre_asap::ColumnRef; + + fn bare_count() -> SketchQuery { + SketchQuery::PointCount { + key: ColumnRef::SampleValue, + value: None, + } + } + + // A bare point count reads the total sample count, and merge adds counts. + #[test] + fn count_and_quantile_survive_merge() { + let (mut a, mut b) = ( + DDSketchAccumulator::new(0.01), + DDSketchAccumulator::new(0.01), + ); + for v in 1..=50 { + a.inner.update(f64::from(v)); + b.inner.update(f64::from(v + 50)); + } + let merged = a.merge_with(&b).unwrap(); + assert_eq!(merged.estimate(&bare_count()).unwrap(), 100.0); + let median = merged.estimate(&SketchQuery::Quantile { q: 0.5 }).unwrap(); + assert!((median - 50.0).abs() <= 1.0, "{median}"); + } + + // An empty DDSketch has no quantile, and unsupported queries are errors. + #[test] + fn empty_quantile_and_unsupported_queries_fail() { + let dd = DDSketchAccumulator::new(0.01); + assert!(dd.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + assert!(dd.estimate(&SketchQuery::Cardinality).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/exact.rs b/crates/asap-physical-operators/src/summary_kernels/exact.rs new file mode 100644 index 00000000..cb9d6099 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/exact.rs @@ -0,0 +1,237 @@ +//! Exact summary state identified by Planner family, independent of keyed layout. +use super::increase::IncreaseAccumulator; +use crate::Statistic; +use crate::{AggregateCore, KeyByLabelValues, Measurement}; +use planner_types::post_asap::{ExactKind, ExactParams, SummaryFamilyType}; +use serde::{Deserialize, Serialize}; +use std::collections::HashMap; + +type Error = Box; + +#[derive(Debug, Clone, Serialize, Deserialize)] +enum ScalarState { + Sum(f64), + Count(u64), + Min(Option), + Max(Option), + Counter(Option), +} + +/// Both the family and population layout survive persistence. Sharing counter +/// arithmetic never authorizes a Rate state to answer an Increase readout. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ExactAccumulator { + family: SummaryFamilyType, + scalar: ScalarState, + keyed: Option>, +} + +/// Planned readout of an exact summary. `lookback_ms` is the logical PromQL +/// counter window; the evaluation range is resolved from it at run time. +#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] +pub struct ExactReadout { + pub statistic: Statistic, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub lookback_ms: Option, +} + +impl ExactAccumulator { + /// Read one population. An empty MIN/MAX population reads as `None`. + /// `range_ms` extrapolates a counter Rate/Increase to that evaluation range. + pub fn readout( + &self, + statistic: Statistic, + range_ms: Option<(i64, i64)>, + key: Option<&KeyByLabelValues>, + ) -> Result, Error> { + if statistic != self.statistic() { + return Err("readout differs from Planner exact family".into()); + } + let state = match (&self.keyed, key) { + (Some(states), Some(key)) => states.get(key).ok_or("unknown exact population")?, + (None, None) => &self.scalar, + _ => return Err("readout population differs from installed layout".into()), + }; + match state { + ScalarState::Sum(sum) => Ok(Some(*sum)), + ScalarState::Count(count) => Ok(Some(*count as f64)), + ScalarState::Min(value) | ScalarState::Max(value) => Ok(*value), + ScalarState::Counter(Some(counter)) => counter + .extrapolated_value(range_ms, statistic == Statistic::Rate) + .map(Some), + ScalarState::Counter(None) => Err("empty counter population".into()), + } + } + + /// Exact integer count of an unkeyed Count state. + pub fn count(&self) -> Option { + match (&self.keyed, &self.scalar) { + (None, ScalarState::Count(count)) => Some(*count), + _ => None, + } + } + + /// Accumulate into run-local scratch state. Persistent input states remain + /// immutable; a failed merge discards this scratch state. + pub(crate) fn merge_from(&mut self, other: &Self) -> Result<(), Error> { + if self.family != other.family || self.is_keyed() != other.is_keyed() { + return Err("cannot merge different Planner families or layouts".into()); + } + if let (Some(target), Some(source)) = (&mut self.keyed, &other.keyed) { + for (key, state) in source { + let combined = match target.get(key) { + Some(old) => merge_scalar(old, state)?, + None => state.clone(), + }; + target.insert(key.clone(), combined); + } + } else { + self.scalar = merge_scalar(&self.scalar, &other.scalar)?; + } + Ok(()) + } + + pub fn new(family: SummaryFamilyType, keyed: bool) -> Result { + use ExactKind as K; + use ExactParams as P; + let scalar = match &family { + SummaryFamilyType::ExactAggregate(K::Sum, P::Sum) => ScalarState::Sum(0.0), + SummaryFamilyType::ExactAggregate(K::Count, P::Count) => ScalarState::Count(0), + SummaryFamilyType::ExactAggregate(K::Min, P::Min) => ScalarState::Min(None), + SummaryFamilyType::ExactAggregate(K::Max, P::Max) => ScalarState::Max(None), + SummaryFamilyType::ExactAggregate(K::Rate, P::Rate) + | SummaryFamilyType::ExactAggregate(K::Increase, P::Increase) => { + ScalarState::Counter(None) + } + _ => return Err(format!("unsupported exact Planner family: {family:?}")), + }; + Ok(Self { + family, + scalar, + keyed: keyed.then(HashMap::new), + }) + } + + pub fn family(&self) -> &SummaryFamilyType { + &self.family + } + pub(crate) fn insufficient_counter_samples( + &self, + statistic: Statistic, + key: &Option, + ) -> bool { + if statistic != self.statistic() { + return false; + } + let state = match (&self.keyed, key) { + (Some(states), Some(key)) => states.get(key), + (None, None) => Some(&self.scalar), + _ => None, + }; + match state { + Some(ScalarState::Counter(None)) => true, + Some(ScalarState::Counter(Some(counter))) => { + counter.sample_count < 2 + || counter.last_seen_timestamp == counter.starting_timestamp + } + _ => false, + } + } + pub fn is_keyed(&self) -> bool { + self.keyed.is_some() + } + + pub fn update(&mut self, key: Option<&KeyByLabelValues>, value: f64, timestamp: i64) { + let state = match (&mut self.keyed, key) { + (Some(states), Some(key)) => states + .entry(key.clone()) + .or_insert_with(|| self.scalar.clone()), + (None, None) => &mut self.scalar, + _ => panic!("exact update population layout differs from installed DAG"), + }; + match state { + ScalarState::Sum(sum) => *sum += value, + ScalarState::Count(count) => { + *count = count.checked_add(1).expect("exact count overflow") + } + ScalarState::Min(current) => { + *current = Some(current.map_or(value, |old| old.min(value))) + } + ScalarState::Max(current) => { + *current = Some(current.map_or(value, |old| old.max(value))) + } + ScalarState::Counter(current) => match current { + Some(counter) => counter.update(Measurement::new(value), timestamp), + None => { + *current = Some(IncreaseAccumulator::new( + Measurement::new(value), + timestamp, + Measurement::new(value), + timestamp, + )) + } + }, + } + } + + fn statistic(&self) -> Statistic { + match self.family { + SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) => Statistic::Sum, + SummaryFamilyType::ExactAggregate(ExactKind::Count, _) => Statistic::Count, + SummaryFamilyType::ExactAggregate(ExactKind::Min, _) => Statistic::Min, + SummaryFamilyType::ExactAggregate(ExactKind::Max, _) => Statistic::Max, + SummaryFamilyType::ExactAggregate(ExactKind::Rate, _) => Statistic::Rate, + SummaryFamilyType::ExactAggregate(ExactKind::Increase, _) => Statistic::Increase, + _ => unreachable!("validated exact family"), + } + } +} + +fn merge_scalar(left: &ScalarState, right: &ScalarState) -> Result { + Ok(match (left, right) { + (ScalarState::Sum(a), ScalarState::Sum(b)) => ScalarState::Sum(a + b), + (ScalarState::Count(a), ScalarState::Count(b)) => { + ScalarState::Count(a.checked_add(*b).ok_or("exact count overflow")?) + } + (ScalarState::Min(a), ScalarState::Min(b)) => { + ScalarState::Min(a.iter().chain(b).copied().reduce(f64::min)) + } + (ScalarState::Max(a), ScalarState::Max(b)) => { + ScalarState::Max(a.iter().chain(b).copied().reduce(f64::max)) + } + (ScalarState::Counter(a), ScalarState::Counter(b)) => ScalarState::Counter(match (a, b) { + (Some(a), Some(b)) => Some(IncreaseAccumulator::merge_pair(a, b)), + (a, b) => a.clone().or_else(|| b.clone()), + }), + _ => return Err("exact scalar state families differ".into()), + }) +} + +impl AggregateCore for ExactAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("merge requires Planner exact state")?; + let mut merged = self.clone(); + merged.merge_from(other)?; + Ok(Box::new(merged)) + } + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::() + + self.keyed.as_ref().map_or(0, |m| { + m.keys() + .map(|k| { + std::mem::size_of::() + + k.labels.iter().map(String::len).sum::() + }) + .sum::() + }) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/factory.rs b/crates/asap-physical-operators/src/summary_kernels/factory.rs new file mode 100644 index 00000000..d44e64ab --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/factory.rs @@ -0,0 +1,799 @@ +use crate::summary_kernels::hll_sketch::HllSketchAccumulator; +use crate::summary_kernels::univmon::UnivMonAccumulator; +use crate::summary_kernels::{ + CountMinSketchAccumulator, CountMinSketchWithHeapAccumulator, CountSketchAccumulator, + CountSketchWithHeapAccumulator, DDSketchAccumulator, DatasketchesKLLAccumulator, + HydraKllSketchAccumulator, +}; +use crate::{AggregateCore, KeyByLabelValues}; +use planner_types::post_asap::{SketchAlgorithm, SketchParams, SummaryFamilyType}; + +/// Generate the clone-based `AccumulatorUpdater` methods for updaters whose +/// inner `acc` field implements `Clone + AggregateCore`. +macro_rules! impl_clone_accumulator_methods { + ($acc_field:ident) => { + fn take_accumulator(&mut self) -> Box { + let result = Box::new(self.$acc_field.clone()); + self.reset(); + result + } + + fn snapshot_accumulator(&self) -> Box { + Box::new(self.$acc_field.clone()) + } + + fn into_accumulator(self: Box) -> Box { + // Consume the updater and MOVE the accumulator out — no clone. + // Avoids a clone when a pane is evicted at window close. + let this = *self; + Box::new(this.$acc_field) + } + }; +} + +/// Shared update interface for query-time and precompute-time accumulation. +/// +/// This provides a uniform interface over all accumulator types so that the +/// operators don't need to know which concrete type they're dealing with. +pub trait AccumulatorUpdater: Send { + /// Validate an immutable precompute input before an updater can silently + /// discard a value outside its representable domain. + fn validate_single_input(&self, value: f64) -> Result<(), String> { + if value.is_finite() { + Ok(()) + } else { + Err("accumulator input must be finite".into()) + } + } + + /// Feed a single (value, timestamp_ms) pair — for SingleSubpopulation types. + fn update_single(&mut self, value: f64, timestamp_ms: i64); + + /// Feed a keyed (key, value, timestamp_ms) triple, e.g. a frequency item or an exact keyed state. + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp_ms: i64); + + /// Extract the final accumulator as a boxed `AggregateCore`. + fn take_accumulator(&mut self) -> Box; + + /// Non-destructive read of the current accumulator state (clone without reset). + /// Used by pane-based sliding windows to read shared panes. + fn snapshot_accumulator(&self) -> Box; + + /// Consume the updater and return its accumulator by move, avoiding the + /// clone that `take_accumulator`/`snapshot_accumulator` pay. Default falls + /// back to a clone for updaters that can't move their inner state out. + fn into_accumulator(self: Box) -> Box { + self.snapshot_accumulator() + } + + /// Reset internal state for reuse (avoids re-allocation). + fn reset(&mut self); + + /// Whether this updater consumes keyed updates. + fn is_keyed(&self) -> bool; + + /// Estimated memory usage in bytes. + fn memory_usage_bytes(&self) -> usize; +} + +// --------------------------------------------------------------------------- +// KllAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct KllAccumulatorUpdater { + acc: DatasketchesKLLAccumulator, + k: u16, +} + +impl KllAccumulatorUpdater { + pub fn new(k: u16) -> Self { + Self { + acc: DatasketchesKLLAccumulator::new(k), + k, + } + } +} + +impl AccumulatorUpdater for KllAccumulatorUpdater { + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = DatasketchesKLLAccumulator::new(self.k); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + // KLL sketch size is hard to estimate precisely; use a rough estimate + std::mem::size_of::() + 4096 + } +} + +// --------------------------------------------------------------------------- +// DDSketchAccumulatorUpdater +// --------------------------------------------------------------------------- +pub struct DDSketchAccumulatorUpdater { + acc: DDSketchAccumulator, + alpha: f64, +} + +impl DDSketchAccumulatorUpdater { + pub fn new(alpha: f64) -> Self { + Self { + acc: DDSketchAccumulator::new(alpha), + alpha, + } + } +} + +impl AccumulatorUpdater for DDSketchAccumulatorUpdater { + fn validate_single_input(&self, value: f64) -> Result<(), String> { + let (minimum, maximum) = + asap_sketchlib::sketches::ddsketch::ddsketch_indexable_bounds(self.alpha); + if value.is_finite() && value > 0.0 && value >= minimum && value <= maximum { + Ok(()) + } else { + Err("DDS maintenance input is outside its positive representable domain".into()) + } + } + + fn update_single(&mut self, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(value); + } + + fn update_keyed(&mut self, _key: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = DDSketchAccumulator::new(self.alpha); + } + + fn is_keyed(&self) -> bool { + false + } + + fn memory_usage_bytes(&self) -> usize { + // Bucket store is variable; rough estimate matches KLL. + std::mem::size_of::() + 4096 + } +} + +// --------------------------------------------------------------------------- +// CmsAccumulatorUpdater (CountMinSketch) +// --------------------------------------------------------------------------- + +/// Keyed weighted-frequency updater. +/// +/// A raw Prometheus sample represents the observed metric value, so a bare CMS +/// adds `value` for its key. Counting each received sample as one is a distinct +/// event-count operation and requires an explicit typed plan contract; it must +/// not be inferred from the sketch algorithm alone. +pub struct CmsAccumulatorUpdater { + acc: CountMinSketchAccumulator, + row_num: usize, + col_num: usize, +} + +impl CmsAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + acc: CountMinSketchAccumulator::new(row_num, col_num), + row_num, + col_num, + } + } +} + +impl AccumulatorUpdater for CmsAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(&key.to_semicolon_str(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountMinSketchAccumulator::new(self.row_num, self.col_num); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// CmsHeapAccumulatorUpdater — value-weighted / count-weighted top-k +// --------------------------------------------------------------------------- + +/// What quantity the top-k heap ranks keys by. +/// +/// These are DIFFERENT query semantics and must be chosen explicitly: +/// +/// * [`TopkWeight::Value`] — accumulate **Σ of the datapoint value** per key. +/// This answers "top-k by total " (e.g. "top-k hosts by +/// total CPU"). The heap value is the summed metric value, so the read-side +/// reducer's "sort heap descending by value" yields the correct ranking. +/// +/// * [`TopkWeight::Count`] — accumulate **+1 per event** per key (occurrence +/// frequency), the textbook heavy-hitter / frequency-top-k semantics +/// ("which keys appear most often"). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum TopkWeight { + /// Σ datapoint value per key (value-weighted top-k). + Value, + /// +1 per event per key (count-weighted / frequency top-k). + Count, +} + +/// Keyed top-k updater backed by a real `CountMinSketchWithHeap` (a CMS +/// matrix PLUS a size-`heap_size` top-k heap). Unlike the heap-LESS +/// `CmsAccumulatorUpdater`, this enumerates top-k keys at read time +/// (`get_topk_keys` / `topk_heap_items`), which is what `topk(...)` queries +/// need. +/// +/// The key is the frequency item supplied by the operator (e.g. a `host` +/// value), not a group-by population. The accumulated quantity is selected +/// by [`TopkWeight`]: +/// * `Value` → `inner.update(key, value)` adds the datapoint value (Σ value). +/// * `Count` → `inner.update(key, 1.0)` adds one per event (Σ count). +/// +/// Both `CountMinSketchWithHeap` and `CountSketchWithHeap` raw-input policies +/// route here; the heap is the shared distinguishing payload. +pub struct CmsHeapAccumulatorUpdater { + acc: CountMinSketchWithHeapAccumulator, + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, +} + +impl CmsHeapAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { + Self { + acc: CountMinSketchWithHeapAccumulator::new(row_num, col_num, heap_size), + row_num, + col_num, + heap_size, + weight, + } + } +} + +impl AccumulatorUpdater for CmsHeapAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + // Heap key = the group-by label-value vector (e.g. `host`), joined the + // same way the read-side `get_topk_keys` splits it back apart (`;`). + let weighted = match self.weight { + // Σ value: feed the datapoint value. sketchlib's CMS-heap + // `update(key, w)` adds `w.round()` occurrences of `key`, so the + // heap value accumulates the (rounded) summed metric value. + TopkWeight::Value => value, + // Σ count: one occurrence per event, regardless of value. + TopkWeight::Count => 1.0, + }; + self.acc.inner.update(&key.to_semicolon_str(), weighted); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = + CountMinSketchWithHeapAccumulator::new(self.row_num, self.col_num, self.heap_size); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + + self.heap_size * (std::mem::size_of::() + 32) + } +} + +// --------------------------------------------------------------------------- +// CountSketchAccumulatorUpdater (real median-of-signed-rows CountSketch) +// --------------------------------------------------------------------------- + +/// Keyed point-frequency updater backed by a real `asap_sketchlib::CountSketch` +/// (signed rows, median-of-rows estimator) — distinct math from +/// `CmsAccumulatorUpdater`'s CMS (min-of-rows). +/// +/// As with bare CMS, each raw Prometheus sample contributes its `value`. +/// Unit event counting must be selected explicitly by a future typed plan +/// contract rather than being implied by `SketchAlgorithm::CountSketch`. +pub struct CountSketchAccumulatorUpdater { + acc: CountSketchAccumulator, + row_num: usize, + col_num: usize, +} + +impl CountSketchAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize) -> Self { + Self { + acc: CountSketchAccumulator::new(row_num, col_num), + row_num, + col_num, + } + } +} + +impl AccumulatorUpdater for CountSketchAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.inner.update(&key.to_semicolon_str(), value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountSketchAccumulator::new(self.row_num, self.col_num); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + } +} + +// --------------------------------------------------------------------------- +// CountSketchWithHeapAccumulatorUpdater (real CountSketch + top-k heap) +// --------------------------------------------------------------------------- + +/// Keyed top-k updater backed by a real `CountSketchWithHeap` (signed-row +/// CountSketch matrix PLUS a size-`heap_size` top-k heap). Distinct math from +/// `CmsHeapAccumulatorUpdater`'s CMS-with-heap (min-of-rows); shares the same +/// [`TopkWeight`] semantics and heap payload shape. +pub struct CountSketchWithHeapAccumulatorUpdater { + acc: CountSketchWithHeapAccumulator, + row_num: usize, + col_num: usize, + heap_size: usize, + weight: TopkWeight, +} + +impl CountSketchWithHeapAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, heap_size: usize, weight: TopkWeight) -> Self { + Self { + acc: CountSketchWithHeapAccumulator::new(row_num, col_num, heap_size), + row_num, + col_num, + heap_size, + weight, + } + } +} + +impl AccumulatorUpdater for CountSketchWithHeapAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + let weighted = match self.weight { + TopkWeight::Value => value, + TopkWeight::Count => 1.0, + }; + self.acc.inner.update(&key.to_semicolon_str(), weighted); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = CountSketchWithHeapAccumulator::new(self.row_num, self.col_num, self.heap_size); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + std::mem::size_of::() + + self.row_num * self.col_num * std::mem::size_of::() + + self.heap_size * (std::mem::size_of::() + 32) + } +} + +// --------------------------------------------------------------------------- +// HydraKllAccumulatorUpdater +// --------------------------------------------------------------------------- + +pub struct HydraKllAccumulatorUpdater { + acc: HydraKllSketchAccumulator, + row_num: usize, + col_num: usize, + k: u16, +} + +impl HydraKllAccumulatorUpdater { + pub fn new(row_num: usize, col_num: usize, k: u16) -> Self { + Self { + acc: HydraKllSketchAccumulator::new(row_num, col_num, k), + row_num, + col_num, + k, + } + } +} + +impl AccumulatorUpdater for HydraKllAccumulatorUpdater { + fn update_single(&mut self, _value: f64, _timestamp_ms: i64) { + debug_assert!( + false, + "update_single called on keyed updater; use update_keyed" + ); + } + + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, _timestamp_ms: i64) { + self.acc.update(key, value); + } + + impl_clone_accumulator_methods!(acc); + + fn reset(&mut self) { + self.acc = HydraKllSketchAccumulator::new(self.row_num, self.col_num, self.k); + } + + fn is_keyed(&self) -> bool { + true + } + + fn memory_usage_bytes(&self) -> usize { + // Rough estimate: each cell is a KLL sketch + std::mem::size_of::() + self.row_num * self.col_num * 4096 + } +} + +// --------------------------------------------------------------------------- +// Config helpers +// --------------------------------------------------------------------------- + +fn cms_dims(params: &SketchParams) -> (usize, usize) { + match params { + SketchParams::Cms { width, depth } | SketchParams::CountSketch { width, depth } => { + (*depth as usize, *width as usize) + } + other => unreachable!( + "accumulator_spec() paired SketchAlgorithm::Cms/CountSketch with unexpected params: {other:?}" + ), + } +} + +/// Read `(rows = depth, columns = width, heap_size)` out of `SketchParams::CmsWithHeap` +/// or `::CountSketchWithHeap`. +fn cms_heap_dims(params: &SketchParams) -> (usize, usize, usize) { + match params { + SketchParams::CmsWithHeap { + width, + depth, + heap_size, + } + | SketchParams::CountSketchWithHeap { + width, + depth, + heap_size, + } => (*depth as usize, *width as usize, *heap_size as usize), + other => unreachable!( + "accumulator_spec() paired a WithHeap SketchAlgorithm with unexpected params: {other:?}" + ), + } +} + +/// Construct the kernel declared by a Planner SummaryAgg. No deployment config +/// tags participate in this dispatch and unsupported payloads are errors. +pub fn create_planner_accumulator( + family: &SummaryFamilyType, + input: &planner_types::post_asap::SummaryUpdate, + grouping: &planner_types::post_asap::GroupingStrategy, +) -> Result, String> { + if input.item.is_some() + && matches!( + input.weight_domain, + planner_types::post_asap::WeightDomain::NonNegative { + proof: + planner_types::post_asap::NonNegativeWeightProof::ResetAwareCounterDerivative + } + ) + { + return Err("window-weighted summaries require typed DAG binding; integer heap updaters cannot consume rates".into()); + } + + crate::capability::validate_summary_kernel(family, input, grouping)?; + use planner_types::post_asap::GroupingStrategy; + if grouping != &GroupingStrategy::PerSubpopulationInstance { + return Err("shared summary grouping requires a supported Planner Hydra kernel".into()); + } + if matches!(family, SummaryFamilyType::ExactAggregate(..)) { + return Ok(Box::new(PlannerExactUpdater { + acc: crate::summary_kernels::exact::ExactAccumulator::new( + family.clone(), + input.item.is_some(), + )?, + })); + } + let SummaryFamilyType::Sketch(kind, family_grouping) = family else { + return Err(format!("unsupported Planner summary family {family:?}")); + }; + if family_grouping != grouping { + return Err("Planner family and operator grouping disagree".into()); + } + let updater: Box = match (kind.algorithm(), kind.params()) { + (SketchAlgorithm::Kll, SketchParams::Kll { k }) => Box::new(KllAccumulatorUpdater::new( + u16::try_from(*k).map_err(|_| "KLL k exceeds runtime bound")?, + )), + (SketchAlgorithm::DDSketch, SketchParams::DDSketch { alpha }) => { + Box::new(DDSketchAccumulatorUpdater::new(*alpha)) + } + (SketchAlgorithm::Cms, params @ SketchParams::Cms { .. }) => { + let (r, c) = cms_dims(params); + Box::new(CmsAccumulatorUpdater::new(r, c)) + } + (SketchAlgorithm::CountSketch, params @ SketchParams::CountSketch { .. }) => { + let (r, c) = cms_dims(params); + Box::new(CountSketchAccumulatorUpdater::new(r, c)) + } + (SketchAlgorithm::CmsWithHeap, params @ SketchParams::CmsWithHeap { .. }) => { + let (r, c, h) = cms_heap_dims(params); + Box::new(CmsHeapAccumulatorUpdater::new(r, c, h, TopkWeight::Value)) + } + ( + SketchAlgorithm::CountSketchWithHeap, + params @ SketchParams::CountSketchWithHeap { .. }, + ) => { + let (r, c, h) = cms_heap_dims(params); + Box::new(CountSketchWithHeapAccumulatorUpdater::new( + r, + c, + h, + TopkWeight::Value, + )) + } + (SketchAlgorithm::Hll, SketchParams::Hll { precision }) => Box::new(HllUpdater { + acc: HllSketchAccumulator::new( + asap_sketchlib::HllVariant::Regular, + u32::from(*precision), + ), + }), + ( + SketchAlgorithm::UnivMon, + SketchParams::UnivMon { + heap_size, + sketch_rows, + sketch_cols, + layers, + }, + ) => Box::new(UnivMonUpdater { + acc: UnivMonAccumulator::new( + *heap_size as usize, + *sketch_rows as usize, + *sketch_cols as usize, + *layers as usize, + ) + .map_err(|e| e.to_string())?, + }), + _ => { + return Err(format!( + "unsupported Planner algorithm/parameters: {kind:?}" + )) + } + }; + if updater.is_keyed() != input.item.is_some() + && !crate::capability::is_unit_sample_frequency(input) + { + return Err("Planner item expression does not match the selected kernel layout".into()); + } + Ok(updater) +} + +struct PlannerExactUpdater { + acc: crate::summary_kernels::exact::ExactAccumulator, +} +impl AccumulatorUpdater for PlannerExactUpdater { + fn update_single(&mut self, value: f64, timestamp: i64) { + self.acc.update(None, value, timestamp); + } + fn update_keyed(&mut self, key: &KeyByLabelValues, value: f64, timestamp: i64) { + self.acc.update(Some(key), value, timestamp); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc = crate::summary_kernels::exact::ExactAccumulator::new( + self.acc.family().clone(), + self.acc.is_keyed(), + ) + .expect("installed exact family"); + } + fn is_keyed(&self) -> bool { + self.acc.is_keyed() + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } +} + +struct UnivMonUpdater { + acc: UnivMonAccumulator, +} + +struct HllUpdater { + acc: HllSketchAccumulator, +} + +impl AccumulatorUpdater for HllUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + if !value.is_nan() { + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.acc.inner.update(&bits.to_le_bytes()); + } + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.inner = + asap_sketchlib::HllSketch::new(self.acc.inner.variant, self.acc.inner.precision); + } +} + +impl AccumulatorUpdater for UnivMonUpdater { + fn is_keyed(&self) -> bool { + false + } + fn memory_usage_bytes(&self) -> usize { + self.acc.approx_memory_bytes() + } + fn update_single(&mut self, value: f64, _: i64) { + self.acc + .insert_sample(value) + .expect("UnivMon sample counter overflow"); + } + fn update_keyed(&mut self, _: &KeyByLabelValues, value: f64, timestamp_ms: i64) { + self.update_single(value, timestamp_ms); + } + impl_clone_accumulator_methods!(acc); + fn reset(&mut self) { + self.acc.clear(); + } +} + +#[cfg(test)] +mod planner_parameter_regression { + use super::*; + use planner_types::post_asap::{SketchKind, SummaryInputExpr, SummaryUpdate}; + + // Planner width is the bucket count; depth is the independent hash-row count. + #[test] + fn planner_sketch_dimensions_are_not_transposed() { + for (algorithm, params) in [ + ( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 128, + depth: 3, + }, + ), + ( + SketchAlgorithm::CountSketch, + SketchParams::CountSketch { + width: 128, + depth: 3, + }, + ), + ( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 128, + depth: 3, + heap_size: 8, + }, + ), + ( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 128, + depth: 3, + heap_size: 8, + }, + ), + ] { + let family = SummaryFamilyType::Sketch( + SketchKind::new(algorithm.clone(), params), + Default::default(), + ); + let update = SummaryUpdate { + item: Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::Named("host".into()), + )), + weight: SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }; + let state = create_planner_accumulator(&family, &update, &Default::default()) + .unwrap() + .snapshot_accumulator(); + let dims = match algorithm { + SketchAlgorithm::Cms => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + SketchAlgorithm::CountSketch => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows, s.inner.cols) + } + SketchAlgorithm::CmsWithHeap => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + SketchAlgorithm::CountSketchWithHeap => { + let s = state + .as_any() + .downcast_ref::() + .unwrap(); + (s.inner.rows(), s.inner.cols()) + } + _ => unreachable!(), + }; + assert_eq!(dims, (3, 128), "{algorithm:?}"); + } + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs new file mode 100644 index 00000000..089261c5 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/hll_sketch.rs @@ -0,0 +1,79 @@ +//! HyperLogLog distinct-count summary over `asap_sketchlib::HllSketch`. +use crate::{AggregateCore, KernelError}; +use asap_sketchlib::{HllSketch, HllVariant}; +use planner_types::post_asap::SketchQuery; + +#[derive(Debug, Clone)] +pub struct HllSketchAccumulator { + pub inner: HllSketch, +} + +impl HllSketchAccumulator { + pub fn new(variant: HllVariant, precision: u32) -> Self { + Self { + inner: HllSketch::new(variant, precision), + } + } +} + +impl AggregateCore for HllSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("HLL merges only with HLL")?; + Ok(Box::new(Self { + inner: HllSketch::merge_refs(&[&self.inner, &other.inner])?, + })) + } + + /// Distinct count. A bare `PointCount` over an HLL also reads the distinct count. + fn estimate(&self, query: &SketchQuery) -> Result { + match query { + SketchQuery::Cardinality | SketchQuery::PointCount { value: None, .. } => { + Ok(self.inner.estimate()) + } + other => Err(format!("HLL does not answer {other:?}").into()), + } + } + + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::().saturating_add(self.inner.registers.capacity()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Distinct count after merge counts overlapping items once. + #[test] + fn merged_cardinality_deduplicates_overlap() { + let (mut a, mut b) = ( + HllSketchAccumulator::new(HllVariant::Regular, 12), + HllSketchAccumulator::new(HllVariant::Regular, 12), + ); + for v in 0..1000u32 { + a.inner.update(&v.to_le_bytes()); + b.inner.update(&(v + 500).to_le_bytes()); + } + let merged = a.merge_with(&b).unwrap(); + let estimate = merged.estimate(&SketchQuery::Cardinality).unwrap(); + assert!((estimate - 1500.0).abs() / 1500.0 < 0.05, "{estimate}"); + } + + // HLL does not answer quantiles. + #[test] + fn rejects_quantile() { + let hll = HllSketchAccumulator::new(HllVariant::Regular, 12); + assert!(hll.estimate(&SketchQuery::Quantile { q: 0.5 }).is_err()); + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs new file mode 100644 index 00000000..c0347e69 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/hydra_kll.rs @@ -0,0 +1,55 @@ +use crate::{AggregateCore, KeyByLabelValues}; +use asap_sketchlib::HydraKllSketch; + +/// HydraKLL (shared-grouping quantiles) over `asap_sketchlib::HydraKllSketch`. +#[derive(Debug, Clone)] +pub struct HydraKllSketchAccumulator { + pub inner: HydraKllSketch, +} + +impl HydraKllSketchAccumulator { + pub fn new(row_num: usize, col_num: usize, k: u16) -> Self { + Self { + inner: HydraKllSketch::new(row_num, col_num, k), + } + } + + pub fn update(&mut self, key: &KeyByLabelValues, value: f64) { + self.inner.update(&key.to_semicolon_str(), value); + } + + pub fn query_key(&self, key: &KeyByLabelValues, quantile: f64) -> f64 { + self.inner.quantile(&key.to_semicolon_str(), quantile) + } +} + +impl AggregateCore for HydraKllSketchAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let hk = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to HydraKllSketchAccumulator")?; + + let mut merged = self.clone(); + merged.inner.merge(&hk.inner)?; + Ok(Box::new(merged)) + } + + fn approx_memory_bytes(&self) -> usize { + // HydraKLL is a row*col grid of KLL sketches; typical instances + // are on the order of tens of KiB. 32 KiB is a conservative + // per-instance default. + 32 * 1024 + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/increase.rs b/crates/asap-physical-operators/src/summary_kernels/increase.rs new file mode 100644 index 00000000..f92ee582 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/increase.rs @@ -0,0 +1,213 @@ +use crate::{AggregateCore, Measurement}; +use serde::{Deserialize, Serialize}; + +/// Accumulator for tracking increases in counter metrics +/// Stores the starting and last seen measurements with timestamps +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct IncreaseAccumulator { + pub starting_measurement: Measurement, + pub starting_timestamp: i64, + pub last_seen_measurement: Measurement, + pub last_seen_timestamp: i64, + /// Sum of monotonic deltas, adding the post-reset value whenever the + /// counter decreases. This is the reset correction Prometheus applies. + #[serde(default)] + pub total_increase: f64, + #[serde(default)] + pub sample_count: u64, +} + +impl IncreaseAccumulator { + /// Merge two counter intervals without a temporary collection. Ties retain + /// the left input, matching the stable ordering of multi-pane merges. + pub(crate) fn merge_pair(left: &Self, right: &Self) -> Self { + let (first, second) = if left.starting_timestamp <= right.starting_timestamp { + (left, right) + } else { + (right, left) + }; + let mut merged = first.clone(); + if second.starting_timestamp > merged.last_seen_timestamp { + merged.total_increase += + if second.starting_measurement.value >= merged.last_seen_measurement.value { + second.starting_measurement.value - merged.last_seen_measurement.value + } else { + second.starting_measurement.value + }; + } + merged.total_increase += second.total_increase; + merged.sample_count = merged.sample_count.saturating_add(second.sample_count); + if second.last_seen_timestamp > merged.last_seen_timestamp { + merged.last_seen_measurement = second.last_seen_measurement.clone(); + merged.last_seen_timestamp = second.last_seen_timestamp; + } + + merged + } + + pub fn new( + starting_measurement: Measurement, + starting_timestamp: i64, + last_seen_measurement: Measurement, + last_seen_timestamp: i64, + ) -> Self { + let total_increase = if last_seen_timestamp <= starting_timestamp { + 0.0 + } else if last_seen_measurement.value >= starting_measurement.value { + last_seen_measurement.value - starting_measurement.value + } else { + last_seen_measurement.value + }; + let sample_count = if last_seen_timestamp > starting_timestamp { + 2 + } else { + 1 + }; + Self { + starting_measurement, + starting_timestamp, + last_seen_measurement, + last_seen_timestamp, + total_increase, + sample_count, + } + } + + pub fn update(&mut self, measurement: Measurement, timestamp: i64) { + if timestamp < self.last_seen_timestamp { + return; + } + if timestamp == self.last_seen_timestamp { + return; + } + if measurement.value >= self.last_seen_measurement.value { + self.total_increase += measurement.value - self.last_seen_measurement.value; + } else { + self.total_increase += measurement.value; + } + self.last_seen_measurement = measurement; + self.last_seen_timestamp = timestamp; + self.sample_count = self.sample_count.saturating_add(1); + } +} + +impl AggregateCore for IncreaseAccumulator { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + // Downcast to IncreaseAccumulator + let other_increase = other + .as_any() + .downcast_ref::() + .ok_or("Failed to downcast to IncreaseAccumulator")?; + + let merged = Self::merge_pair(self, other_increase); + Ok(Box::new(merged)) + } + + fn approx_memory_bytes(&self) -> usize { + // Two Measurements + two i64s. Measurements are a few f64 fields. + std::mem::size_of::() + } +} + +impl IncreaseAccumulator { + /// PromQL-style increase or rate, extrapolated to `range_ms` when given. + pub(crate) fn extrapolated_value( + &self, + range_ms: Option<(i64, i64)>, + is_rate: bool, + ) -> Result> { + if self.sample_count < 2 || self.last_seen_timestamp <= self.starting_timestamp { + return Err("at least two ordered counter samples are required".into()); + } + let sampled_interval = (self.last_seen_timestamp - self.starting_timestamp) as f64 / 1000.0; + let Some((range_start, range_end)) = range_ms else { + return Ok(if is_rate { + self.total_increase / sampled_interval + } else { + self.total_increase + }); + }; + if range_end <= range_start { + return Err("invalid counter evaluation range".into()); + } + + let mut duration_to_start = + (self.starting_timestamp.saturating_sub(range_start)) as f64 / 1000.0; + let duration_to_end = (range_end.saturating_sub(self.last_seen_timestamp)) as f64 / 1000.0; + let average_sample_interval = sampled_interval / (self.sample_count - 1) as f64; + let extrapolation_threshold = average_sample_interval * 1.1; + + if self.total_increase > 0.0 && self.starting_measurement.value >= 0.0 { + let duration_to_zero = + sampled_interval * (self.starting_measurement.value / self.total_increase); + duration_to_start = duration_to_start.min(duration_to_zero); + } + let mut extrapolate_to = sampled_interval; + extrapolate_to += if duration_to_start < extrapolation_threshold { + duration_to_start.max(0.0) + } else { + average_sample_interval / 2.0 + }; + extrapolate_to += if duration_to_end < extrapolation_threshold { + duration_to_end.max(0.0) + } else { + average_sample_interval / 2.0 + }; + let mut factor = extrapolate_to / sampled_interval; + if is_rate { + factor /= (range_end - range_start) as f64 / 1000.0; + } + Ok(self.total_increase * factor) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_increase_accumulator_creation() { + let starting_measurement = Measurement::new(10.0); + let last_seen_measurement = Measurement::new(25.0); + let acc = IncreaseAccumulator::new( + starting_measurement.clone(), + 1000, + last_seen_measurement.clone(), + 2000, + ); + + assert_eq!(acc.starting_measurement.value, 10.0); + assert_eq!(acc.starting_timestamp, 1000); + assert_eq!(acc.last_seen_measurement.value, 25.0); + assert_eq!(acc.last_seen_timestamp, 2000); + } + + #[test] + fn test_increase_accumulator_update() { + let starting_measurement = Measurement::new(10.0); + let mut acc = IncreaseAccumulator::new( + starting_measurement.clone(), + 1000, + starting_measurement.clone(), + 1000, + ); + + let new_measurement = Measurement::new(25.0); + acc.update(new_measurement.clone(), 2000); + + assert_eq!(acc.last_seen_measurement.value, 25.0); + assert_eq!(acc.last_seen_timestamp, 2000); + assert_eq!(acc.starting_measurement.value, 10.0); // Should remain unchanged + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/mod.rs b/crates/asap-physical-operators/src/summary_kernels/mod.rs new file mode 100644 index 00000000..478e75d0 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/mod.rs @@ -0,0 +1,26 @@ +//! In-memory summary state: thin adapters over `asap_sketchlib` and exact Planner state. +pub mod count_min_sketch; +pub mod count_min_sketch_with_heap; +pub mod count_sketch; +pub mod count_sketch_with_heap; +pub mod datasketches_kll; +pub mod dd_sketch; +pub mod exact; +pub mod hll_sketch; +pub mod hydra_kll; +pub mod increase; +pub mod univmon; + +pub use count_min_sketch::*; +pub use count_min_sketch_with_heap::*; +pub use count_sketch::*; +pub use count_sketch_with_heap::*; +pub use datasketches_kll::*; +pub use dd_sketch::*; +pub use hll_sketch::*; +pub use hydra_kll::*; +pub use increase::*; + +pub mod factory; +pub mod traits; +pub mod weighted_frequency; diff --git a/crates/asap-physical-operators/src/summary_kernels/traits.rs b/crates/asap-physical-operators/src/summary_kernels/traits.rs new file mode 100644 index 00000000..7d866077 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/traits.rs @@ -0,0 +1,35 @@ +use planner_types::post_asap::SketchQuery; + +pub type KernelError = Box; + +/// In-memory state of one population's summary. +/// +/// Kernels adapt `asap_sketchlib` structures (or exact Planner state) to the +/// operations physical operators need: merge, typed readout and memory +/// accounting. Grouping belongs to operators; byte encodings belong to +/// `asap_sketchlib` and deployments. +pub trait AggregateCore: Send + Sync { + fn clone_boxed_core(&self) -> Box; + + fn as_any(&self) -> &dyn std::any::Any; + + /// Merge with a state of the same family and shape, leaving both inputs unchanged. + fn merge_with(&self, other: &dyn AggregateCore) -> Result, KernelError>; + + /// Answer a sketch readout. Exact states are read through + /// [`ExactAccumulator::readout`](super::exact::ExactAccumulator::readout). + fn estimate(&self, query: &SketchQuery) -> Result { + Err(format!("{query:?} is not supported by this summary").into()) + } + + /// Approximate in-memory footprint, used for execution memory reservations. + fn approx_memory_bytes(&self) -> usize { + 4096 + } +} + +impl Clone for Box { + fn clone(&self) -> Self { + self.clone_boxed_core() + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/univmon.rs b/crates/asap-physical-operators/src/summary_kernels/univmon.rs new file mode 100644 index 00000000..bffd8afa --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/univmon.rs @@ -0,0 +1,109 @@ +//! One frequency state shared by count, distinct, L2 and entropy readouts. + +use crate::AggregateCore; +use asap_sketchlib::{DataInput, UnivMon}; + +type Error = Box; + +#[derive(Debug, Clone)] +pub struct UnivMonAccumulator { + inner: UnivMon, +} + +impl UnivMonAccumulator { + /// Empty the sketch in place, keeping its shape. + pub(crate) fn clear(&mut self) { + self.inner.free(); + } + + pub fn new(heap_size: usize, rows: usize, cols: usize, layers: usize) -> Result { + if heap_size == 0 || cols == 0 || !(1..=20).contains(&rows) || !(1..=64).contains(&layers) { + return Err("invalid UnivMon dimensions".into()); + } + rows.checked_mul(cols) + .and_then(|n| n.checked_mul(layers)) + .ok_or("UnivMon dimensions overflow")?; + Ok(Self { + inner: UnivMon::init_univmon(heap_size, rows, cols, layers), + }) + } + + /// Each non-NaN sample is one occurrence. Signed zero has one identity. + pub fn insert_sample(&mut self, value: f64) -> Result<(), Error> { + if value.is_nan() { + return Ok(()); + } + self.inner + .bucket_size + .checked_add(1) + .ok_or("UnivMon count overflow")?; + let bits = if value == 0.0 { 0 } else { value.to_bits() }; + self.inner.insert(&DataInput::U64(bits), 1); + Ok(()) + } + + fn compatible(&self, other: &Self) -> bool { + ( + self.inner.heap_size, + self.inner.sketch_row, + self.inner.sketch_col, + self.inner.layer_size, + ) == ( + other.inner.heap_size, + other.inner.sketch_row, + other.inner.sketch_col, + other.inner.layer_size, + ) + } + + pub fn dimensions(&self) -> (usize, usize, usize, usize) { + ( + self.inner.heap_size, + self.inner.sketch_row, + self.inner.sketch_col, + self.inner.layer_size, + ) + } + + pub fn merge_in_place(&mut self, other: &Self) -> Result<(), Error> { + if !self.compatible(other) { + return Err("incompatible UnivMon dimensions".into()); + } + self.inner + .bucket_size + .checked_add(other.inner.bucket_size) + .ok_or("UnivMon count overflow")?; + self.inner.merge(&other.inner); + Ok(()) + } +} + +impl AggregateCore for UnivMonAccumulator { + fn approx_memory_bytes(&self) -> usize { + std::mem::size_of::().saturating_add( + self.inner.layer_size.saturating_mul( + self.inner + .sketch_row + .saturating_mul(self.inner.sketch_col) + .saturating_mul(16) + .saturating_add(self.inner.heap_size.saturating_mul(256)), + ), + ) + } + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + + fn merge_with(&self, other: &dyn AggregateCore) -> Result, Error> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("expected UnivMon state")?; + let mut merged = self.clone(); + merged.merge_in_place(other)?; + Ok(Box::new(merged)) + } +} diff --git a/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs new file mode 100644 index 00000000..6f008532 --- /dev/null +++ b/crates/asap-physical-operators/src/summary_kernels/weighted_frequency.rs @@ -0,0 +1,152 @@ +//! ASAP type and trait adapter for sketchlib's Float64 weighted frequency kernel. +use crate::AggregateCore; +use crate::{values::Value, Error}; +pub use asap_sketchlib::FrequencyAlgorithm; +use asap_sketchlib::{FrequencyIdentity, WeightedFrequency as Kernel, WeightedFrequencyError}; +use serde::{Deserialize, Serialize}; + +fn adapt_error(error: WeightedFrequencyError) -> Error { + match error { + WeightedFrequencyError::Invalid(message) => Error::Invalid(message), + WeightedFrequencyError::Update(message) => Error::Operator(message), + } +} +fn identity(value: &Value) -> Result { + Ok(match value { + Value::Null => FrequencyIdentity::Null, + Value::Bool(v) => FrequencyIdentity::Bool(*v), + Value::Int64(v) => FrequencyIdentity::Int64(*v), + Value::Float64(v) => FrequencyIdentity::Float64(*v), + Value::Utf8(v) => FrequencyIdentity::Utf8(v.to_string()), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency identity".into(), + )) + } + }) +} +fn value(identity: FrequencyIdentity) -> Value { + match identity { + FrequencyIdentity::Null => Value::Null, + FrequencyIdentity::Bool(v) => Value::Bool(v), + FrequencyIdentity::Int64(v) => Value::Int64(v), + FrequencyIdentity::Float64(v) => Value::Float64(v), + FrequencyIdentity::Utf8(v) => Value::Utf8(v.into()), + } +} +#[derive(Clone, Debug, Serialize, Deserialize)] +#[serde(transparent)] +pub struct WeightedFrequency { + inner: Kernel, +} +impl WeightedFrequency { + pub(crate) fn configuration( + kind: &planner_types::post_asap::SketchKind, + ) -> Result<(FrequencyAlgorithm, usize, usize, usize), Error> { + use planner_types::post_asap::{SketchAlgorithm as A, SketchParams as P}; + let (algorithm, width, depth, capacity) = match (kind.algorithm(), kind.params()) { + ( + A::CmsWithHeap, + P::CmsWithHeap { + width, + depth, + heap_size, + }, + ) => (FrequencyAlgorithm::Cms, *width, *depth, *heap_size), + ( + A::CountSketchWithHeap, + P::CountSketchWithHeap { + width, + depth, + heap_size, + }, + ) if depth % 2 == 1 => (FrequencyAlgorithm::CountSketch, *width, *depth, *heap_size), + _ => { + return Err(Error::Invalid( + "unsupported weighted frequency family or depth".into(), + )) + } + }; + if width == 0 || depth == 0 || capacity == 0 { + return Err(Error::Invalid( + "invalid weighted frequency dimensions".into(), + )); + } + Ok((algorithm, width as usize, depth as usize, capacity as usize)) + } + + pub(crate) fn algorithm(&self) -> FrequencyAlgorithm { + self.inner.algorithm() + } + pub(crate) fn shape(&self) -> (usize, usize, usize) { + self.inner.shape() + } + pub fn new( + algorithm: FrequencyAlgorithm, + width: usize, + depth: usize, + capacity: usize, + ) -> Result { + Kernel::new(algorithm, width, depth, capacity) + .map(|inner| Self { inner }) + .map_err(adapt_error) + } + pub fn update(&mut self, values: &[Value], weight: f64) -> Result<(), Error> { + let values = values.iter().map(identity).collect::, _>>()?; + self.inner.update(&values, weight).map_err(adapt_error) + } + pub fn rows(&self, n: usize) -> Vec> { + self.inner + .topk(n) + .into_iter() + .map(|(items, score)| { + let mut row = items.into_iter().map(value).collect::>(); + row.push(Value::Float64(score)); + row + }) + .collect() + } +} +impl AggregateCore for WeightedFrequency { + fn clone_boxed_core(&self) -> Box { + Box::new(self.clone()) + } + fn as_any(&self) -> &dyn std::any::Any { + self + } + fn merge_with( + &self, + other: &dyn AggregateCore, + ) -> Result, Box> { + let other = other + .as_any() + .downcast_ref::() + .ok_or("weighted frequency state type mismatch")?; + Ok(Box::new(Self { + inner: self.inner.merge(&other.inner)?, + })) + } + fn approx_memory_bytes(&self) -> usize { + self.inner.approx_memory_bytes() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + // Merge uses the same Float64 state representation and rejects other shapes. + #[test] + fn compatible_merge_preserves_fractional_weights() { + let mut left = WeightedFrequency::new(FrequencyAlgorithm::Cms, 4096, 5, 8).unwrap(); + let mut right = left.clone(); + left.update(&[Value::Int64(7)], 0.125).unwrap(); + right.update(&[Value::Int64(7)], 0.25).unwrap(); + let merged = left.merge_with(&right).unwrap(); + let merged = merged.as_any().downcast_ref::().unwrap(); + assert!(matches!(merged.rows(1)[0][1], Value::Float64(0.375))); + assert!(left + .merge_with(&WeightedFrequency::new(FrequencyAlgorithm::Cms, 32, 5, 8).unwrap()) + .is_err()); + } +} diff --git a/crates/asap-physical-operators/src/values.rs b/crates/asap-physical-operators/src/values.rs new file mode 100644 index 00000000..56d43633 --- /dev/null +++ b/crates/asap-physical-operators/src/values.rs @@ -0,0 +1,405 @@ +//! Runtime values preserve Planner schemas; summary states are typed values too. +use crate::AggregateCore; +use crate::Error; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::{cmp::Ordering, sync::Arc}; +pub type Schema = Arc; +#[derive(Clone, serde::Serialize, serde::Deserialize)] +pub enum Value { + Null, + Bool(bool), + Int64(i64), + Float64(f64), + Utf8(Arc), + Timestamp(i64), + Date(i32), + Interval { + months: i32, + days: i32, + nanos: i64, + }, + List(Arc<[Value]>), + Struct(Arc<[Value]>), + Map(Arc<[(Value, Value)]>), + #[serde(skip)] + Summary { + family: SummaryFamilyType, + state: Arc, + }, +} +impl std::fmt::Debug for Value { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Summary { family, .. } => f.debug_tuple("Summary").field(family).finish(), + _ => write!(f, "{:?}", self.key()), + } + } +} +impl Value { + pub fn bytes(&self) -> usize { + std::mem::size_of::() + + match self { + Self::Utf8(s) => s.len(), + Self::List(v) | Self::Struct(v) => v.iter().map(Self::bytes).sum(), + Self::Map(v) => v.iter().map(|(k, v)| k.bytes() + v.bytes()).sum(), + Self::Summary { state, .. } => state.approx_memory_bytes(), + _ => 0, + } + } + pub fn matches(&self, dtype: &DataType, nullable: bool) -> bool { + if matches!(self, Self::Null) { + return nullable || matches!(dtype, DataType::Null); + } + match (self, dtype) { + (Self::Bool(_), DataType::Bool) + | (Self::Int64(_), DataType::Int64) + | (Self::Float64(_), DataType::Float64) + | (Self::Utf8(_), DataType::Utf8) + | (Self::Timestamp(_), DataType::Timestamp) + | (Self::Date(_), DataType::Date) + | (Self::Interval { .. }, DataType::Interval) => true, + (Self::List(v), DataType::List { element }) => v + .iter() + .all(|v| v.matches(&element.dtype, element.nullable)), + (Self::Struct(v), DataType::Struct { fields }) => { + v.len() == fields.len() + && v.iter() + .zip(fields) + .all(|(v, f)| v.matches(&f.dtype, f.nullable)) + } + ( + Self::Map(v), + DataType::Map { + key, + value, + value_nullable, + }, + ) => v + .iter() + .all(|(k, v)| k.matches(key, false) && v.matches(value, *value_nullable)), + _ => false, + } + } + /// Stable typed equality key. Zero signs and NaN payloads form one group. + pub fn key(&self) -> Result, Error> { + let mut out = Vec::new(); + macro_rules! number { + ($tag:expr,$v:expr) => {{ + out.push($tag); + out.extend_from_slice(&$v.to_le_bytes()); + }}; + } + match self { + Self::Null => out.push(0), + Self::Bool(v) => out.extend([1, *v as u8]), + Self::Int64(v) => number!(2, v), + Self::Float64(v) => { + let bits = if *v == 0. { + 0 + } else if v.is_nan() { + f64::NAN.to_bits() + } else { + v.to_bits() + }; + number!(3, bits); + } + Self::Utf8(v) => { + out.push(4); + out.extend(v.as_bytes()); + } + Self::Timestamp(v) => number!(5, v), + Self::Date(v) => number!(6, v), + Self::Interval { + months, + days, + nanos, + } => { + number!(7, months); + number!(8, days); + number!(9, nanos); + } + Self::List(v) | Self::Struct(v) => { + out.push(if matches!(self, Self::List(_)) { + 10 + } else { + 11 + }); + for v in v.iter() { + let key = v.key()?; + out.extend((key.len() as u64).to_le_bytes()); + out.extend(key); + } + } + Self::Map(v) => { + out.push(12); + for (k, v) in v.iter() { + for value in [k, v] { + let key = value.key()?; + out.extend((key.len() as u64).to_le_bytes()); + out.extend(key); + } + } + } + Self::Summary { .. } => { + return Err(Error::Invalid( + "summary states cannot be grouping keys".into(), + )) + } + } + Ok(out) + } + pub fn compare(&self, other: &Self) -> Result { + Ok(match (self, other) { + (Self::Null, Self::Null) => Ordering::Equal, + (Self::Int64(a), Self::Int64(b)) | (Self::Timestamp(a), Self::Timestamp(b)) => a.cmp(b), + (Self::Float64(a), Self::Float64(b)) => { + if a == b { + Ordering::Equal + } else { + a.total_cmp(b) + } + } + (Self::Utf8(a), Self::Utf8(b)) => a.cmp(b), + (Self::Bool(a), Self::Bool(b)) => a.cmp(b), + (Self::Date(a), Self::Date(b)) => a.cmp(b), + (Self::Map(left), Self::Map(right)) => { + let mut result = Ordering::Equal; + for ((lk, lv), (rk, rv)) in left.iter().zip(right.iter()) { + result = lk.compare(rk)?; + if result != Ordering::Equal { + break; + } + result = match (lv, rv) { + (Self::Null, Self::Null) => Ordering::Equal, + (Self::Null, _) => Ordering::Greater, + (_, Self::Null) => Ordering::Less, + _ => lv.compare(rv)?, + }; + if result != Ordering::Equal { + break; + } + } + if result == Ordering::Equal { + left.len().cmp(&right.len()) + } else { + result + } + } + _ => { + return Err(Error::Operator( + "values do not have a supported common ordering".into(), + )) + } + }) + } +} +#[derive(Clone, Debug)] +pub struct Batch { + schema: Schema, + rows: Vec>, +} +impl Batch { + pub fn try_new(schema: Schema, rows: Vec>) -> Result { + validate_schema(&schema)?; + for row in &rows { + if row.len() != schema.fields.len() { + return Err(Error::Invalid( + "row width differs from Planner schema".into(), + )); + } + for (value, field) in row.iter().zip(&schema.fields) { + let matches = match (&field.dtype, value) { + (SummaryFamilyType::Plain(dtype), value) => { + value.matches(dtype, field.nullable) + } + (expected, Value::Summary { family, state }) => { + expected == family && validate_state(family, state.as_ref()).is_ok() + } + _ => false, + }; + if !matches { + return Err(Error::Invalid(format!( + "value differs from type of {}", + field.name + ))); + } + } + } + Ok(Self { schema, rows }) + } + pub fn schema(&self) -> &Schema { + &self.schema + } + pub fn rows(&self) -> &[Vec] { + &self.rows + } + pub fn bytes(&self) -> usize { + std::mem::size_of::() + + self.rows.capacity() * std::mem::size_of::>() + + self + .rows + .iter() + .flat_map(|r| r.iter()) + .map(Value::bytes) + .sum::() + } +} +pub(crate) fn group_key(row: &[Value], columns: &[usize]) -> Result>, Error> { + columns + .iter() + .map(|&i| { + row.get(i) + .ok_or_else(|| Error::Invalid("group column out of range".into()))? + .key() + }) + .collect() +} + +pub(crate) use crate::capability::validate_native_family as validate_family; + +fn validate_state(family: &SummaryFamilyType, state: &dyn AggregateCore) -> Result<(), Error> { + use crate::summary_kernels::{ + datasketches_kll::DatasketchesKLLAccumulator, dd_sketch::DDSketchAccumulator, + exact::ExactAccumulator, hll_sketch::HllSketchAccumulator, + }; + use planner_types::post_asap::SketchParams; + validate_family(family)?; + let valid = match family { + SummaryFamilyType::Sketch(kind, _) + if matches!( + kind.params(), + SketchParams::CmsWithHeap { .. } | SketchParams::CountSketchWithHeap { .. } + ) => + { + use crate::summary_kernels::weighted_frequency::WeightedFrequency; + let (algorithm, width, depth, capacity) = WeightedFrequency::configuration(kind)?; + state + .as_any() + .downcast_ref::() + .is_some_and(|state| { + state.algorithm() == algorithm && state.shape() == (width, depth, capacity) + }) + } + + SummaryFamilyType::ExactAggregate(..) => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.family() == family && !s.is_keyed()), + SummaryFamilyType::Sketch(kind, _) => match kind.params() { + SketchParams::Kll { k } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| u32::from(s.inner.k()) == *k), + SketchParams::DDSketch { alpha } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.inner.alpha == *alpha), + SketchParams::Hll { precision } => state + .as_any() + .downcast_ref::() + .is_some_and(|s| s.inner.precision == u32::from(*precision)), + _ => false, + }, + _ => false, + }; + if valid { + Ok(()) + } else { + Err(Error::Invalid( + "state payload differs from declared family, parameters or population layout".into(), + )) + } +} + +pub(crate) fn validate_schema(schema: &Schema) -> Result<(), Error> { + if schema.time_index.is_some_and(|index| { + schema + .fields + .get(index) + .is_none_or(|field| field.dtype != SummaryFamilyType::Plain(DataType::Timestamp)) + }) { + return Err(Error::Invalid( + "time index must name a Timestamp column".into(), + )); + } + for field in &schema.fields { + if !matches!(field.dtype, SummaryFamilyType::Plain(_)) { + validate_family(&field.dtype)?; + if field.nullable { + return Err(Error::Invalid( + "nullable summary states are not supported".into(), + )); + } + } + } + Ok(()) +} + +pub(crate) fn field(schema: &Schema, column: usize) -> Result<&SummaryField, Error> { + schema + .fields + .get(column) + .ok_or_else(|| Error::Invalid("column out of range".into())) +} +pub(crate) fn plain(schema: &Schema, column: usize) -> Result<(&DataType, bool), Error> { + let f = field(schema, column)?; + let SummaryFamilyType::Plain(dtype) = &f.dtype else { + return Err(Error::Invalid("plain value required".into())); + }; + Ok((dtype, f.nullable)) +} + +#[cfg(test)] +mod weighted_state_tests { + use super::*; + use crate::summary_kernels::weighted_frequency::{FrequencyAlgorithm, WeightedFrequency}; + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + + // A state cannot acquire a different family or shape merely by relabeling its batch. + #[test] + fn weighted_state_family_and_shape_must_match() { + let cms = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 32, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let cs = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 32, + depth: 5, + heap_size: 8, + }, + ), + Default::default(), + ); + let state = WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 32, 5, 8).unwrap(); + assert!(validate_state(&cs, &state).is_ok()); + assert!(validate_state(&cms, &state).is_err()); + let wrong_shape = + WeightedFrequency::new(FrequencyAlgorithm::CountSketch, 64, 5, 8).unwrap(); + assert!(validate_state(&cs, &wrong_shape).is_err()); + let even_depth = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 32, + depth: 4, + heap_size: 8, + }, + ), + Default::default(), + ); + assert!(validate_family(&even_depth).is_err()); + } +} diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs new file mode 100644 index 00000000..7a312892 --- /dev/null +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -0,0 +1,242 @@ +//! Blocking operators enforce resources before returning their first batch. +use asap_physical_operators::{ + operators::Operator, + plan::{PhysicalDag, PhysicalOperator}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, + Error, +}; +use futures::{executor::block_on, FutureExt, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{DataType, JoinKind, Predicate, QueryExpr, ScalarValue}, +}; +use std::sync::Arc; + +fn schema(width: usize) -> Schema { + Arc::new(SummarySchema { + fields: (0..width) + .map(|i| SummaryField { + name: format!("v{i}"), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }) + .collect(), + time_index: None, + }) +} +fn context(max_bytes: usize) -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes, + ..Limits::default() + }, + ) + .unwrap() +} +fn source(n: usize) -> PhysicalDag<'static, Batch, Schema> { + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + schema(1), + vec![Batch::try_new(schema(1), vec![vec![Value::Int64(1)]; n]).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag +} +fn cross_join() -> Operator { + Operator::relational_join( + schema(1), + schema(1), + JoinKind::Cross, + &Predicate(std::rc::Rc::new(QueryExpr::Literal(ScalarValue::Boolean( + true, + )))), + schema(2), + ) + .unwrap() +} + +// Even callers starting an operator directly cannot bypass its workspace budget. +#[test] +fn join_reserves_result_growth_before_returning_output() { + let sources = source(64); + let run = context(32 * 1024); + let inputs = sources.execute(&[0, 0], run.clone()).unwrap(); + let join = cross_join(); + let mut output = join.start(inputs, run.clone()).unwrap(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} + +// A single large input batch must not monopolize the worker during a join. +#[test] +fn join_yields_during_computation_and_observes_cancellation() { + let sources = source(64); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0, 0], run.clone()).unwrap(); + let join = cross_join(); + let mut output = join.start(inputs, run.clone()).unwrap(); + assert!( + output.next().now_or_never().is_none(), + "join should yield before producing all 4096 rows" + ); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} + +// Sorting and grouping yield even for one large batch. +#[test] +fn blocking_reductions_yield_and_release_memory_on_cancellation() { + use asap_physical_operators::{ + operators::{Reduction, SortKey}, + plan::PhysicalOperator, + }; + let operators = vec![ + Operator::sort( + schema(1), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + Operator::aggregate(schema(1), vec![], vec![("sum".into(), Reduction::Sum(0))]).unwrap(), + ]; + for operator in operators { + let sources = source(768); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0], run.clone()).unwrap(); + let mut output = operator.start(inputs, run.clone()).unwrap(); + assert!(output.next().now_or_never().is_none()); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); + } +} + +// Merge-sort rounds preserve input order for tied keys across chunk boundaries. +#[test] +fn cooperative_sort_preserves_ties_across_chunks() { + use asap_physical_operators::operators::SortKey; + let batch = Batch::try_new( + schema(2), + (0..1025) + .rev() + .map(|i| vec![Value::Int64(i % 3), Value::Int64(i)]) + .collect(), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(schema(2), vec![batch]).unwrap()) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema(2), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ) + .unwrap(); + let mut output = dag + .execute(&[1], context(16 * 1024 * 1024)) + .unwrap() + .remove(0); + let batch = block_on(output.next()).unwrap().unwrap(); + let expected = (0..3) + .flat_map(|key| (0..1025).rev().filter(move |i| i % 3 == key)) + .collect::>(); + for (row, expected) in batch.rows().iter().zip(expected) { + assert!(matches!(row[1], Value::Int64(i) if i == expected)); + } + assert_eq!(batch.rows().len(), 1025); +} + +// The integrated weighted-summary path obeys the same cooperative cancellation contract. +#[test] +fn weighted_summary_build_yields_within_a_batch() { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "item".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }, + SummaryField { + name: "weight".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: None, + }); + let mut sources = PhysicalDag::default(); + let batch = Batch::try_new( + input.clone(), + (0..1500) + .map(|i| vec![Value::Int64(i % 8), Value::Float64(0.25)]) + .collect(), + ) + .unwrap(); + sources + .add( + 0, + vec![], + Operator::source(input.clone(), vec![batch]).unwrap(), + ) + .unwrap(); + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 64, + depth: 3, + heap_size: 8, + }, + ), + Default::default(), + ); + let operator = Operator::keyed_summary_build(input, family, 1, vec![0], vec![]).unwrap(); + let run = context(16 * 1024 * 1024); + let inputs = sources.execute(&[0], run.clone()).unwrap(); + let mut output = operator.start(inputs, run.clone()).unwrap(); + assert!(output.next().now_or_never().is_none()); + run.cancel(); + assert!(matches!( + block_on(output.next()), + Some(Err(Error::Cancelled)) + )); + drop(output); + assert_eq!(run.retained_bytes(), 0); +} diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs new file mode 100644 index 00000000..322bd6fe --- /dev/null +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -0,0 +1,401 @@ +//! Spatial heap weights come from a fresh instant vector, never sample history. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{ + promql_rows::{decode_series_identity, series_row, SERIES_IDENTITY_COLUMN}, + CompiledPhysicalDag, InputContract, Source, + }, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::DataType}; +use std::{collections::BTreeMap, sync::Arc}; + +fn schema() -> Arc { + Arc::new(SummarySchema { + fields: [ + ("ts", DataType::Timestamp), + ("value", DataType::Float64), + ("job", DataType::Utf8), + (SERIES_IDENTITY_COLUMN, DataType::Utf8), + ] + .into_iter() + .map(|(name, dtype)| SummaryField { + name: name.into(), + dtype: SummaryFamilyType::Plain(dtype), + nullable: false, + }) + .collect(), + time_index: Some(0), + }) +} +fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result, String> { + let recovered = serde_json::from_slice::( + &serde_json::to_vec(&program).map_err(|e| e.to_string())?, + ) + .map_err(|e| e.to_string())?; + let input_id = recovered.input_contracts().next().unwrap().0; + let graph = recovered + .instantiate(BTreeMap::from([( + input_id, + Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: end, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = graph.execute(recovered.roots(), context).unwrap().remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.map_err(|e| e.to_string())?).clone()); + } + Ok(batches) + }) +} +fn input(samples: &[(&str, i64, f64)]) -> Batch { + let schema = schema(); + let rows = samples + .iter() + .map(|(instance, time, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("hidden_instance".into(), (*instance).into()), + ]), + *time, + *value, + ) + .unwrap() + }) + .collect(); + Batch::try_new(schema, rows).unwrap() +} +fn snapshot_plan() -> CompiledPhysicalDag { + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema()))]), + BTreeMap::from([( + 1, + ( + vec![0], + Operator::current_series(schema(), 3, 0, 1, 60_000).unwrap(), + ), + )]), + vec![1], + ) + .unwrap() +} + +// Replacement, expiry and stale markers act before sketch updates. Hidden labels +// survive even when every series has the same projected `job` value. +#[test] +fn latest_snapshot_replaces_decreases_expires_and_retains_full_identity() { + let plan = snapshot_plan(); + let batches = run( + &plan, + input(&[ + ("decrease", 10_000, 100.), + ("decrease", 50_000, 1.), + ("steady", 40_000, 20.), + ("expired", 0, 1_000.), + ("stale", 20_000, 500.), + ("stale", 55_000, f64::from_bits(0x7ff0_0000_0000_0002)), + ("future", 60_001, 2_000.), + ]), + 60_000, + ) + .unwrap(); + let values = batches + .iter() + .flat_map(|batch| batch.rows()) + .map(|row| { + let Value::Utf8(identity) = &row[3] else { + panic!() + }; + let Value::Float64(value) = row[1] else { + panic!() + }; + assert!(matches!(row[0], Value::Timestamp(60_000))); + ( + decode_series_identity(identity).unwrap()["hidden_instance"].clone(), + value, + ) + }) + .collect::>(); + assert_eq!( + values, + BTreeMap::from([("decrease".into(), 1.), ("steady".into(), 20.)]) + ); + assert!(run(&plan, input(&[("steady", 40_000, 20.)]), 100_000) + .unwrap() + .iter() + .all(|batch| batch.rows().is_empty())); + assert!(run( + &plan, + input(&[("conflict", 50_000, 1.), ("conflict", 50_000, 2.)]), + 60_000 + ) + .is_err()); +} + +#[test] +fn spatial_heap_ranks_latest_values_in_independent_runs() { + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let params = match algorithm { + SketchAlgorithm::CmsWithHeap => SketchParams::CmsWithHeap { + width: 2048, + depth: 5, + heap_size: 100, + }, + _ => SketchParams::CountSketchWithHeap { + width: 2048, + depth: 5, + heap_size: 100, + }, + }; + let family = + SummaryFamilyType::Sketch(SketchKind::new(algorithm, params), Default::default()); + let build = Operator::keyed_summary_build(schema(), family, 1, vec![3], vec![2]).unwrap(); + let output = Arc::new(SummarySchema { + fields: vec![ + schema().fields[2].clone(), + schema().fields[3].clone(), + schema().fields[1].clone(), + ], + time_index: None, + }); + let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); + let plan = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema()))]), + BTreeMap::from([ + ( + 1, + ( + vec![0], + Operator::current_series(schema(), 3, 0, 1, 60_000).unwrap(), + ), + ), + (2, (vec![1], build)), + (3, (vec![2], read)), + ]), + vec![3], + ) + .unwrap(); + for (samples, end, winner, score) in [ + ( + vec![("a", 10_000, 100.), ("a", 50_000, 1.), ("b", 50_000, 20.)], + 60_000, + "b", + 20., + ), + ( + vec![("a", 110_000, 3.), ("b", 50_000, 20.)], + 120_000, + "a", + 3., + ), + ] { + let batches = run(&plan, input(&samples), end).unwrap(); + let rows = batches + .iter() + .flat_map(|batch| batch.rows()) + .collect::>(); + assert_eq!(rows.len(), 1); + let Value::Utf8(encoded) = &rows[0][1] else { + panic!() + }; + assert_eq!( + decode_series_identity(encoded).unwrap()["hidden_instance"], + winner + ); + assert!(matches!(rows[0][2], Value::Float64(actual) if actual == score)); + } + } +} + +// Blocking membership selection shares the run's cancellation and byte budget. +#[test] +fn current_series_observes_resource_limits() { + use asap_physical_operators::Error; + let plan = snapshot_plan(); + for cancelled in [false, true] { + let data = input(&[("one", 50_000, 1.)]); + let graph = plan + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) + as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 60_000, + revision: 0, + }, + Limits { + max_bytes: if cancelled { 1 << 20 } else { 1 }, + ..Limits::default() + }, + ) + .unwrap(); + if cancelled { + context.cancel(); + } + let result = match graph.execute(&[1], context.clone()) { + Err(error) => Err(error), + Ok(mut streams) => block_on(streams.remove(0).next()).unwrap().map(|_| ()), + }; + assert!(matches!( + (cancelled, result), + (true, Err(Error::Cancelled)) | (false, Err(Error::MemoryLimit)) + )); + assert_eq!(context.retained_bytes(), 0); + } +} + +#[test] +fn identity_encoding_is_lossless_and_rejects_noncanonical_inputs() { + use asap_physical_operators::physical_planner::promql_rows::encode_series_identity; + let labels = BTreeMap::from([ + ("a".into(), "quote\"slash\\".into()), + ("other".into(), "".into()), + ]); + assert_eq!( + decode_series_identity(&encode_series_identity(&labels).unwrap()).unwrap(), + labels + ); + for invalid in [ + "[]", + "{\"a\":1}", + "{\"a\":\"x\",\"a\":\"x\"}", + "{ \"a\":\"x\"}", + ] { + assert!(decode_series_identity(invalid).is_err(), "{invalid}"); + } +} + +// The actual Planner population candidate lowers to native operators; this +// test does not manually assemble the computation or its dependency edges. +#[test] +fn planner_current_series_candidate_compiles_with_dynamic_identity() { + use asap_physical_operators::physical_planner::{compile, promql_rows::with_series_identity}; + use planner_types::{types::AccuracyTarget, workload::*}; + use std::rc::Rc; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("topk by(job)(1, m)".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(60_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let original = asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0); + let open_root = Rc::new(original.clone()); + let open_selected = + asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&open_root), + ) + .candidate(&open_root) + .unwrap(); + let snapshot_program = + asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + &open_selected, + ) + .unwrap(); + let encoded = String::from_utf8(serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); + assert!( + !encoded.contains("CurrentSeries"), + "maintained input must not be rebuilt" + ); + assert!(encoded.contains("Sort") && encoded.contains("Limit")); + assert_eq!(snapshot_program.input_contracts().count(), 1); + let root = Rc::new(with_series_identity(&original).unwrap()); + let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( + std::slice::from_ref(&root), + ) + .candidate(&root) + .unwrap(); + let logical = compile_post_asap_dag(&selected).unwrap(); + let raw = logical + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::Fallback { .. })) + .unwrap(); + let raw_schema = Arc::new(raw.output_schema.clone()); + let physical = compile( + &logical, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(raw_schema.clone()), + )]), + &[u64::from(logical.root.0)], + ) + .unwrap(); + let bytes = String::from_utf8(serde_json::to_vec(&physical).unwrap()).unwrap(); + assert!(bytes.contains("CurrentSeries")); + assert!(bytes.contains("Sort")); + assert!(bytes.contains("Limit")); + let rows = [("a", 10_000, 100.), ("a", 50_000, 1.), ("b", 50_000, 20.)] + .into_iter() + .map(|(member, at, value)| { + series_row( + &raw_schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("unreferenced".into(), member.into()), + ]), + at, + value, + ) + .unwrap() + }) + .collect(); + let batches = run(&physical, Batch::try_new(raw_schema, rows).unwrap(), 60_000).unwrap(); + let rows = batches + .iter() + .flat_map(|batch| batch.rows()) + .collect::>(); + assert_eq!(rows.len(), 1); + assert!(matches!(rows[0][1], Value::Float64(20.))); + let id = batches[0] + .schema() + .fields + .iter() + .position(|field| field.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let Value::Utf8(encoded) = &rows[0][id] else { + panic!() + }; + assert_eq!( + decode_series_identity(encoded).unwrap()["unreferenced"], + "b" + ); +} diff --git a/crates/asap-physical-operators/tests/deployment.rs b/crates/asap-physical-operators/tests/deployment.rs new file mode 100644 index 00000000..0c261a03 --- /dev/null +++ b/crates/asap-physical-operators/tests/deployment.rs @@ -0,0 +1,90 @@ +//! Exercise the public library without a backend server, store, or scheduler. +use asap_physical_operators::planner::{ + post_asap::{ + GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryFamilyType, + SummaryUpdate, + }, + pre_asap::ColumnRef, +}; +use asap_physical_operators::{factory::create_planner_accumulator, AggregateCore}; + +fn family(k: u32) -> SummaryFamilyType { + SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + GroupingStrategy::PerSubpopulationInstance, + ) +} +fn build(values: &[f64]) -> Box { + let mut operator = create_planner_accumulator( + &family(512), + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ) + .unwrap(); + for (at, value) in values.iter().enumerate() { + operator.validate_single_input(*value).unwrap(); + operator.update_single(*value, at as i64); + } + operator.into_accumulator() +} +fn read(state: &dyn AggregateCore) -> f64 { + state + .estimate(&asap_physical_operators::planner::post_asap::SketchQuery::Quantile { q: 0.5 }) + .unwrap() +} + +// The same kernels work when every build is query-time, when only a prefix +// was precomputed, and when all state was precomputed before the readout. +#[test] +fn raw_partial_and_fully_precomputed_use_the_same_kernels() { + let raw: Vec = (0..128).map(f64::from).collect(); + let raw_only = build(&raw); + let stored_prefix = build(&raw[..64]); + let query_time_suffix = build(&raw[64..]); + let partial = stored_prefix.merge_with(&*query_time_suffix).unwrap(); + let stored_complete = build(&raw); + assert_eq!(read(&*raw_only), read(&*partial)); + assert_eq!(read(&*partial), read(&*stored_complete)); + assert!((read(&*raw_only) - 64.0).abs() <= 1.0); +} + +// A compiler must reject invalid physical parameters before starting execution. +#[test] +fn invalid_kll_parameters_are_rejected_at_binding() { + let result = create_planner_accumulator( + &family(0), + &SummaryUpdate::column(ColumnRef::SampleValue), + &Default::default(), + ); + assert!(result.is_err()); +} + +// Native CountSketch supports the confidence-sized depth used by the backend; +// a packed-wire column-bit budget must not be imposed on this constructor. +#[test] +fn native_count_sketch_dimensions_are_not_packed_wire_dimensions() { + use asap_physical_operators::planner::post_asap::SummaryInputExpr; + use asap_physical_operators::KeyByLabelValues; + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CountSketchWithHeap, + SketchParams::CountSketchWithHeap { + width: 1200, + depth: 55, + heap_size: 3, + }, + ), + Default::default(), + ); + let mut update = SummaryUpdate::column(ColumnRef::SampleValue); + update.item = Some(SummaryInputExpr::Column(ColumnRef::Named("host".into()))); + let mut operator = create_planner_accumulator(&family, &update, &Default::default()).unwrap(); + let key = KeyByLabelValues::new_with_labels(vec!["a".into()]); + operator.update_keyed(&key, 7.0, 1000); + let state = operator.into_accumulator(); + let state = state + .as_any() + .downcast_ref::() + .unwrap(); + assert_eq!(state.query_key(&key), 7.0); +} diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs new file mode 100644 index 00000000..96e74e54 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -0,0 +1,1443 @@ +//! Acceptance tests use the library directly, without either backend engine. +use asap_physical_operators::{ + dag::{ + operators::{Expression, Operator, Reduction, SortKey}, + values::{Batch, Schema, Value}, + Limits, PhysicalDag, RunContext, Scope, + }, + Statistic, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{ExactKind, ExactParams, SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::sync::Arc; +fn schema(fields: &[(&str, DataType, bool)]) -> Schema { + Arc::new(SummarySchema { + fields: fields + .iter() + .map(|(name, dtype, nullable)| SummaryField { + name: (*name).into(), + dtype: SummaryFamilyType::Plain(dtype.clone()), + nullable: *nullable, + }) + .collect(), + time_index: None, + }) +} +fn run(dag: &PhysicalDag<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { + let context = RunContext::new( + scope, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap(); + block_on(async { + let mut stream = dag.execute(&[root], context.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + assert_eq!(context.retained_bytes(), 0); + rows + }) +} +fn query() -> Scope { + Scope::Query { + evaluation_time_ms: 1000, + revision: 2, + } +} +fn floats(rows: &[Vec], column: usize) -> Vec { + rows.iter() + .map(|r| { + if let Value::Float64(v) = r[column] { + v + } else { + panic!("not Float64") + } + }) + .collect() +} + +// Sort followed by partitioned Limit implements ranking independently per group. +#[test] +fn grouped_sort_limit_across_batches() { + let schema = schema(&[ + ("group", DataType::Int64, false), + ("score", DataType::Float64, false), + ]); + let batches = [ + vec![(1, 1.), (2, 4.), (1, 9.)], + vec![(2, 8.), (1, 5.), (2, 2.)], + ] + .into_iter() + .map(|rows| { + Batch::try_new( + schema.clone(), + rows.into_iter() + .map(|(g, v)| vec![Value::Int64(g), Value::Float64(v)]) + .collect(), + ) + .unwrap() + }) + .collect(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), batches).unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema.clone(), + vec![SortKey { + column: 1, + descending: true, + nulls_first: false, + }], + vec![0], + ) + .unwrap(), + ) + .unwrap(); + dag.add(2, vec![1], Operator::limit(schema, 1, 1, vec![0]).unwrap()) + .unwrap(); + assert_eq!(floats(&run(&dag, 2, query()), 1), vec![5., 4.]); +} + +// The same computation runs in either engine scope with fresh per-run state. +#[test] +fn summary_construction_merge_and_readout_at_both_phases() { + let schema = schema(&[("v", DataType::Float64, false)]); + let batches = (1..=20) + .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Float64(v as f64)]]).unwrap()) + .collect(); + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let build = Operator::summary_build(schema.clone(), family, 0, None, vec![]).unwrap(); + let state = build.schema(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(schema, batches).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add(2, vec![1, 1], Operator::union(state.clone(), 2).unwrap()) + .unwrap(); + dag.add( + 3, + vec![2], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add( + 4, + vec![3], + Operator::readout( + state, + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Sum, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + for scope in [ + query(), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 2, + }, + ] { + assert_eq!(floats(&run(&dag, 4, scope), 0), vec![420.]); + } +} + +// A semi-join can consume two branches of one producer with a one-batch buffer. +#[test] +fn diamond_semijoin_preserves_left_values_and_multiplicity() { + let schema = schema(&[("key", DataType::Int64, false)]); + let batches = [1, 2, 2, 3] + .into_iter() + .map(|v| Batch::try_new(schema.clone(), vec![vec![Value::Int64(v)]]).unwrap()) + .collect(); + let filter = Operator::filter( + schema.clone(), + Expression::Equal( + Box::new(Expression::Column(0)), + Box::new(Expression::Literal { + value: Value::Int64(2), + dtype: DataType::Int64, + }), + ), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), batches).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], filter).unwrap(); + dag.add( + 2, + vec![0, 1], + Operator::semi_join(schema.clone(), schema, vec![(0, 0)]).unwrap(), + ) + .unwrap(); + let rows = run(&dag, 2, query()); + assert_eq!(rows.len(), 2); + assert!(rows.iter().all(|r| matches!(r[0], Value::Int64(2)))); +} + +// Integer aggregation must not silently lose precision through Float64. +#[test] +fn exact_integer_and_empty_extrema() { + let schema = schema(&[("v", DataType::Int64, false)]); + let aggregate = Operator::aggregate( + schema.clone(), + vec![], + vec![("sum".into(), Reduction::Sum(0))], + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + let value = 9_007_199_254_740_993; + dag.add( + 0, + vec![], + Operator::source( + schema.clone(), + vec![Batch::try_new( + schema.clone(), + vec![vec![Value::Int64(value)], vec![Value::Int64(2)]], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], aggregate).unwrap(); + assert!(matches!(run(&dag,1,query())[0][0],Value::Int64(v) if v==value+2)); + let mut empty = PhysicalDag::default(); + empty + .add(0, vec![], Operator::source(schema.clone(), vec![]).unwrap()) + .unwrap(); + empty + .add( + 1, + vec![0], + Operator::aggregate(schema, vec![], vec![("min".into(), Reduction::Min(0))]).unwrap(), + ) + .unwrap(); + assert!(matches!(run(&empty, 1, query())[0][0], Value::Null)); +} + +// Plain value operators are library implementations, including NaN comparison. +#[test] +fn scalar_negation_and_vector_conversion() { + let scalar = Operator::scalar(Value::Float64(7.), DataType::Float64).unwrap(); + let project = Operator::project( + scalar.schema(), + vec![( + "v".into(), + Expression::Negate(Box::new(Expression::Column(0))), + )], + ) + .unwrap(); + let convert = Operator::vector_to_scalar(project.schema(), 0).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scalar).unwrap(); + dag.add(1, vec![0], project).unwrap(); + dag.add(2, vec![1], convert).unwrap(); + assert_eq!(floats(&run(&dag, 2, query()), 0), vec![-7.]); + let scalar = Operator::scalar(Value::Float64(f64::NAN), DataType::Float64).unwrap(); + let predicate = Expression::Equal( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(0)), + ); + let filter = Operator::filter(scalar.schema(), predicate).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scalar).unwrap(); + dag.add(1, vec![0], filter).unwrap(); + assert!(run(&dag, 1, query()).is_empty()); +} + +// Invalid operations fail at binding rather than becoming external fallbacks. +#[test] +fn binding_rejects_unsupported_operations() { + let schema = schema(&[("v", DataType::Float64, false)]); + assert!(Operator::summary_build( + schema.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Rate, ExactParams::Rate), + 0, + None, + vec![] + ) + .is_err()); + let sum = Operator::summary_build( + schema.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum), + 0, + None, + vec![], + ) + .unwrap(); + assert!(Operator::readout( + sum.schema(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q: 0.5 } + ) + ) + .is_err()); + assert!(Operator::filter(schema, Expression::Column(0)).is_err()); +} + +// KLL is one family example: precomputation changes input sources, not operators. +#[test] +fn kll_raw_partial_and_precomputed_are_native_dags() { + use planner_types::post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("value", DataType::Float64, false)]); + let family = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), + GroupingStrategy::PerSubpopulationInstance, + ); + let build = Operator::summary_build(input.clone(), family, 0, None, vec![]).unwrap(); + let state = build.schema(); + let build_range = |start: u32, end: u32| { + let mut dag = PhysicalDag::default(); + let batch = Batch::try_new( + input.clone(), + (start..end) + .map(|v| vec![Value::Float64(f64::from(v))]) + .collect(), + ) + .unwrap(); + dag.add( + 0, + vec![], + Operator::source(input.clone(), vec![batch]).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], build.clone()).unwrap(); + run( + &dag, + 1, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 1, + }, + ) + }; + let prefix = build_range(0, 64); + let complete = build_range(0, 128); + let query_plan = |stored: Option>>, raw_start: Option| { + let mut dag = PhysicalDag::default(); + let mut states = vec![]; + if let Some(rows) = stored { + dag.add( + 0, + vec![], + Operator::source( + state.clone(), + vec![Batch::try_new(state.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + states.push(0); + } + if let Some(start) = raw_start { + dag.add( + 1, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new( + input.clone(), + (start..128) + .map(|v| vec![Value::Float64(f64::from(v))]) + .collect(), + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add(2, vec![1], build.clone()).unwrap(); + states.push(2); + } + dag.add( + 3, + states.clone(), + Operator::union(state.clone(), states.len()).unwrap(), + ) + .unwrap(); + dag.add( + 4, + vec![3], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add( + 5, + vec![4], + Operator::readout( + state.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q: 0.5 }, + ), + ) + .unwrap(), + ) + .unwrap(); + floats(&run(&dag, 5, query()), 0)[0] + }; + let raw = query_plan(None, Some(0)); + let partial = query_plan(Some(prefix), Some(64)); + let full = query_plan(Some(complete), None); + assert_eq!(raw, partial); + assert_eq!(partial, full); + assert!((raw - 64.).abs() <= 1.); +} + +// Exact state must match its declared family; a mislabeled state is rejected. +#[test] +fn exact_state_and_family_validation() { + use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let mut acc = ExactAccumulator::new(family.clone(), false).unwrap(); + acc.update(None, 7., 0); + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }], + time_index: None, + }); + let value = Value::Summary { + family: family.clone(), + state: Arc::new(acc), + }; + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + schema.clone(), + vec![Batch::try_new(schema.clone(), vec![vec![value]]).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::readout( + schema.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Sum, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + assert_eq!(floats(&run(&dag, 1, query()), 0), vec![7.]); + let wrong = ExactAccumulator::new( + SummaryFamilyType::ExactAggregate(ExactKind::Max, ExactParams::Max), + false, + ) + .unwrap(); + assert!(Batch::try_new( + schema, + vec![vec![Value::Summary { + family, + state: Arc::new(wrong) + }]] + ) + .is_err()); +} + +// Planner binding rejects unknown computation instead of accepting a fallback. +#[test] +fn bind_post_asap_before_execution() { + use asap_physical_operators::dag::planner::bind; + use planner_types::{ + post_asap::{ + EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, + PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, + WindowEdgeCompatibility, + }, + pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let schema = schema(&[("value", DataType::Float64, false)]); + let node = |id, payload| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*schema).clone(), + guarantee: None, + }; + let mut dag = PostAsapDag { + nodes: vec![ + node( + 0, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(1.), + }, + ), + node( + 1, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Project { + cols: vec![ProjectItem { + alias: None, + expr: QueryExpr::Arithmetic { + op: ArithmeticOpKind::Add, + left: Rc::new(QueryExpr::Column(0)), + right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(2.))), + }, + }], + qualifier: None, + }, + }, + ), + ], + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: (*schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let sources = || -> BTreeMap> { + BTreeMap::from([( + 0, + Box::new( + Operator::source( + schema.clone(), + vec![Batch::try_new(schema.clone(), vec![vec![Value::Float64(1.)]]).unwrap()], + ) + .unwrap(), + ) as asap_physical_operators::dag::planner::Source<'static>, + )]) + }; + let native = bind(&dag, sources(), &[1]).unwrap(); + assert_eq!(floats(&run(&native, 1, query()), 0), vec![3.]); + assert!(bind(&dag, BTreeMap::new(), &[1]).is_err()); + dag.nodes[1].payload = PostAsapOperatorPayload::Value { + operation: ValueOperation::Extension { + name: "unknown".into(), + }, + }; + assert!(bind(&dag, sources(), &[1]).is_err()); +} + +// A completed empty population has an exact zero count, with integer output. +#[test] +fn empty_exact_count_is_an_integer_state_readout() { + let input = schema(&[("value", DataType::Float64, false)]); + let build = Operator::summary_build( + input.clone(), + SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count), + 0, + None, + vec![], + ) + .unwrap(); + let read = Operator::readout( + build.schema(), + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic: Statistic::Count, + lookback_ms: None, + }, + ), + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input, vec![]).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add(2, vec![1], read).unwrap(); + assert!(matches!(run(&dag, 2, query())[0][0], Value::Int64(0))); +} + +// A deployment source cannot pass a different row shape to bound expressions. +#[test] +fn source_batches_must_match_the_bound_schema() { + use asap_physical_operators::dag::{self, PhysicalOperator}; + use planner_types::{ + post_asap::{ + ExecutionDataState, PostAsapDag, PostAsapDagNode, PostAsapNodeId, + PostAsapOperatorPayload, + }, + pre_asap::QueryExpr, + }; + use std::{cell::Cell, collections::BTreeMap, rc::Rc}; + struct WrongSource { + schema: Schema, + starts: Rc>, + } + impl PhysicalOperator for WrongSource { + fn name(&self) -> &str { + "ExternalSource" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.schema.clone() + } + fn output_bytes(&self, value: &Batch) -> usize { + value.bytes() + } + fn start<'a>( + &'a self, + _: Vec>, + _: RunContext, + ) -> Result, dag::Error> { + self.starts.set(self.starts.get() + 1); + Ok( + futures::stream::once(async { Batch::try_new(schema(&[]), vec![vec![]]) }) + .boxed_local(), + ) + } + } + let expected = schema(&[("value", DataType::Float64, false)]); + let starts = Rc::new(Cell::new(0)); + let plan = PostAsapDag { + nodes: vec![PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(1.), + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*expected).clone(), + guarantee: None, + }], + edges: vec![], + root: PostAsapNodeId(0), + }; + let source = Box::new(WrongSource { + schema: expected, + starts: starts.clone(), + }) as dag::planner::Source<'static>; + let native = dag::planner::bind(&plan, BTreeMap::from([(0, source)]), &[0]).unwrap(); + assert_eq!(starts.get(), 0); + let context = RunContext::new(query(), Limits::default()).unwrap(); + let mut output = native.execute(&[0], context).unwrap().remove(0); + assert!(matches!( + block_on(output.next()), + Some(Err(dag::Error::AtNode { node: 0, .. })) + )); + assert_eq!(starts.get(), 1); +} + +// Float extrema have the same NaN behavior as the exact summary kernels. +#[test] +fn extrema_preserve_numeric_values_in_the_presence_of_nan() { + let input = schema(&[("v", DataType::Float64, false)]); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new( + input.clone(), + vec![vec![Value::Float64(-f64::NAN)], vec![Value::Float64(5.)]], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::aggregate( + input, + vec![], + vec![ + ("min".into(), Reduction::Min(0)), + ("max".into(), Reduction::Max(0)), + ], + ) + .unwrap(), + ) + .unwrap(); + let rows = run(&dag, 1, query()); + assert_eq!(floats(&rows, 0), vec![5.]); + assert_eq!(floats(&rows, 1), vec![5.]); +} + +// Planner wire nodes, including grouping and edge roles, are executable at either phase. +#[test] +fn planner_semijoin_sort_limit_contract_at_both_phases() { + use asap_physical_operators::dag::planner::{bind, Source}; + use planner_types::{ + post_asap::*, + pre_asap::{CompareOpKind, GroupKeys, JoinKind, Predicate, QueryExpr, SortKey}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let rows_schema = schema(&[ + ("group", DataType::Utf8, false), + ("key", DataType::Utf8, false), + ("score", DataType::Float64, false), + ]); + let keys_schema = schema(&[("key", DataType::Utf8, false)]); + let node = |id, payload, schema: &Schema| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_schema: (**schema).clone(), + output_state: ExecutionDataState::QUERY_ROWS, + guarantee: None, + }; + let edge = |producer, consumer, role, schema: &Schema| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: (**schema).clone(), + data_state: ExecutionDataState::QUERY_ROWS, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }; + let groups = GroupKeys::by(vec![0]); + let dag = PostAsapDag { + nodes: vec![ + node( + 0, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(0.), + }, + &rows_schema, + ), + node( + 1, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::promql_scalar(0.), + }, + &keys_schema, + ), + node( + 2, + PostAsapOperatorPayload::RelationalJoin { + join_kind: JoinKind::Semi, + pruning: None, + pred: Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(1)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(3)), + })), + }, + &rows_schema, + ), + node( + 3, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Sort { + keys: vec![SortKey { + expr: QueryExpr::Column(2), + ascending: false, + nulls_first: false, + }], + partition_by: groups.clone(), + }, + }, + &rows_schema, + ), + node( + 4, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: groups, + }, + }, + &rows_schema, + ), + ], + // Deliberately put Right before Left: list order must not swap inputs. + edges: vec![ + edge(1, 2, EdgeRole::Right, &keys_schema), + edge(0, 2, EdgeRole::Left, &rows_schema), + edge(2, 3, EdgeRole::Input, &rows_schema), + edge(3, 4, EdgeRole::Input, &rows_schema), + ], + root: PostAsapNodeId(4), + }; + let text = |v: &str| Value::Utf8(v.into()); + for (phase, scope) in [ + (ExecutionTiming::QueryTime, query()), + ( + ExecutionTiming::IngestionTime, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1000, + revision: 2, + }, + ), + ] { + let dag = dag + .with_execution_phases(&dag.nodes.iter().map(|node| (node.id, phase)).collect()) + .unwrap(); + let sources: BTreeMap> = BTreeMap::from([ + ( + 0, + Box::new( + Operator::source( + rows_schema.clone(), + vec![Batch::try_new( + rows_schema.clone(), + vec![ + vec![text("a"), text("x"), Value::Float64(8.)], + vec![text("a"), text("y"), Value::Float64(9.)], + vec![text("b"), text("x"), Value::Float64(2.)], + vec![text("b"), text("z"), Value::Float64(99.)], + ], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'static>, + ), + ( + 1, + Box::new( + Operator::source( + keys_schema.clone(), + vec![Batch::try_new( + keys_schema.clone(), + vec![vec![text("x")], vec![text("y")]], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'static>, + ), + ]); + let native = bind(&dag, sources, &[4]).unwrap(); + let mut scores = floats(&run(&native, 4, scope), 2); + scores.sort_by(f64::total_cmp); + assert_eq!(scores, vec![2., 9.]); + } +} + +// Planner scalar signatures, collection access and null predicates share native execution. +#[test] +fn planner_expressions_preserve_collection_and_nullable_types() { + use asap_physical_operators::dag::expressions::CompiledExpression; + use planner_types::pre_asap::{CompareOpKind, QueryExpr, ScalarValue}; + use std::rc::Rc; + let input_schema = schema(&[( + "items", + DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Int64), + value_nullable: false, + }, + false, + )]); + let access = QueryExpr::FunctionCall { + name: "asap_element_access".into(), + args: vec![ + QueryExpr::Column(0), + QueryExpr::Literal(ScalarValue::Utf8("count".into())), + ], + }; + let project = Operator::project( + input_schema.clone(), + vec![( + "count".into(), + Expression::planner(CompiledExpression::compile(&access, &input_schema).unwrap()), + )], + ) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source( + input_schema.clone(), + vec![Batch::try_new( + input_schema.clone(), + vec![ + vec![Value::Map( + vec![(Value::Utf8("count".into()), Value::Int64(7))].into(), + )], + vec![Value::Map(Arc::from([]))], + ], + ) + .unwrap()], + ) + .unwrap(), + ) + .unwrap(); + let projected = project.schema(); + dag.add(1, vec![0], project).unwrap(); + let predicate = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Ge, + right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(1))), + }; + dag.add( + 2, + vec![1], + Operator::filter( + projected.clone(), + Expression::planner(CompiledExpression::compile(&predicate, &projected).unwrap()), + ) + .unwrap(), + ) + .unwrap(); + let rows = run(&dag, 2, query()); + assert!(matches!(rows.as_slice(),[row] if matches!(row.as_slice(),[Value::Int64(7)]))); + let unknown = QueryExpr::FunctionCall { + name: "unregistered_function".into(), + args: vec![QueryExpr::Column(0)], + }; + assert!(CompiledExpression::compile(&unknown, &input_schema).is_err()); +} + +// Outer, semi and anti joins share Planner predicates and preserve SQL null behavior. +#[test] +fn native_relational_join_kinds_preserve_unmatched_rows() { + use planner_types::pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}; + use std::rc::Rc; + let input = schema(&[("key", DataType::Int64, true)]); + let predicate = Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })); + for (kind, count) in [ + (JoinKind::Inner, 1), + (JoinKind::Left, 3), + (JoinKind::Right, 3), + (JoinKind::Full, 5), + (JoinKind::Semi, 1), + (JoinKind::Anti, 2), + (JoinKind::Cross, 9), + ] { + let output = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + input.clone() + } else { + schema(&[ + ("left", DataType::Int64, true), + ("right", DataType::Int64, true), + ]) + }; + let mut dag = PhysicalDag::default(); + for (id, rows) in [ + ( + 0, + vec![ + vec![Value::Int64(1)], + vec![Value::Int64(2)], + vec![Value::Null], + ], + ), + ( + 1, + vec![ + vec![Value::Int64(2)], + vec![Value::Int64(3)], + vec![Value::Null], + ], + ), + ] { + dag.add( + id, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new(input.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + } + dag.add( + 2, + vec![0, 1], + Operator::relational_join( + input.clone(), + input.clone(), + kind.clone(), + &predicate, + output, + ) + .unwrap(), + ) + .unwrap(); + assert_eq!(run(&dag, 2, query()).len(), count, "{kind:?}"); + } +} + +// Per-series fractional rates feed either weighted frequency family per job, in either scope. +#[test] +fn weighted_rate_topk_preserves_partitions_fractional_scores_and_evaluation_scope() { + for count_sketch in [false, true] { + assert_weighted_rate_topk(count_sketch); + } +} +fn assert_weighted_rate_topk(count_sketch: bool) { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let raw = schema(&[ + ("service", DataType::Utf8, false), + ("job", DataType::Utf8, false), + ("instance", DataType::Int64, false), + ("t", DataType::Timestamp, false), + ("value", DataType::Float64, false), + ]); + let mut rows = Vec::new(); + // Multiple instances of auth accumulate. Batch has a very different scale. + for (service, job, instance, rate) in [ + ("auth", "api", 1, 0.125), + ("auth", "api", 2, 0.25), + ("checkout", "api", 1, 0.3125), + ("search", "api", 1, 0.0625), + ("ingest", "batch", 1, 100.0), + ("export", "batch", 1, 80.0), + ("cleanup", "batch", 1, 20.0), + ] { + for (t, value) in [(0, 0.0), (30_000, rate * 30.0), (60_000, rate * 60.0)] { + rows.push(vec![ + Value::Utf8(service.into()), + Value::Utf8(job.into()), + Value::Int64(instance), + Value::Timestamp(t), + Value::Float64(value), + ]); + } + } + let rates = Operator::window( + raw.clone(), + planner_types::pre_asap::AggIntent::Rate, + 3, + 4, + vec![0, 1, 2], + Some((0, 60_000)), + ) + .unwrap(); + let family = SummaryFamilyType::Sketch( + SketchKind::new( + if count_sketch { + SketchAlgorithm::CountSketchWithHeap + } else { + SketchAlgorithm::CmsWithHeap + }, + if count_sketch { + SketchParams::CountSketchWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + } + } else { + SketchParams::CmsWithHeap { + width: 4096, + depth: 5, + heap_size: 8, + } + }, + ), + Default::default(), + ); + let build = Operator::keyed_summary_build(rates.schema(), family, 3, vec![0], vec![1]).unwrap(); + let output = schema(&[ + ("job", DataType::Utf8, false), + ("service", DataType::Utf8, false), + ("score", DataType::Float64, false), + ]); + let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(raw.clone(), vec![Batch::try_new(raw, rows).unwrap()]).unwrap(), + ) + .unwrap(); + dag.add(1, vec![0], rates).unwrap(); + dag.add(2, vec![1], build).unwrap(); + dag.add(3, vec![2], readout).unwrap(); + dag.add( + 4, + vec![3], + Operator::sort( + output.clone(), + vec![SortKey { + column: 2, + descending: true, + nulls_first: false, + }], + vec![0], + ) + .unwrap(), + ) + .unwrap(); + dag.add(5, vec![4], Operator::limit(output, 2, 0, vec![0]).unwrap()) + .unwrap(); + for scope in [ + query(), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 2, + }, + query(), + ] { + let result = run(&dag, 5, scope); + assert_eq!(result.len(), 4); + assert_eq!(floats(&result, 2), vec![0.375, 0.3125, 100.0, 80.0]); + let services = result + .iter() + .map(|row| match &row[1] { + Value::Utf8(v) => v.as_ref(), + _ => panic!("service"), + }) + .collect::>(); + assert_eq!(services, vec!["auth", "checkout", "ingest", "export"]); + } +} + +// The grouped temporal reducer's sample schema must survive physical Sort/Limit binding. +#[test] +fn grouped_temporal_schema_compiles_and_executes_topk() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::post_asap::{ + ExecutionDataState, PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, + ValueOperation, + }; + use planner_types::pre_asap::{ + aggregate_output_schema, AggIntent, Column, GroupKeys, QueryExpr, Reduction as IrReduction, + Schema as IrSchema, + }; + let grouped = IrSchema::new(vec![ + Column::new("job", DataType::Utf8, false), + Column::new("sum", DataType::Float64, false), + ]); + let output = aggregate_output_schema( + &grouped, + &IrReduction::PerEntity, + &[AggIntent::Avg { col: None }], + &[], + ) + .unwrap(); + let input = schema( + &output + .columns + .iter() + .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) + .collect::>(), + ); + let node = |id, operation| PostAsapDagNode { + id: PostAsapNodeId(id), + payload: PostAsapOperatorPayload::Value { operation }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*input).clone(), + guarantee: None, + }; + let sort = compile_node( + &node( + 1, + ValueOperation::Sort { + keys: vec![planner_types::pre_asap::SortKey { + expr: QueryExpr::Column(1), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::none(), + }, + ), + std::slice::from_ref(&input), + ) + .unwrap(); + let limit = compile_node( + &node( + 2, + ValueOperation::Limit { + n: 1, + offset: 0, + partition_by: GroupKeys::none(), + }, + ), + std::slice::from_ref(&input), + ) + .unwrap(); + let compiled = CompiledPhysicalDag::from_operators( + [(0, InputContract::bounded(input.clone()))].into(), + [(1, (vec![0], sort)), (2, (vec![1], limit))].into(), + vec![2], + ) + .unwrap(); + let recovered = + serde_json::from_slice::(&serde_json::to_vec(&compiled).unwrap()) + .unwrap(); + assert_eq!(recovered.row_source(2), Some(0)); + assert_eq!(recovered.operator_name(2), Some("Limit")); + let expected = vec![Value::Utf8("api".into()), Value::Float64(9.)]; + let batch = Batch::try_new( + input.clone(), + vec![ + vec![Value::Utf8("worker".into()), Value::Float64(2.)], + expected.clone(), + ], + ) + .unwrap(); + let source = Box::new(Operator::source(input, vec![batch]).unwrap()) as Source<'_>; + let physical = recovered.instantiate([(0, source)].into()).unwrap(); + let mut stream = physical + .execute(&[2], RunContext::new(query(), Limits::default()).unwrap()) + .unwrap() + .remove(0); + let rows = block_on(async { + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend_from_slice(batch.unwrap().rows()); + } + rows + }); + assert_eq!(rows.len(), 1); + assert!(matches!(&rows[0][0], Value::Utf8(label) if label.as_ref() == "api")); + assert!(matches!(rows[0][1], Value::Float64(9.))); +} + +// A certified candidate set must have authoritative values for every key, including after recovery. +#[test] +fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::{ + post_asap::*, + pre_asap::{CompareOpKind, JoinKind, Predicate, QueryExpr}, + }; + use std::{collections::BTreeMap, rc::Rc}; + let schema = schema(&[("key", DataType::Utf8, false)]); + for certified in [false, true] { + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + output_schema: (*schema).clone(), + output_state: ExecutionDataState::QUERY_ROWS, + guarantee: None, + payload: PostAsapOperatorPayload::RelationalJoin { + join_kind: JoinKind::Semi, + pred: Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })), + pruning: certified.then_some(CandidateCompleteness::Certified { + guarantee: ResultGuarantee { + metric: ErrorMetric::TopKMembership, + bound: BoundExpr::Zero, + failure_probability: ProbabilityExpr::Constant { value: 0.01 }, + provenance: vec![], + }, + }), + }, + }; + let graph = CompiledPhysicalDag::from_operators( + [ + (0, InputContract::bounded(schema.clone())), + (1, InputContract::bounded(schema.clone())), + ] + .into(), + [( + 2, + ( + vec![0, 1], + compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(), + ), + )] + .into(), + vec![2], + ) + .unwrap(); + let graph = + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + assert_eq!( + graph.certified_pruning_keys(2), + certified.then_some(&[(0, 0)][..]) + ); + for complete in [false, true] { + let sources = [ + vec!["a"], + if complete { + vec!["a"] + } else { + vec!["a", "missing"] + }, + ] + .into_iter() + .enumerate() + .map(|(i, keys)| { + let batch = Batch::try_new( + schema.clone(), + keys.into_iter() + .map(|k| vec![Value::Utf8(k.into())]) + .collect(), + ) + .unwrap(); + ( + i as u64, + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect::>(); + let bound = graph.instantiate(sources).unwrap(); + let result = block_on(async { + let mut stream = bound + .execute( + graph.roots(), + RunContext::new(query(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok::<_, asap_physical_operators::Error>(rows) + }); + if certified && !complete { + assert!(result + .unwrap_err() + .to_string() + .contains("no authoritative value")); + } else { + let rows = result.unwrap(); + assert_eq!(rows.len(), 1); + assert!(matches!(&rows[0][0], Value::Utf8(key) if key.as_ref() == "a")); + } + } + } +} + +// Precompute arithmetic must match population/window identities, never zip arrival order. +#[test] +fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { + use asap_physical_operators::physical_planner::{ + compile_node, CompiledPhysicalDag, InputContract, Source, + }; + use planner_types::{ + post_asap::*, + pre_asap::{ArithmeticOpKind, BinaryOpKind}, + }; + use std::collections::BTreeMap; + let input = schema(&[ + ("population", DataType::Utf8, false), + ("time", DataType::Timestamp, false), + ("value", DataType::Float64, false), + ]); + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + output_schema: (*input).clone(), + output_state: ExecutionDataState::INGESTION_ROWS, + guarantee: None, + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + }; + let program = CompiledPhysicalDag::from_operators( + [ + (0, InputContract::bounded(input.clone())), + (1, InputContract::bounded(input.clone())), + ] + .into(), + [( + 2, + ( + vec![0, 1], + compile_node(&node, &[input.clone(), input.clone()]).unwrap(), + ), + )] + .into(), + vec![2], + ) + .unwrap(); + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + for (right, expected) in [ + (vec![("b", 2, 3.), ("a", 1, 2.)], Some(vec![8., 17.])), + (vec![("b", 2, 3.)], None), + (vec![("a", 1, 2.), ("a", 1, 2.)], None), + (vec![("a", 2, 2.), ("b", 1, 3.)], None), + (vec![("a", 1, f64::NAN), ("b", 2, 3.)], None), + ] { + let sources = [vec![("a", 1, 10.), ("b", 2, 20.)], right] + .into_iter() + .enumerate() + .map(|(i, rows)| { + let rows = rows + .into_iter() + .map(|(group, time, value)| { + vec![ + Value::Utf8(group.into()), + Value::Timestamp(time), + Value::Float64(value), + ] + }) + .collect(); + let batch = Batch::try_new(input.clone(), rows).unwrap(); + ( + i as u64, + Box::new(Operator::source(input.clone(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect::>(); + let graph = program.instantiate(sources).unwrap(); + let result = block_on(async { + let mut stream = graph + .execute( + program.roots(), + RunContext::new(query(), Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok::<_, asap_physical_operators::Error>(rows) + }); + match expected { + Some(values) => assert_eq!(floats(&result.unwrap(), 2), values), + None => assert!(result.is_err()), + } + } +} diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs new file mode 100644 index 00000000..5a920a8d --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -0,0 +1,105 @@ +//! Deserialized physical plans recover selected operators without logical lowering. +//! Deployments choose the encoding; JSON is used here only as a test format. +use asap_physical_operators::{ + operators::{Operator, SortKey}, + physical_planner::{CompiledPhysicalDag, InputContract}, +}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::DataType, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn sorted() -> CompiledPhysicalDag { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }); + CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + BTreeMap::from([( + 1, + ( + vec![0], + Operator::sort( + schema, + vec![SortKey { + column: 0, + descending: true, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ), + )]), + vec![1], + ) + .unwrap() +} + +#[test] +fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { + let bytes = serde_json::to_vec(&sorted()).unwrap(); + let recovered = serde_json::from_slice::(&bytes).unwrap(); + assert_eq!(serde_json::to_vec(&recovered).unwrap(), bytes); + for mutation in ["column", "edge", "output"] { + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + match mutation { + "column" => { + wire["nodes"]["1"]["Operator"]["operator"]["kind"]["Sort"]["keys"][0]["column"] = + 7.into() + } + "edge" => wire["nodes"]["1"]["Operator"]["inputs"][0] = 999.into(), + "output" => { + wire["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][0]["dtype"] = + serde_json::json!({"Plain":"utf8"}) + } + _ => unreachable!(), + } + assert!( + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) + .is_err(), + "accepted {mutation}" + ); + } +} + +#[test] +fn candidate_recovery_preserves_materialization_boundary() { + use asap_physical_operators::physical_planner::PhysicalCandidate; + let precompute = sorted(); + let output = InputContract::bounded(precompute.output_contract(1).unwrap().schema); + let query = CompiledPhysicalDag::from_operators( + BTreeMap::from([(1, output.clone())]), + BTreeMap::from([( + 2, + ( + vec![1], + Operator::limit(output.schema.clone(), 3, 0, vec![]).unwrap(), + ), + )]), + vec![2], + ) + .unwrap(); + let candidate = PhysicalCandidate { + precompute: Some(precompute), + query, + materialized_outputs: BTreeMap::from([(1, output)]), + }; + let bytes = serde_json::to_vec(&candidate).unwrap(); + let restored = serde_json::from_slice::(&bytes).unwrap(); + assert_eq!(restored.precompute.as_ref().unwrap().roots(), &[1]); + assert_eq!(restored.query.roots(), &[2]); + assert_eq!(serde_json::to_vec(&restored).unwrap(), bytes); + let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); + wire["materialized_outputs"]["1"]["schema"]["fields"][0]["dtype"] = + serde_json::json!({"Plain":"utf8"}); + assert!( + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()).is_err() + ); +} diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs new file mode 100644 index 00000000..5770d7d6 --- /dev/null +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -0,0 +1,695 @@ +//! Contract tests inspired by DataFusion's limit, sort and join test matrices. +//! Expectations follow ASAP's IR (notably row-count and IEEE NaN equality). +//! Reference: apache/datafusion e2ca7f3, physical-plan/src/{limit.rs,sorts/sort.rs}. +use asap_physical_operators::{ + expressions::CompiledExpression, + operators::{Expression, Operator, Reduction, SortKey}, + plan::PhysicalDag, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{CompareOpKind, DataType, JoinKind, Predicate, QueryExpr}, +}; +use std::{rc::Rc, sync::Arc}; + +fn schema(fields: &[(&str, DataType, bool)]) -> Schema { + Arc::new(SummarySchema { + fields: fields + .iter() + .map(|(name, dtype, nullable)| SummaryField { + name: (*name).into(), + dtype: SummaryFamilyType::Plain(dtype.clone()), + nullable: *nullable, + }) + .collect(), + time_index: None, + }) +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap() +} +fn collect(dag: &PhysicalDag<'_, Batch, Schema>, root: u64) -> Vec> { + let run = context(); + let rows = block_on(async { + let mut stream = dag.execute(&[root], run.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = stream.next().await { + rows.extend_from_slice(batch.unwrap().rows()); + } + rows + }); + assert_eq!(run.retained_bytes(), 0); + rows +} +fn unary(input: Schema, batches: Vec>>, op: Operator) -> Vec> { + let mut dag = PhysicalDag::default(); + let batches = batches + .into_iter() + .map(|rows| Batch::try_new(input.clone(), rows).unwrap()) + .collect(); + dag.add(0, vec![], Operator::source(input, batches).unwrap()) + .unwrap(); + dag.add(1, vec![0], op).unwrap(); + collect(&dag, 1) +} +fn keys(rows: &[Vec]) -> Vec>> { + rows.iter() + .map(|r| r.iter().map(|v| v.key().unwrap()).collect()) + .collect() +} +fn eq_predicate() -> Predicate { + Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Eq, + right: Rc::new(QueryExpr::Column(1)), + })) +} +fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec> { + let input = schema(&[("key", DataType::Float64, true)]); + let output = if matches!(kind, JoinKind::Semi | JoinKind::Anti) { + input.clone() + } else { + schema(&[ + ("left", DataType::Float64, true), + ("right", DataType::Float64, true), + ]) + }; + let op = if keyed { + Operator::semi_join(input.clone(), input.clone(), vec![(0, 0)]).unwrap() + } else { + Operator::relational_join(input.clone(), input.clone(), kind, &eq_predicate(), output) + .unwrap() + }; + let mut dag = PhysicalDag::default(); + for (id, values) in [(0, left), (1, right)] { + let batches = values + .into_iter() + .map(|v| Batch::try_new(input.clone(), vec![vec![v]]).unwrap()) + .collect(); + dag.add( + id, + vec![], + Operator::source(input.clone(), batches).unwrap(), + ) + .unwrap(); + } + dag.add(2, vec![0, 1], op).unwrap(); + collect(&dag, 2) +} + +// OFFSET/FETCH must be invariant to empty batches and input batch boundaries. +#[test] +fn limit_offset_fetch_matrix() { + let input = schema(&[("v", DataType::Int64, false)]); + for chunk in [1, 2, 5, 12] { + let values = (0..9).map(|n| vec![Value::Int64(n)]).collect::>(); + let mut batches = vec![vec![]]; + for rows in values.chunks(chunk) { + batches.push(rows.to_vec()); + batches.push(vec![]); + } + for offset in [0, 1, 8, 9, 10, u64::MAX] { + for n in [0, 1, 3, 12, u64::MAX] { + let rows = unary( + input.clone(), + batches.clone(), + Operator::limit(input.clone(), n, offset, vec![]).unwrap(), + ); + let expected = values + .iter() + .skip(offset.min(9) as usize) + .take(n.min(9) as usize) + .cloned() + .collect::>(); + assert_eq!( + keys(&rows), + keys(&expected), + "chunk={chunk}, offset={offset}, n={n}" + ); + } + } + } +} + +// Zero-column batches still have rows: LIMIT must not infer cardinality from columns. +#[test] +fn limit_preserves_zero_column_row_count() { + let input = schema(&[]); + let rows = unary( + input.clone(), + vec![vec![vec![]; 5], vec![vec![]; 5]], + Operator::limit(input, 4, 3, vec![]).unwrap(), + ); + assert_eq!(rows.len(), 4); +} + +// NULL placement is independent of sort direction; ties retain original row order. +#[test] +fn sort_direction_null_placement_and_ties() { + let input = schema(&[("v", DataType::Int64, true), ("id", DataType::Int64, false)]); + let values = [Some(2), None, Some(1), Some(2), None]; + let rows = values + .iter() + .enumerate() + .map(|(i, v)| { + vec![ + v.map(Value::Int64).unwrap_or(Value::Null), + Value::Int64(i as i64), + ] + }) + .collect::>(); + for (descending, nulls_first, expected) in [ + (false, false, vec![2, 0, 3, 1, 4]), + (false, true, vec![1, 4, 2, 0, 3]), + (true, false, vec![0, 3, 2, 1, 4]), + (true, true, vec![1, 4, 0, 3, 2]), + ] { + let op = Operator::sort( + input.clone(), + vec![SortKey { + column: 0, + descending, + nulls_first, + }], + vec![], + ) + .unwrap(); + let result = unary( + input.clone(), + vec![rows[..2].to_vec(), vec![], rows[2..].to_vec()], + op, + ); + let ids = result + .iter() + .map(|r| match r[1] { + Value::Int64(n) => n, + _ => unreachable!(), + }) + .collect::>(); + assert_eq!(ids, expected); + } +} + +// Outer joins preserve unmatched NULLs, while semi/anti joins preserve left multiplicity. +#[test] +fn joins_nulls_duplicates_and_empty_sides() { + for (kind, expected_len) in [ + (JoinKind::Inner, 4), + (JoinKind::Left, 6), + (JoinKind::Right, 6), + (JoinKind::Full, 8), + (JoinKind::Semi, 2), + (JoinKind::Anti, 2), + ] { + let left = vec![ + Value::Float64(1.), + Value::Float64(1.), + Value::Float64(2.), + Value::Null, + ]; + let right = vec![ + Value::Float64(1.), + Value::Float64(1.), + Value::Float64(3.), + Value::Null, + ]; + let result = join(left, right, kind.clone(), false); + assert_eq!(result.len(), expected_len, "{kind:?}"); + } + for (kind, expected_len) in [ + (JoinKind::Inner, 0), + (JoinKind::Left, 1), + (JoinKind::Right, 0), + (JoinKind::Full, 1), + (JoinKind::Semi, 0), + (JoinKind::Anti, 1), + ] { + assert_eq!( + join(vec![Value::Float64(7.)], vec![], kind.clone(), false).len(), + expected_len, + "{kind:?}" + ); + } + let result = join(vec![Value::Float64(7.)], vec![], JoinKind::Left, false); + assert!(matches!( + result[0].as_slice(), + [Value::Float64(7.), Value::Null] + )); +} + +// Changing the semi-join algorithm must not turn IEEE NaN != NaN into a match. +#[test] +fn keyed_semijoin_obeys_ieee_equality_for_nan_and_zero() { + let left = vec![ + Value::Float64(f64::NAN), + Value::Float64(-0.), + Value::Float64(0.), + Value::Null, + ]; + let right = vec![Value::Float64(f64::NAN), Value::Float64(0.), Value::Null]; + let keyed = join(left, right, JoinKind::Semi, true); + let expected = vec![vec![Value::Float64(-0.)], vec![Value::Float64(0.)]]; + assert_eq!(keys(&keyed), keys(&expected)); +} + +// Group equality intentionally differs from predicate equality: NULL and NaNs group together. +#[test] +fn grouping_canonicalizes_null_nan_and_signed_zero() { + let input = schema(&[("v", DataType::Float64, true)]); + let op = Operator::aggregate( + input.clone(), + vec![0], + vec![("count".into(), Reduction::Count)], + ) + .unwrap(); + let values = vec![ + Value::Null, + Value::Null, + Value::Float64(0.), + Value::Float64(-0.), + Value::Float64(f64::NAN), + Value::Float64(f64::from_bits(0x7ff8000000000001)), + ]; + let result = unary( + input, + values.into_iter().map(|v| vec![vec![v]]).collect(), + op, + ); + assert_eq!(result.len(), 3); + assert!(result.iter().all(|r| matches!(r[1], Value::Int64(2)))); +} + +// Global empty input yields one aggregate row; grouped empty input yields none. +#[test] +fn aggregate_empty_and_all_null_follow_asap_contract() { + let input = schema(&[("v", DataType::Int64, true)]); + for batches in [ + vec![], + vec![vec![]], + vec![vec![vec![Value::Null], vec![Value::Null]]], + ] { + let n = batches.iter().map(Vec::len).sum::(); + let op = Operator::aggregate( + input.clone(), + vec![], + vec![ + ("count".into(), Reduction::Count), + ("min".into(), Reduction::Min(0)), + ("max".into(), Reduction::Max(0)), + ], + ) + .unwrap(); + let result = unary(input.clone(), batches, op); + assert_eq!(result.len(), 1); + assert!(matches!(result[0][0], Value::Int64(v) if v == n as i64)); + assert!(matches!(result[0][1], Value::Null)); + assert!(matches!(result[0][2], Value::Null)); + } + let op = Operator::aggregate( + input.clone(), + vec![0], + vec![("count".into(), Reduction::Count)], + ) + .unwrap(); + assert!(unary(input, vec![], op).is_empty()); +} + +// A precompiled expression with a different input contract must fail during binding. +#[test] +fn projection_rejects_expression_bound_to_another_schema() { + let original = schema(&[("a", DataType::Int64, false), ("b", DataType::Int64, false)]); + let current = schema(&[("a", DataType::Int64, false)]); + let expr = CompiledExpression::compile(&QueryExpr::Column(1), &original).unwrap(); + assert!(Operator::project(current, vec![("b".into(), Expression::planner(expr))]).is_err()); +} + +// A valid Planner MIN/MAX schema must bind even for a non-null input column. +#[test] +fn global_extrema_bind_with_planner_derived_schema() { + use asap_physical_operators::physical_planner::compile_node; + use planner_types::{ + post_asap::*, + pre_asap::{AggIntent, Column, GroupKeys, Reduction as PlanReduction}, + }; + let input = schema(&[("v", DataType::Int64, false)]); + for measure in [ + AggIntent::Min { col: Some(0) }, + AggIntent::Max { col: Some(0) }, + ] { + let planner_input = + planner_types::pre_asap::Schema::new(vec![Column::new("v", DataType::Int64, false)]); + let derived = planner_types::pre_asap::query_expr::aggregate_output_schema( + &planner_input, + &PlanReduction::Reduce(GroupKeys::by(vec![])), + std::slice::from_ref(&measure), + &[], + ) + .unwrap(); + let result = derived.columns[0].clone(); + let output = schema(&[(&result.name, result.dtype, result.nullable)]); + let node = PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::Exact(ExactOperation::Aggregate { + reduction: PlanReduction::Reduce(GroupKeys::by(vec![])), + measures: vec![measure], + output_names: vec![result.name], + having: None, + }), + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*output).clone(), + guarantee: None, + }; + let operator = compile_node(&node, std::slice::from_ref(&input)) + .expect("global extremum should bind to its Planner schema"); + assert!(operator.schema().fields[0].nullable); + let empty = unary(input.clone(), vec![], operator.clone()); + assert!(matches!(empty[0][0], Value::Null)); + let nonempty = unary(input.clone(), vec![vec![vec![Value::Int64(7)]]], operator); + assert!(matches!(nonempty[0][0], Value::Int64(7))); + } +} + +// NaN is a valid numeric input, not a schema error; all six comparisons obey IEEE rules. +#[test] +fn planner_comparisons_handle_nan_without_execution_errors() { + let input = schema(&[ + ("a", DataType::Float64, false), + ("b", DataType::Float64, false), + ]); + for op in [ + CompareOpKind::Eq, + CompareOpKind::Ne, + CompareOpKind::Lt, + CompareOpKind::Le, + CompareOpKind::Gt, + CompareOpKind::Ge, + ] { + let expression = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: op.clone(), + right: Rc::new(QueryExpr::Column(1)), + }; + let compiled = CompiledExpression::compile(&expression, &input).unwrap(); + for row in [ + [Value::Float64(f64::NAN), Value::Float64(1.)], + [Value::Float64(1.), Value::Float64(f64::NAN)], + [Value::Float64(f64::NAN), Value::Float64(f64::NAN)], + ] { + let actual = compiled.evaluate(&row).unwrap(); + assert!(matches!(actual,Value::Bool(value) if value == (op == CompareOpKind::Ne))); + } + } +} + +// A bounded LIMIT branch must unsubscribe so another branch can drain the producer. +#[test] +fn limit_branch_finishes_without_blocking_shared_sibling() { + let input = schema(&[("v", DataType::Int64, false)]); + let mut dag = PhysicalDag::default(); + let batches = (0..100) + .map(|v| Batch::try_new(input.clone(), vec![vec![Value::Int64(v)]]).unwrap()) + .collect(); + dag.add(0, vec![], Operator::source(input.clone(), batches).unwrap()) + .unwrap(); + dag.add( + 1, + vec![0], + Operator::limit(input.clone(), 1, 0, vec![]).unwrap(), + ) + .unwrap(); + dag.add(2, vec![0, 1], Operator::union(input, 2).unwrap()) + .unwrap(); + // Bound polls as well as rows so a backpressure regression cannot hang the suite. + use futures::{task::noop_waker_ref, Stream}; + use std::{ + pin::Pin, + task::{Context, Poll}, + }; + let run = context(); + let mut stream = dag.execute(&[2], run.clone()).unwrap().remove(0); + let mut cx = Context::from_waker(noop_waker_ref()); + let mut count = 0; + for _ in 0..2000 { + match Pin::new(&mut stream).poll_next(&mut cx) { + Poll::Ready(Some(batch)) => count += batch.unwrap().rows().len(), + Poll::Ready(None) => { + assert_eq!(count, 101); + drop(stream); + assert_eq!(run.retained_bytes(), 0); + return; + } + Poll::Pending => {} + } + } + panic!("shared LIMIT/Union failed to make progress"); +} + +// Mixed numeric comparisons must not round Int64 values through f64 before comparing. +#[test] +fn mixed_numeric_comparisons_preserve_large_integer_precision() { + let input = schema(&[ + ("a", DataType::Int64, false), + ("b", DataType::Float64, false), + ]); + let expr = QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Gt, + right: Rc::new(QueryExpr::Column(1)), + }; + let compiled = CompiledExpression::compile(&expr, &input).unwrap(); + for (a, b, expected) in [ + (9_007_199_254_740_993, 9_007_199_254_740_992.0, true), + (i64::MAX, 9_223_372_036_854_775_808.0, false), + (i64::MIN, f64::NEG_INFINITY, true), + ] { + assert!( + matches!(compiled.evaluate(&[Value::Int64(a),Value::Float64(b)]).unwrap(), Value::Bool(v) if v == expected) + ); + } +} + +// Both expression paths must implement all nine combinations of three-valued booleans. +#[test] +fn boolean_truth_tables_agree_between_expression_paths() { + let input = schema(&[("a", DataType::Bool, true), ("b", DataType::Bool, true)]); + for and in [true, false] { + for a in [None, Some(false), Some(true)] { + for b in [None, Some(false), Some(true)] { + let parts = vec![QueryExpr::Column(0), QueryExpr::Column(1)]; + let planner = if and { + QueryExpr::BoolAnd(parts) + } else { + QueryExpr::BoolOr(parts) + }; + let native = if and { + Expression::And( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(1)), + ) + } else { + Expression::Or( + Box::new(Expression::Column(0)), + Box::new(Expression::Column(1)), + ) + }; + let expected = match (a, b, and) { + (Some(false), _, true) | (_, Some(false), true) => Some(false), + (Some(true), _, false) | (_, Some(true), false) => Some(true), + (None, _, _) | (_, None, _) => None, + (Some(a), Some(b), true) => Some(a && b), + (Some(a), Some(b), false) => Some(a || b), + } + .map(Value::Bool) + .unwrap_or(Value::Null); + let row = vec![ + a.map(Value::Bool).unwrap_or(Value::Null), + b.map(Value::Bool).unwrap_or(Value::Null), + ]; + let compiled = CompiledExpression::compile(&planner, &input).unwrap(); + assert_eq!( + compiled.evaluate(&row).unwrap().key().unwrap(), + expected.key().unwrap() + ); + let op = Operator::project(input.clone(), vec![("result".into(), native)]).unwrap(); + let result = unary(input.clone(), vec![vec![row]], op); + assert_eq!(result[0][0].key().unwrap(), expected.key().unwrap()); + } + } + } +} + +// Partial/final execution must agree with one build for an uncompacted KLL population. +#[test] +fn kll_partial_merge_and_multiple_readouts_preserve_population() { + use planner_types::post_asap::{SketchAlgorithm, SketchKind, SketchParams}; + let input = schema(&[("v", DataType::Float64, false)]); + let family = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), + Default::default(), + ); + let mut dag = PhysicalDag::default(); + for (id, range) in [(0, 0..64), (1, 64..128), (2, 0..128)] { + let rows = range.map(|n| vec![Value::Float64(n as f64)]).collect(); + dag.add( + id, + vec![], + Operator::source( + input.clone(), + vec![Batch::try_new(input.clone(), rows).unwrap()], + ) + .unwrap(), + ) + .unwrap(); + dag.add( + id + 3, + vec![id], + Operator::summary_build(input.clone(), family.clone(), 0, None, vec![]).unwrap(), + ) + .unwrap(); + } + let state = Operator::summary_build(input, family, 0, None, vec![]) + .unwrap() + .schema(); + dag.add(6, vec![3, 4], Operator::union(state.clone(), 2).unwrap()) + .unwrap(); + dag.add( + 7, + vec![6], + Operator::summary_merge(state.clone(), 0, vec![]).unwrap(), + ) + .unwrap(); + let mut roots = vec![]; + for (i, q) in [0.0, 0.5, 1.0].into_iter().enumerate() { + for (j, build) in [5, 7].into_iter().enumerate() { + let id = 8 + (i * 2 + j) as u64; + dag.add( + id, + vec![build], + Operator::readout( + state.clone(), + 0, + asap_physical_operators::operators::ReadoutQuery::Sketch( + planner_types::post_asap::SketchQuery::Quantile { q }, + ), + ) + .unwrap(), + ) + .unwrap(); + roots.push(id); + } + } + for _ in 0..2 { + let run = context(); + let outputs = block_on(futures::future::join_all( + dag.execute(&roots, run.clone()) + .unwrap() + .into_iter() + .map(|s| s.collect::>()), + )); + for (pair, expected) in outputs.chunks(2).zip([0., 64., 127.]) { + let value = |batches: &[Result< + asap_physical_operators::runtime::SharedValue, + asap_physical_operators::Error, + >]| { + assert_eq!(batches.len(), 1); + match batches[0].as_ref().unwrap().rows()[0][0] { + Value::Float64(v) => v, + _ => panic!("quantile must be Float64"), + } + }; + assert_eq!(value(&pair[0]), value(&pair[1])); + assert!((value(&pair[0]) - expected).abs() <= 1.); + } + drop(outputs); + assert_eq!(run.retained_bytes(), 0); + } +} + +// Retained zero-column rows still own Vec headers and must consume the output budget. +#[test] +fn zero_column_output_obeys_memory_limit() { + use asap_physical_operators::Error; + let input = schema(&[]); + let batch = Batch::try_new(input.clone(), vec![vec![]; 200]).unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input, vec![batch]).unwrap()) + .unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_bytes: 1024, + ..Limits::default() + }, + ) + .unwrap(); + let mut stream = dag.execute(&[0], run.clone()).unwrap().remove(0); + assert!(matches!( + block_on(stream.next()), + Some(Err(Error::MemoryLimit)) + )); + drop(stream); + assert_eq!(run.retained_bytes(), 0); +} + +// Empty exact-state finalization must preserve ordinary global MIN/MAX null semantics. +#[test] +fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { + use asap_physical_operators::Statistic; + use planner_types::post_asap::{ExactKind, ExactParams}; + let input = schema(&[("v", DataType::Float64, false)]); + for (kind, params, statistic) in [ + (ExactKind::Min, ExactParams::Min, Statistic::Min), + (ExactKind::Max, ExactParams::Max, Statistic::Max), + ] { + let build = Operator::summary_build( + input.clone(), + SummaryFamilyType::ExactAggregate(kind, params), + 0, + None, + vec![], + ) + .unwrap(); + let state = build.schema(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], Operator::source(input.clone(), vec![]).unwrap()) + .unwrap(); + dag.add(1, vec![0], build).unwrap(); + dag.add( + 2, + vec![1], + Operator::readout( + state, + 0, + asap_physical_operators::operators::ReadoutQuery::Exact( + asap_physical_operators::summary_kernels::exact::ExactReadout { + statistic, + lookback_ms: None, + }, + ), + ) + .unwrap(), + ) + .unwrap(); + let rows = collect(&dag, 2); + assert_eq!(rows.len(), 1); + assert!(matches!(rows[0][0], Value::Null)); + } +} diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs new file mode 100644 index 00000000..f9a509c2 --- /dev/null +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -0,0 +1,151 @@ +//! Finite-input contracts are validated before source execution. +use asap_physical_operators::{ + operators::{Operator, SortKey}, + plan::{Boundedness, Emission, PhysicalDag}, + runtime::{Limits, OutputStream, RunContext, Scope}, + sources::{DataSources, RawSource}, + values::{Batch, Schema}, + Error, +}; +use planner_types::{ + post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, + pre_asap::{Column, DataType, QueryExpr, Schema as LogicalSchema, Source}, +}; +use std::sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, +}; +struct DeclaredSource { + schema: Schema, + boundedness: Boundedness, + opens: Arc, +} +impl RawSource for DeclaredSource { + fn schema(&self) -> Schema { + self.schema.clone() + } + fn boundedness(&self) -> Boundedness { + self.boundedness + } + fn scan(&self, _: RunContext) -> Result, Error> { + self.opens.fetch_add(1, Ordering::SeqCst); + Ok(Box::pin(futures::stream::empty())) + } +} +// A blocking parent must reject unknown and unbounded Scan inputs without opening a reader. +#[test] +fn blocking_inputs_require_an_explicit_finite_source() { + let schema = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "v".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: false, + }], + time_index: None, + }); + for boundedness in [ + Boundedness::Unknown, + Boundedness::Unbounded, + Boundedness::Bounded, + ] { + let opens = Arc::new(AtomicUsize::new(0)); + let mut registry = DataSources::default(); + let identity = Source::Table { + table_ref: "t".into(), + }; + registry + .register( + identity.clone(), + Arc::new(DeclaredSource { + schema: schema.clone(), + boundedness, + opens: opens.clone(), + }), + ) + .unwrap(); + let scan = registry + .bind(&QueryExpr::Scan { + source: identity, + schema: LogicalSchema::new(vec![Column::new("v", DataType::Int64, false)]), + predicates: vec![], + }) + .unwrap(); + let mut dag = PhysicalDag::default(); + dag.add(0, vec![], scan).unwrap(); + dag.add( + 1, + vec![0], + Operator::sort( + schema.clone(), + vec![SortKey { + column: 0, + descending: false, + nulls_first: false, + }], + vec![], + ) + .unwrap(), + ) + .unwrap(); + let run = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + if boundedness == Boundedness::Bounded { + let properties = dag.properties(&[1]).unwrap(); + assert_eq!(properties[&1].emission, Emission::AfterInput); + assert_eq!(properties[&1].boundedness, Boundedness::Bounded); + assert!(dag.execute(&[1], run).is_ok()); + } else { + assert!( + matches!(dag.execute(&[1], run), Err(Error::Invalid(message)) if message.contains("requires bounded inputs")) + ); + } + assert_eq!(opens.load(Ordering::SeqCst), 0); + } +} + +// Kernel support must not be mistaken for executable native state/readout support. +#[test] +fn summary_capability_levels_are_distinct() { + use asap_physical_operators::{ + capability::{validate_native_family, validate_sketch_readout, validate_summary_kernel}, + planner::post_asap::SketchQuery, + }; + use planner_types::{ + post_asap::{GroupingStrategy, SketchAlgorithm, SketchKind, SketchParams, SummaryUpdate}, + pre_asap::ColumnRef, + }; + let grouping = GroupingStrategy::default(); + let cms = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::Cms, + SketchParams::Cms { + width: 64, + depth: 4, + }, + ), + grouping.clone(), + ); + let update = SummaryUpdate { + item: Some(planner_types::post_asap::SummaryInputExpr::Column( + ColumnRef::Named("host".into()), + )), + weight: planner_types::post_asap::SummaryInputExpr::Constant(1.0), + weight_domain: Default::default(), + }; + assert!(validate_summary_kernel(&cms, &update, &grouping).is_ok()); + assert!(validate_native_family(&cms).is_err()); + let kll = SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 128 }), + grouping, + ); + assert!(validate_native_family(&kll).is_ok()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Quantile { q: 1.5 }).is_err()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Cardinality).is_err()); + assert!(validate_sketch_readout(&kll, &SketchQuery::Quantile { q: 0.5 }).is_ok()); +} diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs new file mode 100644 index 00000000..c00b388f --- /dev/null +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -0,0 +1,548 @@ +//! Materialized frontiers are compiled by Planner, never rewritten by deployment. +use asap_aware_mapping::{cost_model::DefaultCostModel, search_workload}; +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{ + compile_candidates, select_candidate, CandidateCost, CompiledPhysicalDag, InputContract, + Source, + }, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{post_asap::*, pre_asap::DataType, types::AccuracyTarget, workload::*}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +fn grouped_rate_space() -> asap_aware_mapping::PlanSpace<&'static str> { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query("sum by(job)(rate(m[1m]))".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Exact), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let root = Rc::new( + asap_frontend_promql::lower_promql_workload(&workload, 0) + .unwrap() + .remove(0), + ); + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity(&root) + .unwrap(), + ); + search_workload(vec![("grouped-rate", root)]) +} + +fn grouped_rate() -> PostAsapDag { + let space = grouped_rate_space(); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + compile_post_asap_dag(&selected).unwrap() +} +fn run(plan: &CompiledPhysicalDag, inputs: BTreeMap, scope: Scope) -> Vec { + let sources = inputs + .into_iter() + .map(|(id, batch)| { + let source = Operator::source(batch.schema().clone(), vec![batch]).unwrap(); + (id, Box::new(source) as Source<'_>) + }) + .collect(); + let dag = plan.instantiate(sources).unwrap(); + block_on(async { + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut output = dag.execute(plan.roots(), context).unwrap().remove(0); + let mut batches = vec![]; + while let Some(batch) = output.next().await { + batches.push((*batch.unwrap()).clone()); + } + batches + }) +} + +/// Rate readouts and grouped Sum can run together during bounded precompute; +/// storing per-series rates instead leaves the same Sum in the query DAG. +#[test] +fn grouped_rate_can_be_materialized_before_or_after_grouped_sum() { + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let readout = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator + } + ) + }) + .unwrap(); + let input_schema = Arc::new(state.output_schema.clone()); + let (family, update, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + let range_ms = Some((-58_000, 2_000)); + let mut expected_rate_sum = 0.; + let rows = [[100., 0., 100.], [100., 200., 0.]] + .into_iter() + .enumerate() + .map(|(index, values)| { + let mut accumulator = create_planner_accumulator(family, update, grouping).unwrap(); + for (i, value) in values.into_iter().enumerate() { + accumulator.update_single(value, i as i64 * 1000); + } + let state = accumulator.into_accumulator(); + expected_rate_sum += state + .as_any() + .downcast_ref::() + .unwrap() + .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .unwrap() + .unwrap(); + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(state), + }; + input_schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(2000), + SummaryFamilyType::Plain(DataType::Utf8) => { + Value::Utf8(if field.name == "job" { + "api".into() + } else { + format!("series-{index}").into() + }) + } + _ => panic!("unexpected input field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(input_schema.clone(), rows).unwrap(); + let root = u64::from(dag.root.0); + let state_id = u64::from(state.id.0); + let rate_id = u64::from(readout.id.0); + let frontiers = asap_physical_operators::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 128, + ) + .unwrap(); + assert!(frontiers.contains(&vec![])); + assert!(frontiers.contains(&vec![rate_id])); + assert!(frontiers.contains(&vec![root])); + assert!(!frontiers.contains(&vec![root, rate_id])); + assert!( + asap_physical_operators::physical_planner::enumerate_frontiers( + &dag, + &BTreeMap::from([(state_id, InputContract::bounded(input_schema.clone()))]), + &[root], + 1, + ) + .is_err() + ); + let candidates = compile_candidates( + &dag, + BTreeMap::from([(state_id, InputContract::bounded(input_schema))]), + &[root], + &[vec![], vec![rate_id], vec![root]], + ); + // Scoped cost fixtures select either precompute boundary. No readers are + // opened during candidate construction or selection. + for prefer_grouped in [false, true] { + let inventory = compile_candidates( + &dag, + BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[root], + &[vec![999], vec![rate_id], vec![root]], + ); + assert!(inventory[0].is_err()); + let mut evaluated = 0; + let selected = select_candidate(inventory, |candidate| { + evaluated += 1; + let grouped = candidate.materialized_outputs.contains_key(&root); + Ok(Some(CandidateCost { + workload_scope: "reset-counter-workload".into(), + horizon_seconds: 300., + total_cost: if grouped == prefer_grouped { 1. } else { 100. }, + })) + }) + .unwrap(); + assert_eq!( + selected.candidate.materialized_outputs.contains_key(&root), + prefer_grouped + ); + assert_eq!(selected.cost.total_cost, 1.); + assert_eq!(evaluated, 2, "uncompilable candidates must never be priced"); + let candidate = selected.candidate; + let precompute = candidate.precompute.as_ref().unwrap(); + let stored = run( + precompute, + BTreeMap::from([(state_id, batch.clone())]), + Scope::Ingestion { + window_start_ms: -58_000, + window_end_ms: 2000, + revision: 1, + }, + ); + let output = run( + &candidate.query, + BTreeMap::from([(precompute.roots()[0], stored[0].clone())]), + Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }, + ); + assert!( + matches!(output[0].rows()[0][1], Value::Float64(value) if value == expected_rate_sum) + ); + } + let contracts = BTreeMap::from([( + state_id, + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + for frontier in [vec![rate_id, rate_id], vec![root, rate_id], vec![999]] { + assert!( + asap_physical_operators::physical_planner::compile_candidate( + &dag, + contracts.clone(), + &[root], + &frontier + ) + .is_err() + ); + } + let inventory = compile_candidates( + &dag, + contracts.clone(), + &[root], + &[vec![rate_id], vec![root]], + ); + let selected = select_candidate(inventory, |candidate| { + if candidate.materialized_outputs.contains_key(&root) { + return Ok(None); + } + Ok(Some(CandidateCost { + workload_scope: "same-workload".into(), + horizon_seconds: 300., + total_cost: 100., + })) + }) + .unwrap(); + assert!(selected + .candidate + .materialized_outputs + .contains_key(&rate_id)); + let inventory = compile_candidates(&dag, contracts, &[root], &[vec![rate_id], vec![root]]); + assert!( + select_candidate(inventory, |candidate| Ok(Some(CandidateCost { + workload_scope: "same-workload".into(), + horizon_seconds: if candidate.materialized_outputs.contains_key(&root) { + 60. + } else { + 300. + }, + total_cost: 1., + }))) + .is_err() + ); + let query_scope = Scope::Query { + evaluation_time_ms: 2000, + revision: 1, + }; + let maintenance_scope = Scope::Ingestion { + window_start_ms: -58_000, + window_end_ms: 2000, + revision: 1, + }; + let mut results = vec![]; + for candidate in candidates { + let candidate = candidate.unwrap(); + let inputs = if let Some(precompute) = &candidate.precompute { + let source = Operator::source(batch.schema().clone(), vec![batch.clone()]).unwrap(); + let invalid = precompute + .instantiate(BTreeMap::from([(state_id, Box::new(source) as Source<'_>)])) + .unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + assert!(invalid.execute(precompute.roots(), context).is_err()); + let stored = run( + precompute, + BTreeMap::from([(state_id, batch.clone())]), + maintenance_scope.clone(), + ); + assert_eq!(stored.len(), 1); + let boundary = precompute.roots()[0]; + assert_eq!( + candidate.materialized_outputs[&boundary].schema, + *stored[0].schema() + ); + BTreeMap::from([(boundary, stored[0].clone())]) + } else { + BTreeMap::from([(state_id, batch.clone())]) + }; + let output = run(&candidate.query, inputs, query_scope.clone()); + assert_eq!(output.len(), 1); + assert_eq!(output[0].rows().len(), 1); + assert!(matches!(&output[0].rows()[0][0], Value::Utf8(job) if job.as_ref() == "api")); + assert!( + matches!(output[0].rows()[0][1], Value::Float64(value) if value == expected_rate_sum) + ); + results.push( + output[0].rows()[0] + .iter() + .map(|value| value.key().unwrap()) + .collect::>(), + ); + } + assert_eq!(results[0], results[1]); + assert_eq!(results[1], results[2]); + let mut wrong_order = create_planner_accumulator(family, update, grouping).unwrap(); + for (i, value) in [200., 200., 100.].into_iter().enumerate() { + wrong_order.update_single(value, i as i64 * 1000); + } + let rate_of_sum = wrong_order + .into_accumulator() + .as_any() + .downcast_ref::() + .unwrap() + .readout(asap_physical_operators::Statistic::Rate, range_ms, None) + .unwrap() + .unwrap(); + assert_ne!( + expected_rate_sum, rate_of_sum, + "counter resets prohibit moving Sum before Rate" + ); +} + +/// Enumerated frontiers include both grouped-result and per-series readout +/// persistence; an explicit Rate-state input retains its original semantics. +#[test] +fn bounded_inventory_exposes_grouped_rate_physical_frontiers() { + use asap_physical_operators::physical_planner::enumerate_frontiers; + let dag = grouped_rate(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let inputs = BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]); + let roots = [u64::from(dag.root.0)]; + let frontiers = enumerate_frontiers(&dag, &inputs, &roots, 4096).unwrap(); + let candidates = compile_candidates(&dag, inputs.clone(), &roots, &frontiers) + .into_iter() + .collect::, _>>() + .unwrap(); + assert!(candidates.iter().any(|c| c.precompute.is_none())); + assert!(candidates + .iter() + .any(|c| c.materialized_outputs.contains_key(&roots[0]))); + assert!(candidates + .iter() + .any(|c| !c.materialized_outputs.is_empty() + && !c.materialized_outputs.contains_key(&roots[0]))); + assert!(enumerate_frontiers(&dag, &inputs, &roots, 1).is_err()); +} + +#[test] +fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { + let inventory = grouped_rate_space().enumerate_candidate_dags(4096).unwrap(); + let mut executed = 0; + for forest in inventory.candidates { + let root = &forest[0].1; + let dag = compile_post_asap_dag(root).unwrap(); + let Some(state) = dag.nodes.iter().find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) else { + continue; + }; + let boundary = dag + .nodes + .iter() + .find(|node| { + matches!( + node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Sum, _), + .. + } + ) + }) + .map(|node| u64::from(node.id.0)) + .unwrap_or(u64::from(dag.root.0)); + let physical_candidates = compile_candidates( + &dag, + BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[vec![], vec![boundary]], + ); + let (family, input, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + let schema = Arc::new(state.output_schema.clone()); + let rows = ["a", "b"] + .into_iter() + .map(|instance| { + let mut accumulator = create_planner_accumulator(family, input, grouping).unwrap(); + for (timestamp, value) in [(1_000, 1.), (31_000, 31.), (59_000, 59.)] { + accumulator.update_single(value, timestamp); + } + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(accumulator.into_accumulator()), + }; + schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + SummaryFamilyType::Plain(DataType::Utf8) + if field.name == "$promql_series_identity" => + { + Value::Utf8( + serde_json::to_string(&BTreeMap::from([ + ("job", "api"), + ("instance", instance), + ])) + .unwrap() + .into(), + ) + } + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + _ => panic!("unexpected input field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema, rows).unwrap(); + for physical in physical_candidates { + let physical = physical.unwrap(); + let inputs = if let Some(precompute) = &physical.precompute { + let source_id = precompute.input_contracts().next().unwrap().0; + let stored = run( + precompute, + BTreeMap::from([(source_id, batch.clone())]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 1, + }, + ); + assert_eq!(stored.len(), 1); + BTreeMap::from([(precompute.roots()[0], stored[0].clone())]) + } else { + BTreeMap::from([( + physical.query.input_contracts().next().unwrap().0, + batch.clone(), + )]) + }; + let output = run( + &physical.query, + inputs, + Scope::Query { + evaluation_time_ms: 60_000, + revision: 1, + }, + ); + assert_eq!(output.len(), 1); + assert_eq!(output[0].rows().len(), 1); + assert!(output[0] + .schema() + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + assert!( + output[0].rows()[0] + .iter() + .any(|value| matches!(value, Value::Float64(x) if (*x - 2.).abs() < 1e-12)), + "{:?}", + output[0].rows() + ); + executed += 1; + } + } + assert!( + executed >= 2, + "must execute both stored and query-time grouped Rate candidates: {executed}" + ); +} diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs new file mode 100644 index 00000000..3bb3af6c --- /dev/null +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -0,0 +1,425 @@ +//! Persisted precompute graphs preserve group/window identity and execute state-to-state computation. +use asap_physical_operators::{ + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{precompute, CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, + Statistic, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{ArithmeticOpKind, BinaryOpKind, ColumnRef, DataType, GroupKeys, Reduction}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +#[test] +fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = |dtype| SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype, + nullable: false, + }], + time_index: None, + }; + let state_schema = schema(family.clone()); + let mut value_schema = schema(SummaryFamilyType::Plain(DataType::Float64)); + value_schema.fields.push(SummaryField { + name: "time".into(), + dtype: SummaryFamilyType::Plain(DataType::Timestamp), + nullable: false, + }); + value_schema.time_index = Some(1); + for (weight, expected) in [ + (SummaryInputExpr::Column(ColumnRef::SampleValue), 60.), + (SummaryInputExpr::Constant(1.), 4.), + ] { + let nodes = vec![ + PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: state_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: value_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(2), + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: value_schema.clone(), + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(3), + payload: PostAsapOperatorPayload::SummaryAgg { + family: family.clone(), + input: SummaryUpdate { + weight, + ..SummaryUpdate::column(ColumnRef::SampleValue) + }, + reduction: Reduction::Reduce(GroupKeys::by(vec![])), + grouping: GroupingStrategy::PerSubpopulationInstance, + }, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: state_schema.clone(), + guarantee: None, + }, + ]; + let edges = [ + (0, 1, EdgeRole::Input), + (1, 2, EdgeRole::Left), + (1, 2, EdgeRole::Right), + (2, 3, EdgeRole::Input), + ] + .into_iter() + .map(|(producer, consumer, role)| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role, + intermediate_schema: nodes[producer as usize].output_schema.clone(), + data_state: nodes[producer as usize].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }) + .collect(); + let dag = PostAsapDag { + nodes, + edges, + root: PostAsapNodeId(3), + }; + let mut invalid_grouping = dag.clone(); + let PostAsapOperatorPayload::SummaryAgg { reduction, .. } = + &mut invalid_grouping.nodes[3].payload + else { + unreachable!() + }; + *reduction = Reduction::Reduce(GroupKeys::by(vec![0])); + assert!( + precompute::compile(&invalid_grouping, &[0], &[3]).is_err(), + "numeric values cannot be reinterpreted as population labels" + ); + let program = precompute::compile(&dag, &[0], &[3]).unwrap(); + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + assert_eq!(program.input_contracts().count(), 1); + for revision in [1, 2] { + let rows = [ + ("a", 1000, 2.), + ("a", 2000, 4.), + ("b", 1000, 8.), + ("b", 2000, 16.), + ] + .into_iter() + .map(|(group, time, value)| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + state.update_single(value, time); + vec![ + Value::Map( + vec![(Value::Utf8("instance".into()), Value::Utf8(group.into()))].into(), + ), + Value::Timestamp(time), + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let input = + Batch::try_new(precompute::population_schema(family.clone()), rows).unwrap(); + let sources = BTreeMap::from([( + 0, + Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) + as Source<'_>, + )]); + let graph = program.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision, + }, + Limits::default(), + ) + .unwrap(); + let output = block_on(async { + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let batch = stream.next().await.unwrap().unwrap(); + assert!(stream.next().await.is_none()); + batch + }); + assert_eq!(output.rows().len(), 1); + assert!(matches!(&output.rows()[0][0], Value::Map(labels) if labels.is_empty())); + assert!(matches!(&output.rows()[0][1], Value::Timestamp(2000))); + let Value::Summary { state, .. } = &output.rows()[0][2] else { + panic!("state expected") + }; + assert_eq!( + state + .as_any() + .downcast_ref::() + .unwrap() + .readout(Statistic::Sum, None, None) + .unwrap() + .unwrap(), + expected + ); + } + } +} + +fn logical_schema(family: SummaryFamilyType) -> SummarySchema { + SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: family, + nullable: false, + }], + time_index: None, + } +} +fn state_graph( + family: SummaryFamilyType, + target: Option, + merge: bool, +) -> CompiledPhysicalDag { + let mut nodes = vec![PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: logical_schema(family.clone()), + guarantee: None, + }]; + if merge { + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::SummaryMerge, + ..nodes[0].clone() + }); + } + let read_id = nodes.len() as u32; + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(read_id), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::FinalizeExactAccumulator, + }, + output_state: ExecutionDataState::INGESTION_ROWS, + output_schema: logical_schema(SummaryFamilyType::Plain(DataType::Float64)), + guarantee: None, + }); + if let Some(target) = target { + nodes.push(PostAsapDagNode { + id: PostAsapNodeId(nodes.len() as u32), + payload: PostAsapOperatorPayload::SummaryAgg { + family: target.clone(), + input: SummaryUpdate::column(ColumnRef::SampleValue), + reduction: Reduction::by(vec![]), + grouping: GroupingStrategy::default(), + }, + output_state: ExecutionDataState::INGESTION_SUMMARY, + output_schema: logical_schema(target), + guarantee: None, + }); + } + let edges = (1..nodes.len()) + .map(|i| PostAsapDagEdge { + producer: nodes[i - 1].id, + consumer: nodes[i].id, + role: EdgeRole::Input, + intermediate_schema: nodes[i - 1].output_schema.clone(), + data_state: nodes[i - 1].output_state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }) + .collect(); + let root = nodes.last().unwrap().id; + precompute::compile( + &PostAsapDag { nodes, edges, root }, + &[0], + &[u64::from(root.0)], + ) + .unwrap() +} +fn native_run( + program: &CompiledPhysicalDag, + family: SummaryFamilyType, + states: Vec>, + context: RunContext, +) -> Result>, asap_physical_operators::Error> { + let program = + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + .unwrap(); + let rows = states + .into_iter() + .enumerate() + .map(|(i, state)| { + vec![ + Value::Map(vec![].into()), + Value::Timestamp((i as i64 + 1) * 1000), + Value::Summary { + family: family.clone(), + state, + }, + ] + }) + .collect(); + let input = Batch::try_new(precompute::population_schema(family), rows)?; + let graph = program.instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(input.schema().clone(), vec![input])?) as Source<'_>, + )]))?; + block_on(async { + let mut rows = Vec::new(); + let mut stream = graph.execute(program.roots(), context)?.remove(0); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} +fn ingestion_context(limits: Limits) -> RunContext { + RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 2000, + revision: 1, + }, + limits, + ) + .unwrap() +} +fn sum_state(value: f64) -> Arc { + let mut state = asap_physical_operators::summary_kernels::exact::ExactAccumulator::new( + planner_types::post_asap::SummaryFamilyType::ExactAggregate( + planner_types::post_asap::ExactKind::Sum, + planner_types::post_asap::ExactParams::Sum, + ), + false, + ) + .unwrap(); + state.update(None, value, 0); + Arc::new(state) +} + +// Only an explicit merge may collapse distinct pane updates before finalization. +#[test] +fn explicit_merge_changes_pane_cardinality() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + for (merge, expected) in [(false, vec![2., 7.]), (true, vec![9.])] { + let program = state_graph(family.clone(), None, merge); + let rows = native_run( + &program, + family.clone(), + vec![sum_state(2.), sum_state(7.)], + ingestion_context(Limits::default()), + ) + .unwrap(); + let values = rows + .iter() + .map(|row| match row[2] { + Value::Float64(v) => v, + _ => panic!("numeric readout expected"), + }) + .collect::>(); + assert_eq!(values, expected); + assert!(matches!(rows.last().unwrap()[1], Value::Timestamp(2000))); + } +} + +// Typed updates reject invalid domains before publishing any target state. +#[test] +fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { + let source = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let target = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::DDSketch, + SketchParams::DDSketch { alpha: 0.01 }, + ), + GroupingStrategy::default(), + ); + let program = state_graph(source.clone(), Some(target), false); + assert!(native_run( + &program, + source.clone(), + vec![sum_state(20.)], + ingestion_context(Limits::default()) + ) + .is_ok()); + for value in [-20., 0., f64::MAX, f64::NAN, f64::INFINITY] { + assert!(native_run( + &program, + source.clone(), + vec![sum_state(value)], + ingestion_context(Limits::default()) + ) + .is_err()); + } +} + +// An exact count must not silently lose units when exposed through Float64 rows. +#[test] +fn precompute_count_conversion_checks_precision() { + use asap_physical_operators::summary_kernels::exact::ExactAccumulator; + let family = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); + let program = state_graph(family.clone(), None, false); + for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { + let mut state = + serde_json::to_value(ExactAccumulator::new(family.clone(), false).unwrap()).unwrap(); + state["scalar"]["Count"] = count.into(); + let state: ExactAccumulator = serde_json::from_value(state).unwrap(); + let result = native_run( + &program, + family.clone(), + vec![Arc::new(state)], + ingestion_context(Limits::default()), + ); + assert_eq!(result.is_ok(), valid); + } +} + +// Graph execution retains terminal cancellation and shared workspace limits. +#[test] +fn precompute_graph_enforces_cancellation_and_budget() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let program = state_graph(family.clone(), None, true); + let context = ingestion_context(Limits::default()); + context.cancel(); + let error = native_run(&program, family.clone(), vec![sum_state(1.)], context).unwrap_err(); + assert!(format!("{error:?}").contains("Cancelled")); + let error = native_run( + &program, + family, + vec![sum_state(1.)], + ingestion_context(Limits { + max_bytes: 1, + ..Limits::default() + }), + ) + .unwrap_err(); + assert!(format!("{error:?}").contains("MemoryLimit")); +} diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs new file mode 100644 index 00000000..24844b5e --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -0,0 +1,270 @@ +//! Binary computation must be fully compiled before deployment binds values. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile_node, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Schema, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::{ + BinaryOperator, ExecutionDataState, PostAsapDagNode, PostAsapNodeId, + PostAsapOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, + }, + pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +fn schema() -> Schema { + Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "labels".into(), + dtype: SummaryFamilyType::Plain(DataType::Map { + key: Box::new(DataType::Utf8), + value: Box::new(DataType::Utf8), + value_nullable: false, + }), + nullable: false, + }, + SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }, + ], + time_index: None, + }) +} +fn row(name: &str, job: &str, value: f64) -> Vec { + vec![ + Value::Map( + vec![ + (Value::Utf8("__name__".into()), Value::Utf8(name.into())), + (Value::Utf8("job".into()), Value::Utf8(job.into())), + ] + .into(), + ), + Value::Float64(value), + ] +} +fn program() -> CompiledPhysicalDag { + let schema = schema(); + let node = PostAsapDagNode { + id: PostAsapNodeId(2), + payload: PostAsapOperatorPayload::Binary { + operator: BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + vector_match: None, + checked_relative_division: true, + checked_finite_division: false, + }, + }, + output_state: ExecutionDataState::QUERY_ROWS, + output_schema: (*schema).clone(), + guarantee: None, + }; + let operator = compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(); + let graph = CompiledPhysicalDag::from_operators( + BTreeMap::from([ + (0, InputContract::bounded(schema.clone())), + (1, InputContract::bounded(schema)), + ]), + BTreeMap::from([(2, (vec![0, 1], operator))]), + vec![2], + ) + .unwrap(); + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()).unwrap() +} +fn evaluate( + left: Vec>, + right: Vec>, +) -> Result>, asap_physical_operators::Error> { + let graph = program(); + let sources = [left, right] + .into_iter() + .enumerate() + .map(|(id, rows)| { + let batch = Batch::try_new(schema(), rows).unwrap(); + ( + id as u64, + Box::new(Operator::source(schema(), vec![batch]).unwrap()) as Source<'_>, + ) + }) + .collect(); + let bound = graph.instantiate(sources)?; + let ctx = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits::default(), + )?; + block_on(async { + let mut stream = bound.execute(&[2], ctx)?.remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} + +#[test] +fn compiled_binary_matches_series_and_preserves_checked_division() { + let rows = evaluate( + vec![row("left", "api", 6.), row("left", "unmatched", 8.)], + vec![row("right", "api", 2.)], + ) + .unwrap(); + assert_eq!( + serde_json::to_value(&rows).unwrap(), + serde_json::to_value(vec![vec![ + Value::Map(vec![(Value::Utf8("job".into()), Value::Utf8("api".into()))].into()), + Value::Float64(3.) + ]]) + .unwrap() + ); + assert!(evaluate(vec![row("a", "api", 1.)], vec![row("b", "api", 0.)]).is_err()); +} + +#[test] +fn duplicate_matching_identity_is_rejected() { + assert!(evaluate( + vec![row("a", "api", 1.)], + vec![row("b", "api", 2.), row("c", "api", 3.)] + ) + .is_err()); +} + +// Scalar broadcasting and comparison filtering keep the vector operand's value. +#[test] +fn scalar_broadcast_and_bool_comparison_are_distinct() { + use asap_physical_operators::physical_planner::promql_values; + use planner_types::pre_asap::CompareOpKind; + for return_bool in [false, true] { + let graph = promql_values::compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Lt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + true, + false, + ) + .unwrap(); + let graph = + serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + let scalar = promql_values::scalar_schema(); + let vector = promql_values::vector_schema(); + let sources = BTreeMap::from([ + ( + 0, + Box::new( + Operator::source( + scalar.clone(), + vec![Batch::try_new(scalar, vec![vec![Value::Float64(2.)]]).unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ), + ( + 1, + Box::new( + Operator::source( + vector.clone(), + vec![Batch::try_new( + vector, + vec![row("requests", "api", 4.), row("requests", "worker", 1.)], + ) + .unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ), + ]); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let result = block_on(async { + bound + .execute(&[2], context) + .unwrap() + .remove(0) + .next() + .await + .unwrap() + .unwrap() + }); + assert_eq!(result.rows().len(), if return_bool { 2 } else { 1 }); + assert!( + matches!(result.rows()[0].last(), Some(Value::Float64(v)) if *v == if return_bool { 1. } else { 4. }) + ); + let Value::Map(labels) = &result.rows()[0][0] else { + panic!("missing labels") + }; + assert_eq!( + labels + .iter() + .any(|(key, _)| matches!(key, Value::Utf8(s) if s.as_ref() == "__name__")), + !return_bool + ); + } +} + +// Terminal request controls retain their native error classification. +#[test] +fn binary_obeys_memory_and_cancellation() { + for cancel in [false, true] { + let graph = program(); + let sources = (0..2) + .map(|id| { + ( + id, + Box::new( + Operator::source( + schema(), + vec![Batch::try_new(schema(), vec![row("x", "api", 1.)]).unwrap()], + ) + .unwrap(), + ) as Source<'_>, + ) + }) + .collect(); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 0, + }, + Limits { + max_bytes: if cancel { 10000 } else { 1 }, + ..Limits::default() + }, + ) + .unwrap(); + if cancel { + context.cancel(); + } + let result = block_on(async { + match bound.execute(&[2], context) { + Err(error) => Err(error), + Ok(mut streams) => streams.remove(0).next().await.unwrap().map(|_| ()), + } + }); + assert!(matches!( + (cancel, result), + (true, Err(asap_physical_operators::Error::Cancelled)) + | (false, Err(asap_physical_operators::Error::MemoryLimit)) + )); + } +} diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs new file mode 100644 index 00000000..886b7d00 --- /dev/null +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -0,0 +1,492 @@ +//! Compile, persist and rebind dynamic-label computation without deployment lowering. +use asap_physical_operators::{ + operators::Operator, + physical_planner::{promql_values::*, CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::pre_asap::{AggIntent, ColumnRef, GroupKeys}; +use std::collections::BTreeMap; + +fn row(labels: &[(&str, &str)], value: f64) -> Vec { + vec![ + Value::Map( + labels + .iter() + .map(|(k, v)| (Value::Utf8((*k).into()), Value::Utf8((*v).into()))) + .collect::>() + .into(), + ), + Value::Float64(value), + ] +} +fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { + run_inputs(graph, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() +} +fn run_inputs( + graph: CompiledPhysicalDag, + batches: Vec, +) -> Result>, asap_physical_operators::Error> { + let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + let sources = batches + .into_iter() + .enumerate() + .map(|(id, batch)| { + ( + id as u64, + Box::new(Operator::source(batch.schema().clone(), vec![batch]).unwrap()) + as Source<'_>, + ) + }) + .collect::>(); + let bound = graph.instantiate(sources).unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut stream = bound.execute(graph.roots(), context).unwrap().remove(0); + let mut rows = Vec::new(); + while let Some(batch) = stream.next().await { + rows.extend(batch?.rows().iter().cloned()); + } + Ok(rows) + }) +} +fn equal_rows(actual: Vec>, expected: Vec>) { + let mut actual = actual + .into_iter() + .map(|r| serde_json::to_string(&r).unwrap()) + .collect::>(); + let mut expected = expected + .into_iter() + .map(|r| serde_json::to_string(&r).unwrap()) + .collect::>(); + actual.sort(); + expected.sort(); + assert_eq!(actual, expected); +} + +#[test] +fn grouping_preserves_unenumerated_labels_and_empty_label_semantics() { + let rows = vec![ + row(&[("__name__", "m"), ("instance", "a"), ("job", "api")], 1.), + row(&[("__name__", "m"), ("instance", "b"), ("job", "api")], 2.), + row(&[("instance", "c"), ("job", "")], 4.), + row(&[("instance", "d")], 8.), + ]; + equal_rows( + run( + compile_aggregate( + &AggIntent::Sum { col: None }, + &GroupKeys::without(vec![ColumnRef::Named("instance".into())]), + ) + .unwrap(), + rows.clone(), + ), + vec![row(&[("job", "api")], 3.), row(&[], 12.)], + ); + equal_rows( + run( + compile_aggregate( + &AggIntent::Count { + accuracy: planner_types::types::AccuracyTarget::Exact, + }, + &GroupKeys::by(vec![ColumnRef::Named("job".into())]), + ) + .unwrap(), + rows, + ), + vec![row(&[("job", "api")], 2.), row(&[], 2.)], + ); +} + +#[test] +fn ranking_and_grouped_limit_preserve_full_selected_series() { + let grouping = GroupKeys::by(vec![ColumnRef::Named("job".into())]); + let rows = vec![ + row(&[("instance", "a"), ("job", "api")], 1.), + row(&[("instance", "b"), ("job", "api")], 3.), + row(&[("instance", "c"), ("job", "worker")], 2.), + ]; + let sorted = run(compile_sort(true, &grouping).unwrap(), rows); + let selected = run(compile_limit(1, 0, &grouping).unwrap(), sorted); + equal_rows( + selected, + vec![ + row(&[("instance", "b"), ("job", "api")], 3.), + row(&[("instance", "c"), ("job", "worker")], 2.), + ], + ); + assert!(run(compile_limit(0, 0, &grouping).unwrap(), vec![row(&[], 1.)]).is_empty()); +} + +#[test] +fn empty_vector_aggregation_stays_empty() { + assert!(run( + compile_aggregate(&AggIntent::Sum { col: None }, &GroupKeys::default()).unwrap(), + vec![] + ) + .is_empty()); + let scalar = run(compile_vector_to_scalar().unwrap(), vec![]); + assert!(matches!(scalar[0][0],Value::Float64(v) if v.is_nan())); +} + +// One persisted temporal graph accepts different request windows and detects resets. +#[test] +fn temporal_graph_uses_bound_window_without_recompilation() { + let graph = compile_temporal(&AggIntent::Rate, false).unwrap(); + for start in [0, 60_000] { + let labels = row(&[("__name__", "counter"), ("job", "api")], 0.)[0].clone(); + let samples = [(0, 5.), (30_000, 1.), (60_000, 7.)]; + let rows = samples + .into_iter() + .map(|(time, value)| { + vec![ + labels.clone(), + Value::Timestamp(start + time), + Value::Float64(value), + Value::Timestamp(start), + Value::Timestamp(start + 60_000), + ] + }) + .collect(); + let output = run_inputs( + graph.clone(), + vec![Batch::try_new(matrix_schema(), rows).unwrap()], + ) + .unwrap(); + equal_rows(output, vec![row(&[("job", "api")], 7. / 60.)]); + } + let labels = row(&[("job", "api")], 0.)[0].clone(); + let rows = vec![ + vec![ + labels.clone(), + Value::Timestamp(0), + Value::Float64(1.), + Value::Timestamp(0), + Value::Timestamp(1000), + ], + vec![ + labels, + Value::Timestamp(1000), + Value::Float64(2.), + Value::Timestamp(0), + Value::Timestamp(2000), + ], + ]; + assert!(run_inputs(graph, vec![Batch::try_new(matrix_schema(), rows).unwrap()]).is_err()); +} + +// The quantile is an ordinary scalar input, and bucket labels are native computation. +#[test] +fn histogram_quantile_keeps_each_label_group() { + let graph = compile_histogram_quantile().unwrap(); + let buckets = vec![ + row(&[("job", "api"), ("le", "1")], 2.), + row(&[("job", "api"), ("le", "2")], 4.), + row(&[("job", "api"), ("le", "+Inf")], 4.), + ]; + let output = run_inputs( + graph, + vec![ + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(0.75)]]).unwrap(), + Batch::try_new(vector_schema(), buckets).unwrap(), + ], + ) + .unwrap(); + equal_rows(output, vec![row(&[("job", "api")], 1.5)]); +} + +// Linking an ensemble preserves its shared producer and every selected operator. +#[test] +fn composed_ensemble_shares_a_producer_across_roots() { + use asap_physical_operators::{ + physical_planner::InputContract, + plan::{PhysicalOperator, PlanProperties}, + runtime::{Input, OutputStream}, + values::Schema, + }; + use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{ArithmeticOpKind, BinaryOpKind}, + }; + struct Counted { + source: Operator, + starts: std::rc::Rc>, + } + impl PhysicalOperator for Counted { + fn name(&self) -> &str { + "CountedInput" + } + fn input_schemas(&self) -> Vec { + vec![] + } + fn output_schema(&self) -> Schema { + self.source.schema() + } + fn output_bytes(&self, batch: &Batch) -> usize { + batch.bytes() + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + self.source.properties(inputs) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, asap_physical_operators::Error> { + self.starts.set(self.starts.get() + 1); + self.source.start(inputs, context) + } + } + let aggregate = compile_aggregate( + &AggIntent::Sum { col: None }, + &GroupKeys::by(vec![ColumnRef::Named("job".into())]), + ) + .unwrap(); + let binary = compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + vector_match: None, + checked_finite_division: false, + checked_relative_division: false, + }, + false, + false, + false, + ) + .unwrap(); + let graph = CompiledPhysicalDag::compose( + BTreeMap::from([(0, InputContract::bounded(vector_schema()))]), + BTreeMap::from([ + (10, (vec![0], aggregate)), + (20, (vec![10, 10], binary)), + (30, (vec![10], compile_negate(false).unwrap())), + ]), + vec![20, 30], + ) + .unwrap(); + let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) + .unwrap(); + assert_eq!(graph.input_contracts().count(), 1); + let starts = std::rc::Rc::new(std::cell::Cell::new(0)); + for _ in 0..2 { + let input = Batch::try_new(vector_schema(), vec![row(&[("job", "api")], 3.)]).unwrap(); + let source = Counted { + source: Operator::source(vector_schema(), vec![input]).unwrap(), + starts: starts.clone(), + }; + let bound = graph + .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 0, + }, + Limits { + max_buffered_batches: 1, + ..Limits::default() + }, + ) + .unwrap(); + let results = + block_on(futures::future::join_all( + bound + .execute(graph.roots(), context) + .unwrap() + .into_iter() + .map(|mut stream| async move { + stream.next().await.unwrap().unwrap().rows().to_vec() + }), + )); + equal_rows(results[0].clone(), vec![row(&[("job", "api")], 6.)]); + equal_rows(results[1].clone(), vec![row(&[("job", "api")], -3.)]); + } + assert_eq!(starts.get(), 2); +} + +#[test] +fn compiled_constant_needs_no_deployment_source() { + let graph = compile_scalar(3.).unwrap(); + assert_eq!(graph.input_contracts().count(), 0); + let result = run_inputs(graph, vec![]).unwrap(); + assert!(matches!(result[0][0], Value::Float64(3.))); +} + +// Scalar broadcasting cannot silently create duplicate result identities when +// arithmetic or bool comparisons remove the metric name. +#[test] +fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { + use planner_types::{ + post_asap::BinaryOperator, + pre_asap::{ArithmeticOpKind, BinaryOpKind, CompareOpKind}, + }; + for left_scalar in [false, true] { + for names in [["a", "a"], ["a", "b"]] { + for (kind, return_bool) in [ + (BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), false), + (BinaryOpKind::Compare(CompareOpKind::Gt), true), + ] { + let graph = compile_binary( + &BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + left_scalar, + !left_scalar, + ) + .unwrap(); + let vector = Batch::try_new( + vector_schema(), + vec![ + row(&[("__name__", names[0]), ("job", "api")], 2.), + row(&[("__name__", names[1]), ("job", "api")], 3.), + ], + ) + .unwrap(); + let scalar = + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(); + let result = run_inputs( + graph, + if left_scalar { + vec![scalar, vector] + } else { + vec![vector, scalar] + }, + ); + assert!(result.is_err(), "duplicate output label sets were accepted"); + } + } + } + let graph = compile_binary( + &BinaryOperator { + kind: BinaryOpKind::Compare(CompareOpKind::Gt), + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + false, + false, + true, + ) + .unwrap(); + let rows = vec![ + row(&[("__name__", "a"), ("job", "api")], 2.), + row(&[("__name__", "b"), ("job", "api")], 3.), + ]; + equal_rows( + run_inputs( + graph, + vec![ + Batch::try_new(vector_schema(), rows.clone()).unwrap(), + Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(), + ], + ) + .unwrap(), + rows, + ); +} + +// Persisted exact readout graphs, rather than the storage adapter, merge panes, +// finalize each population, and preserve the requested metric-name semantics. +#[test] +fn exact_state_readouts_recover_and_finalize_panes() { + use asap_physical_operators::factory::create_planner_accumulator; + use planner_types::post_asap::*; + use std::sync::Arc; + for (kind, params, expected) in [ + (ExactKind::Sum, ExactParams::Sum, 12.), + (ExactKind::Count, ExactParams::Count, 4.), + (ExactKind::Min, ExactParams::Min, 1.), + (ExactKind::Max, ExactParams::Max, 5.), + ] { + let family = SummaryFamilyType::ExactAggregate(kind, params); + for preserve in [false, true] { + let rows = [[1., 2.], [4., 5.]] + .into_iter() + .map(|samples| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + for sample in samples { + state.update_single(sample, 0); + } + let labels = row(&[("__name__", "m"), ("instance", "a")], 0.).remove(0); + vec![ + labels, + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let output = run_inputs( + compile_exact_readout(family.clone(), 60_000, preserve).unwrap(), + vec![Batch::try_new(exact_state_schema(family.clone()).unwrap(), rows).unwrap()], + ) + .unwrap(); + let labels = if preserve { + vec![("__name__", "m"), ("instance", "a")] + } else { + vec![("instance", "a")] + }; + equal_rows(output, vec![row(&labels, expected)]); + } + } +} + +#[test] +fn recovered_exact_counter_uses_window_and_omits_insufficient_samples() { + use asap_physical_operators::factory::create_planner_accumulator; + use planner_types::post_asap::*; + use std::sync::Arc; + for (kind, params, expected) in [ + (ExactKind::Rate, ExactParams::Rate, 1.), + (ExactKind::Increase, ExactParams::Increase, 60.), + ] { + let family = SummaryFamilyType::ExactAggregate(kind, params); + let rows = [1, 2] + .into_iter() + .map(|count| { + let mut state = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + state.update_single(100., -50_000); + if count == 2 { + state.update_single(140., -10_000); + } + vec![ + row(&[("instance", if count == 1 { "one" } else { "two" })], 0.).remove(0), + Value::Summary { + family: family.clone(), + state: Arc::from(state.into_accumulator()), + }, + ] + }) + .collect(); + let output = run_inputs( + compile_exact_readout(family.clone(), 60_000, false).unwrap(), + vec![Batch::try_new(exact_state_schema(family).unwrap(), rows).unwrap()], + ) + .unwrap(); + equal_rows(output, vec![row(&[("instance", "two")], expected)]); + } +} diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs new file mode 100644 index 00000000..78e4c8c0 --- /dev/null +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -0,0 +1,386 @@ +//! Scan acceptance uses the public connector contract and Planner physical DAGs. +use asap_physical_operators::dag::{ + planner::bind_with_data_sources, + scan::{DataSources, MemorySource, RawSource}, + values::{Batch, Schema, Value}, + Error, Limits, OutputStream, RunContext, Scope, +}; +use futures::{executor::block_on, stream, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{Column, DataType, GroupKeys, Predicate, QueryExpr, Source}, +}; +use std::{ + collections::BTreeMap, + rc::Rc, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, +}; + +fn fixture() -> (QueryExpr, Schema, Vec) { + let schema = + planner_types::pre_asap::Schema::new(vec![Column::new("value", DataType::Int64, true)]); + let output = Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Int64), + nullable: true, + }], + time_index: None, + }); + let scan = QueryExpr::Scan { + source: Source::Table { + table_ref: "numbers".into(), + }, + predicates: vec![Predicate(Rc::new(QueryExpr::IsNotNull(Rc::new( + QueryExpr::Column(0), + ))))], + schema, + }; + let batches = vec![ + Batch::try_new( + output.clone(), + vec![vec![Value::Int64(3)], vec![Value::Null]], + ) + .unwrap(), + Batch::try_new( + output.clone(), + vec![vec![Value::Int64(9)], vec![Value::Int64(2)]], + ) + .unwrap(), + ]; + (scan, output, batches) +} +fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDag { + let node = |id, payload| PostAsapDagNode { + id: PostAsapNodeId(id), + payload, + output_state: state, + output_schema: (**schema).clone(), + guarantee: None, + }; + let edge = |producer, consumer| PostAsapDagEdge { + producer: PostAsapNodeId(producer), + consumer: PostAsapNodeId(consumer), + role: EdgeRole::Input, + intermediate_schema: (**schema).clone(), + data_state: state, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }; + PostAsapDag { + nodes: vec![ + node(0, PostAsapOperatorPayload::Fallback { expression: scan }), + node( + 1, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Sort { + keys: vec![planner_types::pre_asap::SortKey { + expr: QueryExpr::Column(0), + ascending: false, + nulls_first: false, + }], + partition_by: GroupKeys::by(vec![]), + }, + }, + ), + node( + 2, + PostAsapOperatorPayload::Value { + operation: ValueOperation::Limit { + n: 2, + offset: 0, + partition_by: GroupKeys::by(vec![]), + }, + }, + ), + ], + edges: vec![edge(0, 1), edge(1, 2)], + root: PostAsapNodeId(2), + } +} +fn registry(source: Arc) -> DataSources { + let mut r = DataSources::default(); + r.register( + Source::Table { + table_ref: "numbers".into(), + }, + source, + ) + .unwrap(); + r +} +fn context() -> RunContext { + RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits::default(), + ) + .unwrap() +} + +// Raw-only execution filters nulls and ranks across batches at either phase. +#[test] +fn raw_scan_to_sort_limit_at_both_phases() { + let (scan, schema, batches) = fixture(); + let sources = registry(Arc::new( + MemorySource::new(schema.clone(), batches).unwrap(), + )); + for state in [ + ExecutionDataState::QUERY_ROWS, + ExecutionDataState::INGESTION_ROWS, + ] { + let dag = plan(scan.clone(), &schema, state); + let bound = bind_with_data_sources(&dag, BTreeMap::new(), &[2], &sources).unwrap(); + let ctx = if state == ExecutionDataState::QUERY_ROWS { + context() + } else { + RunContext::new( + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 1, + revision: 1, + }, + Limits::default(), + ) + .unwrap() + }; + let rows = block_on(async { + let mut output = bound.execute(&[2], ctx.clone()).unwrap().remove(0); + let mut rows = vec![]; + while let Some(batch) = output.next().await { + rows.extend(batch.unwrap().rows().iter().cloned()); + } + rows + }); + assert!( + matches!(rows.as_slice(), [a,b] if matches!(a.as_slice(), [Value::Int64(9)]) && matches!(b.as_slice(), [Value::Int64(3)])) + ); + assert_eq!(ctx.retained_bytes(), 0); + } +} +struct CountingSource { + schema: Schema, + opened: Arc, + fail: bool, +} +impl RawSource for CountingSource { + fn boundedness(&self) -> asap_physical_operators::plan::Boundedness { + asap_physical_operators::plan::Boundedness::Bounded + } + fn schema(&self) -> Schema { + self.schema.clone() + } + fn scan(&self, _: RunContext) -> Result, Error> { + self.opened.fetch_add(1, Ordering::SeqCst); + if self.fail { + return Err(Error::Operator("reader failed".into())); + } + Ok(stream::iter(vec![Batch::try_new( + self.schema.clone(), + vec![vec![Value::Int64(7)]], + )]) + .boxed_local()) + } +} +// Binding and cancellation do not perform I/O; fan-out opens one cursor per run. +#[test] +fn lazy_open_shared_producer_and_cancellation() { + let (scan, schema, _) = fixture(); + let opened = Arc::new(AtomicUsize::new(0)); + let sources = registry(Arc::new(CountingSource { + schema: schema.clone(), + opened: opened.clone(), + fail: false, + })); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let bound = bind_with_data_sources(&plan, BTreeMap::new(), &[0, 2], &sources).unwrap(); + let ctx = context(); + let streams = bound.execute(&[0, 2], ctx.clone()).unwrap(); + assert_eq!(opened.load(Ordering::SeqCst), 0); + ctx.cancel(); + drop(streams); + assert_eq!(opened.load(Ordering::SeqCst), 0); + for _ in 0..2 { + block_on(async { + let streams = bound.execute(&[0, 2], context()).unwrap(); + let all = + futures::future::join_all(streams.into_iter().map(|s| s.collect::>())).await; + assert!(all.iter().flatten().all(Result::is_ok)); + }); + } + assert_eq!(opened.load(Ordering::SeqCst), 2); +} +// Unavailable sources and unsupported predicates fail before opening any cursor. +#[test] +fn binding_errors_and_reader_errors_are_not_empty_results() { + let (mut scan, schema, _) = fixture(); + assert!(DataSources::default().bind(&scan).is_err()); + let opened = Arc::new(AtomicUsize::new(0)); + let sources = registry(Arc::new(CountingSource { + schema: schema.clone(), + opened: opened.clone(), + fail: true, + })); + if let QueryExpr::Scan { predicates, .. } = &mut scan { + predicates.push(Predicate(Rc::new(QueryExpr::Column(0)))); + } + assert!(sources.bind(&scan).is_err()); + assert_eq!(opened.load(Ordering::SeqCst), 0); + let (scan, _, _) = fixture(); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let bound = bind_with_data_sources(&plan, BTreeMap::new(), &[2], &sources).unwrap(); + block_on(async { + let mut stream = bound.execute(&[2], context()).unwrap().remove(0); + assert!(stream.next().await.unwrap().is_err()); + }); +} + +// Schema drift cannot enter the DAG, and connector batches obey execution limits. +#[test] +fn schema_drift_and_memory_limits_fail_the_scan() { + struct Drift { + expected: Schema, + batch: Batch, + } + impl RawSource for Drift { + fn schema(&self) -> Schema { + self.expected.clone() + } + fn scan(&self, _: RunContext) -> Result, Error> { + Ok(stream::once(async { Ok(self.batch.clone()) }).boxed_local()) + } + } + let (scan, schema, batches) = fixture(); + let mut different = (*schema).clone(); + different.fields[0].name = "wrong".into(); + let bad = Batch::try_new(Arc::new(different), vec![vec![Value::Int64(1)]]).unwrap(); + let sources = registry(Arc::new(Drift { + expected: schema.clone(), + batch: bad, + })); + let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + block_on(async { + let mut s = graph.execute(&[0], context()).unwrap().remove(0); + assert!(s.next().await.unwrap().is_err()); + }); + let sources = registry(Arc::new(MemorySource::new(schema, batches).unwrap())); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let ctx = RunContext::new( + Scope::Query { + evaluation_time_ms: 0, + revision: 1, + }, + Limits { + max_bytes: 1, + max_buffered_batches: 1, + }, + ) + .unwrap(); + block_on(async { + let mut s = graph.execute(&[0], ctx.clone()).unwrap().remove(0); + assert!(s.next().await.unwrap().is_err()); + }); + assert_eq!(ctx.retained_bytes(), 0); +} + +// An empty table is a valid empty scan; nullable comparisons retain only TRUE. +#[test] +fn empty_sources_and_three_valued_predicates() { + use planner_types::pre_asap::{CompareOpKind, ScalarValue}; + let (mut scan, schema, batches) = fixture(); + if let QueryExpr::Scan { + predicates, source, .. + } = &mut scan + { + *source = Source::TimeSeries { + metric: "samples".into(), + }; + *predicates = vec![Predicate(Rc::new(QueryExpr::Compare { + left: Rc::new(QueryExpr::Column(0)), + op: CompareOpKind::Gt, + right: Rc::new(QueryExpr::Literal(ScalarValue::Int64(2))), + }))]; + } + for (batches, expected) in [(vec![], 0), (batches, 2)] { + let mut sources = DataSources::default(); + sources + .register( + Source::TimeSeries { + metric: "samples".into(), + }, + Arc::new(MemorySource::new(schema.clone(), batches).unwrap()), + ) + .unwrap(); + let plan = plan(scan.clone(), &schema, ExecutionDataState::QUERY_ROWS); + let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + block_on(async { + let mut s = graph.execute(&[0], context()).unwrap().remove(0); + let mut count = 0; + while let Some(b) = s.next().await { + count += b.unwrap().rows().len(); + } + assert_eq!(count, expected); + }); + } +} + +// A physical candidate can be compiled once without readers and rebound per run. +#[test] +fn compile_without_readers_and_rebind_inputs() { + use asap_physical_operators::{ + operators::Operator, + physical_planner::{compile, InputContract, Source}, + }; + let (scan, schema, batches) = fixture(); + let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let compiled = compile( + &dag, + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + &[2], + ) + .unwrap(); + assert_eq!(compiled.input_contracts().count(), 1); + for _ in 0..2 { + let sources = BTreeMap::from([( + 0, + Box::new(Operator::source(schema.clone(), batches.clone()).unwrap()) as Source<'_>, + )]); + let graph = compiled.instantiate(sources).unwrap(); + let mut outputs = graph.execute(compiled.roots(), context()).unwrap(); + let result = block_on(outputs.remove(0).collect::>()); + assert!(result.iter().all(Result::is_ok)); + assert_eq!( + result + .iter() + .map(|b| b.as_ref().unwrap().rows().len()) + .sum::(), + 2 + ); + } + assert!(compiled.instantiate(BTreeMap::new()).is_err()); +} + +// Input boundedness must be proved during compilation, before readers exist. +#[test] +fn compilation_rejects_unknown_boundedness_for_sort() { + use asap_physical_operators::{ + physical_planner::{compile, InputContract}, + plan::{Boundedness, Emission, PlanProperties}, + }; + let (scan, schema, _) = fixture(); + let dag = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); + let input = InputContract { + schema, + properties: PlanProperties { + boundedness: Boundedness::Unknown, + emission: Emission::Unknown, + }, + }; + assert!(compile(&dag, BTreeMap::from([(0, input)]), &[2]).is_err()); +} diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs new file mode 100644 index 00000000..a61a596c --- /dev/null +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -0,0 +1,161 @@ +//! Opaque state travels through a retained physical projection without scalar decoding. +use asap_physical_operators::{ + expressions::Expression, + factory::create_planner_accumulator, + operators::Operator, + physical_planner::{compile, CompiledPhysicalDag, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + values::{Batch, Value}, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{ColumnRef, DataType, ProjectItem, QueryExpr}, +}; +use std::{collections::BTreeMap, sync::Arc}; + +// A Post-ASAP projection may reorder/rename summary columns; recovery must retain +// the family and pass through the same immutable state, without decoding the payload. +#[test] +fn post_asap_summary_projection_survives_recovery() { + let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); + let schema = Arc::new(SummarySchema { + fields: vec![ + SummaryField { + name: "state".into(), + dtype: family.clone(), + nullable: false, + }, + SummaryField { + name: "service".into(), + dtype: SummaryFamilyType::Plain(DataType::Utf8), + nullable: false, + }, + ], + time_index: None, + }); + let output = SummarySchema { + fields: vec![ + schema.fields[1].clone(), + SummaryField { + name: "renamed".into(), + ..schema.fields[0].clone() + }, + ], + time_index: None, + }; + let dag = PostAsapDag { + nodes: vec![ + PostAsapDagNode { + id: PostAsapNodeId(0), + payload: PostAsapOperatorPayload::SummaryMerge, + output_schema: (*schema).clone(), + output_state: ExecutionDataState::INGESTION_SUMMARY, + guarantee: None, + }, + PostAsapDagNode { + id: PostAsapNodeId(1), + payload: PostAsapOperatorPayload::Value { + operation: ValueOperation::Project { + cols: vec![1, 0] + .into_iter() + .map(|index| ProjectItem { + alias: None, + expr: QueryExpr::Column(index), + }) + .collect(), + qualifier: None, + }, + }, + output_schema: output.clone(), + output_state: ExecutionDataState::INGESTION_SUMMARY, + guarantee: None, + }, + ], + edges: vec![PostAsapDagEdge { + producer: PostAsapNodeId(0), + consumer: PostAsapNodeId(1), + role: EdgeRole::Input, + intermediate_schema: (*schema).clone(), + data_state: ExecutionDataState::INGESTION_SUMMARY, + grouping: GroupingEdgeCompatibility::NotApplicable, + window: WindowEdgeCompatibility::NotApplicable, + }], + root: PostAsapNodeId(1), + }; + let program = compile( + &dag, + BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), + &[1], + ) + .unwrap(); + let encoded = serde_json::to_vec(&program).unwrap(); + let program = serde_json::from_slice::(&encoded).unwrap(); + let mut forged: serde_json::Value = serde_json::from_slice(&encoded).unwrap(); + forged["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][1]["dtype"] = + serde_json::json!({"Plain": "float64"}); + assert!( + serde_json::from_slice::(&serde_json::to_vec(&forged).unwrap()) + .is_err() + ); + assert!(Operator::project( + schema.clone(), + vec![("invalid".into(), Expression::Column(2))] + ) + .is_err()); + assert!(Operator::project( + schema.clone(), + vec![( + "invalid".into(), + Expression::Negate(Box::new(Expression::Column(0))) + )] + ) + .is_err()); + assert_eq!(*program.output_contract(1).unwrap().schema, output); + let mut accumulator = create_planner_accumulator( + &family, + &SummaryUpdate::column(ColumnRef::SampleValue), + &GroupingStrategy::PerSubpopulationInstance, + ) + .unwrap(); + accumulator.update_single(7., 1); + let state = Arc::from(accumulator.into_accumulator()); + let batch = Batch::try_new( + schema.clone(), + vec![vec![ + Value::Summary { + family, + state: Arc::clone(&state), + }, + Value::Utf8("api".into()), + ]], + ) + .unwrap(); + let graph = program + .instantiate(BTreeMap::from([( + 0, + Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, + )])) + .unwrap(); + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 1, + revision: 1, + }, + Limits::default(), + ) + .unwrap(); + block_on(async { + let mut output = graph.execute(&[1], context).unwrap().remove(0); + let batch = output.next().await.unwrap().unwrap(); + assert!(matches!(&batch.rows()[0][0], Value::Utf8(label) if label.as_ref() == "api")); + let Value::Summary { + state: projected, .. + } = &batch.rows()[0][1] + else { + panic!("missing summary") + }; + assert!(Arc::ptr_eq(&state, projected)); + assert!(output.next().await.is_none()); + }); +} diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs new file mode 100644 index 00000000..664ae799 --- /dev/null +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -0,0 +1,1018 @@ +//! Planner output binds directly to the shared runtime at a declared rate-value frontier. +use asap_aware_mapping::{ + accuracy::{ + AccuracyEvidenceProvider, DefaultAccuracyModel, EqualSplitAllocator, PropagationStats, + }, + cost_model::DefaultCostModel, + Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG, +}; +use asap_physical_operators::dag::{ + operators::Operator, + planner::{compile, InputContract, Source}, + values::{Batch, Value}, + Limits, RunContext, Scope, +}; +use futures::{executor::block_on, StreamExt}; +use planner_types::{ + post_asap::*, + pre_asap::{DataType, QueryExpr}, + types::AccuracyTarget, +}; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; +struct Evidence; +impl AccuracyEvidenceProvider for Evidence { + fn topk_max_distinct_items(&self, _: &QueryExpr) -> Option { + Some(1000) + } + fn propagation_stats( + &self, + op: &CompositionOperator, + _: &SummaryFamilyType, + _: Option<&SketchQuery>, + ) -> PropagationStats { + if matches!(op, CompositionOperator::TopKSelection) { + PropagationStats { + topk_selected_lower_bound: Some(101.), + topk_excluded_upper_bound: Some(100.), + topk_interval_failure_probability: Some(0.001), + ..Default::default() + } + } else { + Default::default() + } + } +} +// The evidence here exercises binding; it is not inferred from the sample data. +#[test] +fn planner_weighted_topk_binds_at_either_deployment_phase() { + assert_weighted_binding(&Evidence, SketchAlgorithm::CmsWithHeap); + assert_weighted_binding(&Evidence, SketchAlgorithm::CountSketchWithHeap); +} + +// Binding validates representation, while deployment owns evidence acceptance. +#[test] +fn physical_binding_does_not_impose_an_accuracy_acceptance_policy() { + assert_weighted_binding( + &asap_aware_mapping::accuracy::NoAccuracyEvidence, + SketchAlgorithm::CmsWithHeap, + ); + assert_weighted_binding( + &asap_aware_mapping::accuracy::NoAccuracyEvidence, + SketchAlgorithm::CountSketchWithHeap, + ); +} + +fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: SketchAlgorithm) { + let root = Rc::new( + lower_promql( + "topk by(job)(2, sum by(service, job)(rate(m[1m])))", + AccuracyTarget::Epsilon(0.1), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + evidence, + ); + let plan = strategy + .replacements(&TargetSubDAG::new(&root)) + .into_iter() + .find_map(|candidate| match candidate.replacement { + Replacement::Summary(node) + if candidate.rationale.contains(&format!("{algorithm:?}")) => + { + Some(node) + } + _ => None, + }) + .unwrap(); + let dag = compile_post_asap_dag(&plan).unwrap(); + let build=dag.nodes.iter().find(|node|matches!(&node.payload,PostAsapOperatorPayload::SummaryAgg{family:SummaryFamilyType::Sketch(kind,_),..}if kind.algorithm()==&algorithm)).unwrap(); + let rate_id = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let rates = Arc::new( + dag.nodes + .iter() + .find(|node| node.id == rate_id) + .unwrap() + .output_schema + .clone(), + ); + let rows = [ + ("auth", "api", 0.125), + ("auth", "api", 0.25), + ("checkout", "api", 0.3125), + ("search", "api", 0.0625), + ("ingest", "batch", 100.), + ("export", "batch", 80.), + ("cleanup", "batch", 20.), + ] + .into_iter() + .map(|(service, job, value)| { + rates + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8(job.into()), + "value" => Value::Float64(value), + _ => match field.dtype { + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(60_000), + _ => panic!("unexpected rate column {field:?}"), + }, + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(rates.clone(), rows).unwrap(); + for (phase, scope) in [ + ( + ExecutionTiming::IngestionTime, + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 60_000, + revision: 1, + }, + ), + ( + ExecutionTiming::QueryTime, + Scope::Query { + evaluation_time_ms: 60_000, + revision: 1, + }, + ), + ] { + let placed = dag + .with_execution_phases(&dag.nodes.iter().map(|node| (node.id, phase)).collect()) + .unwrap(); + let source = Box::new(Operator::source(rates.clone(), vec![batch.clone()]).unwrap()) + as Source<'static>; + let compiled = compile( + &placed, + BTreeMap::from([(rate_id.0 as u64, InputContract::bounded(rates.clone()))]), + &[dag.root.0 as u64], + ) + .unwrap(); + let graph = compiled + .instantiate(BTreeMap::from([(rate_id.0 as u64, source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let output = block_on(async { + let mut output = Vec::new(); + let mut stream = graph + .execute(&[dag.root.0 as u64], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + output.extend(batch.unwrap().rows().iter().cloned()); + } + output + }); + assert_eq!(output.len(), 4); + let mut scores = output + .iter() + .map(|row| { + row.iter() + .find_map(|v| { + if let Value::Float64(v) = v { + Some(*v) + } else { + None + } + }) + .unwrap() + }) + .collect::>(); + scores.sort_by(f64::total_cmp); + assert_eq!(scores, vec![0.3125, 0.375, 80., 100.]); + } +} + +use asap_frontend_promql::lower_promql_workload; +use planner_types::workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence as WorkloadEvidence, + PlanningWorkload, Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, + TimeSelection, +}; +pub fn lower_promql( + query: &str, + accuracy: AccuracyTarget, +) -> Result { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: WorkloadEvidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let mut lowered = lower_promql_workload(&workload, 0)?; + Ok(lowered.remove(0)) +} + +// The old untyped heap updater must not silently round a Planner rate update. +#[test] +fn rate_updates_cannot_enter_integer_heap_factory() { + let family = SummaryFamilyType::Sketch( + SketchKind::new( + SketchAlgorithm::CmsWithHeap, + SketchParams::CmsWithHeap { + width: 272, + depth: 5, + heap_size: 100, + }, + ), + Default::default(), + ); + let input = SummaryUpdate { + item: Some(SummaryInputExpr::Column( + planner_types::pre_asap::ColumnRef::Named("service".into()), + )), + weight: SummaryInputExpr::Column(planner_types::pre_asap::ColumnRef::SampleValue), + weight_domain: WeightDomain::NonNegative { + proof: NonNegativeWeightProof::ResetAwareCounterDerivative, + }, + }; + assert!( + asap_physical_operators::factory::create_planner_accumulator( + &family, + &input, + &Default::default() + ) + .is_err() + ); +} + +/// A catalog-resolved per-series rate can feed a heap sketch directly, without +/// requiring an otherwise unnecessary grouped Sum between Rate and TopK. +#[test] +fn direct_rate_topk_exposes_heap_candidates_with_complete_series_identity() { + check_direct_rate_topk(false); +} + +// Unreferenced labels still distinguish series throughout Rate and heap readout. +#[test] +fn direct_rate_topk_preserves_dynamic_unreferenced_labels() { + check_direct_rate_topk(true); +} + +fn check_direct_rate_topk(dynamic: bool) { + use asap_physical_operators::physical_planner::promql_rows::{ + decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, + }; + let mut logical = + lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(); + fn resolve_catalog(node: &mut QueryExpr) { + match node { + QueryExpr::Aggregate { child, .. } | QueryExpr::TimeRange { child, .. } => { + resolve_catalog(Rc::make_mut(child)) + } + QueryExpr::Scan { schema, .. } => { + schema.closed = true; + schema + .columns + .push(planner_types::pre_asap::schema::Column::new( + "service", + DataType::Utf8, + false, + )); + } + _ => panic!("unexpected input shape: {node:?}"), + } + } + if dynamic { + logical = with_series_identity(&logical).unwrap(); + } else { + resolve_catalog(&mut logical); + } + let root = Rc::new(logical); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy.replacements(&TargetSubDAG::new(&root)); + for algorithm in [ + SketchAlgorithm::CmsWithHeap, + SketchAlgorithm::CountSketchWithHeap, + ] { + let candidate = candidates + .iter() + .find_map(|candidate| match &candidate.replacement { + Replacement::Summary(node) + if candidate.rationale.contains(&format!("{algorithm:?}")) => + { + Some(node) + } + _ => None, + }) + .unwrap_or_else(|| panic!("missing {algorithm:?} over direct Rate")); + if dynamic { + let (source, ranked) = + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking( + candidate, + ) + .unwrap(); + assert!(matches!( + source.expr, + SummaryExpr::ValueOperation { + operation: ValueOperation::FinalizeExactAccumulator, + .. + } + )); + assert_eq!(ranked.input_contracts().count(), 1); + let encoded = String::from_utf8(serde_json::to_vec(&ranked).unwrap()).unwrap(); + assert!(encoded.contains("KeyedSummaryBuild")); + assert!(encoded.contains("KeyedReadout")); + assert!( + !encoded.contains("\"Rate\""), + "Rate must be supplied by its exact stored-state readout" + ); + } + let dag = compile_post_asap_dag(candidate).unwrap(); + assert!(dag.nodes.iter().any(|node| matches!(&node.payload, + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm))); + let build = dag.nodes.iter().find(|node| matches!(&node.payload, + PostAsapOperatorPayload::SummaryAgg { family: SummaryFamilyType::Sketch(kind, _), .. } if kind.algorithm() == &algorithm)).unwrap(); + let input_id = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let schema = Arc::new( + dag.nodes + .iter() + .find(|node| node.id == input_id) + .unwrap() + .output_schema + .clone(), + ); + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) + }) + .unwrap_or_else(|| panic!("no raw counter source: {dag:?}")); + let raw_schema = Arc::new(raw.output_schema.clone()); + let raw_compiled = compile( + &dag, + BTreeMap::from([( + u64::from(raw.id.0), + InputContract::bounded(raw_schema.clone()), + )]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + let bytes = serde_json::to_vec(&raw_compiled).unwrap(); + let raw_compiled = serde_json::from_slice::< + asap_physical_operators::physical_planner::CompiledPhysicalDag, + >(&bytes) + .unwrap(); + // Each evaluation receives a complete raw window. A reset, a stopped + // series and an expired leader must not retain last run's heap weights. + for (end, series, expected) in [ + ( + 60_000, + vec![ + ("auth", vec![10., 30., 50.]), + ("checkout", vec![10., 50., 90.]), + ("search", vec![10., 70., 130.]), + ], + vec![11. / 6., 8. / 3.], + ), + ( + 120_000, + vec![ + ("auth", vec![100., 10., 50.]), + ("checkout", vec![100., 100., 100.]), + ], + vec![0., 1.25], + ), + ] { + let mut raw_rows = Vec::new(); + for (service, samples) in series { + for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + if dynamic { + raw_rows.push( + series_row( + &raw_schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("service".into(), service.into()), + ("unreferenced".into(), format!("{service}-extra")), + ]), + end - 60_000 + offset, + value, + ) + .unwrap(), + ); + continue; + } + raw_rows.push( + raw_schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8("api".into()), + "value" => Value::Float64(value), + "ts" => Value::Timestamp(end - 60_000 + offset), + _ => panic!("unexpected raw field"), + }) + .collect(), + ); + } + } + let raw_batch = Batch::try_new(raw_schema.clone(), raw_rows).unwrap(); + for scope in [ + Scope::Ingestion { + window_start_ms: end - 60_000, + window_end_ms: end, + revision: 1, + }, + Scope::Query { + evaluation_time_ms: end, + revision: 1, + }, + ] { + let source = Box::new( + Operator::source(raw_schema.clone(), vec![raw_batch.clone()]).unwrap(), + ) as Source<'static>; + let graph = raw_compiled + .instantiate(BTreeMap::from([(u64::from(raw.id.0), source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut raw_scores = block_on(async { + let mut scores = Vec::new(); + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + for row in batch.rows() { + if dynamic { + let column = batch + .schema() + .fields + .iter() + .position(|field| field.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let Value::Utf8(encoded) = &row[column] else { + panic!("identity lost"); + }; + let labels = decode_series_identity(encoded).unwrap(); + assert_eq!(labels["job"], "api"); + assert_eq!( + labels["unreferenced"], + format!("{}-extra", labels["service"]) + ); + } + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(time) if *time == end) + )); + scores.extend(row.iter().filter_map(|value| match value { + Value::Float64(value) => Some(*value), + _ => None, + })); + } + } + scores + }); + raw_scores.sort_by(f64::total_cmp); + assert_eq!(raw_scores.len(), expected.len()); + for (actual, expected) in raw_scores.iter().zip(&expected) { + assert!( + (actual - expected).abs() < 1e-12, + "raw counter semantics must precede heap ranking: {raw_scores:?}" + ); + } + } + } + let compiled = compile( + &dag, + BTreeMap::from([( + u64::from(input_id.0), + InputContract::bounded(schema.clone()), + )]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + for (time, values, expected) in [ + ( + 60_000, + vec![("auth", 3.), ("checkout", 2.), ("search", 1.)], + vec![2., 3.], + ), + ( + 61_000, + vec![("auth", 0.), ("checkout", 2.), ("search", 4.)], + vec![2., 4.], + ), + (62_000, vec![("auth", 0.), ("checkout", 2.)], vec![0., 2.]), + ] { + let rows = values + .into_iter() + .map(|(service, value)| { + if dynamic { + return series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("service".into(), service.into()), + ]), + time, + value, + ) + .unwrap(); + } + schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "job" => Value::Utf8("api".into()), + "value" => Value::Float64(value), + "ts" => Value::Timestamp(time), + _ => panic!("unexpected rate field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + for scope in [ + Scope::Query { + evaluation_time_ms: time, + revision: 1, + }, + Scope::Ingestion { + window_start_ms: time - 60_000, + window_end_ms: time, + revision: 1, + }, + ] { + let source = + Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) + as Source<'static>; + let graph = compiled + .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) + .unwrap(); + let context = RunContext::new(scope, Limits::default()).unwrap(); + let mut scores = block_on(async { + let mut scores = vec![]; + let mut stream = graph + .execute(&[u64::from(dag.root.0)], context) + .unwrap() + .remove(0); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + for row in batch.rows() { + assert!(row.iter().any( + |value| matches!(value, Value::Timestamp(actual) if *actual == time) + )); + scores.push( + row.iter() + .find_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .unwrap(), + ); + } + } + scores + }); + scores.sort_by(f64::total_cmp); + assert_eq!( + scores, expected, + "heap snapshots must not accumulate across evaluations" + ); + } + } + } +} + +// Spatial ranking consumes one eligible instant vector. Signed values require +// CountSketch; a raw metric does not establish the non-negative CMS contract. +#[test] +fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { + use asap_physical_operators::physical_planner::promql_rows::{ + decode_series_identity, series_row, with_series_identity, SERIES_IDENTITY_COLUMN, + }; + let logical = lower_promql("topk by(job)(1, m)", AccuracyTarget::Epsilon(0.1)).unwrap(); + let root = Rc::new(with_series_identity(&logical).unwrap()); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy + .current_series_topk_candidates(&root, &AccuracyTarget::Epsilon(0.1)) + .candidates; + assert!(!candidates + .iter() + .any(|c| c.rationale.contains("CmsWithHeap"))); + let selected = candidates + .iter() + .find_map(|candidate| match &candidate.replacement { + Replacement::Summary(node) if candidate.rationale.contains("CountSketchWithHeap") => { + Some(node) + } + _ => None, + }) + .expect("signed spatial TopK must expose CountSketch with heap"); + let dag = compile_post_asap_dag(selected).unwrap(); + let raw = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::TimeRange { .. } + } + ) + }) + .unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let program = compile( + &dag, + BTreeMap::from([(u64::from(raw.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + let snapshot_program = + asap_physical_operators::physical_planner::promql_rows::compile_current_series_readout( + selected, + ) + .unwrap(); + let encoded: serde_json::Value = + serde_json::from_slice(&serde_json::to_vec(&snapshot_program).unwrap()).unwrap(); + assert!(!encoded.to_string().contains("CurrentSeries")); + assert!(encoded.to_string().contains("KeyedSummaryBuild")); + assert!(encoded.to_string().contains("KeyedReadout")); + for (values, expected, score) in [ + ([100., 20.], "a", 100.), + ([1., 20.], "b", 20.), + ([-10., -2.], "b", -2.), + ] { + let rows = ["a", "b"] + .into_iter() + .zip(values) + .map(|(instance, value)| { + series_row( + &schema, + &BTreeMap::from([ + ("job".into(), "api".into()), + ("unreferenced".into(), instance.into()), + ]), + 60_000, + value, + ) + .unwrap() + }) + .collect(); + let batch = Batch::try_new(schema.clone(), rows).unwrap(); + let graph = program + .instantiate(BTreeMap::from([( + u64::from(raw.id.0), + Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, + )])) + .unwrap(); + block_on(async { + let context = RunContext::new( + Scope::Query { + evaluation_time_ms: 60_000, + revision: 0, + }, + Limits::default(), + ) + .unwrap(); + let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut result = Vec::new(); + while let Some(batch) = stream.next().await { + let batch = batch.unwrap(); + let identity = batch + .schema() + .fields + .iter() + .position(|f| f.name == SERIES_IDENTITY_COLUMN) + .unwrap(); + let value = batch + .schema() + .fields + .iter() + .position(|f| f.name == "value") + .unwrap(); + for row in batch.rows() { + let Value::Utf8(labels) = &row[identity] else { + panic!() + }; + let Value::Float64(v) = row[value] else { + panic!() + }; + result.push(( + decode_series_identity(labels).unwrap()["unreferenced"].clone(), + v, + )); + } + } + assert_eq!(result, vec![(expected.into(), score)]); + }); + } +} + +// Placement changes execution ownership only. Every fixed-window candidate +// contains Rate finalization before a fresh heap, with query readout downstream. +#[test] +fn planner_exposes_fixed_window_rate_heap_precompute_candidates() { + use asap_physical_operators::physical_planner::{ + compile_candidate, promql_rows::with_series_identity, + }; + let root = Rc::new( + with_series_identity( + &lower_promql("topk by(job)(2, rate(m[1m]))", AccuracyTarget::Epsilon(0.1)).unwrap(), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let candidates = strategy.fixed_window_rate_candidates(&root).candidates; + assert_eq!(candidates.len(), 2); + for candidate in candidates { + let Replacement::Summary(root) = candidate.replacement else { + panic!() + }; + let dag = compile_post_asap_dag(&root).unwrap(); + let state = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::ExactAggregate(ExactKind::Rate, _), + .. + } + ) + }) + .unwrap(); + let heap = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::SummaryAgg { + family: SummaryFamilyType::Sketch(..), + .. + } + ) + }) + .unwrap(); + assert_eq!(heap.output_state.timing, ExecutionTiming::IngestionTime); + let physical = compile_candidate( + &dag, + BTreeMap::from([( + u64::from(state.id.0), + InputContract::bounded(Arc::new(state.output_schema.clone())), + )]), + &[u64::from(dag.root.0)], + &[u64::from(heap.id.0)], + ) + .unwrap(); + let exported = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&root).unwrap(); + assert_eq!( + serde_json::to_vec(&exported).unwrap(), + serde_json::to_vec(&physical).unwrap() + ); + assert!( + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(&root) + .is_err(), + "query binding must not move the selected precompute frontier" + ); + // Execute the selected split across a state serialization boundary. + // Each run builds fresh weights from that window's counters. + let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag, + input: Batch, + scope: Scope| { + let id = plan.input_contracts().next().unwrap().0; + let source = Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) + as Source<'static>; + let graph = plan.instantiate(BTreeMap::from([(id, source)])).unwrap(); + block_on(async { + let mut stream = graph + .execute( + plan.roots(), + RunContext::new(scope, Limits::default()).unwrap(), + ) + .unwrap() + .remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.unwrap()).clone()); + } + assert_eq!(batches.len(), 1); + batches.remove(0) + }) + }; + let (family, input, grouping) = match &state.payload { + PostAsapOperatorPayload::SummaryAgg { + family, + input, + grouping, + .. + } => (family, input, grouping), + _ => unreachable!(), + }; + for (end, samples, leader) in [ + ( + 60_000, + [[0., 100., 200.], [0., 10., 20.], [0., 1., 2.]], + "a", + ), + ( + 120_000, + [[200., 200., 200.], [100., 0., 300.], [2., 3., 4.]], + "b", + ), + ] { + let schema = Arc::new(state.output_schema.clone()); + let rows = samples + .into_iter() + .zip(["a", "b", "c"]) + .map(|(samples, label)| { + let mut accumulator = + asap_physical_operators::factory::create_planner_accumulator( + family, input, grouping, + ) + .unwrap(); + for (offset, value) in [10_000, 30_000, 50_000].into_iter().zip(samples) { + accumulator.update_single(value, end - 60_000 + offset); + } + let summary = Value::Summary { + family: family.clone(), + state: Arc::from(accumulator.into_accumulator()), + }; + schema + .fields + .iter() + .map(|field| match &field.dtype { + SummaryFamilyType::ExactAggregate(..) => summary.clone(), + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(end), + SummaryFamilyType::Plain(DataType::Utf8) + if field.name == "$promql_series_identity" => + { + Value::Utf8( + serde_json::to_string(&BTreeMap::from([ + ("job", "api"), + ("instance", label), + ])) + .unwrap() + .into(), + ) + } + SummaryFamilyType::Plain(DataType::Utf8) => Value::Utf8("api".into()), + _ => panic!("unexpected state field {field:?}"), + }) + .collect() + }) + .collect(); + let batch = Batch::try_new(schema, rows).unwrap(); + let precompute = physical.precompute.as_ref().unwrap(); + let heap = execute( + precompute, + batch, + Scope::Ingestion { + window_start_ms: end - 60_000, + window_end_ms: end, + revision: 1, + }, + ); + let result = execute( + &physical.query, + heap, + Scope::Query { + evaluation_time_ms: end, + revision: 1, + }, + ); + let identity = result + .schema() + .fields + .iter() + .position(|f| f.name == "$promql_series_identity") + .unwrap(); + let Value::Utf8(encoded) = &result.rows()[0][identity] else { + panic!() + }; + let labels: BTreeMap = serde_json::from_str(encoded).unwrap(); + assert_eq!(labels["instance"], leader); + assert_eq!(result.rows().len(), 2); + } + let precompute = + String::from_utf8(serde_json::to_vec(&physical.precompute.unwrap()).unwrap()).unwrap(); + assert!(precompute.contains("KeyedSummaryBuild")); + assert!(precompute.contains("Rate")); + assert!( + !String::from_utf8(serde_json::to_vec(&physical.query).unwrap()) + .unwrap() + .contains("KeyedSummaryBuild") + ); + } +} + +// Grouped Rate has a legal stored Sum candidate as well as query-time reduction. +#[test] +fn grouped_rate_exposes_precomputed_sum_with_query_readout() { + let root = Rc::new( + asap_physical_operators::physical_planner::promql_rows::with_series_identity( + &lower_promql("sum by(job)(rate(m[1m]))", AccuracyTarget::Exact).unwrap(), + ) + .unwrap(), + ); + let strategy = SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence( + &DefaultCostModel, + &DefaultAccuracyModel, + &EqualSplitAllocator, + &Evidence, + ); + let direct = strategy.query_time_rate_aggregation_candidates(&root); + assert!( + direct.candidates.iter().any(|candidate| { + let Replacement::Summary(root) = &candidate.replacement else { + return false; + }; + let Ok((_, program)) = + asap_physical_operators::physical_planner::promql_rows::compile_rate_ranking(root) + else { + return false; + }; + let output = program.output_contract(program.roots()[0]).unwrap(); + output + .schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_))) + }), + "query-time grouped Rate must finalize Sum inside the physical graph" + ); + let candidates = strategy.fixed_window_rate_candidates(&root).candidates; + assert!( + !candidates.is_empty(), + "Planner must expose Rate -> grouped Sum at ingestion" + ); + for candidate in candidates { + let Replacement::Summary(root) = candidate.replacement else { + panic!() + }; + let physical = asap_physical_operators::physical_planner::promql_rows::compile_fixed_window_rate_aggregation(&root).unwrap(); + let precompute = + String::from_utf8(serde_json::to_vec(&physical.precompute.unwrap()).unwrap()).unwrap(); + assert!( + precompute.contains("SummaryBuild") + && precompute.contains("Rate") + && precompute.contains("Sum") + ); + let query = String::from_utf8(serde_json::to_vec(&physical.query).unwrap()).unwrap(); + assert!(query.contains("Readout") && !query.contains("SummaryBuild")); + } +} diff --git a/crates/integration-tests/Cargo.toml b/crates/integration-tests/Cargo.toml index 7b0c0d3e..5a5de990 100644 --- a/crates/integration-tests/Cargo.toml +++ b/crates/integration-tests/Cargo.toml @@ -13,3 +13,6 @@ asap-aware-mapping = { path = "../asap-aware-mapping" } asap_sketchlib = { workspace = true } serde_json = "1" tokio = { version = "1", features = ["rt", "macros", "rt-multi-thread"] } + +asap-physical-operators = { path = "../asap-physical-operators" } +futures = "0.3" diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs new file mode 100644 index 00000000..e818796f --- /dev/null +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -0,0 +1,286 @@ +//! Maintenance -> stored pane state -> independently bound query execution. +mod physical_common; +use asap_physical_operators::{ + operators::{Operator, ReadoutQuery}, + physical_planner::{CompiledPhysicalDag, InputContract, Source}, + plan::{PhysicalDag, PhysicalOperator, PlanProperties}, + runtime::{Input, Limits, OutputStream, RunContext, Scope}, + summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, + values::{Batch, Schema, Value}, + AggregateCore, Error, +}; +use asap_types::{ + post_asap::{ + SketchAlgorithm, SketchKind, SketchParams, SketchQuery, SummaryFamilyType, SummaryField, + SummarySchema, + }, + pre_asap::DataType, +}; +use futures::{executor::block_on, StreamExt}; +use std::{ + collections::BTreeMap, + sync::{ + atomic::{AtomicUsize, Ordering}, + Arc, + }, +}; + +fn family(k: u32) -> SummaryFamilyType { + SummaryFamilyType::Sketch( + SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k }), + Default::default(), + ) +} +fn raw_schema() -> Schema { + Arc::new(SummarySchema { + fields: vec![SummaryField { + name: "value".into(), + dtype: SummaryFamilyType::Plain(DataType::Float64), + nullable: false, + }], + time_index: None, + }) +} +fn query_scope() -> Scope { + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + } +} +fn fixture() -> (CompiledPhysicalDag, Operator, Schema) { + let raw = raw_schema(); + let build = Operator::summary_build(raw.clone(), family(200), 0, None, vec![]).unwrap(); + let state = build.schema(); + let maintenance = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(raw))]), + BTreeMap::from([(1, (vec![0], build))]), + vec![1], + ) + .unwrap(); + let merge = Operator::summary_merge(state.clone(), 0, vec![]).unwrap(); + (maintenance, merge, state) +} +fn pane_state(maintenance: &CompiledPhysicalDag, pane: i64) -> Arc { + // Twenty samples in each (start,end] one-minute pane; k=200 avoids + // compaction so quantiles and sample counts have deterministic oracles. + let raw = raw_schema(); + let rows = (0..20) + .map(|i| vec![Value::Float64((pane * 20 + i) as f64)]) + .collect(); + let state = physical_common::execute( + maintenance, + BTreeMap::from([(0, Batch::try_new(raw, rows).unwrap())]), + Scope::Ingestion { + window_start_ms: pane * 60_000, + window_end_ms: (pane + 1) * 60_000, + revision: 1, + }, + ); + let Value::Summary { state, .. } = &state[0][0].rows()[0][0] else { + panic!("missing KLL") + }; + state.clone() +} +fn restore(schema: Schema, states: &[Arc]) -> Batch { + Batch::try_new( + schema, + states + .iter() + .map(|state| { + vec![Value::Summary { + family: family(200), + state: state.clone(), + }] + }) + .collect(), + ) + .unwrap() +} +fn readout(schema: Schema, q: f64) -> Operator { + Operator::readout(schema, 0, ReadoutQuery::Sketch(SketchQuery::Quantile { q })).unwrap() +} +struct CountStarts { + operator: Operator, + starts: Arc, +} +impl PhysicalOperator for CountStarts { + fn name(&self) -> &str { + self.operator.name() + } + fn properties(&self, inputs: &[PlanProperties]) -> PlanProperties { + self.operator.properties(inputs) + } + fn requires_bounded_input(&self) -> bool { + self.operator.requires_bounded_input() + } + fn input_schemas(&self) -> Vec { + self.operator.input_schemas() + } + fn output_schema(&self) -> Schema { + self.operator.output_schema() + } + fn output_bytes(&self, batch: &Batch) -> usize { + self.operator.output_bytes(batch) + } + fn start<'a>( + &'a self, + inputs: Vec>, + context: RunContext, + ) -> Result, Error> { + self.starts.fetch_add(1, Ordering::SeqCst); + self.operator.start(inputs, context) + } +} + +/// Actual codec bytes survive destruction of maintenance state; one shared +/// native merge supplies p50, p99 and the population-count oracle per run. +#[test] +fn five_panes_roundtrip_and_shared_merge_runs_once() { + let (maintenance, merge, schema) = fixture(); + let panes: Vec<_> = (0..6).map(|pane| pane_state(&maintenance, pane)).collect(); + drop(maintenance); + let compiled = CompiledPhysicalDag::from_operators( + (0..5) + .map(|id| (id, InputContract::bounded(schema.clone()))) + .collect(), + BTreeMap::from([ + ( + 5, + ( + vec![0, 1, 2, 3, 4], + Operator::union(schema.clone(), 5).unwrap(), + ), + ), + (6, (vec![5], merge.clone())), + (7, (vec![6], readout(schema.clone(), 0.5))), + (8, (vec![6], readout(schema.clone(), 0.99))), + ]), + vec![6, 7, 8], + ) + .unwrap(); + let layout = asap_types::post_asap::PaneLayout { + pane_width_ms: 60_000, + pane_origin_ms: Some(0), + }; + assert!( + asap_types::post_asap::validate_pane_coverage( + &layout, + Some(330_000), + &asap_types::post_asap::WindowEdgeCoverage::PaneAligned + ) + .is_err(), + "moving window edges require residual computation" + ); + for offset in [0, 1] { + let restored = restore(schema.clone(), &panes[offset..offset + 5]); + let evaluation_time_ms = (5 + offset as i64) * 60_000; + asap_types::post_asap::validate_pane_coverage( + &layout, + Some(evaluation_time_ms), + &asap_types::post_asap::WindowEdgeCoverage::PaneAligned, + ) + .unwrap(); + let inputs: BTreeMap<_, _> = (0..5) + .map(|id| { + ( + id as u64, + restore(schema.clone(), &panes[offset + id..offset + id + 1]), + ) + }) + .collect(); + // A five-pane deployment cannot bind only four state slots. + let incomplete: BTreeMap<_, _> = inputs + .iter() + .take(4) + .map(|(&id, batch)| { + ( + id, + Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) + as Source<'_>, + ) + }) + .collect(); + assert!(compiled.instantiate(incomplete).is_err()); + let result = physical_common::execute( + &compiled, + inputs, + Scope::Query { + evaluation_time_ms, + revision: 1, + }, + ); + let Value::Summary { state, .. } = &result[0][0].rows()[0][0] else { + panic!("missing merged state") + }; + let kll = state + .as_any() + .downcast_ref::() + .unwrap(); + assert_eq!(kll.inner.count(), 100); + let value = |index: usize| match result[index][0].rows()[0][0] { + Value::Float64(value) => value, + _ => panic!("missing quantile"), + }; + assert!((value(1) - (50 + offset * 20) as f64).abs() <= 1.); + assert!((value(2) - (99 + offset * 20) as f64).abs() <= 1.); + let starts = Arc::new(AtomicUsize::new(0)); + let mut dag = PhysicalDag::default(); + dag.add( + 0, + vec![], + Operator::source(schema.clone(), vec![restored]).unwrap(), + ) + .unwrap(); + dag.add( + 1, + vec![0], + CountStarts { + operator: merge.clone(), + starts: starts.clone(), + }, + ) + .unwrap(); + dag.add(2, vec![1], readout(schema.clone(), 0.5)).unwrap(); + dag.add(3, vec![1], readout(schema.clone(), 0.99)).unwrap(); + let outputs = block_on(futures::future::join_all( + dag.execute( + &[2, 3], + RunContext::new(query_scope(), Limits::default()).unwrap(), + ) + .unwrap() + .into_iter() + .map(|stream| stream.collect::>()), + )); + assert!(outputs + .iter() + .all(|output| output.len() == 1 && output[0].is_ok())); + assert_eq!(starts.load(Ordering::SeqCst), 1); + } +} + +/// Relabelled KLL parameters and missing bindings fail explicitly. +#[test] +fn panes_reject_parameters_schema_and_missing_binding() { + let (_maintenance, merge, schema) = fixture(); + let wrong = Value::Summary { + family: family(200), + state: Arc::new(DatasketchesKLLAccumulator::new(128)), + }; + assert!(Batch::try_new(schema.clone(), vec![vec![wrong]]).is_err()); + let compiled = CompiledPhysicalDag::from_operators( + BTreeMap::from([(0, InputContract::bounded(schema))]), + BTreeMap::from([(1, (vec![0], merge))]), + vec![1], + ) + .unwrap(); + assert!(compiled.instantiate(BTreeMap::new()).is_err()); + let raw = raw_schema(); + let source = Operator::source( + raw.clone(), + vec![Batch::try_new(raw, vec![vec![Value::Float64(1.)]]).unwrap()], + ) + .unwrap(); + assert!(compiled + .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) + .is_err()); +} diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs new file mode 100644 index 00000000..aca4d4ef --- /dev/null +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -0,0 +1,42 @@ +use asap_physical_operators::{ + operators::Operator, + physical_planner::{CompiledPhysicalDag, Source}, + runtime::{Limits, RunContext, Scope}, + values::Batch, +}; +use futures::{executor::block_on, StreamExt}; +use std::collections::BTreeMap; + +pub fn execute( + plan: &CompiledPhysicalDag, + inputs: BTreeMap, + scope: Scope, +) -> Vec> { + let sources = inputs + .into_iter() + .map(|(id, batch)| { + ( + id, + Box::new(Operator::source(batch.schema().clone(), vec![batch]).unwrap()) + as Source<'_>, + ) + }) + .collect(); + let dag = plan.instantiate(sources).unwrap(); + block_on(async { + let streams = dag + .execute( + plan.roots(), + RunContext::new(scope, Limits::default()).unwrap(), + ) + .unwrap(); + futures::future::join_all(streams.into_iter().map(|mut stream| async move { + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push((*batch.unwrap()).clone()); + } + batches + })) + .await + }) +} diff --git a/crates/integration-tests/tests/sql_to_physical.rs b/crates/integration-tests/tests/sql_to_physical.rs new file mode 100644 index 00000000..be96107e --- /dev/null +++ b/crates/integration-tests/tests/sql_to_physical.rs @@ -0,0 +1,165 @@ +//! SQL frontend, candidate selection, physical compilation and fresh-run execution. +use asap_aware_mapping::{search_workload, DefaultCostModel}; +use asap_frontend_sql::{lower_sql, SqlCatalog}; +use asap_physical_operators::{ + physical_planner::{compile, InputContract, Source}, + runtime::{Limits, RunContext, Scope}, + sources::{DataSources, MemorySource}, + values::{Batch, Value}, +}; +use asap_types::{ + post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, + pre_asap::{Column, DataType, QueryExpr, Schema}, + types::AccuracyTarget, +}; +use futures::StreamExt; +use std::{collections::BTreeMap, rc::Rc, sync::Arc}; + +/// SQL filtering and grouped aggregation survive logical/physical lowering; +/// rebinding the compiled DAG runs against new data rather than cached results. +#[tokio::test] +async fn sql_filter_grouped_sum_executes_and_rebinds() { + let catalog = SqlCatalog::new().with_table( + "metrics", + Schema::new(vec![ + Column::new("service", DataType::Utf8, false), + Column::new("value", DataType::Float64, true), + ]), + ); + for query in [ + "SELECT service, SUM(value) AS total FROM metrics WHERE value > 1 GROUP BY service", + "SELECT service, SUM(value) AS total FROM metrics GROUP BY service", + ] { + let logical = Rc::new( + lower_sql(query, &catalog, AccuracyTarget::Exact) + .await + .unwrap(), + ); + let space = search_workload(vec![("sql", logical)]); + let selected = space + .global_selection(&DefaultCostModel) + .assemble_selected_dag(&space.roots[0].1) + .unwrap() + .unwrap(); + let dag = compile_post_asap_dag(&selected).unwrap(); + let scan = dag + .nodes + .iter() + .find(|node| { + matches!( + &node.payload, + PostAsapOperatorPayload::Fallback { + expression: QueryExpr::Scan { .. } + } + ) + }) + .expect("raw SQL scan"); + let schema = Arc::new(scan.output_schema.clone()); + assert!(schema + .fields + .iter() + .all(|field| matches!(field.dtype, SummaryFamilyType::Plain(_)))); + let plan = compile( + &dag, + BTreeMap::from([(u64::from(scan.id.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + ) + .unwrap(); + for multiplier in [1., 2.] { + let rows = [ + ("api", Some(2.)), + ("api", Some(3.)), + ("api", None), + ("batch", Some(4.)), + ("batch", Some(1.)), + ] + .into_iter() + .map(|(service, value)| { + schema + .fields + .iter() + .map(|field| match field.name.as_str() { + "service" => Value::Utf8(service.into()), + "value" => { + value.map_or(Value::Null, |value| Value::Float64(value * multiplier)) + } + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let PostAsapOperatorPayload::Fallback { expression } = &scan.payload else { + unreachable!() + }; + let QueryExpr::Scan { source, .. } = expression else { + unreachable!() + }; + let mut sources = DataSources::default(); + sources + .register( + source.clone(), + Arc::new( + MemorySource::new( + schema.clone(), + vec![Batch::try_new(schema.clone(), rows).unwrap()], + ) + .unwrap(), + ), + ) + .unwrap(); + let bound = plan + .instantiate(BTreeMap::from([( + u64::from(scan.id.0), + Box::new(sources.bind(expression).unwrap()) as Source<'_>, + )])) + .unwrap(); + let mut stream = bound + .execute( + plan.roots(), + RunContext::new( + Scope::Query { + evaluation_time_ms: 300_000, + revision: 1, + }, + Limits::default(), + ) + .unwrap(), + ) + .unwrap() + .remove(0); + let mut batches = Vec::new(); + while let Some(batch) = stream.next().await { + batches.push(batch.unwrap()); + } + let mut actual: Vec<_> = batches + .iter() + .flat_map(|batch| batch.rows()) + .map(|row| { + let Value::Utf8(service) = &row[0] else { + panic!("missing service") + }; + let Value::Float64(value) = row[1] else { + panic!("missing sum") + }; + (service.to_string(), value) + }) + .collect(); + actual.sort_by(|a, b| a.0.cmp(&b.0)); + assert_eq!( + actual, + vec![ + ("api".into(), 5. * multiplier), + ( + "batch".into(), + 4. * multiplier + + if query.contains("WHERE") && multiplier == 1. { + 0. + } else { + multiplier + } + ) + ] + ); + } + } +} diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index 186dd04b..d5406f23 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -119,52 +119,7 @@ fn dashboard_workload() -> PlanningWorkload { #[test] fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() { let workload = dashboard_workload(); - workload.validate().unwrap(); - - let lowered = lower_promql_workload(&workload, 0) - .expect("valid PromQL workload") - .into_iter() - .next() - .expect("one normalized workload entry"); - let root = Rc::new(lowered); - let strategies = asap_aware_mapping::default_strategies_with(&FullyCostedRuntime); - let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); - let target = Rc::clone(&space.roots[0].1); - let capabilities = SummaryMaintenanceLifecycleCapabilities { - supports_ephemeral: true, - supports_prepared: false, - supports_shared: false, - supports_continuously_maintained: true, - }; - - let selection = global_selection_with_summary_maintenance_lifecycles( - &space, - WorkloadDemand { - workload: &workload.query_workload, - data_workload: workload.data_workload.as_ref(), - entry_indices: &[1], - }, - NOW_MS, - Some(Horizon(100.0)), - capabilities, - &FullyCostedRuntime, - ) - .unwrap(); - let plan = assemble_selected_dag_with_summary_maintenance_lifecycles( - &selection, - &target, - WorkloadDemand::new_with_data( - &workload.query_workload, - workload.data_workload.as_ref().unwrap(), - &[1], - ), - NOW_MS, - Some(Horizon(100.0)), - capabilities, - &FullyCostedRuntime, - ) - .unwrap() - .expect("selected summary plan"); + let plan = selected_plan(&workload); assert!(!plan.selected_raw_recompute); assert_eq!(plan.expected_reads, Some(100.0)); @@ -231,3 +186,243 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() "continuously_maintained" ); } + +fn selected_plan( + workload: &PlanningWorkload, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + selected_plan_with_model(workload, &FullyCostedRuntime) +} + +fn selected_plan_with_model( + workload: &PlanningWorkload, + model: &dyn CostModel, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + selected_plan_with_horizon(workload, model, Horizon(100.)) +} + +fn selected_plan_with_horizon( + workload: &PlanningWorkload, + model: &dyn CostModel, + horizon: Horizon, +) -> asap_aware_mapping::SummaryMaintenanceLifecyclePlan { + workload.validate().unwrap(); + + let lowered = lower_promql_workload(workload, 0) + .expect("valid PromQL workload") + .into_iter() + .next() + .expect("one normalized workload entry"); + let root = Rc::new(lowered); + let strategies = asap_aware_mapping::default_strategies_with(model); + let space = search_workload_with(vec![("dashboard", Rc::clone(&root))], &strategies); + let target = Rc::clone(&space.roots[0].1); + let capabilities = SummaryMaintenanceLifecycleCapabilities { + supports_ephemeral: true, + supports_prepared: false, + supports_shared: false, + supports_continuously_maintained: true, + }; + + let selection = global_selection_with_summary_maintenance_lifecycles( + &space, + WorkloadDemand { + workload: &workload.query_workload, + data_workload: workload.data_workload.as_ref(), + entry_indices: &[1], + }, + NOW_MS, + Some(horizon), + capabilities, + model, + ) + .unwrap(); + assemble_selected_dag_with_summary_maintenance_lifecycles( + &selection, + &target, + WorkloadDemand::new_with_data( + &workload.query_workload, + workload.data_workload.as_ref().unwrap(), + &[1], + ), + NOW_MS, + Some(horizon), + capabilities, + model, + ) + .unwrap() + .expect("selected summary plan") +} + +mod physical_common; + +/// A selected continuous lifecycle supplies a materialization boundary; its +/// maintenance and query DAGs execute the selected KLL computation in fresh runs. +#[test] +fn continuous_lifecycle_compiles_and_executes_spatial_kll() { + use asap_physical_operators::{ + physical_planner::{compile_candidate, InputContract}, + runtime::Scope, + values::{Batch, Value}, + }; + use asap_types::{ + post_asap::{compile_post_asap_dag, PostAsapOperatorPayload, SummaryFamilyType}, + pre_asap::DataType, + }; + use std::{collections::BTreeMap, sync::Arc}; + let mut workload = dashboard_workload(); + workload.query_workload.query_batch.as_mut().unwrap()[0].query = + Query("quantile(0.99, latency)".into()); + workload.query_workload.repeating_queries.as_mut().unwrap()[0].query = + Query("quantile(0.99, latency)".into()); + let selected = selected_plan(&workload); + assert_eq!( + selected.deployments[0] + .summary_maintenance_lifecycle_guarantee + .as_ref() + .unwrap() + .summary_maintenance_lifecycle, + SummaryMaintenanceLifecycle::ContinuouslyMaintained + ); + let dag = compile_post_asap_dag(&selected.root).unwrap(); + let build = dag + .nodes + .iter() + .find(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) + .unwrap(); + let input = dag + .edges + .iter() + .find(|edge| edge.consumer == build.id) + .unwrap() + .producer; + let raw = dag.nodes.iter().find(|node| node.id == input).unwrap(); + let schema = Arc::new(raw.output_schema.clone()); + let candidate = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &[u64::from(build.id.0)], + ) + .unwrap(); + + // A continuous input without a finite pane boundary cannot implement this + // blocking builder. Retain lifecycle ownership in the candidate payload; + // only the legal bounded request candidate reaches workload pricing. + let mut unbounded = InputContract::bounded(schema.clone()); + unbounded.properties.boundedness = asap_physical_operators::plan::Boundedness::Unbounded; + let rejected = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), unbounded)]), + &[u64::from(dag.root.0)], + &[u64::from(build.id.0)], + ); + assert!(rejected.is_err()); + let request = compile_candidate( + &dag, + BTreeMap::from([(u64::from(input.0), InputContract::bounded(schema.clone()))]), + &[u64::from(dag.root.0)], + &[], + ) + .unwrap(); + let mut priced = 0; + let feedback = asap_physical_operators::physical_planner::select_candidate( + vec![ + rejected.map(|candidate| { + ( + SummaryMaintenanceLifecycle::ContinuouslyMaintained, + candidate, + ) + }), + Ok((SummaryMaintenanceLifecycle::Ephemeral, request)), + ], + |_| { + priced += 1; + Ok(Some( + asap_physical_operators::physical_planner::CandidateCost { + workload_scope: "dashboard".into(), + horizon_seconds: 100., + total_cost: 1000., + }, + )) + }, + ) + .unwrap(); + assert_eq!(priced, 1); + assert_eq!(feedback.candidate.0, SummaryMaintenanceLifecycle::Ephemeral); + for revision in [1, 2] { + let rows = (1..=100) + .map(|value| { + schema + .fields + .iter() + .map(|field| match field.dtype { + SummaryFamilyType::Plain(DataType::Float64) => { + Value::Float64(f64::from(value)) + } + SummaryFamilyType::Plain(DataType::Timestamp) => Value::Timestamp(300_000), + _ => panic!("unexpected field {field:?}"), + }) + .collect() + }) + .collect(); + let raw_batch = Batch::try_new(schema.clone(), rows).unwrap(); + let direct = physical_common::execute( + &feedback.candidate.1.query, + BTreeMap::from([(u64::from(input.0), raw_batch.clone())]), + Scope::Query { + evaluation_time_ms: 300_000, + revision, + }, + ); + let state = physical_common::execute( + candidate.precompute.as_ref().unwrap(), + BTreeMap::from([(u64::from(input.0), raw_batch)]), + Scope::Ingestion { + window_start_ms: 0, + window_end_ms: 300_000, + revision, + }, + ); + let result = physical_common::execute( + &candidate.query, + BTreeMap::from([(u64::from(build.id.0), state[0][0].clone())]), + Scope::Query { + evaluation_time_ms: 300_000, + revision, + }, + ); + let values: Vec<_> = result[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .collect(); + let direct_values: Vec<_> = direct[0] + .iter() + .flat_map(|batch| batch.rows()) + .flat_map(|row| row.iter()) + .filter_map(|value| { + if let Value::Float64(value) = value { + Some(*value) + } else { + None + } + }) + .collect(); + assert_eq!( + values, direct_values, + "maintenance and request candidates preserve the same population" + ); + assert_eq!(values.len(), 1); + assert!( + (98. ..=100.).contains(&values[0]), + "p99 rank must reflect the supplied population" + ); + } +} diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md new file mode 100644 index 00000000..fe71c8c3 --- /dev/null +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -0,0 +1,504 @@ +# Physical Planning, Summary Maintenance, and Deployment + +## 1. Architecture + +A Post-ASAP computation is progressively realized through four layers: + +```mermaid +flowchart LR + L["Logical Post-ASAP DAG
What computation?"] + M["Summary Maintenance Lifecycle
How is state maintained?"] + P["Physical DAG(s)
How is it executed?"] + D["Deployment Plan / DAG
How is it instantiated?"] + + L -->|"Summary Maintenance
Candidate Generation"| M + M -->|"Physical Plan
Compiler"| P + P -->|"Deployment Plan
Compiler"| D +``` + +| Layer | Defines | +| --- | --- | +| **Logical Post-ASAP DAG** | Computation semantics | +| **Summary Maintenance Lifecycle** | Build, retention, reuse, and window strategy | +| **Physical DAG(s)** | Supported physical candidates, executable operators and typed input boundaries | +| **Deployment Plan / DAG** | Selected candidate, concrete data/state bindings and operational lifecycle | + +ASAPPlanner owns the first three layers and the shared physical operator +implementation library. Deployment systems such as ASAPQuery and asap-fusion +own deployment compilation and operation. The lifecycle is a planning contract +associated with the logical DAG, not a separate computation IR. + +The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the +language-independent query semantics before summary selection. Both are +logical. Planning builds Post-ASAP `SummaryNode` trees; `compile_post_asap_dag` +exports the selected tree as a `PostAsapDag`, which is the Physical Plan +Compiler's input. Its per-node execution phase (ingestion or query time) is an +initial placement: compilation places ingestion-time nodes in the precompute DAG, +while frontier enumeration proposes alternative materialization splits. Which +layer owns placement is an open design question, deferred to a later change. + +### Candidate generation and deployment selection + +Planner exposes the supported, semantically legal **physical plan candidates**. +It does not discard a computation family or materialization placement merely +because a deployment-independent cost estimate prefers another candidate. +Logical candidates are an internal search stage, not the deployment handoff. + +```text +Query semantics + accuracy and lifecycle requirements + ↓ Planner +Supported Physical DAG candidates + typed inputs/outputs + requirements + ↓ backend +Binding feasibility + runtime statistics + resource limits + ERP + ↓ backend deployment compiler +Selected PrecomputePlan + QueryPlan + StoredOutputReferences +``` + +Planner owns operators, dependencies, sharing, and each candidate's +materialization frontier. The backend rejects candidates it cannot realize and +prices feasible candidates over a comparable workload and time horizon. It binds +the selected candidate; it does not lower the logical computation again, exchange +operators, or move an operator across the selected frontier. A missing quote is +not a zero-cost implementation. ERP evidence cannot authorize an illegal rewrite. + +The candidate inventory must identify its supported search scope and budget. +If a configured exhaustive enumeration exceeds its budget, planning fails +explicitly instead of selecting from an undisclosed partial inventory. Reports +separate unsupported compilation, deployment infeasibility, missing evidence, +and a feasible candidate that loses on cost. Absence is not a cost comparison. + +For `sum by(job)(rate(m[1m]))`, Rate remains per series before grouped Sum. +When lifecycle requirements permit it, a candidate may finalize Rate and Sum +within a bounded precompute run and persist the grouped value. Another may leave +those operators in the query DAG. Storing a value requires its exact evaluation +window, revision, readiness and serving cadence to match the query contract. + +For instant-vector TopK, CMS/CountSketch with a candidate heap requires explicit +series identity and a supported latest-value input protocol. Appending historical +sample values does not preserve instant-vector semantics. Replacement, rank +decrease, expiry, grouping and the required approximation guarantee must be +validated before admitting that physical candidate. + +This is the target ownership contract. A backend path that still reconstructs +operators from logical candidates has not completed this integration. + +### Input semantics and summary semantics + +`source`, `filter`, `grouping` and `window` describe input-data semantics: +where records originate, which records qualify, how they are grouped and which +time interval applies. They are not a complete description of arbitrary summary +computation. In particular, the same four fields can summarize different value +expressions or produce different states. + +| Concern | Required semantic information | +| --- | --- | +| Input computation | Source identities and schemas, filters, joins/transforms and their order, or a reference to the canonical input sub-DAG | +| Values and grouping | Value expressions, item identities and weights where applicable, group keys and types, and operation-defined null/duplicate handling | +| Time | Time column and interpretation, interval bounds, evaluation alignment, and distinction between query range and maintained panes | +| Summary computation | Exact operation or sketch family, algorithm and parameters, and supported build/merge behavior | +| Output | State versus finalized value, output schema/type, and readout parameters when part of the output computation | + +For example, KLL over `latency_seconds` and KLL over `log(latency_seconds)` differ +even with identical source, filter, grouping and window. Likewise, weighted +frequency state needs both item and weight expressions. More complex inputs +must retain their computation DAG; four descriptive fields cannot replace it. + +The canonical selected computation is authoritative. These categories describe +what must be preserved, not a new flat IR or a second expression language. +Operator-defined behavior should be referenced through its canonical contract, +not independently configured in deployment metadata. Unsupported or unresolved +semantics cannot be treated as compatible. + +Logical planning defines the semantics; physical compilation realizes them as +operators and typed boundaries. Deployment binds concrete readers and state +records that satisfy those requirements. A stored summary definition records or +references the relevant semantics for compatibility checks. Matching a definition +alone does not establish actual window coverage, revision compatibility or +readiness; those require runtime checks. Physical location, encoding, scheduling +and retention are separate execution/deployment contracts. + +Persisted semantic identity, its wire format and any tenant or dataset binding +belong to the deployment. Planner provides the typed `PostAsapDag` that a +deployment canonicalizes; it does not define a stored-definition format. + +### Running example + +Suppose p50 and p99 are requested over the same latency samples in a five-minute window, +and one Planner candidate uses KLL with `k=200`. Assume query windows align with one-minute +pane boundaries and that the selected parameters satisfy the required guarantees. +Operator names below are illustrative; the example defines the design, not a +claim that the entire deployment integration is implemented. + +The data source identifies where samples come from. Filters, grouping and the +window determine which samples enter each summary. Here `pane_duration: 1m` +means each stored pane covers one minute; the query range is five minutes. +Neither duration specifies how often maintenance runs or how long state is kept. + +The example evolves through the architecture as follows: + +```text +1. Logical Post-ASAP DAG + +raw latency + ↓ +KLLBuild(k=200) + ↓ +KLLMerge + ┌─┴─────┐ + ↓ ↓ + p50 p99 + + │ + │ Summary Maintenance Candidate Generation + ▼ + +2. Summary Maintenance Lifecycle + +KLLBuild(k=200) + strategy = continuously maintain + window = 1-minute panes + reuse = p50 + p99 + query = merge panes covering requested aligned 5 minutes + + │ + │ Physical Plan Compiler + ▼ + +3. Physical DAGs + +Precompute DAG: +RawInput + ↓ +NativeKllBuild(k=200) + ↓ +KllStateOutput + +Query DAG: +InputSlot[5 panes] + ↓ +NativeKllMerge(k=200) + ┌─┴────────┐ + ↓ ↓ + NativeP50 NativeP99 + + │ + │ Deployment Plan Compiler + ▼ + +4. Deployment Plan / DAG + +Precompute: +OTLP latency source + ↓ +run KLL build over each complete 1-minute input pane + ↓ +store as latency-kll-1m/ + +Query: +resolve five latency-kll-1m states + ↓ +execute query Physical DAG + ↓ +return p50 / p99 +``` + +Each stage adds a different class of decision while preserving the preceding +contracts. Here, continuous maintenance means recurring production of pane state; +the bounded build DAG does not itself implement an unbounded streaming window. + +## 2. Logical Post-ASAP DAG → Summary Maintenance Lifecycle + +The **Logical Post-ASAP DAG** defines computation semantics: + +```text +Scan(latency) + ↓ +KLLBuild(k=200) + ↓ +KLLMerge + ┌─┴────────────┐ + ↓ ↓ +Quantile(.5) Quantile(.99) +``` + +It establishes that KLL with `k=200` is used and that the merge is shared by the +two readouts. It does not determine when KLL states are built or retained. + +**Summary Maintenance Candidate Generation** enumerates legal lifecycle choices +using workload demand, window/freshness requirements and supported physical +implementations. Backend selection uses runtime feasibility and cost after +physical compilation. The following example follows one candidate. + +For the running example, assume it selects: + +```text +producer: KLLBuild(k=200) + +strategy: + continuously maintain + +window realization: + 1-minute panes + +query requirement: + combine panes covering the requested aligned 5-minute range + +reuse: + one merged state serves p50 and p99 +``` + +This produces the **Summary Maintenance Lifecycle**. + +The lifecycle specifies how the selected logical summary should be maintained, +but not its concrete operator implementation or storage location. + +Physical feasibility may feed back into selection. For example, if the required +pane-based maintenance cannot be implemented, this lifecycle candidate cannot be +selected. One-minute panes alone also cannot cover an arbitrarily phased query +window; that requires supported boundary handling or a different candidate. + +## 3. Summary Maintenance Lifecycle → Physical DAG + +The **Physical Plan Compiler** consumes both computation semantics and maintenance +requirements: + +```text +Logical Post-ASAP DAG (PostAsapDag) ++ Summary Maintenance Lifecycle ++ physical capabilities + ↓ +Physical Plan Compiler + ↓ +Physical DAG(s) +``` + +For the running example, the lifecycle creates two execution boundaries. + +These two halves are named as `PhysicalCandidate` names them, `precompute` +and `query`. *Maintenance* stays the lifecycle's word (section 2): it covers +how state is built, retained, reused and scheduled. A precompute DAG is the +physical object that a maintenance lifecycle compiles to, so reusing +*maintenance* for it collapses two layers that the crates keep apart: +`asap-aware-mapping::summary_maintenance_*` owns the lifecycle, and +`asap-physical-operators::physical_planner` owns the DAGs. + +### Precompute Physical DAG + +```text +RawInputSlot( + window = 1m, + bounded = true +) + ↓ +NativeKllBuild(k=200) + ↓ +KllStateOutput(k=200) +``` + +This DAG computes the state of one maintained one-minute pane. Its input +contract requires all input samples matching the source, filters and group within that pane; the deployment supplies that +bounded input from its source integration. + +### Query Physical DAG + +```text +InputSlot( + k = 200, + coverage = requested aligned 5m +) + ↓ +NativeKllMerge(k=200) + ┌─┴──────────────────┐ + ↓ ↓ +NativeQuantile(.50) NativeQuantile(.99) +``` + +The Physical Plan Compiler chooses `NativeKllBuild`, `NativeKllMerge`, and the +physical quantile implementations, validates state compatibility, and preserves +the shared merge. It also resolves expressions, schemas, ordered dependencies +and execution properties. + +The resulting Physical DAGs know that compatible KLL states are required, but +do not know where those states are stored. + +For example: + +```text +InputSlot +``` + +is physical, while: + +```text +s3://.../latency-kll/12:01 +``` + +is deployment-specific. Placement and scheduling also remain outside the Physical +DAG. If the required behavior cannot be realized, physical compilation fails. + +### Physical candidates include precompute computation + +Materialization frontiers are Planner decisions. A candidate records both the +precompute Physical DAG and the query Physical DAG, with typed outputs connecting +them. The deployment compiler binds those outputs; it does not move operators. + +For `sum by(job)(rate(m[1m]))`, legal physical candidates can include: + +```text +Candidate A: + precompute: compatible per-series counter states → per-series Rate + materialized output: per-series rate values for window/evaluation/revision + query: stored per-series rate values → grouped Sum + +Candidate B: + precompute: compatible per-series counter states → per-series Rate → grouped Sum + materialized output: grouped values for window/evaluation/revision + query: stored grouped values → result +``` + +Both preserve reset-aware Rate before Sum. Summing raw counters before Rate is +not equivalent. The counter-state build may be another precompute DAG; typed +state inputs do not imply that a deployment can construct or bind those states. + +The shared library exposes `physical_planner::compile_candidates(...)` to lower +explicit frontier candidates to `PhysicalCandidate { precompute, query, +materialized_outputs }`. `select_candidate(...)` accepts deployment feasibility +and scoped complete-workload costs and chooses the lowest-cost feasible +candidate. Costs must describe the same workload and planning horizon; missing +feasibility is rejected before pricing. The optimizer supplies candidate +frontiers and cost evidence, including updates, retention, recurrence and sharing. +`enumerate_frontiers` constructs bounded, reachable antichain frontiers above explicit input boundaries, including query-only and fully precomputed results. It fails explicitly when the candidate budget is exceeded. Maintenance selection must still reject frontiers that violate window, freshness, or reuse requirements; deployment feasibility is checked before pricing. + +Physical compilation opens no readers. Bounded precompute outputs become typed +query inputs. Their source, filters, grouping, build window, evaluation time, readiness and +revision contracts must accompany the selected lifecycle and be checked during +deployment binding. Type compatibility alone does not establish reuse legality. + +The Planner integration test executes both candidates through the shared runtime +and reverses the selected frontier with two controlled cost fixtures. It also +rejects shadowed/duplicate boundaries and incomparable planning horizons. This +establishes Planner capability; it does not establish that ASAPQuery currently +supports persisting every scalar/result-output frontier. + +## 4. Physical DAG → Deployment Plan / DAG + +The **Deployment Plan Compiler** binds the Physical DAGs to the concrete deployment: + +```text +Physical DAGs ++ Summary Maintenance Lifecycle ++ deployment catalog/state ++ sources/materializations ++ operational policy + ↓ +Deployment Plan Compiler + ↓ +Deployment Plan / DAG +``` + +For the precompute DAG, it may produce: + +```text +Source: + RawInputSlot + → complete bounded panes from the OTLP latency source + +Schedule: + each 1-minute pane, once its completion requirements are met + +Execution: + RawInput → NativeKllBuild(k=200) + +Output: + KllStateOutput + → latency-kll-1m/ +``` + +For a query over `(12:00, 12:05]`, its input-binding rule resolves: + +```text +InputSlot[5 panes] + ├── latency-kll-1m/(12:00,12:01] + ├── latency-kll-1m/(12:01,12:02] + ├── latency-kll-1m/(12:02,12:03] + ├── latency-kll-1m/(12:03,12:04] + └── latency-kll-1m/(12:04,12:05] + ↓ + Query Physical DAG + ↓ + p50, p99 +``` + +The Deployment Plan Compiler establishes bindings and checks that their contracts +satisfy the physical inputs and selected lifecycle, including KLL parameters, +source, filters, grouping, window coverage and revision scope. The deployment engine +resolves request-specific states and checks their actual coverage, revisions and +readiness at execution time. A compiled plan cannot establish future readiness. + +The compiler does not replace `NativeKllMerge`, choose another sketch, or decide +to maintain different windows. Such changes require replanning. A Deployment +Plan / DAG is an operational instantiation, not another computation IR. + +## 5. Responsibility Boundary + +The complete example makes the ownership boundary explicit: + +| Stage | KLL example decision | +| --- | --- | +| **Logical Post-ASAP DAG** | Use `KLL(k=200)` with shared merge for p50/p99 | +| **Summary Maintenance Candidate Generation** | Maintain 1-minute panes and reuse them for aligned five-minute queries | +| **Summary Maintenance Lifecycle** | Record pane/window/freshness/reuse requirements | +| **Physical Plan Compiler** | Lower to native KLL build, merge, and readout operators | +| **Physical DAG** | Define precompute and query DAGs with typed input/output boundaries | +| **Deployment Plan Compiler** | Bind raw input and KLL state slots to concrete sources/materializations | +| **Deployment Plan / DAG** | Specify maintenance schedules, stored-pane resolution and query execution | + +```text +Logical: + "Use KLL for p50/p99." + +Lifecycle: + "Maintain reusable 1-minute KLL panes." + +Physical: + "Execute NativeKllBuild and + NativeKllMerge → {p50, p99}." + +Deployment: + "Read OTLP here, store panes here, + and bind these five panes for this aligned query." +``` + +The deployment engine executes the bound Physical DAGs through ASAPPlanner's +shared physical operator implementation library, `asap-physical-operators`, and +its DAG runtime. The merge executes once per run for both consumers. Execution +does not introduce additional planning decisions. + +The physical layer does not own raw ingestion, pane construction or geometry, +storage formats, or decoding persisted bytes into typed state. It compiles +computation over typed input contracts: summary build, merge (for example KLL +merge), sketch estimates and exact finalization. The deployment constructs panes, +reads and decodes stored state, and binds the typed values to input slots. +Compiled physical plans are Planner outputs and keep their own serialized form. + +Each maintained pane contributes its input samples once. A replacement snapshot +replaces that pane's state; query merging must not count both the old and new +snapshots as separate inputs. + +## 6. Executable acceptance coverage + +The tests cover optimizer-selected lifecycle execution alongside independent +operator/runtime fixtures: + +| Test | Contract exercised | +| --- | --- | +| `summary_maintenance_lifecycle_e2e::continuous_lifecycle_compiles_and_executes_spatial_kll` | PromQL workload → selected continuous lifecycle → logical DAG → compiled precompute/query candidate → results in independent revisions; an unbounded candidate fails before pricing, and a bounded request candidate summarizes the same input samples | +| `kll_pane_execution::five_panes_roundtrip_and_shared_merge_runs_once` | Explicit one-minute precompute DAGs → real MessagePack state bytes → five required query inputs → shared native merge → p50/p99; counts every sample once, checks adjacent aligned windows and instruments one merge start per run | +| `kll_pane_execution::restored_panes_reject_corruption_parameters_schema_and_missing_binding` | Corrupt bytes, parameter relabelling, incompatible schemas and absent bindings fail explicitly | +| `precompute_candidates::grouped_rate_can_be_materialized_before_or_after_grouped_sum` | Cost changes select different legal precompute frontiers; both selected candidates execute with the same reset-sensitive result; uncompilable candidates are not priced | +| `sql_to_physical::sql_filter_grouped_sum_executes_and_rebinds` | SQL text → candidate search → physical compilation → shared Scan predicates and grouped summary execution; NULL samples are ignored and fresh bindings produce new results | + +Pane construction, pane timestamp checks, stored identity, revisions, readiness +and complete coverage of required input samples are deployment +responsibilities. Real storage and HTTP execution belong to +deployment-repository E2E tests. diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md new file mode 100644 index 00000000..aa1a52b6 --- /dev/null +++ b/docs/develop_docs/native-promql-inputs.md @@ -0,0 +1,55 @@ +# Native PromQL source rows + +Audience: source-adapter and physical-executor developers. + +A PromQL query only names some labels. Those columns cannot establish series +identity for Rate or TopK: two series with the same `job` may have different +unreferenced instance labels. + +`physical_planner::promql_rows::with_series_identity` resolves supported unary +PromQL computations to a bounded row representation before candidate search. +It appends `$promql_series_identity`, a non-null UTF-8 column containing the +canonical JSON encoding of the full label map. The name cannot collide with a +legal PromQL label. The resulting schema is closed over physical columns; the +label map remains dynamic and is not restricted to labels named in the query. +This realization rejects unsupported label rewriting, implicit vector matching, +and `without` operations rather than dropping hidden labels. + +Source adapters construct batches with `series_row`. Named label columns are +projections of the same complete identity; absent named labels project to empty +strings. `decode_series_identity` restores all labels on result conversion and +rejects noncanonical encodings. A query adapter must still apply the selected +operator's metric-name/result-label rules. Source selection, complete window +coverage and revision admission remain deployment responsibilities. + +Planner's maintained-population candidate recognizes this explicit identity +representation. Its TopK readout compiles automatically to `CurrentSeries`, +`Sort`, and `Limit`; deployment supplies the raw boundary or an already maintained +population boundary. Compilation does not open either source. + +The native `CurrentSeries` operator selects the latest sample per complete +identity in `(evaluation_time - lookback, evaluation_time]`. It removes stale +markers after selecting the latest sample, so an older value cannot reappear. +It rejects conflicting values at one series timestamp and emits the evaluation +timestamp. Each run builds a new snapshot; decreased values and expired series +cannot retain earlier heap weights. It reserves workspace and observes the +run's cancellation and byte budget. Precompute scopes must match the declared +lookback before any input is polled. + +CMS/CountSketch heap operators can consume this snapshot. CMS still requires +nonnegative weights; legal approximate TopK admission still requires the +Planner's accuracy/membership evidence. Executing a heap does not establish +that its result satisfies a query's accuracy requirements. + +Tests cover open-label Rate → CMS/CountSketch heaps, hidden-label round trips, +reset and zero-rate cases, snapshot replacement/decrease/expiry/staleness, +serialized physical recovery, and resource rejection. These are shared-library +tests, not proof of Backend candidate selection or durable deployment execution. + +Spatial heap candidates use the same complete series identity. Planner's +`current_series_topk_candidates` explores a CountSketch-with-heap realization +of canonical Sort/Limit under an explicit accuracy target. The physical graph +selects the latest eligible samples before building a fresh heap. A maintained +population boundary can supply that snapshot directly. Arbitrary signed metric +values do not authorize CMS; counter Rate's non-negative proof is separate. +These candidates still require membership/score evidence for deployment admission.